diff --git a/ggml/CMakeLists.txt b/ggml/CMakeLists.txt index b7110fa1..1df78fdd 100644 --- a/ggml/CMakeLists.txt +++ b/ggml/CMakeLists.txt @@ -4,8 +4,8 @@ project("ggml" C CXX ASM) ### GGML Version set(GGML_VERSION_MAJOR 0) -set(GGML_VERSION_MINOR 20) -set(GGML_VERSION_PATCH 2) +set(GGML_VERSION_MINOR 25) +set(GGML_VERSION_PATCH 1) set(GGML_VERSION_BASE "${GGML_VERSION_MAJOR}.${GGML_VERSION_MINOR}.${GGML_VERSION_PATCH}") list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake/") @@ -200,12 +200,12 @@ option(GGML_CUDA "ggml: use CUDA" option(GGML_MUSA "ggml: use MUSA" OFF) option(GGML_CUDA_FORCE_MMQ "ggml: use mmq kernels instead of cuBLAS" OFF) option(GGML_CUDA_FORCE_CUBLAS "ggml: always use cuBLAS instead of mmq kernels" OFF) -set (GGML_CUDA_PEER_MAX_BATCH_SIZE "128" CACHE STRING - "ggml: max. batch size for using peer access") option(GGML_CUDA_NO_PEER_COPY "ggml: do not use peer to peer copies" OFF) option(GGML_CUDA_NO_VMM "ggml: do not try to use CUDA VMM" OFF) option(GGML_CUDA_FA "ggml: compile ggml FlashAttention CUDA kernels" ON) option(GGML_CUDA_FA_ALL_QUANTS "ggml: compile all quants for FlashAttention" OFF) +set (GGML_CUDA_FA_QUANTS "q4_0-q4_0;q8_0-q8_0;f16-f16;bf16-bf16" CACHE STRING + "ggml: FlashAttention K-V type combinations to compile, \"all\" or a list such as \"q8_0-q8_0;q8_0-q4_0\"") option(GGML_CUDA_GRAPHS "ggml: use CUDA graphs (llama.cpp only)" ${GGML_CUDA_GRAPHS_DEFAULT}) option(GGML_CUDA_NCCL "ggml: use NVIDIA Collective Comm. Library" ON) set (GGML_CUDA_COMPRESSION_MODE "size" CACHE STRING @@ -242,7 +242,10 @@ option(GGML_METAL_EMBED_LIBRARY "ggml: embed Metal library" set (GGML_METAL_MACOSX_VERSION_MIN "" CACHE STRING "ggml: metal minimum macOS version") set (GGML_METAL_STD "" CACHE STRING "ggml: metal standard version (-std flag)") +set (GGML_METAL_TARGET_OS "macos" CACHE STRING + "ggml: metal -mtargetos OS name (macos, ios, xros, tvos)") option(GGML_OPENMP "ggml: use OpenMP" ON) +option(GGML_OPENMP_FETCH "ggml: fetch LLVM OpenMP" OFF) option(GGML_RPC "ggml: use RPC" OFF) option(GGML_SYCL "ggml: use SYCL" OFF) option(GGML_SYCL_F16 "ggml: use 16 bit floats for sycl calculations" OFF) @@ -341,9 +344,6 @@ set(GGML_PUBLIC_HEADERS include/gguf.h) set_target_properties(ggml PROPERTIES PUBLIC_HEADER "${GGML_PUBLIC_HEADERS}") -#if (GGML_METAL) -# set_target_properties(ggml PROPERTIES RESOURCE "${CMAKE_CURRENT_SOURCE_DIR}/src/ggml-metal.metal") -#endif() install(TARGETS ggml LIBRARY PUBLIC_HEADER) install(TARGETS ggml-base LIBRARY) @@ -406,10 +406,6 @@ write_basic_package_version_file( VERSION ${GGML_INSTALL_VERSION} COMPATIBILITY SameMajorVersion) -target_compile_definitions(ggml-base PRIVATE - GGML_VERSION="${GGML_INSTALL_VERSION}" - GGML_COMMIT="${GGML_BUILD_COMMIT}" -) message(STATUS "ggml version: ${GGML_INSTALL_VERSION}") message(STATUS "ggml commit: ${GGML_BUILD_COMMIT}") diff --git a/ggml/README.md b/ggml/README.md index 455c8125..3178f3a5 100644 --- a/ggml/README.md +++ b/ggml/README.md @@ -1,50 +1,49 @@ # ggml -[Manifesto](https://github.com/ggerganov/llama.cpp/discussions/205) +
-Tensor library for machine learning +ggml logo -***Note that this project is under active development. \ -Some of the development is currently happening in the [llama.cpp](https://github.com/ggerganov/llama.cpp) and [whisper.cpp](https://github.com/ggerganov/whisper.cpp) repos*** +Tensor library for machine learning -## Features +[![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://opensource.org/licenses/MIT) +[![Release](https://img.shields.io/github/v/release/ggml-org/ggml?filter=v*)](https://github.com/ggml-org/ggml/releases) +[![CI](https://github.com/ggml-org/ggml/actions/workflows/build-cpu.yml/badge.svg)](https://github.com/ggml-org/ggml/actions/workflows/build-cpu.yml) -- Low-level cross-platform implementation -- Integer quantization support -- Broad hardware support -- Automatic differentiation -- ADAM and L-BFGS optimizers -- No third-party dependencies -- Zero memory allocations during runtime +
+ +## Quick start -## Build +Build from source: ```bash git clone https://github.com/ggml-org/ggml cd ggml -# install python dependencies in a virtual environment -python3.10 -m venv .venv -source .venv/bin/activate -pip install -r requirements.txt - -# build the examples mkdir build && cd build cmake .. cmake --build . --config Release -j 8 ``` -## GPT inference (example) +For a minimal, fully commented example (matrix multiplication), see [examples/simple](examples/simple). -```bash -# run the GPT-2 small 117M model -../examples/gpt-2/download-ggml-model.sh 117M -./bin/gpt-2-backend -m models/gpt-2-117M/ggml-model.bin -p "This is an example" -``` +## Description -For more information, checkout the corresponding programs in the [examples](examples) folder. +The main goal of `ggml` is to be a simple, portable, and efficient tensor library for machine learning with minimal setup. -## Resources +- Plain C/C++ implementation without any dependencies +- Cross-platform - x86, ARM, RISC-V, LoongArch, PowerPC, s390x, and WebAssembly +- SIMD-optimized kernels for x86, ARM, and RISC-V +- Broad backend support - CPU, GPU, NPU, and browser +- 2- to 8-bit integer quantization, plus MXFP4 and NVFP4 microscaling formats +- Zero memory allocations during runtime + +## Documentation +- [The GGUF file format](docs/gguf.md) - [Introduction to ggml](https://huggingface.co/blog/introduction-to-ggml) -- [The GGUF file format](https://github.com/ggerganov/ggml/blob/master/docs/gguf.md) +- [GGML tips & tricks](https://github.com/ggml-org/llama.cpp/wiki/GGML-Tips-&-Tricks) + +## Contributing + +- For changes to the core `ggml` library (including to the CMake build system), please open a PR in [llama.cpp](https://github.com/ggml-org/llama.cpp) - doing so will make your PR more visible, better tested, and more likely to be reviewed diff --git a/ggml/UPSTREAM b/ggml/UPSTREAM index 75f5980b..20a32545 100644 --- a/ggml/UPSTREAM +++ b/ggml/UPSTREAM @@ -1,5 +1,5 @@ repo: git@github.com:ggml-org/ggml.git -sha: 8c63e70982c95ceb862e3a1073a2c1beef75d60a +sha: e565a8f4ce2e462c4973a24c51098dd3c81c0256 patches: patches/ggml/0001-fix-threadpool-oversubscription.patch diff --git a/ggml/cmake/common.cmake b/ggml/cmake/common.cmake index cb663883..25eff7a5 100644 --- a/ggml/cmake/common.cmake +++ b/ggml/cmake/common.cmake @@ -48,3 +48,74 @@ function(ggml_get_system_arch) set(GGML_SYSTEM_ARCH "UNKNOWN" PARENT_SCOPE) endif() endfunction() + +# Determines which FlashAttention vector kernel template instances to compile, returns them in OUT_SRCS. +function(ggml_cuda_fattn_vec_instances DIR OUT_SRCS) + set(FA_TYPES q4_0 q4_1 q5_0 q5_1 q8_0 bf16 f16) + + string(TOLOWER "${GGML_CUDA_FA_QUANTS}" FA_QUANTS) + string(STRIP "${FA_QUANTS}" FA_QUANTS) + if (GGML_CUDA_FA_ALL_QUANTS) + message(WARNING "GGML_CUDA_FA_ALL_QUANTS is deprecated, use GGML_CUDA_FA_QUANTS=all instead") + set(FA_QUANTS all) + endif() + if (NOT FA_QUANTS) + message(FATAL_ERROR "GGML_CUDA_FA_QUANTS must not be empty") + endif() + + if (FA_QUANTS STREQUAL "all") + set(FA_COMBINATIONS "") + foreach (TYPE_V IN LISTS FA_TYPES) + foreach (TYPE_K IN LISTS FA_TYPES) + list(APPEND FA_COMBINATIONS ${TYPE_K}-${TYPE_V}) + endforeach() + endforeach() + else() + set(FA_COMBINATIONS f16-f16) + + string(REPLACE "," ";" FA_SELECTED "${FA_QUANTS}") + foreach (COMBINATION IN LISTS FA_SELECTED) + string(STRIP "${COMBINATION}" COMBINATION) + if (NOT COMBINATION MATCHES "^([a-z0-9_]+)-([a-z0-9_]+)$") + message(FATAL_ERROR "GGML_CUDA_FA_QUANTS: \"${COMBINATION}\" is not \"all\" or a - combination") + endif() + set(TYPE_K ${CMAKE_MATCH_1}) + set(TYPE_V ${CMAKE_MATCH_2}) + foreach (TYPE ${TYPE_K} ${TYPE_V}) + if (NOT TYPE IN_LIST FA_TYPES) + message(FATAL_ERROR + "GGML_CUDA_FA_QUANTS: unknown type \"${TYPE}\" in \"${COMBINATION}\", must be one of: ${FA_TYPES}") + endif() + endforeach() + list(APPEND FA_COMBINATIONS ${TYPE_K}-${TYPE_V}) + endforeach() + endif() + list(REMOVE_DUPLICATES FA_COMBINATIONS) + + string(REPLACE ";" "," FA_QUANTS_DEFINE "${FA_QUANTS}") + add_compile_definitions(GGML_CUDA_FA_QUANTS="${FA_QUANTS_DEFINE}") + foreach (TYPE_V IN LISTS FA_TYPES) + foreach (TYPE_K IN LISTS FA_TYPES) + if ("${TYPE_K}-${TYPE_V}" IN_LIST FA_COMBINATIONS) + set(COMPILED 1) + else() + set(COMPILED 0) + endif() + string(TOUPPER "GGML_CUDA_FA_${TYPE_K}_${TYPE_V}" COMBINATION_DEF) + add_compile_definitions(${COMBINATION_DEF}=${COMPILED}) + endforeach() + endforeach() + + message(STATUS "FlashAttention K-V type combinations: ${FA_COMBINATIONS}") + + set(SRCS "") + foreach (COMBINATION IN LISTS FA_COMBINATIONS) + set(SRC "${DIR}/template-instances/fattn-vec-instance-${COMBINATION}.cu") + if (NOT EXISTS "${SRC}") + message(FATAL_ERROR "FlashAttention template instance \"${SRC}\" does not exist") + endif() + list(APPEND SRCS "${SRC}") + endforeach() + + set(${OUT_SRCS} ${SRCS} PARENT_SCOPE) +endfunction() diff --git a/ggml/cmake/ggml-config.cmake.in b/ggml/cmake/ggml-config.cmake.in index abe17804..a28e49e8 100644 --- a/ggml/cmake/ggml-config.cmake.in +++ b/ggml/cmake/ggml-config.cmake.in @@ -110,6 +110,16 @@ set_and_check(GGML_INCLUDE_DIR "@PACKAGE_GGML_INCLUDE_INSTALL_DIR@") set_and_check(GGML_LIB_DIR "@PACKAGE_GGML_LIB_INSTALL_DIR@") #set_and_check(GGML_BIN_DIR "@PACKAGE_GGML_BIN_INSTALL_DIR@") +if (NOT GGML_SHARED_LIB AND GGML_CPU_KLEIDIAI) + unset(KLEIDIAI_LIBRARY CACHE) + unset(KLEIDIAI_LIBRARY) + find_library(KLEIDIAI_LIBRARY kleidiai + REQUIRED + HINTS ${GGML_LIB_DIR} + NO_CMAKE_FIND_ROOT_PATH) + list(APPEND GGML_CPU_INTERFACE_LINK_LIBRARIES ${KLEIDIAI_LIBRARY}) +endif() + if(NOT TARGET ggml::ggml) find_package(Threads REQUIRED) diff --git a/ggml/examples/common.cpp b/ggml/examples/common.cpp index 8eb633e5..1ff8b43a 100644 --- a/ggml/examples/common.cpp +++ b/ggml/examples/common.cpp @@ -406,6 +406,7 @@ gpt_vocab::id gpt_sample_top_k_top_p( double temp, std::mt19937 & rng) { int n_logits = vocab.id_to_token.size(); + top_k = std::min(top_k, n_logits); std::vector> logits_id; logits_id.reserve(n_logits); @@ -491,6 +492,7 @@ gpt_vocab::id gpt_sample_top_k_top_p_repeat( std::mt19937 & rng) { int n_logits = vocab.id_to_token.size(); + top_k = std::min(top_k, n_logits); const auto * plogits = logits; diff --git a/ggml/include/ggml-rpc.h b/ggml/include/ggml-rpc.h index 276aea00..1f8cb790 100644 --- a/ggml/include/ggml-rpc.h +++ b/ggml/include/ggml-rpc.h @@ -6,7 +6,7 @@ extern "C" { #endif -#define RPC_PROTO_MAJOR_VERSION 5 +#define RPC_PROTO_MAJOR_VERSION 7 #define RPC_PROTO_MINOR_VERSION 0 #define RPC_PROTO_PATCH_VERSION 0 diff --git a/ggml/include/ggml-sycl.h b/ggml/include/ggml-sycl.h index 418a7ba9..1e353ffa 100644 --- a/ggml/include/ggml-sycl.h +++ b/ggml/include/ggml-sycl.h @@ -25,7 +25,7 @@ GGML_BACKEND_API bool ggml_backend_is_sycl(ggml_backend_t backend); GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_buffer_type(int device); // split tensor buffer that splits matrices by rows across multiple devices -GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(const float * tensor_split); +GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(int main_device, const float * tensor_split); // Tensor parallelism (--split-mode tensor): comm_init/free/allreduce_tensor // trio queried by the meta-backend via ggml_backend_reg_get_proc_address. @@ -36,6 +36,8 @@ GGML_BACKEND_API void ggml_backend_sycl_comm_free(void * comm_ctx); GGML_BACKEND_API bool ggml_backend_sycl_comm_allreduce_tensor(void * comm_ctx, struct ggml_tensor ** tensors); // pinned host buffer for use with the CPU backend for faster copies between CPU and GPU +// pins on device 0 - a copy between another device and this memory can fail, +// use ggml_backend_dev_host_buffer_type to pin on the device that does the copy GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type(void); GGML_BACKEND_API void ggml_backend_sycl_print_sycl_devices(void); @@ -43,7 +45,7 @@ GGML_BACKEND_API void ggml_backend_sycl_get_gpu_list(int *id_list, int max_len); GGML_BACKEND_API void ggml_backend_sycl_get_device_description(int device, char *description, size_t description_size); -GGML_BACKEND_API int ggml_backend_sycl_get_device_count(); +GGML_BACKEND_API int ggml_backend_sycl_get_device_count(void); GGML_BACKEND_API void ggml_backend_sycl_get_device_memory(int device, size_t *free, size_t *total); // SYCL doesn't support registering host memory, keep here for reference diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h index c2ccd972..224bdef9 100644 --- a/ggml/include/ggml.h +++ b/ggml/include/ggml.h @@ -433,10 +433,21 @@ extern "C" { GGML_TYPE_COUNT = 43, }; - // precision + // [TAG_GGML_PREC] + // this enum is used to declare the allowed numerical precision/data-types types that can be used during the compute of an op + // the declared types can be: + // - result accumulation type + // - source tensor data representation type + // - etc. + // the precision parameters are stored as ggml_tensor.op_params to the respective ops enum ggml_prec { - GGML_PREC_DEFAULT = 0, // stored as ggml_tensor.op_params, 0 by default - GGML_PREC_F32 = 10, + GGML_PREC_UNDEFINED = 0, + GGML_PREC_DEFAULT = 0, // note: deprecated, use GGML_PREC_UNDEFINED + GGML_PREC_F32 = 10, + GGML_PREC_BF16 = 15, + GGML_PREC_F16 = 20, + GGML_PREC_Q8 = 30, + GGML_PREC_Q4 = 40, }; // op hint @@ -627,6 +638,7 @@ extern "C" { GGML_GLU_OP_SWIGLU_OAI, GGML_GLU_OP_GEGLU_ERF, GGML_GLU_OP_GEGLU_QUICK, + GGML_GLU_OP_SWIGLU_CLAMP, GGML_GLU_OP_COUNT, }; @@ -1367,6 +1379,12 @@ extern "C" { float alpha, float limit); + GGML_API struct ggml_tensor * ggml_swiglu_clamp( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b, + float limit); + // normalize along rows GGML_API struct ggml_tensor * ggml_norm( struct ggml_context * ctx, @@ -1422,6 +1440,42 @@ extern "C" { struct ggml_tensor * b, float eps); + // [TAG_GGML_PREC] + // set the minimum required accumulator type for the implementation to use during the compute + // for example: + // - GGML_PREC_F32 - requires accumulation of the results in F32 + // - GGML_PREC_BF16 - can accumulate the results in BF16, F32 + // - GGML_PREC_F16 - can accumulate the results in F16, F32 + // - GGML_PREC_Q8 - not allowed + // - GGML_PREC_Q4 - not allowed + // + // return false on faliure + GGML_API bool ggml_prec_set_acc( + struct ggml_tensor * a, + enum ggml_prec prec); + + // [TAG_GGML_PREC] + // set the smallest rank that the implementation can use to internally convert the src[idx] data to + // ranks in decreasing order: + // - GGML_PREC_F32 - GGML_TYPE_F32 + // - GGML_PREC_BF16 - GGML_TYPE_BF16 + // - GGML_PREC_F16 - GGML_TYPE_F16, + // - GGML_PREC_Q8 - GGML_TYPE_Q8_0, GGML_TYPE_Q8_1, GGML_TYPE_Q8_K, etc. + // - GGML_PREC_Q4 - GGML_TYPE_Q4_0, GGML_TYPE_Q4_1, GGML_TYPE_Q4_K, GGML_TYPE_NVFP4, GGML_TYPE_MXFP4, etc. + // + // for example: + // - ggml_prec_set_src(a, GGML_PREC_Q8, 1): + // - allows the implementation to quantize F32, BF16, F16 data of src[1] down to GGML_TYPE_Q8_0 + // - cannot quantize it down to GGML_TYPE_Q4_0 or GGML_TYPE_NVFP4 + // - ggml_prec_set_src(a, GGML_PREC_Q4, 1): + // - allows the implementation to quantize F32, BF16, F16 data of src[1] down to 4-bit datatypes such as GGML_TYPE_Q4_K, GGML_TYPE_NVFP4 etc. + // + // return false on faliure + GGML_API bool ggml_prec_set_src( + struct ggml_tensor * a, + enum ggml_prec prec, + int idx); + // A: k columns, n rows => [ne03, ne02, n, k] // B: k columns, m rows (i.e. we transpose it internally) => [ne03 * x, ne02 * y, m, k] // result is n columns, m rows => [ne03 * x, ne02 * y, m, n] @@ -1432,9 +1486,10 @@ extern "C" { // change the precision of a matrix multiplication // set to GGML_PREC_F32 for higher precision (useful for phi-2) - GGML_API void ggml_mul_mat_set_prec( + GGML_DEPRECATED(GGML_API void ggml_mul_mat_set_prec( struct ggml_tensor * a, - enum ggml_prec prec); + enum ggml_prec prec), + "use ggml_prec_set_acc() instead"); // change the hint of a matrix multiplication GGML_API void ggml_mul_mat_set_hint( @@ -1724,6 +1779,19 @@ extern "C" { struct ggml_tensor * a, int n_past); + GGML_API struct ggml_tensor * ggml_clamp( + struct ggml_context * ctx, + struct ggml_tensor * a, + float min, + float max); + + // in-place, returns view(a) + GGML_API struct ggml_tensor * ggml_clamp_inplace( + struct ggml_context * ctx, + struct ggml_tensor * a, + float min, + float max); + GGML_API struct ggml_tensor * ggml_soft_max( struct ggml_context * ctx, struct ggml_tensor * a); @@ -1981,14 +2049,14 @@ extern "C" { float beta_fast, float beta_slow); - - // clamp - // in-place, returns view(a) - GGML_API struct ggml_tensor * ggml_clamp( - struct ggml_context * ctx, + // set the offset dims for RoPE + // a must be GGML_OP_ROPE or GGML_OP_ROPE_BACK + // vision RoPE is not supported + // example: (marking: x = rotated, 0 = unrotated) + // n_embd = 10, n_dims = 4, offset = 2 --> [00xxxx0000] + GGML_API struct ggml_tensor * ggml_rope_set_offset( struct ggml_tensor * a, - float min, - float max); + int n_offs); // im2col // converts data into a format that effectively results in a convolution when combined with matrix multiplication @@ -2426,13 +2494,20 @@ extern "C" { float max_bias, float logit_softcap); - GGML_API void ggml_flash_attn_ext_set_prec( + GGML_DEPRECATED(GGML_API void ggml_flash_attn_ext_set_prec( struct ggml_tensor * a, - enum ggml_prec prec); + enum ggml_prec prec), + "use ggml_prec_set_acc() instead"); GGML_API enum ggml_prec ggml_flash_attn_ext_get_prec( const struct ggml_tensor * a); + // Use finite mask entries as a sparse K/V set. Set 0 to disable. + // n_kv_max must bound the number of finite entries in every mask row. + GGML_API void ggml_flash_attn_ext_set_n_kv_max( + struct ggml_tensor * a, + int32_t n_kv_max); + GGML_API void ggml_flash_attn_ext_add_sinks( struct ggml_tensor * a, struct ggml_tensor * sinks); @@ -2628,11 +2703,21 @@ extern "C" { struct ggml_tensor * x, struct ggml_tensor * weights); + // hc_pre with a per-element gate (Qwen3.8-Flash-Next): gate [n_embd, hc, n_tokens] + // result[i, t] = scale*sum_h x[i, h, t]*sigmoid(gate[i, h, t]) + // + GGML_API struct ggml_tensor * ggml_dsv4_hc_pre_gated( + struct ggml_context * ctx, + struct ggml_tensor * x, + struct ggml_tensor * gate, + float scale); + // hc_post: x [n_embd, n_tokens], residual [n_embd, hc, n_tokens], // post [hc, n_tokens], comb [dst_hc, src_hc, n_tokens] // -> [n_embd, hc, n_tokens] // result[i, dst, t] = x[i, t]*post[dst, t] // + sum_src residual[i, src, t]*comb[dst, src, t] + // comb == NULL uses the identity: result[i, dst, t] = x[i, t]*post[dst, t] + residual[i, dst, t] // GGML_API struct ggml_tensor * ggml_dsv4_hc_post( struct ggml_context * ctx, diff --git a/ggml/scripts/make-release-checks.sh b/ggml/scripts/make-release-checks.sh new file mode 100755 index 00000000..c18cd998 --- /dev/null +++ b/ggml/scripts/make-release-checks.sh @@ -0,0 +1,96 @@ +#!/bin/bash +# Run all pre-release checks and determine the release version. +# +# Usage: make-release-checks.sh [--dry-run] +# --dry-run: warn on failures instead of aborting +# +# Env (when running in GitHub Actions): +# GH_TOKEN, GITHUB_REPOSITORY, GITHUB_OUTPUT +# RELEASE_BRANCH: when set, HEAD must belong to origin/RELEASE_BRANCH and must +# not be older than 3 days from the branch HEAD (skipped when unset) +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" + +DRY_RUN=false +CHECKS_PASSED=true +for arg in "$@"; do + case "$arg" in + --dry-run) DRY_RUN=true ;; + *) echo "Unknown argument: $arg"; exit 1 ;; + esac +done + +MAJOR=$(sed -n 's/^set(GGML_VERSION_MAJOR \([0-9][0-9]*\)).*/\1/p' "$REPO_ROOT/CMakeLists.txt") +MINOR=$(sed -n 's/^set(GGML_VERSION_MINOR \([0-9][0-9]*\)).*/\1/p' "$REPO_ROOT/CMakeLists.txt") +PATCH=$(sed -n 's/^set(GGML_VERSION_PATCH \([0-9][0-9]*\)).*/\1/p' "$REPO_ROOT/CMakeLists.txt") +VERSION="v${MAJOR}.${MINOR}.${PATCH}" +echo "Determined version: ${VERSION}" +if [[ -n "${GITHUB_OUTPUT:-}" ]]; then + echo "version=${VERSION}" >> "$GITHUB_OUTPUT" +fi + +SHA=$(git rev-parse HEAD) + +echo "Checking that commit ${SHA} belongs to the release branch..." +if [[ -z "${RELEASE_BRANCH:-}" ]]; then + echo "Warning: RELEASE_BRANCH not set - skipping commit check (local run)" +else + TIP="origin/${RELEASE_BRANCH}" + COMMIT_ERR="" + if ! git rev-parse --verify "${TIP}" >/dev/null 2>&1; then + COMMIT_ERR="branch ${RELEASE_BRANCH} not found on remote" + elif ! git merge-base --is-ancestor "${SHA}" "${TIP}"; then + COMMIT_ERR="commit ${SHA} is not part of branch ${RELEASE_BRANCH}" + else + COMMIT_TS=$(git show -s --format=%ct "${SHA}") + TIP_TS=$(git show -s --format=%ct "${TIP}") + AGE_DAYS=$(( (TIP_TS - COMMIT_TS) / 86400 )) + if (( TIP_TS - COMMIT_TS > 3 * 86400 )); then + COMMIT_ERR="commit ${SHA} is ${AGE_DAYS} day(s) older than the HEAD of ${RELEASE_BRANCH} (max: 3)" + fi + fi + if [[ -n "${COMMIT_ERR}" ]]; then + if [[ "$DRY_RUN" == "true" ]]; then + echo "Warning: ${COMMIT_ERR} (dry run, continuing)." + CHECKS_PASSED=false + else + echo "Error: ${COMMIT_ERR}" + exit 1 + fi + else + echo "Commit ${SHA} is on branch ${RELEASE_BRANCH} and within 3 days of its HEAD - OK" + fi +fi + +echo "Checking that tag ${VERSION} does not already exist..." +if git ls-remote --tags origin "${VERSION}" | grep -q "${VERSION}"; then + echo "Error: tag ${VERSION} already exists on remote" + exit 1 +fi +echo "Tag ${VERSION} does not exist on remote - OK" + +echo "Checking release.yml status for commit ${SHA}..." +if [[ -z "${GITHUB_REPOSITORY:-}" ]]; then + echo "Warning: GITHUB_REPOSITORY not set - skipping CI check (local run)" +else + RUNS=$(gh api "repos/${GITHUB_REPOSITORY}/actions/workflows/release.yml/runs?per_page=100" \ + --jq "[.workflow_runs[] | select(.head_sha == \"${SHA}\" and .conclusion == \"success\")] | length") + if [[ "$RUNS" -eq 0 ]]; then + if [[ "$DRY_RUN" == "true" ]]; then + echo "Warning: no successful release.yml run found for HEAD (${SHA}) (dry run, continuing)." + CHECKS_PASSED=false + else + echo "Error: no successful release.yml run found for HEAD (${SHA})" + echo "The release workflow must complete successfully before making a release." + exit 1 + fi + else + echo "Found successful release.yml run for HEAD." + fi +fi + +if [[ -n "${GITHUB_OUTPUT:-}" ]]; then + echo "checks_passed=${CHECKS_PASSED}" >> "$GITHUB_OUTPUT" +fi diff --git a/ggml/scripts/make-release-desc.sh b/ggml/scripts/make-release-desc.sh new file mode 100755 index 00000000..2a7b6af4 --- /dev/null +++ b/ggml/scripts/make-release-desc.sh @@ -0,0 +1,69 @@ +#!/bin/bash +# Generate the description of a release: the previous release version and +# the change log. +# +# Usage: make-release-desc.sh +# : current release version (v.., the leading v is optional) +# +# The previous version is the highest plain semver tag (v..) +# strictly below . The change log lists all commits between the +# previous version tag and the release commit, one line per commit. +# +# The release commit is the commit points at when the tag exists, +# HEAD otherwise. +# +# Env (when running in GitHub Actions): +# GITHUB_OUTPUT: previous_tag, changelog_title and changelog are written here +set -euo pipefail + +if [[ $# -ne 1 ]]; then + echo "Usage: $(basename "$0") " + exit 1 +fi +VERSION="$1" + +# Accept the version with or without the leading v, reject anything else +if [[ "${VERSION}" =~ ^[0-9]+\.[0-9]+\.[0-9]+$ ]]; then + VERSION="v${VERSION}" +elif [[ ! "${VERSION}" =~ ^v[0-9]+\.[0-9]+\.[0-9]+$ ]]; then + echo "Error: invalid version '${VERSION}' (expected v..)" + exit 1 +fi + +# Make sure all remote tags are available locally (skipped on local runs without origin) +if ! git fetch --tags origin 2>/dev/null; then + echo "Warning: could not fetch tags from origin (local run?)" +fi + +# Release commit: the commit points at when the tag exists, HEAD otherwise. +if ! RELEASE_COMMIT="$(git rev-parse -q --verify "refs/tags/${VERSION}^{commit}" 2>/dev/null)"; then + RELEASE_COMMIT="$(git rev-parse HEAD)" +fi + +echo "Release commit: $(git rev-parse --short "${RELEASE_COMMIT}")" + +PREV="$( { git tag --list; echo "${VERSION}"; } \ + | grep -E '^v[0-9]+\.[0-9]+\.[0-9]+$' \ + | sort -V \ + | awk -v cur="${VERSION}" '$0 == cur { exit } { prev = $0 } END { print prev }')" + +if [[ -n "${PREV}" ]]; then + CHANGELOG="$(git log --oneline "${PREV}..${RELEASE_COMMIT}")" + CHANGELOG_TITLE="Changelog since ${PREV}" +else + CHANGELOG="(no previous release tag found)" + CHANGELOG_TITLE="Changelog" +fi + +echo "Previous version: ${PREV:-none}" +echo "${CHANGELOG}" + +if [[ -n "${GITHUB_OUTPUT:-}" ]]; then + { + echo "previous_tag=${PREV}" + echo "changelog_title=${CHANGELOG_TITLE}" + echo "changelog<> "${GITHUB_OUTPUT}" +fi diff --git a/ggml/scripts/make-release-summary.txt b/ggml/scripts/make-release-summary.txt new file mode 100644 index 00000000..8db8545e --- /dev/null +++ b/ggml/scripts/make-release-summary.txt @@ -0,0 +1,35 @@ +Take a look at the changelog between the current version and the previous version - use the `./scripts/make-release-desc.sh [current-version]` to obtain it. + +Write a summary of the change log in a few sections: + +``` +## Overview + +[a single paragraph overview of all changes] + +### API changes (if applicable) + +[summarize the changes to `/include/*`] + +### Core changes (if applicable) + +[summarize the changes to `/src/ggml*` + +### Backend changes (if applicable) + +[summarize the changes to `/src/ggml-*`, organize in sub-sections per backend] + +``` + +For all changes, you can refer to the `llama.cpp` and `whisper.cpp` repositories for more information. For example, a change comming from `(llama/XXXX)` has more information in the respective PR in the `llama.cpp` repository. + +Guidelines: + +- All bullet point in the summary should be concise and rarely exceed a single line of 120 characters (excluding PR links) +- Avoid mentioning that the changes are synchronized - this is not relevant +- Avoid mentioning specific models - focus on generality of the ggml framework +- Replace `(llama/XXXX)` and `(whisper/XXXX)` with a link to the respective PR +- Combine related topics (a full list of commits will be appended independently at end of the summary) +- Skip minor-impact notes (e.g. "fix compile warnings", "refactored code", ...) + +Output just the summary in a markdown block, without any extra text. Save it to a local text file called `release-notes-vX.Y.Z.txt`. diff --git a/ggml/scripts/release.sh b/ggml/scripts/release.sh index e8fcea2c..91741e46 100755 --- a/ggml/scripts/release.sh +++ b/ggml/scripts/release.sh @@ -1,37 +1,24 @@ #!/bin/bash # -# Automated release script for ggml. +# Release preparation script for ggml. # -# Note: Sync from llama.cpp should be done separately via PR process -# prior to running this script. +# Bumps the version in CMakeLists.txt on a release candidate branch. +# The branch should then be pushed and a PR created, reviewed, and +# merged. After the PR is merged and the build-cpu workflow has +# completed successfully, the release is finalized by the make-release +# workflow (.github/workflows/make-release.yml), which creates the tag. # # Usage: -# ./scripts/release.sh prepare [major|minor|patch] [--dry-run] -# ./scripts/release.sh finalize [--dry-run] +# ./scripts/release.sh [major|minor|patch] [--dry-run] # -# Two-stage release process: +# Example: +# $ ./scripts/release.sh minor # -# Stage 1 - Prepare: -# $ ./scripts/release.sh prepare minor -# This creates a release candidate branch with version bump and removes -dev suffix. -# The branch should then be manually pushed and a PR created, reviewed, and merged. -# -# Stage 2 - Finalize: -# $ ./scripts/release.sh finalize -# After the RC PR is merged, this reads the current version from CMakeLists.txt, -# creates the release tag, and prepares the next development cycle. -# -# Prepare stage: -# 1. Creates release candidate branch -# 2. Updates version and removes -dev suffix +# The script: +# 1. Creates a release candidate branch (ggml-rc-v..) +# 2. Bumps the version in CMakeLists.txt # 3. Commits the version bump # -# Finalize stage: -# 1. Reads current release version from CMakeLists.txt -# 2. Creates signed git tag on master -# 3. Adds -dev suffix back for next development cycle -# 4. Creates branch and commit for development version -# set -e @@ -41,61 +28,27 @@ if [ ! -f "CMakeLists.txt" ] || [ ! -d "scripts" ]; then fi # Parse command line arguments -COMMAND="" VERSION_TYPE="" DRY_RUN=false -# First argument should be the command -if [ $# -eq 0 ]; then - echo "Error: Missing command" - echo "Usage: $0 prepare [major|minor|patch] [--dry-run]" - echo " $0 finalize [--dry-run]" - exit 1 -fi - -COMMAND="$1" -shift - -# Parse remaining arguments for arg in "$@"; do case $arg in --dry-run) DRY_RUN=true ;; major|minor|patch) - if [ "$COMMAND" = "prepare" ]; then - VERSION_TYPE="$arg" - else - echo "Error: Version type only valid for 'prepare' command" - exit 1 - fi + VERSION_TYPE="$arg" ;; *) echo "Error: Unknown argument '$arg'" - echo "Usage: $0 prepare [major|minor|patch] [--dry-run]" - echo " $0 finalize [--dry-run]" + echo "Usage: $0 [major|minor|patch] [--dry-run]" exit 1 ;; esac done -# Validate command -if [[ ! "$COMMAND" =~ ^(prepare|finalize)$ ]]; then - echo "Error: Command must be 'prepare' or 'finalize'" - echo "Usage: $0 prepare [major|minor|patch] [--dry-run]" - echo " $0 finalize [--dry-run]" - exit 1 -fi - -# For prepare command, default to patch if no version type specified -if [ "$COMMAND" = "prepare" ]; then - VERSION_TYPE="${VERSION_TYPE:-patch}" - if [[ ! "$VERSION_TYPE" =~ ^(major|minor|patch)$ ]]; then - echo "Error: Version type must be 'major', 'minor', or 'patch'" - echo "Usage: $0 prepare [major|minor|patch] [--dry-run]" - exit 1 - fi -fi +# Default to patch if no version type specified +VERSION_TYPE="${VERSION_TYPE:-patch}" # Common validation functions check_git_status() { @@ -233,64 +186,10 @@ prepare_release() { echo "Next steps:" echo " • Push branch to remote: git push origin $RC_BRANCH" echo " • Create a Pull Request from $RC_BRANCH to master" - echo " • After PR is merged, run: ./scripts/release.sh finalize" - fi -} - -finalize_release() { - if [ "$DRY_RUN" = true ]; then - echo "[dry-run] Finalizing release (no changes will be made)" - else - echo "Starting release finalization..." - fi - echo "" - - check_git_status - check_master_branch - check_master_up_to_date - - # Read current version from CMakeLists.txt - echo "Step 1: Reading current release version..." - MAJOR=$(grep "set(GGML_VERSION_MAJOR" CMakeLists.txt | sed 's/.*MAJOR \([0-9]*\).*/\1/') - MINOR=$(grep "set(GGML_VERSION_MINOR" CMakeLists.txt | sed 's/.*MINOR \([0-9]*\).*/\1/') - PATCH=$(grep "set(GGML_VERSION_PATCH" CMakeLists.txt | sed 's/.*PATCH \([0-9]*\).*/\1/') - - RELEASE_VERSION="$MAJOR.$MINOR.$PATCH" - echo "Release version: $RELEASE_VERSION" - echo "" - - # Create git tag - echo "Step 2: Creating signed git tag..." - if [ "$DRY_RUN" = true ]; then - echo " [dry-run] Would create signed tag: v$RELEASE_VERSION with message 'Release version $RELEASE_VERSION'" - else - git tag -s "v$RELEASE_VERSION" -m "Release version $RELEASE_VERSION" - echo "✓ Created signed tag: v$RELEASE_VERSION" - fi - echo "" - - - echo "" - if [ "$DRY_RUN" = true ]; then - echo "[dry-run] Summary (no changes were made):" - echo " • Would have created tag: v$RELEASE_VERSION" - else - echo "Release finalization completed!" - echo "Summary:" - echo " • Created signed tag: v$RELEASE_VERSION" - echo "" - echo "Next steps:" - echo " • Push tag to remote: git push origin v$RELEASE_VERSION" - echo " • The release is now complete!" + echo " • After the PR is merged and the build-cpu workflow has passed," + echo " create the release with the make-release workflow" + echo " (.github/workflows/make-release.yml)" fi } -# Execute the appropriate command -case $COMMAND in - prepare) - prepare_release - ;; - finalize) - finalize_release - ;; -esac +prepare_release diff --git a/ggml/scripts/sync-llama.last b/ggml/scripts/sync-llama.last index e3baac0a..dfd80514 100644 --- a/ggml/scripts/sync-llama.last +++ b/ggml/scripts/sync-llama.last @@ -1 +1 @@ -27e345b574dd8c8838e2c06e47699a3135f16ec9 +66fba63af1f4161052c33024d150cac31f46ff37 diff --git a/ggml/src/CMakeLists.txt b/ggml/src/CMakeLists.txt index 82e9480c..94773200 100644 --- a/ggml/src/CMakeLists.txt +++ b/ggml/src/CMakeLists.txt @@ -213,7 +213,9 @@ set_target_properties(ggml-base PROPERTIES SOVERSION ${GGML_VERSION_MAJOR} ) -target_include_directories(ggml-base PRIVATE .) +configure_file(ggml-version.h.in ${CMAKE_CURRENT_BINARY_DIR}/ggml-version.h @ONLY) + +target_include_directories(ggml-base PRIVATE . ${CMAKE_CURRENT_BINARY_DIR}) if (GGML_BACKEND_DL) target_compile_definitions(ggml-base PUBLIC GGML_BACKEND_DL) endif() @@ -222,9 +224,123 @@ if (GGML_SCHED_NO_REALLOC) target_compile_definitions(ggml-base PUBLIC GGML_SCHED_NO_REALLOC) endif() -if (GGML_OPENMP) +if (GGML_OPENMP_FETCH) + if (NOT GGML_OPENMP) + message(FATAL_ERROR "GGML_OPENMP_FETCH requires GGML_OPENMP") + elseif (NOT WIN32 OR NOT (CMAKE_C_COMPILER_ID MATCHES "Clang")) + message(FATAL_ERROR "GGML_OPENMP_FETCH currently requires Clang on Windows") + endif() + + set(GGML_OPENMP_LLVM_VERSION "20.1.8") + string(REGEX MATCH "^[0-9]+" GGML_OPENMP_LLVM_VERSION_MAJOR "${GGML_OPENMP_LLVM_VERSION}") + string(REGEX MATCH "^[0-9]+" GGML_OPENMP_COMPILER_VERSION_MAJOR "${CMAKE_C_COMPILER_VERSION}") + if (NOT GGML_OPENMP_COMPILER_VERSION_MAJOR STREQUAL GGML_OPENMP_LLVM_VERSION_MAJOR) + message(FATAL_ERROR "LLVM OpenMP ${GGML_OPENMP_LLVM_VERSION} requires Clang ${GGML_OPENMP_LLVM_VERSION_MAJOR}.x") + endif() + + string(TOLOWER "${CMAKE_SYSTEM_PROCESSOR}" GGML_OPENMP_SYSTEM_PROCESSOR) + if (GGML_OPENMP_SYSTEM_PROCESSOR MATCHES "^(amd64|x86_64)$") + set(GGML_OPENMP_ARCH "x64") + set(GGML_OPENMP_INSTALLER_SUFFIX "win64") + set(GGML_OPENMP_INSTALLER_SHA256 "3197846a2b19063687dd56e93e34cd941e3548d907f23a6131571321bdf9fe7b") + elseif (GGML_OPENMP_SYSTEM_PROCESSOR MATCHES "^(aarch64|arm64)$") + set(GGML_OPENMP_ARCH "arm64") + set(GGML_OPENMP_INSTALLER_SUFFIX "woa64") + set(GGML_OPENMP_INSTALLER_SHA256 "7c4ac97eb2ae6b960ca5f9caf3ff6124c8d2a18cc07a7840a4d2ea15537bad8e") + else() + message(FATAL_ERROR "GGML_OPENMP_FETCH does not support ${CMAKE_SYSTEM_PROCESSOR}") + endif() + + set(GGML_OPENMP_CACHE_DIR "${CMAKE_BINARY_DIR}/_deps") + set(GGML_OPENMP_ROOT "${GGML_OPENMP_CACHE_DIR}/llvm-openmp-${GGML_OPENMP_LLVM_VERSION}-${GGML_OPENMP_ARCH}") + set(GGML_OPENMP_LIBRARY "${GGML_OPENMP_ROOT}/lib/libomp.lib") + set(GGML_OPENMP_RUNTIME "${GGML_OPENMP_ROOT}/bin/libomp.dll") + set(GGML_OPENMP_HEADER "${GGML_OPENMP_ROOT}/include/omp.h") + set(GGML_OPENMP_LICENSE "${GGML_OPENMP_ROOT}/LICENSE.TXT") + set(GGML_OPENMP_LICENSE_SHA256 "fdad1758a9e1f9d5a81e18879b3406772115edc92c24bfa36b70c654f325e8e4") + + if (NOT EXISTS "${GGML_OPENMP_LIBRARY}" OR NOT EXISTS "${GGML_OPENMP_RUNTIME}" OR NOT EXISTS "${GGML_OPENMP_HEADER}") + find_program(GGML_OPENMP_7Z NAMES 7z 7zz 7za) + if (NOT GGML_OPENMP_7Z) + message(FATAL_ERROR "GGML_OPENMP_FETCH requires 7-Zip to extract the LLVM installer") + endif() + + set(GGML_OPENMP_INSTALLER "${GGML_OPENMP_ROOT}/LLVM-${GGML_OPENMP_LLVM_VERSION}-${GGML_OPENMP_INSTALLER_SUFFIX}.exe") + set(GGML_OPENMP_EXTRACT_DIR "${GGML_OPENMP_ROOT}/extract") + set(GGML_OPENMP_INSTALLER_URL "https://github.com/llvm/llvm-project/releases/download/llvmorg-${GGML_OPENMP_LLVM_VERSION}/LLVM-${GGML_OPENMP_LLVM_VERSION}-${GGML_OPENMP_INSTALLER_SUFFIX}.exe") + + file(MAKE_DIRECTORY "${GGML_OPENMP_EXTRACT_DIR}") + file(DOWNLOAD "${GGML_OPENMP_INSTALLER_URL}" "${GGML_OPENMP_INSTALLER}" + EXPECTED_HASH "SHA256=${GGML_OPENMP_INSTALLER_SHA256}" + SHOW_PROGRESS + STATUS GGML_OPENMP_DOWNLOAD_STATUS) + list(GET GGML_OPENMP_DOWNLOAD_STATUS 0 GGML_OPENMP_DOWNLOAD_RESULT) + if (NOT GGML_OPENMP_DOWNLOAD_RESULT EQUAL 0) + list(GET GGML_OPENMP_DOWNLOAD_STATUS 1 GGML_OPENMP_DOWNLOAD_ERROR) + message(FATAL_ERROR "Failed to download LLVM OpenMP: ${GGML_OPENMP_DOWNLOAD_ERROR}") + endif() + + execute_process( + COMMAND "${GGML_OPENMP_7Z}" e -y "-o${GGML_OPENMP_EXTRACT_DIR}" "${GGML_OPENMP_INSTALLER}" -r libomp.lib libomp.dll omp.h + RESULT_VARIABLE GGML_OPENMP_EXTRACT_RESULT + OUTPUT_QUIET) + if (NOT GGML_OPENMP_EXTRACT_RESULT EQUAL 0 OR + NOT EXISTS "${GGML_OPENMP_EXTRACT_DIR}/libomp.lib" OR + NOT EXISTS "${GGML_OPENMP_EXTRACT_DIR}/libomp.dll" OR + NOT EXISTS "${GGML_OPENMP_EXTRACT_DIR}/omp.h") + message(FATAL_ERROR "Failed to extract libomp from ${GGML_OPENMP_INSTALLER}") + endif() + + file(MAKE_DIRECTORY "${GGML_OPENMP_ROOT}/lib" "${GGML_OPENMP_ROOT}/bin" "${GGML_OPENMP_ROOT}/include") + file(COPY "${GGML_OPENMP_EXTRACT_DIR}/libomp.lib" DESTINATION "${GGML_OPENMP_ROOT}/lib") + file(COPY "${GGML_OPENMP_EXTRACT_DIR}/libomp.dll" DESTINATION "${GGML_OPENMP_ROOT}/bin") + file(COPY "${GGML_OPENMP_EXTRACT_DIR}/omp.h" DESTINATION "${GGML_OPENMP_ROOT}/include") + file(REMOVE_RECURSE "${GGML_OPENMP_INSTALLER}" "${GGML_OPENMP_EXTRACT_DIR}") + endif() + + # The NSIS installer embeds LLVM's general license in its UI but does not install it as a file; use OpenMP's license to include its additional notices. + if (EXISTS "${GGML_OPENMP_LICENSE}") + file(SHA256 "${GGML_OPENMP_LICENSE}" GGML_OPENMP_LICENSE_ACTUAL_SHA256) + endif() + if (NOT GGML_OPENMP_LICENSE_ACTUAL_SHA256 STREQUAL GGML_OPENMP_LICENSE_SHA256) + file(DOWNLOAD "https://raw.githubusercontent.com/llvm/llvm-project/llvmorg-${GGML_OPENMP_LLVM_VERSION}/openmp/LICENSE.TXT" "${GGML_OPENMP_LICENSE}" + EXPECTED_HASH "SHA256=${GGML_OPENMP_LICENSE_SHA256}") + endif() + + if (COMMAND license_add_file) + license_add_file("LLVM OpenMP" "${GGML_OPENMP_LICENSE}") + endif() + + add_library(ggml-openmp-c INTERFACE) + target_compile_options(ggml-openmp-c INTERFACE "$<$:-fopenmp=libomp>") + target_include_directories(ggml-openmp-c SYSTEM INTERFACE "${GGML_OPENMP_ROOT}/include") + target_link_libraries(ggml-openmp-c INTERFACE "${GGML_OPENMP_LIBRARY}") + + add_library(ggml-openmp-cxx INTERFACE) + target_compile_options(ggml-openmp-cxx INTERFACE "$<$:-fopenmp=libomp>") + target_include_directories(ggml-openmp-cxx SYSTEM INTERFACE "${GGML_OPENMP_ROOT}/include") + target_link_libraries(ggml-openmp-cxx INTERFACE "${GGML_OPENMP_LIBRARY}") + + set(GGML_OPENMP_RUNTIME_OUTPUT_DIR "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}") + if (CMAKE_CONFIGURATION_TYPES) + string(APPEND GGML_OPENMP_RUNTIME_OUTPUT_DIR "/$") + endif() + add_custom_target(ggml-openmp-runtime ALL + COMMAND ${CMAKE_COMMAND} -E make_directory "${GGML_OPENMP_RUNTIME_OUTPUT_DIR}" + COMMAND ${CMAKE_COMMAND} -E copy_if_different "${GGML_OPENMP_RUNTIME}" "${GGML_OPENMP_RUNTIME_OUTPUT_DIR}/libomp.dll" + COMMAND ${CMAKE_COMMAND} -E copy_if_different "${GGML_OPENMP_LICENSE}" "${GGML_OPENMP_RUNTIME_OUTPUT_DIR}/LICENSE-LLVM-OpenMP") + add_dependencies(ggml-base ggml-openmp-runtime) + install(FILES "${GGML_OPENMP_RUNTIME}" DESTINATION ${CMAKE_INSTALL_BINDIR}) + install(FILES "${GGML_OPENMP_LICENSE}" DESTINATION ${CMAKE_INSTALL_BINDIR} RENAME LICENSE-LLVM-OpenMP) + + set(GGML_OPENMP_TARGET_C ggml-openmp-c) + set(GGML_OPENMP_TARGET_CXX ggml-openmp-cxx) + set(GGML_OPENMP_ENABLED "ON" CACHE INTERNAL "") +elseif (GGML_OPENMP) find_package(OpenMP) if (OpenMP_FOUND) + set(GGML_OPENMP_TARGET_C OpenMP::OpenMP_C) + set(GGML_OPENMP_TARGET_CXX OpenMP::OpenMP_CXX) set(GGML_OPENMP_ENABLED "ON" CACHE INTERNAL "") else() set(GGML_OPENMP_ENABLED "OFF" CACHE INTERNAL "") @@ -236,7 +352,7 @@ endif() if (GGML_OPENMP_ENABLED) target_compile_definitions(ggml-base PRIVATE GGML_USE_OPENMP) - target_link_libraries(ggml-base PRIVATE OpenMP::OpenMP_C OpenMP::OpenMP_CXX) + target_link_libraries(ggml-base PRIVATE ${GGML_OPENMP_TARGET_C} ${GGML_OPENMP_TARGET_CXX}) endif() add_library(ggml diff --git a/ggml/src/ggml-alloc.c b/ggml/src/ggml-alloc.c index 3bda9abb..a71838ea 100644 --- a/ggml/src/ggml-alloc.c +++ b/ggml/src/ggml-alloc.c @@ -40,6 +40,7 @@ bool ggml_op_can_inplace(enum ggml_op op) { case GGML_OP_SILU_BACK: case GGML_OP_RMS_NORM: case GGML_OP_RMS_NORM_BACK: + case GGML_OP_CLAMP: case GGML_OP_SOFT_MAX: case GGML_OP_SOFT_MAX_BACK: return true; diff --git a/ggml/src/ggml-backend-impl.h b/ggml/src/ggml-backend-impl.h index 9c56ec30..ef05905c 100644 --- a/ggml/src/ggml-backend-impl.h +++ b/ggml/src/ggml-backend-impl.h @@ -34,6 +34,11 @@ extern "C" { void * context; }; + // [TAG_ALLOC_SIZE_EXPAND] + // returns true for ops that may require additional memory for fleeting data on some backends, + // i.e. the backend buffer type's get_alloc_size may return more than ggml_nbytes for the output tensor + GGML_API bool ggml_op_alloc_size_may_expand(enum ggml_op op); + // // Backend buffer // @@ -83,6 +88,7 @@ extern "C" { GGML_API ggml_backend_buffer_t ggml_backend_multi_buffer_alloc_buffer(ggml_backend_buffer_t * buffers, size_t n_buffers); GGML_API bool ggml_backend_buffer_is_multi_buffer(ggml_backend_buffer_t buffer); GGML_API void ggml_backend_multi_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage); + GGML_API void ggml_backend_meta_buffer_set_usage (ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage); // // Backend (meta) @@ -102,6 +108,16 @@ extern "C" { // Backend (stream) // + // passed to graph_optimize so the backend can add allocation dependencies: + // if the backend executes parts of the graph out of order (e.g. on concurrent streams), + // it must keep the affected tensors allocated until a node where execution is known to have joined + struct ggml_backend_graph_optimize_params { + // keep `tensor` allocated at least until `until` (a node of the same graph) has been computed + // can be called multiple times for the same tensor: the longest lifetime applies + void (*add_alloc_dep)(void * user_data, struct ggml_tensor * tensor, struct ggml_tensor * until); + void * user_data; + }; + struct ggml_backend_i { const char * (*get_name)(ggml_backend_t backend); @@ -136,7 +152,7 @@ extern "C" { void (*event_wait) (ggml_backend_t backend, ggml_backend_event_t event); // (optional) sort/optimize the nodes in the graph - void (*graph_optimize) (ggml_backend_t backend, struct ggml_cgraph * cgraph); + void (*graph_optimize) (ggml_backend_t backend, struct ggml_cgraph * cgraph, struct ggml_backend_graph_optimize_params * params); }; struct ggml_backend { diff --git a/ggml/src/ggml-backend-meta.cpp b/ggml/src/ggml-backend-meta.cpp index 7654ea1f..7c1c0b86 100644 --- a/ggml/src/ggml-backend-meta.cpp +++ b/ggml/src/ggml-backend-meta.cpp @@ -592,7 +592,18 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( GGML_ASSERT(split_states_equal(src_ss[0], src_ss[1])); return {assume_sync ? GGML_BACKEND_SPLIT_AXIS_MIRRORED : GGML_BACKEND_SPLIT_AXIS_PARTIAL, {0}, {1}, 1}; } - GGML_ABORT("fatal error"); + if (src_ss[0].axis == src_ss[1].axis && src_ss[0].axis >= GGML_BACKEND_SPLIT_AXIS_2 && + src_ss[0].axis < GGML_MAX_DIMS) { + GGML_ASSERT(split_states_equal(src_ss[0], src_ss[1])); + return src_ss[0]; + } + // batched matmul with the batches split across devices and a replicated activation + if (src_ss[0].axis >= GGML_BACKEND_SPLIT_AXIS_2 && src_ss[0].axis < GGML_MAX_DIMS && + src_ss[1].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED) { + return src_ss[0]; + } + GGML_ABORT("unsupported mul_mat split states: node=%s src0=%s axis=%d src1=%s axis=%d", + tensor->name, tensor->src[0]->name, (int) src_ss[0].axis, tensor->src[1]->name, (int) src_ss[1].axis); //return {GGML_BACKEND_SPLIT_AXIS_UNKNOWN, {0}, {1}, 1}; }; @@ -602,27 +613,40 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( case GGML_BACKEND_SPLIT_AXIS_1: case GGML_BACKEND_SPLIT_AXIS_2: case GGML_BACKEND_SPLIT_AXIS_3: { - GGML_ASSERT(src_ss[0].n_segments == 1); - if (src_ss[0].axis == ggml_n_dims(tensor->src[0]) - 1 && src_ss[0].nr[0] == 1) { - return {ggml_backend_meta_split_axis(ggml_n_dims(tensor) - 1), {0}, {1}, 1}; - } - int64_t base_ne_in = tensor->src[0]->ne[0]; - for (int dim = 1; dim <= src_ss[0].axis; dim++) { + int64_t base_ne_in = 1; + for (int dim = 0; dim <= src_ss[0].axis; dim++) { base_ne_in *= tensor->src[0]->ne[dim]; } - base_ne_in /= src_ss[0].nr[0]; + if (src_ss[0].n_segments == 1) { + base_ne_in /= src_ss[0].nr[0]; + if (src_ss[0].axis == ggml_n_dims(tensor->src[0]) - 1 && src_ss[0].nr[0] == 1) { + return {ggml_backend_meta_split_axis(ggml_n_dims(tensor) - 1), {0}, {1}, 1}; + } + if (src_ss[0].axis == GGML_BACKEND_SPLIT_AXIS_0 && tensor->ne[0] == tensor->src[0]->ne[0] && + tensor->ne[1] == 1 && src_ss[0].nr[0] == 1) { + bool complete_rows = true; + for (size_t j = 0; j < n_bufs; j++) { + const int64_t ne = src_ss[0].ne[j]; + complete_rows = complete_rows && (ne == 0 || ne == tensor->src[0]->ne[0]); + } + if (complete_rows) { + // Move a complete dim-0 split to the following singleton dimension. + return {GGML_BACKEND_SPLIT_AXIS_1, {0}, {1}, 1}; + } + } + } + // Reshape outputs use one segment; split-state propagation merges source segments. int64_t base_ne_out = 1; for (int dim = 0; dim < GGML_MAX_DIMS; dim++) { - const int64_t base_ne_out_next = base_ne_out *= tensor->ne[dim]; - if (base_ne_out_next % base_ne_in == 0) { - return {ggml_backend_meta_split_axis(dim), {0}, {uint32_t(base_ne_out_next/base_ne_in)}, 1}; + base_ne_out *= tensor->ne[dim]; + if (base_ne_out % base_ne_in == 0) { + return {ggml_backend_meta_split_axis(dim), {0}, {uint32_t(base_ne_out/base_ne_in)}, 1}; } - if (base_ne_out_next > base_ne_in) { + if (base_ne_out > base_ne_in) { GGML_ASSERT(src_ss[0].n_segments == 1); GGML_ASSERT(src_ss[0].nr[0] == 1); return {ggml_backend_meta_split_axis(dim), {0}, {1}, 1}; } - base_ne_out = base_ne_out_next; } GGML_ABORT("shape mismatch for %s", ggml_op_name(tensor->op)); } @@ -747,14 +771,33 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( }; auto handle_flash_attn_ext = [&](const std::vector & src_ss) -> ggml_backend_meta_split_state { - GGML_ASSERT( src_ss[0].axis == GGML_BACKEND_SPLIT_AXIS_2); - GGML_ASSERT( src_ss[1].axis == GGML_BACKEND_SPLIT_AXIS_2); - GGML_ASSERT( src_ss[2].axis == GGML_BACKEND_SPLIT_AXIS_2); - GGML_ASSERT(tensor->src[4] == nullptr || src_ss[3].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + GGML_ASSERT(tensor->src[3] == nullptr || src_ss[3].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + + if (src_ss[0].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED) { + GGML_ASSERT(src_ss[1].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + GGML_ASSERT(src_ss[2].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + GGML_ASSERT(tensor->src[4] == nullptr || src_ss[4].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + return {GGML_BACKEND_SPLIT_AXIS_MIRRORED, {0}, {1}, 1}; + } + + GGML_ASSERT(src_ss[0].axis == GGML_BACKEND_SPLIT_AXIS_2); + const bool kv_split = src_ss[1].axis == GGML_BACKEND_SPLIT_AXIS_2 && + src_ss[2].axis == GGML_BACKEND_SPLIT_AXIS_2; + const bool kv_mirrored = src_ss[1].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED && + src_ss[2].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED; + GGML_ASSERT(kv_split || kv_mirrored); GGML_ASSERT(tensor->src[4] == nullptr || src_ss[4].axis == GGML_BACKEND_SPLIT_AXIS_0); return {GGML_BACKEND_SPLIT_AXIS_1, {0}, {1}, 1}; }; + auto handle_lightning_indexer = [&]( + const std::vector & src_ss) -> ggml_backend_meta_split_state { + for (size_t i = 0; i < 4; i++) { + GGML_ASSERT(src_ss[i].axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + } + return {GGML_BACKEND_SPLIT_AXIS_MIRRORED, {0}, {1}, 1}; + }; + auto handle_ssm_conv = [&](const std::vector & src_ss) -> ggml_backend_meta_split_state { if (src_ss[0].axis == src_ss[1].axis) { if (src_ss[0].axis == GGML_BACKEND_SPLIT_AXIS_0) { @@ -792,7 +835,7 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( ggml_backend_dev_t dev = ggml_backend_buft_get_device(ggml_backend_buffer_get_type(tensor->buffer)); const ggml_backend_meta_device_context * dev_ctx = (const ggml_backend_meta_device_context *) dev->context; ggml_backend_meta_split_state ret = dev_ctx->get_split_state(tensor, dev_ctx->get_split_state_ud); - if (ret.axis >= 0 && ret.axis <= GGML_MAX_DIMS) { + if (ret.axis >= 0 && ret.axis < GGML_MAX_DIMS) { const int64_t granularity = ret.axis == GGML_BACKEND_SPLIT_AXIS_0 ? ggml_blck_size(tensor->type) : 1; int64_t ne_sum = 0; for (size_t s = 0; s < ret.n_segments; s++) { @@ -802,6 +845,9 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( } } GGML_ASSERT(ne_sum == tensor->ne[ret.axis]); + } else if (ret.axis == GGML_BACKEND_SPLIT_AXIS_PARTIAL) { + GGML_ASSERT(ret.n_segments == 1); + GGML_ASSERT(ret.nr[0] == 1); } return ret; } @@ -922,7 +968,7 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( split_state = handle_rope(src_ss); } break; case GGML_OP_ROPE_BACK: { - split_state = handle_generic(src_ss, /*scalar_only =*/ true); + split_state = handle_rope(src_ss); } break; case GGML_OP_CLAMP: { split_state = handle_generic(src_ss, /*scalar_only =*/ false); @@ -986,6 +1032,9 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( case GGML_OP_GATED_DELTA_NET: { split_state = handle_gated_delta_net(src_ss); } break; + case GGML_OP_LIGHTNING_INDEXER: { + split_state = handle_lightning_indexer(src_ss); + } break; case GGML_OP_DSV4_HC_COMB: case GGML_OP_DSV4_HC_PRE: case GGML_OP_DSV4_HC_POST: { @@ -1070,13 +1119,14 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( if (buf_ctx->debug > 0) { std::string srcs_info; for (size_t i = 0; i < GGML_MAX_SRC; i++) { - if (tensor->src[i] == nullptr) { + if (tensor->src[i] == nullptr || tensor->src[i] == tensor) { continue; } if (!srcs_info.empty()) { srcs_info += ", "; } - const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor->src[0], true); + const ggml_backend_meta_split_state split_state = + ggml_backend_meta_get_split_state(tensor->src[i], true); GGML_ASSERT(split_state.n_segments == 1); const char * axis_name = ggml_backend_meta_split_axis_name(split_state.axis); std::string ne_info; @@ -1118,7 +1168,6 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state( } static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(const struct ggml_tensor * tensor, bool assume_sync) { - GGML_ASSERT(ggml_backend_buffer_is_meta(tensor->buffer)); ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) tensor->buffer->context; return ggml_backend_meta_get_split_state(buf_ctx->get_simple_tensor_container(tensor), tensor, assume_sync); } @@ -1151,11 +1200,6 @@ static enum ggml_status ggml_backend_meta_buffer_init_tensor_impl(ggml_backend_m ggml_context * simple_ctx = stc.ctxs[j].get(); ggml_backend_buffer_t simple_buf = buf_ctx->bufs[j].get(); - if ((simple_buf != nullptr) && ggml_backend_buffer_is_multi_buffer(simple_buf)) { - // see https://github.com/ggml-org/llama.cpp/issues/22197 - GGML_ABORT("multi buffers are not supported by the meta backend"); - } - if (split_dim >= 0 && split_dim < GGML_MAX_DIMS) { // TODO: the following assert fails for llama-parallel even though the results are correct: // GGML_ASSERT(ggml_is_contiguously_allocated(tensor)); @@ -1203,13 +1247,31 @@ static enum ggml_status ggml_backend_meta_buffer_init_tensor_impl(ggml_backend_m } } } + // TODO: revisit once the graph allocator has been refactored, see https://github.com/ggml-org/llama.cpp/pull/25051#issuecomment-4842873396 + ggml_backend_buffer_t init_buf = simple_buf; if (t_ij->view_src != nullptr) { t_ij->data = (char *) t_ij->view_src->data + t_ij->view_offs; + // views inherit the source slice's concrete sub-buffer (issue 22197) + if (tensor->view_src != nullptr && ggml_backend_buffer_is_meta(tensor->view_src->buffer) + && t_ij->view_src->buffer != nullptr) { + t_ij->buffer = t_ij->view_src->buffer; + init_buf = t_ij->view_src->buffer; + } } else if (simple_buf != nullptr) { + if (ggml_backend_buffer_is_multi_buffer(simple_buf)) { + GGML_ABORT("multi buffers are not supported by the meta backend"); + } t_ij->data = (char *) ggml_backend_buffer_get_base(simple_buf) + size_t(tensor->data) - size_t(ggml_backend_buffer_get_base(tensor->buffer)); } - t_ij->extra = tensor->extra; + + if (init_buf) { + // the backend that owns the buffer will set .extra + ggml_backend_buffer_init_tensor(init_buf, t_ij); + } else { + t_ij->extra = tensor->extra; + } + for (int i = 0; i < GGML_MAX_SRC; i++) { t_ij->src[i] = tensor->src[i]; if (tensor->src[i] == tensor) { @@ -1255,6 +1317,108 @@ static enum ggml_status ggml_backend_meta_buffer_init_tensor(ggml_backend_buffer return ggml_backend_meta_buffer_init_tensor_impl(buf_ctx->get_simple_tensor_container(tensor), tensor); } +static void ggml_backend_meta_buffer_memset_tensor( + ggml_backend_buffer_t buffer, ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) { + const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); + const ggml_backend_meta_split_state split_state = + ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); + GGML_ASSERT(ggml_is_contiguous(tensor) || split_state.axis == GGML_BACKEND_SPLIT_AXIS_MIRRORED); + + if (split_state.n_segments != 1 || split_state.nr[0] != 1) { + GGML_ASSERT(split_state.axis >= 0 && split_state.axis < GGML_MAX_DIMS); + GGML_ASSERT(split_state.nr[0] != 0); + GGML_ASSERT(tensor->ne[3] == 1); + + std::vector simple_offsets(n_bufs, 0); + if (split_state.axis == GGML_BACKEND_SPLIT_AXIS_0) { + GGML_ASSERT(tensor->ne[2] == 1); + + const size_t row_stride = tensor->nb[1]; + GGML_ASSERT(offset % row_stride == 0); + GGML_ASSERT(size % row_stride == 0); + const int64_t row_start = offset / row_stride; + const int64_t row_count = size / row_stride; + GGML_ASSERT(row_start + row_count <= tensor->ne[1]); + + const int64_t blck_size = ggml_blck_size(tensor->type); + for (size_t s = 0; s < split_state.n_segments; s++) { + for (size_t r = 0; r < split_state.nr[s]; r++) { + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + GGML_ASSERT(split_state.ne[s*n_bufs + j] % blck_size == 0); + const size_t nbytes = split_state.ne[s*n_bufs + j]/blck_size * tensor->nb[0]; + for (int64_t row = 0; row < row_count; row++) { + ggml_backend_tensor_memset(simple_tensor, value, + simple_offsets[j] + (row_start + row)*simple_tensor->nb[1], nbytes); + } + simple_offsets[j] += nbytes; + } + } + } + return; + } + + GGML_ASSERT(split_state.axis == GGML_BACKEND_SPLIT_AXIS_1); + + const size_t row_stride = tensor->nb[2]; + GGML_ASSERT(offset % row_stride == 0); + GGML_ASSERT(size % row_stride == 0); + const int64_t row_start = offset / row_stride; + const int64_t row_count = size / row_stride; + GGML_ASSERT(row_start + row_count <= tensor->ne[2]); + + for (size_t s = 0; s < split_state.n_segments; s++) { + for (size_t r = 0; r < split_state.nr[s]; r++) { + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t nbytes = split_state.ne[s*n_bufs + j] * tensor->nb[1]; + for (int64_t row = 0; row < row_count; row++) { + ggml_backend_tensor_memset(simple_tensor, value, + simple_offsets[j] + (row_start + row)*simple_tensor->nb[2], nbytes); + } + simple_offsets[j] += nbytes; + } + } + } + return; + } + + switch (split_state.axis) { + case GGML_BACKEND_SPLIT_AXIS_0: + case GGML_BACKEND_SPLIT_AXIS_1: + case GGML_BACKEND_SPLIT_AXIS_2: { + const size_t chunk_size_full = tensor->nb[split_state.axis + 1]; + GGML_ASSERT(offset % chunk_size_full == 0); + GGML_ASSERT(size % chunk_size_full == 0); + const int64_t i_start = offset / chunk_size_full; + const int64_t i_stop = (offset + size) / chunk_size_full; + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + const size_t chunk_size = simple_tensor->nb[split_state.axis + 1]; + if (chunk_size == 0) { + continue; + } + for (int64_t i = i_start; i < i_stop; i++) { + ggml_backend_tensor_memset(simple_tensor, value, i*chunk_size, chunk_size); + } + } + } break; + case GGML_BACKEND_SPLIT_AXIS_PARTIAL: { + GGML_ASSERT(value == 0); + [[fallthrough]]; + } + case GGML_BACKEND_SPLIT_AXIS_MIRRORED: { + for (size_t j = 0; j < n_bufs; j++) { + ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); + ggml_backend_tensor_memset(simple_tensor, value, offset, size); + } + } break; + default: { + GGML_ABORT("fatal error"); + } + } +} + static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { const size_t n_bufs = ggml_backend_meta_buffer_n_bufs(buffer); const ggml_backend_meta_split_state split_state = ggml_backend_meta_get_split_state(tensor, /*assume_sync =*/ false); @@ -1352,15 +1516,29 @@ static void ggml_backend_meta_buffer_set_tensor(ggml_backend_buffer_t buffer, gg } break; case GGML_BACKEND_SPLIT_AXIS_PARTIAL: { GGML_ASSERT(tensor->type == GGML_TYPE_F32); - const int64_t ne = ggml_nelements(tensor); - std::vector tmp; - tmp.reserve(ne); - for (int64_t i = 0; i < ne; i++) { - tmp.push_back(((const float *) data)[i] / n_bufs); + GGML_ASSERT(offset % sizeof(float) == 0); + GGML_ASSERT(size % sizeof(float) == 0); + const size_t n_values = size / sizeof(float); + size_t n_contributors = 0; + for (size_t j = 0; j < n_bufs; j++) { + n_contributors += split_state.ne[j] != 0; + } + const bool has_contributor_mask = n_contributors != 0; + if (!has_contributor_mask) { + n_contributors = n_bufs; + } + std::vector tmp(n_values); + for (size_t i = 0; i < n_values; i++) { + tmp[i] = ((const float *) data)[i] / n_contributors; + } + std::vector zero; + if (has_contributor_mask) { + zero.resize(n_values, 0.0f); } for (size_t j = 0; j < n_bufs; j++) { ggml_tensor * simple_tensor = ggml_backend_meta_buffer_simple_tensor(tensor, j); - ggml_backend_tensor_set(simple_tensor, tmp.data(), offset, size); + const float * partial = has_contributor_mask && split_state.ne[j] == 0 ? zero.data() : tmp.data(); + ggml_backend_tensor_set(simple_tensor, partial, offset, size); } } break; default: { @@ -1488,7 +1666,7 @@ static const ggml_backend_buffer_i ggml_backend_meta_buffer_iface = { /* .free_buffer = */ ggml_backend_meta_buffer_free_buffer, /* .get_base = */ ggml_backend_meta_buffer_get_base, /* .init_tensor = */ ggml_backend_meta_buffer_init_tensor, - /* .memset_tensor = */ nullptr, // TODO implement + /* .memset_tensor = */ ggml_backend_meta_buffer_memset_tensor, /* .set_tensor = */ ggml_backend_meta_buffer_set_tensor, /* .get_tensor = */ ggml_backend_meta_buffer_get_tensor, /* .set_tensor_2d = */ nullptr, @@ -1502,6 +1680,16 @@ bool ggml_backend_buffer_is_meta(ggml_backend_buffer_t buf) { return buf != nullptr && buf->iface.free_buffer == ggml_backend_meta_buffer_iface.free_buffer; } +void ggml_backend_meta_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage) { + GGML_ASSERT(ggml_backend_buffer_is_meta(buffer)); + ggml_backend_meta_buffer_context * buf_ctx = (ggml_backend_meta_buffer_context *) buffer->context; + for (size_t i = 0; i < buf_ctx->bufs.size(); i++) { + if (buf_ctx->bufs[i]) { + ggml_backend_buffer_set_usage(buf_ctx->bufs[i].get(), usage); + } + } +} + static ggml_backend_buffer_t ggml_backend_meta_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) { const size_t n_simple_bufts = ggml_backend_meta_buft_n_bufts(buft); @@ -1841,7 +2029,7 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend, { // For MoE models it may make sense to delay the AllReduce in order to reduce I/O: - auto get_i_delayed = [&](const int i) -> int { + auto get_i_delayed_branch = [&](const int i) -> int { int id = i; // i_delayed int idr = i; // i_delayed return, last safe return value @@ -1941,6 +2129,62 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend, return idr; }; + // AllReduce(a) + AllReduce(b) == AllReduce(a + b) for independent partial branches. + auto get_i_delayed = [&](const int i) -> int { + const int i_delayed = get_i_delayed_branch(i); + ggml_tensor * node = cgraph->nodes[i_delayed]; + + if (ggml_node_get_use_count(cgraph, i_delayed) != 1) { + return i_delayed; + } + + for (int id = i_delayed + 1; id < cgraph->n_nodes; id++) { + ggml_tensor * next = cgraph->nodes[id]; + if (next->view_src == node) { + return i_delayed; + } + for (int s = 0; s < GGML_MAX_SRC; s++) { + if (next->src[s] == node) { + return i_delayed; + } + } + + if (next->view_src != nullptr && next->view_src->op == GGML_OP_NONE && ggml_backend_buffer_is_host(next->view_src->buffer)) { + continue; + } + if (ggml_backend_meta_get_split_state(next, false).axis != GGML_BACKEND_SPLIT_AXIS_PARTIAL) { + continue; + } + + const int i_other = id; + const int i_other_delayed = get_i_delayed_branch(i_other); + ggml_tensor * other = cgraph->nodes[i_other_delayed]; + if (ggml_node_get_use_count(cgraph, i_other_delayed) != 1 || i_other_delayed + 1 >= cgraph->n_nodes) { + return i_delayed; + } + + ggml_tensor * sum = cgraph->nodes[i_other_delayed + 1]; + if (sum->op != GGML_OP_ADD || + !ggml_are_same_shape(node, other) || node->type != other->type || sum->type != node->type || + !((sum->src[0] == node && sum->src[1] == other) || + (sum->src[0] == other && sum->src[1] == node)) || + ggml_backend_meta_get_split_state(sum, false).axis != GGML_BACKEND_SPLIT_AXIS_MIRRORED) { + return i_delayed; + } + + for (size_t j = 0; j < n_backends; j++) { + auto & bcj = backend_ctx->backend_configs[j]; + const bool compute = bcj.nodes[i]->flags & GGML_TENSOR_FLAG_COMPUTE; + const bool compute_other = bcj.nodes[i_other]->flags & GGML_TENSOR_FLAG_COMPUTE; + if (compute != compute_other) { + return i_delayed; + } + } + return i_other_delayed + 1; + } + return i_delayed; + }; + int i_start = 0; for (int i = 0; i < cgraph->n_nodes; i++) { ggml_tensor * node = cgraph->nodes[i]; diff --git a/ggml/src/ggml-backend-reg.cpp b/ggml/src/ggml-backend-reg.cpp index e5959467..1c18b82c 100644 --- a/ggml/src/ggml-backend-reg.cpp +++ b/ggml/src/ggml-backend-reg.cpp @@ -490,7 +490,13 @@ static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, #endif // default search paths: executable directory, current directory search_paths.push_back(get_executable_path()); - search_paths.push_back(fs::current_path()); + std::error_code cwd_ec; + const fs::path cwd = fs::current_path(cwd_ec); + if (cwd_ec) { + GGML_LOG_DEBUG("%s: current_path() failure, error-message: %s\n", __func__, cwd_ec.message().c_str()); + } else { + search_paths.push_back(cwd); + } } else { search_paths.push_back(fs::u8path(user_search_path)); } @@ -508,8 +514,14 @@ static ggml_backend_reg_t ggml_backend_load_best(const char * name, bool silent, } continue; } - fs::directory_iterator dir_it(search_path, fs::directory_options::skip_permission_denied); - for (const auto & entry : dir_it) { + std::error_code dir_ec; + fs::directory_iterator dir_it(search_path, fs::directory_options::skip_permission_denied, dir_ec); + if (dir_ec) { + GGML_LOG_DEBUG("%s: failed to enumerate %s: %s\n", __func__, path_str(search_path).c_str(), dir_ec.message().c_str()); + continue; + } + for (const fs::directory_iterator end; dir_it != end; dir_it.increment(dir_ec)) { + const auto & entry = *dir_it; if (entry.is_regular_file(ec)) { auto filename = entry.path().filename(); auto ext = entry.path().extension(); diff --git a/ggml/src/ggml-backend.cpp b/ggml/src/ggml-backend.cpp index f6fb9179..20bf9650 100644 --- a/ggml/src/ggml-backend.cpp +++ b/ggml/src/ggml-backend.cpp @@ -20,6 +20,7 @@ #include #include #include +#include #include #ifdef __APPLE__ @@ -64,6 +65,14 @@ size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, const s if (buft->iface.get_alloc_size) { size_t size = buft->iface.get_alloc_size(buft, tensor); assert(size >= ggml_nbytes(tensor)); + + // [TAG_ALLOC_SIZE_EXPAND] + // if you hit this assert, update ggml_backend_op_alloc_size_may_expand() accordingly + GGML_ASSERT(size <= ggml_nbytes(tensor) || + ggml_op_is_empty(tensor->op) || + ggml_is_quantized(tensor->type) || // [TAG_ALLOC_SIZE_EXPAND] + ggml_op_alloc_size_may_expand(tensor->op)); + return size; } return ggml_nbytes(tensor); @@ -182,6 +191,8 @@ void ggml_backend_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backe // FIXME: add a generic callback to the buffer interface if (ggml_backend_buffer_is_multi_buffer(buffer)) { ggml_backend_multi_buffer_set_usage(buffer, usage); + } else if (ggml_backend_buffer_is_meta(buffer)) { + ggml_backend_meta_buffer_set_usage(buffer, usage); } } @@ -556,10 +567,10 @@ void ggml_backend_event_wait(ggml_backend_t backend, ggml_backend_event_t event) backend->iface.event_wait(backend, event); } -static void ggml_backend_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * cgraph) { +static void ggml_backend_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * cgraph, struct ggml_backend_graph_optimize_params * params) { GGML_ASSERT(backend); if (backend->iface.graph_optimize != NULL) { - backend->iface.graph_optimize(backend, cgraph); + backend->iface.graph_optimize(backend, cgraph, params); } } @@ -838,7 +849,7 @@ static void ggml_backend_sched_split_inputs_grow(struct ggml_backend_sched_split int new_cap = GGML_SCHED_MAX_SPLIT_INPUTS; if (split->inputs_capacity > 0) { new_cap = 2*split->inputs_capacity; - GGML_LOG_WARN("%s: increasing split inputs capacity from %d to %d\n", __func__, split->inputs_capacity, new_cap); + GGML_LOG_DEBUG("%s: increasing split inputs capacity from %d to %d\n", __func__, split->inputs_capacity, new_cap); } auto * pnew = (struct ggml_tensor **) realloc((void *) split->inputs, new_cap * sizeof(struct ggml_tensor *)); if (pnew == NULL) { @@ -853,7 +864,7 @@ static void ggml_backend_sched_graph_inputs_grow(ggml_backend_sched_t sched) { int new_cap = GGML_SCHED_MAX_SPLIT_INPUTS; if (sched->graph_inputs_capacity > 0) { new_cap = 2*sched->graph_inputs_capacity; - GGML_LOG_WARN("%s: increasing graph inputs capacity from %d to %d\n", __func__, sched->graph_inputs_capacity, new_cap); + GGML_LOG_DEBUG("%s: increasing graph inputs capacity from %d to %d\n", __func__, sched->graph_inputs_capacity, new_cap); } auto * pnew = (struct ggml_tensor **) realloc((void *) sched->graph_inputs, new_cap * sizeof(struct ggml_tensor *)); if (pnew == NULL) { @@ -1327,17 +1338,6 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra break; } } - // check if the split has too many inputs - // FIXME: count the number of inputs instead of only checking when full - if (split->n_inputs >= split->inputs_capacity) { - const size_t id = hash_id(src); - int src_backend_id = sched->hv_tensor_backend_ids[id]; - bool supported = ggml_backend_sched_buffer_supported(sched, src, cur_backend_id); - if (src_backend_id != cur_backend_id && tensor_id_copy(id, cur_backend_id, 0) == NULL && !supported) { - need_new_split = true; - break; - } - } } } @@ -1439,11 +1439,40 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra sched->prev_leaf_backend_ids = tmp; } + // optimize the split graphs and collect the allocation dependencies added by the backends + // this needs to happen before we make graph_copy, so they are in sync + // TODO: this may create many small allocations in the scheduler, restructure to use a flat array + std::unordered_map> alloc_deps; + + struct ggml_backend_graph_optimize_params opt_params = { + /* .add_alloc_dep = */ [](void * user_data, ggml_tensor * tensor, ggml_tensor * until) { + auto & deps = *(std::unordered_map> *) user_data; + std::vector & keep = deps[until]; + if (std::find(keep.begin(), keep.end(), tensor) == keep.end()) { + keep.push_back(tensor); + } + }, + /* .user_data = */ &alloc_deps, + }; + + for (int i = 0; i < sched->n_splits; i++) { + struct ggml_backend_sched_split * split = &sched->splits[i]; + split->graph = ggml_graph_view(graph, split->i_start, split->i_end); + + ggml_backend_graph_optimize(sched->backends[split->backend_id], &split->graph, &opt_params); + } + + // each dep is added to graph_copy as a GGML_OP_NONE node with the kept tensors as srcs + int n_dep_nodes = 0; + for (const auto & it : alloc_deps) { + n_dep_nodes += (it.second.size() + GGML_MAX_SRC - 1) / GGML_MAX_SRC; + } + int total_inputs = sched->n_graph_inputs; for (int i = 0; i < sched->n_splits; i++) { total_inputs += sched->splits[i].n_inputs; } - int graph_size = std::max(graph->n_nodes, graph->n_leafs) + total_inputs * 2 * sched->n_copies; + int graph_size = std::max(graph->n_nodes, graph->n_leafs) + total_inputs * 2 * sched->n_copies + n_dep_nodes; // remember the actual graph_size for performing reallocation checks later [GGML_SCHED_DEBUG_REALLOC] sched->debug_prev_graph_size = sched->debug_graph_size; @@ -1461,13 +1490,10 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra struct ggml_cgraph * graph_copy = &sched->graph; + int n_dep_nodes_added = 0; + for (int i = 0; i < sched->n_splits; i++) { struct ggml_backend_sched_split * split = &sched->splits[i]; - split->graph = ggml_graph_view(graph, split->i_start, split->i_end); - - // Optimize this split of the graph. This needs to happen before we make graph_copy, - // so they are in sync. - ggml_backend_graph_optimize(sched->backends[split->backend_id], &split->graph); // add inputs to the graph copy so that they are allocated by ggml-alloc at the start of the split for (int j = 0; j < split->n_inputs; j++) { @@ -1492,9 +1518,32 @@ void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgra assert(graph_copy->size > graph_copy->n_nodes); sched->node_backend_ids[graph_copy->n_nodes] = tensor_backend_id(graph->nodes[j]); graph_copy->nodes[graph_copy->n_nodes++] = graph->nodes[j]; + + if (alloc_deps.empty()) { + continue; + } + + // add a dependency node so that the kept tensors are not freed before this node is computed + auto it = alloc_deps.find(graph->nodes[j]); + if (it != alloc_deps.end()) { + const std::vector & keep = it->second; + for (size_t k = 0; k < keep.size(); k += GGML_MAX_SRC) { + struct ggml_tensor * dep = ggml_view_tensor(sched->ctx, keep[k]); + for (size_t s = 0; s < GGML_MAX_SRC && k + s < keep.size(); s++) { + dep->src[s] = keep[k + s]; + } + assert(graph_copy->size > graph_copy->n_nodes); + sched->node_backend_ids[graph_copy->n_nodes] = split->backend_id; + graph_copy->nodes[graph_copy->n_nodes++] = dep; + n_dep_nodes_added++; + } + } } } + // a mismatch means a backend added a dep with an `until` tensor that is not a node of the optimized graph + GGML_ASSERT(n_dep_nodes_added == n_dep_nodes); + if (sched->n_copies > 1) { // add input copies as leafs so that they are allocated first for (int i = 0; i < sched->n_graph_inputs; i++) { @@ -1581,7 +1630,10 @@ static bool ggml_backend_sched_alloc_splits(ggml_backend_sched_t sched) { ggml_backend_synchronize(sched->backends[i]); } - ggml_gallocr_reserve_n(sched->galloc, &sched->graph, sched->node_backend_ids, sched->leaf_backend_ids); + if (!ggml_gallocr_reserve_n(sched->galloc, &sched->graph, sched->node_backend_ids, sched->leaf_backend_ids)) { + GGML_LOG_ERROR("%s: failed to reserve graph buffers\n", __func__); + return false; + } if (!ggml_gallocr_alloc_graph(sched->galloc, &sched->graph)) { GGML_LOG_ERROR("%s: failed to allocate graph\n", __func__); return false; @@ -1599,11 +1651,23 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s std::vector ids; std::vector used_ids; + int prev_backend_id = -1; + for (int split_id = 0; split_id < sched->n_splits; split_id++) { struct ggml_backend_sched_split * split = &splits[split_id]; int split_backend_id = split->backend_id; ggml_backend_t split_backend = sched->backends[split_backend_id]; + // ensure the previous split's async work has completed before we start + // this split, the allocator may have reused buffer regions across splits + if (split->n_inputs == 0 && prev_backend_id >= 0 && prev_backend_id != split_backend_id) { + if (sched->events[prev_backend_id][sched->cur_copy] != NULL) { + ggml_backend_event_synchronize(sched->events[prev_backend_id][sched->cur_copy]); + } else { + ggml_backend_synchronize(sched->backends[prev_backend_id]); + } + } + // copy the input tensors to the split backend for (int input_id = 0; input_id < split->n_inputs; input_id++) { ggml_backend_t input_backend = ggml_backend_sched_get_tensor_backend(sched, split->inputs[input_id]); @@ -1644,6 +1708,10 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s ggml_tensor * ids_tensor = node->src[2]; ggml_backend_t ids_backend = split_backend; + if (ggml_nelements(ids_tensor) == 0) { + continue; + } + // if the ids tensor is also an input of the split, it may not have been copied yet to the split backend // in that case, we use the original ids tensor for (int i = input_id + 1; i < split->n_inputs; i++) { @@ -1766,12 +1834,12 @@ static enum ggml_status ggml_backend_sched_compute_splits(ggml_backend_sched_t s } } - // record the event of this copy - if (split->n_inputs > 0) { - if (sched->events[split_backend_id][sched->cur_copy] != NULL) { - ggml_backend_event_record(sched->events[split_backend_id][sched->cur_copy], split_backend); - } + // record the event of this split + if (sched->events[split_backend_id][sched->cur_copy] != NULL) { + ggml_backend_event_record(sched->events[split_backend_id][sched->cur_copy], split_backend); } + + prev_backend_id = split_backend_id; } return GGML_STATUS_SUCCESS; @@ -2037,6 +2105,20 @@ ggml_backend_t ggml_backend_sched_get_tensor_backend(ggml_backend_sched_t sched, // utils +bool ggml_op_alloc_size_may_expand(enum ggml_op op) { + switch (op) { + case GGML_OP_FLASH_ATTN_EXT: + case GGML_OP_MUL_MAT: + case GGML_OP_MUL_MAT_ID: + case GGML_OP_CUMSUM: + case GGML_OP_ARGSORT: + case GGML_OP_TOP_K: + return true; + default: + return false; + } +} + enum ggml_status ggml_backend_view_init(struct ggml_tensor * tensor) { GGML_ASSERT(tensor); GGML_ASSERT(tensor->buffer == NULL); diff --git a/ggml/src/ggml-cann/aclnn_ops.cpp b/ggml/src/ggml-cann/aclnn_ops.cpp index 2dc0f409..902d2eda 100644 --- a/ggml/src/ggml-cann/aclnn_ops.cpp +++ b/ggml/src/ggml-cann/aclnn_ops.cpp @@ -211,6 +211,50 @@ void ggml_cann_swiglu(ggml_backend_cann_context & ctx, ggml_tensor * dst) { GGML_CANN_CALL_ACLNN_OP(ctx, SwiGlu, acl_src.get(), (int64_t)2, acl_dst.get()); } +void ggml_cann_swiglu_clamp(ggml_backend_cann_context & ctx, ggml_tensor * dst) { + ggml_tensor * src0 = dst->src[0]; + ggml_tensor * src1 = dst->src[1]; + + GGML_ASSERT(ggml_is_contiguous_1(src0)); + GGML_ASSERT(ggml_is_contiguous_1(dst)); + + const int32_t swapped = ggml_get_op_params_i32(dst, 1); + acl_tensor_ptr acl_gate; + acl_tensor_ptr acl_up; + if (src1) { + GGML_ASSERT(ggml_is_contiguous_1(src1)); + GGML_ASSERT(src0->type == src1->type); + acl_gate = ggml_cann_create_tensor(src0); + acl_up = ggml_cann_create_tensor(src1); + } else { + int64_t ne[] = { src0->ne[0] / 2, src0->ne[1], src0->ne[2], src0->ne[3] }; + size_t nb[] = { src0->nb[0], src0->nb[1], src0->nb[2], src0->nb[3] }; + acl_gate = ggml_cann_create_tensor(src0, ne, nb, GGML_MAX_DIMS, ACL_FORMAT_ND, 0); + acl_up = ggml_cann_create_tensor(src0, ne, nb, GGML_MAX_DIMS, ACL_FORMAT_ND, ne[0] * ggml_element_size(src0)); + if (swapped) { + std::swap(acl_gate, acl_up); + } + } + + ggml_cann_pool_alloc temp_alloc(ctx.pool(), ggml_nbytes(dst)); + acl_tensor_ptr acl_temp = ggml_cann_create_tensor(temp_alloc.get(), ggml_cann_type_mapping(dst->type), + ggml_element_size(dst), dst->ne, dst->nb, GGML_MAX_DIMS); + acl_tensor_ptr acl_dst = ggml_cann_create_tensor(dst); + + const float limit = ggml_get_op_params_f32(dst, 3); + float min_gate = -INFINITY; + float min_up = -limit; + float max_value = limit; + acl_scalar_ptr acl_min_gate = ggml_cann_create_scalar(&min_gate, ACL_FLOAT); + acl_scalar_ptr acl_min_up = ggml_cann_create_scalar(&min_up, ACL_FLOAT); + acl_scalar_ptr acl_limit = ggml_cann_create_scalar(&max_value, ACL_FLOAT); + + GGML_CANN_CALL_ACLNN_OP(ctx, Clamp, acl_gate.get(), acl_min_gate.get(), acl_limit.get(), acl_temp.get()); + GGML_CANN_CALL_ACLNN_OP(ctx, Silu, acl_temp.get(), acl_dst.get()); + GGML_CANN_CALL_ACLNN_OP(ctx, Clamp, acl_up.get(), acl_min_up.get(), acl_limit.get(), acl_temp.get()); + GGML_CANN_CALL_ACLNN_OP(ctx, InplaceMul, acl_dst.get(), acl_temp.get()); +} + // Fused GeGLU using aclnnGeGluV3: splits input along ne[0] (CANN last dim), // activates the LEFT half with GELU, multiplies by right half. // approximate: 0=tanh, 1=none(erf). activateLeft=true matches GGML convention. @@ -4433,4 +4477,3 @@ void ggml_cann_gated_linear_attn(ggml_backend_cann_context & ctx, ggml_tensor * } } } - diff --git a/ggml/src/ggml-cann/aclnn_ops.h b/ggml/src/ggml-cann/aclnn_ops.h index cdbf9260..678f4d65 100644 --- a/ggml/src/ggml-cann/aclnn_ops.h +++ b/ggml/src/ggml-cann/aclnn_ops.h @@ -76,6 +76,7 @@ void ggml_cann_repeat(ggml_backend_cann_context & ctx, ggml_tensor * dst); void ggml_cann_swiglu(ggml_backend_cann_context & ctx, ggml_tensor * dst); +void ggml_cann_swiglu_clamp(ggml_backend_cann_context & ctx, ggml_tensor * dst); void ggml_cann_geglu(ggml_backend_cann_context & ctx, ggml_tensor * dst, int64_t approximate); /** diff --git a/ggml/src/ggml-cann/ggml-cann.cpp b/ggml/src/ggml-cann/ggml-cann.cpp index ffa361af..c2745014 100644 --- a/ggml/src/ggml-cann/ggml-cann.cpp +++ b/ggml/src/ggml-cann/ggml-cann.cpp @@ -1872,6 +1872,9 @@ static bool ggml_cann_compute_forward(ggml_backend_cann_context & ctx, struct gg case GGML_GLU_OP_SWIGLU: ggml_cann_swiglu(ctx, dst); break; + case GGML_GLU_OP_SWIGLU_CLAMP: + ggml_cann_swiglu_clamp(ctx, dst); + break; case GGML_GLU_OP_GEGLU_QUICK: ggml_cann_geglu_quick(ctx, dst); break; @@ -2428,6 +2431,7 @@ static bool ggml_backend_cann_supports_op(ggml_backend_dev_t dev, const ggml_ten case GGML_GLU_OP_SWIGLU: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return true; default: return false; @@ -2534,6 +2538,9 @@ static bool ggml_backend_cann_supports_op(ggml_backend_dev_t dev, const ggml_ten } case GGML_OP_ROPE: { + if (((const int32_t *) op->op_params)[15] != 0) { + return false; // FIXME: support ggml_rope_set_offset + } if (op->src[0]->ne[0] > 896) { return false; } diff --git a/ggml/src/ggml-common.h b/ggml/src/ggml-common.h index 83f9118d..1dbbe326 100644 --- a/ggml/src/ggml-common.h +++ b/ggml/src/ggml-common.h @@ -1131,7 +1131,7 @@ GGML_TABLE_END() #define NGRID_IQ1S 2048 #define IQ1S_DELTA 0.125f #define IQ1M_DELTA 0.125f -#if defined(GGML_COMMON_IMPL_C) +#if defined(GGML_COMMON_IMPL_C) || defined(GGML_COMMON_IMPL_CPP) GGML_TABLE_BEGIN(uint64_t, iq1s_grid, NGRID_IQ1S) 0xffffffffffffffff, 0xffffffffffffff01, 0xffffffffffff0000, 0xffffffffffff01ff, 0xffffffffffff0101, 0xffffffffff00ff00, 0xffffffffff000000, 0xffffffffff01ffff, diff --git a/ggml/src/ggml-cpu/CMakeLists.txt b/ggml/src/ggml-cpu/CMakeLists.txt index 836bae4d..1c7338ee 100644 --- a/ggml/src/ggml-cpu/CMakeLists.txt +++ b/ggml/src/ggml-cpu/CMakeLists.txt @@ -31,6 +31,8 @@ function(ggml_add_cpu_backend_variant_impl tag_name) ggml-cpu/ggml-cpu.cpp ggml-cpu/repack.cpp ggml-cpu/repack.h + ggml-cpu/iqp.cpp + ggml-cpu/iqp.h ggml-cpu/hbm.cpp ggml-cpu/hbm.h ggml-cpu/quants.c @@ -74,7 +76,7 @@ function(ggml_add_cpu_backend_variant_impl tag_name) if (GGML_OPENMP_ENABLED) target_compile_definitions(${GGML_CPU_NAME} PRIVATE GGML_USE_OPENMP) - target_link_libraries(${GGML_CPU_NAME} PRIVATE OpenMP::OpenMP_C OpenMP::OpenMP_CXX) + target_link_libraries(${GGML_CPU_NAME} PRIVATE ${GGML_OPENMP_TARGET_C} ${GGML_OPENMP_TARGET_CXX}) endif() if (GGML_LLAMAFILE) @@ -453,12 +455,16 @@ function(ggml_add_cpu_backend_variant_impl tag_name) ggml-cpu/spacemit/repack.h ggml-cpu/spacemit/ime_env.cpp ggml-cpu/spacemit/ime_env.h - ggml-cpu/spacemit/ime1_kernels.cpp - ggml-cpu/spacemit/ime2_kernels.cpp ggml-cpu/spacemit/ime_kernels.h ggml-cpu/spacemit/rvv_kernels.cpp ggml-cpu/spacemit/rvv_kernels.h ) + if ("RISCV64_SPACEMIT_IME1" IN_LIST RISCV64_SPACEMIT_IME_SPEC) + list(APPEND GGML_CPU_SOURCES ggml-cpu/spacemit/ime1_kernels.cpp) + endif() + if ("RISCV64_SPACEMIT_IME2" IN_LIST RISCV64_SPACEMIT_IME_SPEC) + list(APPEND GGML_CPU_SOURCES ggml-cpu/spacemit/ime2_kernels.cpp) + endif() endif() if(NOT GGML_CPU_ALL_VARIANTS) set(MARCH_STR "rv64gc") @@ -514,7 +520,9 @@ function(ggml_add_cpu_backend_variant_impl tag_name) elseif (GGML_SYSTEM_ARCH STREQUAL "s390x") message(STATUS "s390x detected") list(APPEND GGML_CPU_SOURCES - ggml-cpu/arch/s390/quants.c) + ggml-cpu/arch/s390/quants.c + ggml-cpu/arch/s390/repack.cpp + ) # for native compilation if (GGML_NATIVE) @@ -576,10 +584,25 @@ function(ggml_add_cpu_backend_variant_impl tag_name) endif() if (GGML_CPU_KLEIDIAI) - message(STATUS "Using KleidiAI optimized kernels if applicable") + # upstream repo requires at least cmake 3.16 + if (CMAKE_VERSION VERSION_LESS 3.16) + message(FATAL_ERROR "GGML_CPU_KLEIDIAI requires CMake >= 3.16") + endif() - # Disable the KleidiAI tests - set(KLEIDIAI_BUILD_TESTS OFF) + set(GGML_CPU_KLEIDIAI_AARCH64 OFF) + if (GGML_SYSTEM_ARCH STREQUAL "ARM" AND + (APPLE OR WIN32 OR CMAKE_SYSTEM_NAME MATCHES "^(Linux|Android)$") AND + (CMAKE_SYSTEM_PROCESSOR MATCHES "^(aarch64|arm64|ARM64|arm64-v8a)$" OR + CMAKE_OSX_ARCHITECTURES MATCHES "arm64" OR + CMAKE_GENERATOR_PLATFORM_LWR STREQUAL "arm64" OR + CMAKE_ANDROID_ARCH_ABI STREQUAL "arm64-v8a")) + set(GGML_CPU_KLEIDIAI_AARCH64 ON) + endif() + if (NOT GGML_CPU_KLEIDIAI_AARCH64) + message(FATAL_ERROR "GGML_CPU_KLEIDIAI requires a Linux, Android, Apple, or Windows AArch64/arm64 target") + endif() + + message(STATUS "Using KleidiAI optimized kernels if applicable") # Fetch KleidiAI sources: include(FetchContent) @@ -595,31 +618,49 @@ function(ggml_add_cpu_backend_variant_impl tag_name) list(APPEND KLEIDIAI_FETCH_ARGS DOWNLOAD_EXTRACT_TIMESTAMP NEW) endif() - if (CMAKE_VERSION VERSION_GREATER_EQUAL "3.28") - FetchContent_Declare(KleidiAI_Download - ${KLEIDIAI_FETCH_ARGS} - EXCLUDE_FROM_ALL - ) + FetchContent_Declare(kleidiai + ${KLEIDIAI_FETCH_ARGS} + ) - FetchContent_MakeAvailable(KleidiAI_Download) - FetchContent_GetProperties(KleidiAI_Download SOURCE_DIR KLEIDIAI_SRC) - else() - FetchContent_Declare(KleidiAI_Download - ${KLEIDIAI_FETCH_ARGS} - ) + # Disable tests and benchmark building + set(KLEIDIAI_BUILD_TESTS OFF CACHE BOOL "" FORCE) + set(KLEIDIAI_BUILD_BENCHMARK OFF CACHE BOOL "" FORCE) - FetchContent_GetProperties(KleidiAI_Download + # Use the Populate/add_subdirectory flow for compatibility with CMake 3.16. + FetchContent_GetProperties(kleidiai + SOURCE_DIR KLEIDIAI_SRC + BINARY_DIR KLEIDIAI_BIN + POPULATED KLEIDIAI_POPULATED + ) + if (NOT KLEIDIAI_POPULATED) + FetchContent_Populate(kleidiai) + FetchContent_GetProperties(kleidiai SOURCE_DIR KLEIDIAI_SRC - POPULATED KLEIDIAI_POPULATED + BINARY_DIR KLEIDIAI_BIN ) + endif() - if (NOT KLEIDIAI_POPULATED) - FetchContent_Populate(KleidiAI_Download) - FetchContent_GetProperties(KleidiAI_Download SOURCE_DIR KLEIDIAI_SRC) + if (NOT TARGET kleidiai) + add_subdirectory( + "${CMAKE_CURRENT_SOURCE_DIR}/ggml-cpu/kleidiai" + "${CMAKE_CURRENT_BINARY_DIR}/kleidiai-wrapper" + EXCLUDE_FROM_ALL + ) + if (NOT CMAKE_SKIP_INSTALL_RULES AND + (NOT DEFINED BUILD_SHARED_LIBS OR NOT BUILD_SHARED_LIBS)) + install(TARGETS kleidiai ARCHIVE) endif() endif() - add_compile_definitions(GGML_USE_CPU_KLEIDIAI) + if (NOT TARGET kleidiai) + message(FATAL_ERROR "KleidiAI target was not created") + endif() + + set_target_properties(kleidiai PROPERTIES POSITION_INDEPENDENT_CODE ON) + + target_link_libraries(${GGML_CPU_NAME} PRIVATE kleidiai) + + target_compile_definitions(${GGML_CPU_NAME} PRIVATE GGML_USE_CPU_KLEIDIAI) list(APPEND GGML_CPU_SOURCES ggml-cpu/kleidiai/kleidiai.cpp @@ -627,105 +668,6 @@ function(ggml_add_cpu_backend_variant_impl tag_name) ggml-cpu/kleidiai/kleidiai.h ggml-cpu/kleidiai/kernels.h ) - - # KleidiAI - include_directories( - ${KLEIDIAI_SRC}/ - ${KLEIDIAI_SRC}/kai/ - ${KLEIDIAI_SRC}/kai/ukernels/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_fp32_bf16p_bf16p/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f16p_qsi4c32p/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/ - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/) - - set(ARCH_FLAGS_TEMP "${ARCH_FLAGS}") - if (NOT ARCH_FLAGS_TEMP) - string(REGEX MATCH "-march=[^ ]+" ARCH_FLAGS_TEMP "${CMAKE_C_FLAGS}") - endif() - string(FIND "${ARCH_FLAGS_TEMP}" "+dotprod" DOTPROD_ENABLED) - string(FIND "${ARCH_FLAGS_TEMP}" "+i8mm" I8MM_ENABLED) - string(FIND "${ARCH_FLAGS_TEMP}" "+sme" SME_ENABLED) - string(FIND "${ARCH_FLAGS_TEMP}" "+sve" SVE_ENABLED) - - set(PRIVATE_ARCH_FLAGS ${ARCH_FLAGS_TEMP}) - - list(APPEND GGML_KLEIDIAI_SOURCES - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p_f32.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p4x8sb_f32_neon.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi4c32ps1s0scalef16_qsu4c32s16s0_neon.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p_f32_neon.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi4c32pscalef16_qsu4c32s16s0.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_quant_pack_qai8dxp_f32.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi8cxp_qsi8cx_neon.c) - - if (NOT DOTPROD_ENABLED MATCHES -1) - list(APPEND GGML_KLEIDIAI_SOURCES - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4x4_1x4_neon_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x4_qsi4c32p4x4_16x4_neon_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp4x4_qsi8cxp4x4_16x4_neon_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4x4_1x4_neon_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x8_qsi8cxp4x8_1x4_neon_dotprod.c) - endif() - - if (NOT I8MM_ENABLED MATCHES -1) - list(APPEND GGML_KLEIDIAI_SOURCES - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p4x8_16x4_neon_i8mm.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp4x8_qsi8cxp4x8_16x4_neon_i8mm.c) - endif() - - if (NOT SME_ENABLED MATCHES -1) - list(APPEND GGML_KLEIDIAI_SME_SOURCES - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme_mopa_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme_dot.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme_dot_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1b_2vlx2vl_sme_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1b_2vlx2vl_sme_mopa_asm.S) - set_source_files_properties(${GGML_KLEIDIAI_SME_SOURCES} - PROPERTIES COMPILE_OPTIONS "-fno-tree-vectorize;${ARCH_FLAGS_TEMP}+sve+sve2+sme") - list(APPEND GGML_CPU_SOURCES ${GGML_KLEIDIAI_SME_SOURCES}) - - list(APPEND GGML_KLEIDIAI_SME2_SOURCES - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4vlx4_1x4vl_sme2_sdot.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_fp32_bf16p_bf16p/kai_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_fp32_bf16p_bf16p/kai_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f16p_qsi4c32p/kai_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f16p_qsi4c32p/kai_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_pack_bf16p2vlx2_f32_sme.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_kxn_bf16p2vlx2b_f32_x32_sme.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_pack_f16pmrx2_f32_neon.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_pack_f32p2vlx1_f32_sme.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_lhs_pack_f32p2vlx1_f32_sme_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_nxk_f32p2vlx1biasf32_f32_f32_sme.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/pack/kai_rhs_pack_nxk_f32p2vlx1biasf32_f32_f32_sme_asm.S - ${KLEIDIAI_SRC}/kai/kai_common_sme_asm.S) - set_source_files_properties(${GGML_KLEIDIAI_SME2_SOURCES} - PROPERTIES COMPILE_OPTIONS "-fno-tree-vectorize;${ARCH_FLAGS_TEMP}+sve+sve2+sme2+fp16") - list(APPEND GGML_CPU_SOURCES ${GGML_KLEIDIAI_SME2_SOURCES}) - set(PRIVATE_ARCH_FLAGS "-fno-tree-vectorize;${PRIVATE_ARCH_FLAGS}") - endif() - - if (NOT SVE_ENABLED MATCHES -1) - list(APPEND GGML_KLEIDIAI_SOURCES - ${KLEIDIAI_SRC}/kai/kai_common_sve_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p8x8_1x8_sve_dotprod_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p8x8_1x8_sve_dotprod.c - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p8x8_16x8_sve_i8mm_asm.S - ${KLEIDIAI_SRC}/kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p8x8_16x8_sve_i8mm.c) - endif() - - set_source_files_properties(${GGML_KLEIDIAI_SOURCES} PROPERTIES COMPILE_OPTIONS "${PRIVATE_ARCH_FLAGS}") - list(APPEND GGML_CPU_SOURCES ${GGML_KLEIDIAI_SOURCES}) endif() message(STATUS "Adding CPU backend variant ${GGML_CPU_NAME}: ${ARCH_FLAGS} ${ARCH_DEFINITIONS}") @@ -737,8 +679,9 @@ function(ggml_add_cpu_backend_variant_impl tag_name) set_target_properties(${GGML_CPU_NAME} PROPERTIES COMPILE_FLAGS "-msimd128") endif() - if (CMAKE_CXX_COMPILER_ID STREQUAL "IntelLLVM") - # The compiler automatically enables "-ffast-math" which can cause NaNs in tests due to "-fassociative-math" - target_compile_options(${GGML_CPU_NAME} PRIVATE "-fno-associative-math") - endif() + if (CMAKE_C_COMPILER_ID STREQUAL "IntelLLVM" OR CMAKE_CXX_COMPILER_ID STREQUAL "IntelLLVM") + # The compiler automatically enables "-ffast-math" which can cause NaNs in tests due to "-fassociative-math" + target_compile_options(${GGML_CPU_NAME} PRIVATE "$<$,$>:$<$:/clang:>-fno-associative-math>") + endif() + endfunction() diff --git a/ggml/src/ggml-cpu/arch-fallback.h b/ggml/src/ggml-cpu/arch-fallback.h index 152e0bac..4dbd1982 100644 --- a/ggml/src/ggml-cpu/arch-fallback.h +++ b/ggml/src/ggml-cpu/arch-fallback.h @@ -39,6 +39,8 @@ #define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0 @@ -55,6 +57,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0 @@ -87,6 +91,8 @@ // repack.cpp #define ggml_quantize_mat_q8_0_4x4_generic ggml_quantize_mat_q8_0_4x4 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_K_8x4_q8_K_generic ggml_gemv_q4_K_8x4_q8_K @@ -98,6 +104,8 @@ #define ggml_gemv_mxfp4_4x4_q8_0_generic ggml_gemv_mxfp4_4x4_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_K_8x4_q8_K_generic ggml_gemm_q4_K_8x4_q8_K @@ -124,6 +132,8 @@ #define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0 @@ -140,6 +150,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0 @@ -171,6 +183,8 @@ #define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0 @@ -187,6 +201,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0 @@ -213,6 +229,8 @@ #define ggml_quantize_mat_q8_K_4x1_generic ggml_quantize_mat_q8_K_4x1 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q2_K_8x8_q8_K_generic ggml_gemv_q2_K_8x8_q8_K @@ -228,6 +246,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q2_K_8x8_q8_K_generic ggml_gemm_q2_K_8x8_q8_K @@ -247,7 +267,6 @@ // quants.c #define quantize_row_q8_K_generic quantize_row_q8_K #define ggml_vec_dot_nvfp4_q8_0_generic ggml_vec_dot_nvfp4_q8_0 -#define ggml_vec_dot_q1_0_q8_0_generic ggml_vec_dot_q1_0_q8_0 #define ggml_vec_dot_q2_0_q8_0_generic ggml_vec_dot_q2_0_q8_0 #define ggml_vec_dot_tq1_0_q8_K_generic ggml_vec_dot_tq1_0_q8_K #define ggml_vec_dot_tq2_0_q8_K_generic ggml_vec_dot_tq2_0_q8_K @@ -260,11 +279,11 @@ #define ggml_vec_dot_iq1_s_q8_K_generic ggml_vec_dot_iq1_s_q8_K #define ggml_vec_dot_iq1_m_q8_K_generic ggml_vec_dot_iq1_m_q8_K // repack.cpp -#define ggml_quantize_mat_q8_0_4x4_generic ggml_quantize_mat_q8_0_4x4 #define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 -#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0 #define ggml_gemv_q2_K_8x8_q8_K_generic ggml_gemv_q2_K_8x8_q8_K @@ -280,7 +299,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 -#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0 #define ggml_gemm_q2_K_8x8_q8_K_generic ggml_gemm_q2_K_8x8_q8_K @@ -318,6 +338,8 @@ #define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8 #define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4 #define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8 +#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0 +#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0 #define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0 #define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0 #define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0 @@ -334,6 +356,8 @@ #define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0 #define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0 #define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0 +#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0 +#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0 #define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0 #define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0 #define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0 diff --git a/ggml/src/ggml-cpu/arch/arm/repack.cpp b/ggml/src/ggml-cpu/arch/arm/repack.cpp index a7534443..ad0e5cca 100644 --- a/ggml/src/ggml-cpu/arch/arm/repack.cpp +++ b/ggml/src/ggml-cpu/arch/arm/repack.cpp @@ -48,6 +48,24 @@ static inline void decode_q_Kx8_6bit_scales(const uint8_t * scales_in, int16x8_t } #endif +#if defined(__aarch64__) && defined(__ARM_NEON) && (defined(__ARM_FEATURE_DOTPROD) || defined(__ARM_FEATURE_MATMUL_INT8)) +#define B1(c,s,n) 0x ## n ## c , 0x ## n ## s +#define B2(c,s,n) B1(c,s,n ## c), B1(c,s,n ## s) +#define B3(c,s,n) B2(c,s,n ## c), B2(c,s,n ## s) +#define B4(c,s,n) B3(c,s,n ## c), B3(c,s,n ## s) +#define B5(c,s,n) B4(c,s,n ## c), B4(c,s,n ## s) +#define B6(c,s,n) B5(c,s,n ## c), B5(c,s,n ## s) +#define B7(c,s,n) B6(c,s,n ## c), B6(c,s,n ## s) +#define B8(c,s ) B7(c,s, c), B7(c,s, s) + +static const uint64_t table_q1_signs[256] = { B8(ff, 01) }; + +static inline int8x16_t ggml_q1_0_unpack_pair(uint8_t bits0, uint8_t bits1) { + return vreinterpretq_s8_u8(vcombine_u8(vcreate_u8(table_q1_signs[bits0]), + vcreate_u8(table_q1_signs[bits1]))); +} +#endif + void ggml_quantize_mat_q8_0_4x4(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k) { assert(QK8_0 == 32); assert(k % QK8_0 == 0); @@ -1823,6 +1841,132 @@ void ggml_gemv_q8_0_4x8_q8_0(int n, ggml_gemv_q8_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc); } +void ggml_gemv_q1_0_4x4_q8_0(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD) + for (int c = 0; c < nc; c += ncols_interleaved) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (c / ncols_interleaved) * nb; + const block_q8_0 * a_ptr = (const block_q8_0 *) vy; + float32x4_t acc = vdupq_n_f32(0); + + for (int l = 0; l < nb; l++) { + const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d)); + float32x4_t accb = vdupq_n_f32(0); + + for (int k = 0; k < 4; k++) { + const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * 4 + k; + const float ad = GGML_CPU_FP16_TO_FP32(a_blk->d); + int32x4_t ret = vdupq_n_s32(0); + + for (int tile = 0; tile < 8; tile += 4) { + const int8x16_t signs0 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 0) + 0], + b_ptr[l].qs[k * 16 + 2 * (tile + 0) + 1]); + const int8x16_t signs1 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 1) + 0], + b_ptr[l].qs[k * 16 + 2 * (tile + 1) + 1]); + const int8x16_t signs2 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 2) + 0], + b_ptr[l].qs[k * 16 + 2 * (tile + 2) + 1]); + const int8x16_t signs3 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 3) + 0], + b_ptr[l].qs[k * 16 + 2 * (tile + 3) + 1]); + const int8x16_t q_tiles = vld1q_s8(a_blk->qs + tile * 4); + + ret = vdotq_laneq_s32(ret, signs0, q_tiles, 0); + ret = vdotq_laneq_s32(ret, signs1, q_tiles, 1); + ret = vdotq_laneq_s32(ret, signs2, q_tiles, 2); + ret = vdotq_laneq_s32(ret, signs3, q_tiles, 3); + } + + accb = vfmaq_n_f32(accb, vcvtq_f32_s32(ret), ad); + } + acc = vfmaq_f32(acc, accb, b_d); + } + vst1q_f32(s, acc); + s += ncols_interleaved; + } + return; +#endif + ggml_gemv_q1_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc); +} + +void ggml_gemv_q1_0_4x8_q8_0(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD) + for (int c = 0; c < nc; c += ncols_interleaved) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (c / ncols_interleaved) * nb; + const block_q8_0 * a_ptr = (const block_q8_0 *) vy; + float32x4_t acc = vdupq_n_f32(0); + + for (int l = 0; l < nb; l++) { + const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d)); + float32x4_t accb = vdupq_n_f32(0); + + for (int k = 0; k < 4; ++k) { + const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * 4 + k; + const uint8_t * GGML_RESTRICT b_qs = (const uint8_t *) b_ptr[l].qs + k * 16; + const float ad = GGML_CPU_FP16_TO_FP32(a_blk->d); + + int8x8x4_t a_chunks = vld1_s8_x4(a_blk->qs); + int8x16_t a0 = vcombine_s8(a_chunks.val[0], a_chunks.val[0]); + int8x16_t a1 = vcombine_s8(a_chunks.val[1], a_chunks.val[1]); + int8x16_t a2 = vcombine_s8(a_chunks.val[2], a_chunks.val[2]); + int8x16_t a3 = vcombine_s8(a_chunks.val[3], a_chunks.val[3]); + + int32x4_t ret0 = vdupq_n_s32(0); + int32x4_t ret1 = vdupq_n_s32(0); + + ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[0], b_qs[1]), a0); + ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[2], b_qs[3]), a0); + ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[4], b_qs[5]), a1); + ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[6], b_qs[7]), a1); + ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[8], b_qs[9]), a2); + ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[10], b_qs[11]), a2); + ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[12], b_qs[13]), a3); + ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[14], b_qs[15]), a3); + + accb = vfmaq_n_f32(accb, vcvtq_f32_s32(vpaddq_s32(ret0, ret1)), ad); + } + + acc = vfmaq_f32(acc, accb, b_d); + } + + vst1q_f32(s, acc); + s += ncols_interleaved; + } + return; +#endif + + ggml_gemv_q1_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc); +} + void ggml_gemm_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) { const int qk = QK8_0; const int nb = n / qk; @@ -5154,3 +5298,168 @@ void ggml_gemm_q8_0_4x8_q8_0(int n, #endif // defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8) ggml_gemm_q8_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc); } + +void ggml_gemm_q1_0_4x4_q8_0(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(n % qk == 0); + assert(nr % 4 == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD) + for (int y = 0; y < nr / 4; y++) { + const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb); + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + float32x4_t sumf[4]; + for (int m = 0; m < 4; m++) { + sumf[m] = vdupq_n_f32(0); + } + + for (int l = 0; l < nb; l++) { + float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d)); + float32x4_t blockf_0 = vdupq_n_f32(0); + float32x4_t blockf_1 = vdupq_n_f32(0); + float32x4_t blockf_2 = vdupq_n_f32(0); + float32x4_t blockf_3 = vdupq_n_f32(0); + + for (int k = 0; k < 4; ++k) { + const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k; + float32x4_t a_d = vcvt_f32_f16(vld1_f16((const float16_t *) a_blk->d)); + + int32x4_t sumi_0 = vdupq_n_s32(0); + int32x4_t sumi_1 = vdupq_n_s32(0); + int32x4_t sumi_2 = vdupq_n_s32(0); + int32x4_t sumi_3 = vdupq_n_s32(0); + + for (int tile = 0; tile < 8; ++tile) { + const int8x16_t signs = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * tile + 0], + b_ptr[l].qs[k * 16 + 2 * tile + 1]); + const int8x16_t a_tile = vld1q_s8(a_blk->qs + tile * 16); + + sumi_0 = vdotq_laneq_s32(sumi_0, signs, a_tile, 0); + sumi_1 = vdotq_laneq_s32(sumi_1, signs, a_tile, 1); + sumi_2 = vdotq_laneq_s32(sumi_2, signs, a_tile, 2); + sumi_3 = vdotq_laneq_s32(sumi_3, signs, a_tile, 3); + } + + blockf_0 = vfmaq_laneq_f32(blockf_0, vcvtq_f32_s32(sumi_0), a_d, 0); + blockf_1 = vfmaq_laneq_f32(blockf_1, vcvtq_f32_s32(sumi_1), a_d, 1); + blockf_2 = vfmaq_laneq_f32(blockf_2, vcvtq_f32_s32(sumi_2), a_d, 2); + blockf_3 = vfmaq_laneq_f32(blockf_3, vcvtq_f32_s32(sumi_3), a_d, 3); + } + + sumf[0] = vfmaq_f32(sumf[0], blockf_0, b_d); + sumf[1] = vfmaq_f32(sumf[1], blockf_1, b_d); + sumf[2] = vfmaq_f32(sumf[2], blockf_2, b_d); + sumf[3] = vfmaq_f32(sumf[3], blockf_3, b_d); + } + + for (int m = 0; m < 4; m++) { + vst1q_f32(s + (y * 4 + m) * bs + x * 4, sumf[m]); + } + } + } + return; +#endif + ggml_gemm_q1_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc); +} + +void ggml_gemm_q1_0_4x8_q8_0(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(n % qk == 0); + assert(nr % 4 == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8) + for (int y = 0; y < nr / 4; y++) { + const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb); + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + float32x4_t sumf[4]; + for (int m = 0; m < 4; ++m) { + sumf[m] = vdupq_n_f32(0); + } + + for (int l = 0; l < nb; l++) { + const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d)); + float32x4_t blockf[4]; + for (int m = 0; m < 4; ++m) { + blockf[m] = vdupq_n_f32(0); + } + + for (int k = 0; k < 4; ++k) { + const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k; + const uint8_t * GGML_RESTRICT b_qs = (const uint8_t *) b_ptr[l].qs + k * 16; + + int32x4_t acc[4]; + for (int i = 0; i < 4; ++i) { + acc[i] = vdupq_n_s32(0); + } + + for (int chunk = 0; chunk < 4; ++chunk) { + const int8x16_t a01 = vld1q_s8(a_blk->qs + chunk * 32); + const int8x16_t a23 = vld1q_s8(a_blk->qs + chunk * 32 + 16); + const int8x16_t b01 = ggml_q1_0_unpack_pair(b_qs[chunk * 4 + 0], b_qs[chunk * 4 + 1]); + const int8x16_t b23 = ggml_q1_0_unpack_pair(b_qs[chunk * 4 + 2], b_qs[chunk * 4 + 3]); + + acc[0] = vmmlaq_s32(acc[0], a01, b01); + acc[1] = vmmlaq_s32(acc[1], a01, b23); + acc[2] = vmmlaq_s32(acc[2], a23, b01); + acc[3] = vmmlaq_s32(acc[3], a23, b23); + } + + const int32x4_t row0 = vcombine_s32(vget_low_s32(acc[0]), vget_low_s32(acc[1])); + const int32x4_t row1 = vcombine_s32(vget_high_s32(acc[0]), vget_high_s32(acc[1])); + const int32x4_t row2 = vcombine_s32(vget_low_s32(acc[2]), vget_low_s32(acc[3])); + const int32x4_t row3 = vcombine_s32(vget_high_s32(acc[2]), vget_high_s32(acc[3])); + const float32x4_t a_d = vcvt_f32_f16(vld1_f16((const float16_t *) a_blk->d)); + + blockf[0] = vfmaq_laneq_f32(blockf[0], vcvtq_f32_s32(row0), a_d, 0); + blockf[1] = vfmaq_laneq_f32(blockf[1], vcvtq_f32_s32(row1), a_d, 1); + blockf[2] = vfmaq_laneq_f32(blockf[2], vcvtq_f32_s32(row2), a_d, 2); + blockf[3] = vfmaq_laneq_f32(blockf[3], vcvtq_f32_s32(row3), a_d, 3); + } + + sumf[0] = vfmaq_f32(sumf[0], blockf[0], b_d); + sumf[1] = vfmaq_f32(sumf[1], blockf[1], b_d); + sumf[2] = vfmaq_f32(sumf[2], blockf[2], b_d); + sumf[3] = vfmaq_f32(sumf[3], blockf[3], b_d); + } + + for (int m = 0; m < 4; ++m) { + vst1q_f32(s + (y * 4 + m) * bs + x * 4, sumf[m]); + } + } + } + return; +#endif + + ggml_gemm_q1_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc); +} diff --git a/ggml/src/ggml-cpu/arch/s390/quants.c b/ggml/src/ggml-cpu/arch/s390/quants.c index 50085757..52344828 100644 --- a/ggml/src/ggml-cpu/arch/s390/quants.c +++ b/ggml/src/ggml-cpu/arch/s390/quants.c @@ -146,6 +146,74 @@ void quantize_row_q8_1(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, i //===================================== Dot products ================================= +void ggml_vec_dot_q1_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, size_t bx, const void * GGML_RESTRICT vy, size_t by, int nrc) { + const int qk = QK1_0; // 128 + const int nb = n / qk; + + assert(n % qk == 0); + assert(nrc == 1); + UNUSED(nrc); + UNUSED(bx); + UNUSED(by); + UNUSED(bs); + + const block_q1_0 * GGML_RESTRICT x = vx; + const block_q8_0 * GGML_RESTRICT y = vy; + +#if defined(__VXE__) || defined(__VXE2__) + float32x4_t v_sumf = vec_splats(0.0f); + + const uint8x16_t v_zero = vec_splats((uint8_t)0x00); // zero + const uint8x16_t v_bias = vec_splats((uint8_t)0x80); // bias from signed to unsigned + // v ^ 0x80 == v + 128 + + const uint8x16_t v_idx = (const uint8x16_t){ 0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1 }; + const uint8x16_t v_bit = (const uint8x16_t){ 1, 2, 4, 8, 16, 32, 64, 128, 1, 2, 4, 8, 16, 32, 64, 128 }; + + for (int i = 0; i < nb; ++i) { + const uint8x16_t v_x = vec_xl(0, (const uint8_t *)x[i].qs); + const float32x4_t v_xd = vec_splats(GGML_CPU_FP16_TO_FP32(x[i].d)); + + for (int k = 0; k < 4; ++k) { + // sub-block k holds elements 32k .. 32k+31 + const block_q8_0 * GGML_RESTRICT yb = &y[i*4 + k]; + const float32x4_t v_yd = vec_splats(GGML_CPU_FP16_TO_FP32(yb->d)); + + const uint8x16_t v_xrl = vec_perm(v_x, v_x, vec_add(v_idx, vec_splats((uint8_t)(k*4 + 0)))); + const uint8x16_t v_xrh = vec_perm(v_x, v_x, vec_add(v_idx, vec_splats((uint8_t)(k*4 + 2)))); + + // isolate each lane's bit, then set all ones where that bit is clear, the -d case + const int8x16_t v_ml = (int8x16_t)vec_cmpeq(vec_and(v_xrl, v_bit), v_zero); + const int8x16_t v_mh = (int8x16_t)vec_cmpeq(vec_and(v_xrh, v_bit), v_zero); + + const int8x16_t v_yl = vec_xl(0, (const int8_t *)yb->qs); + const int8x16_t v_yh = vec_xl(QK8_0/2, (const int8_t *)yb->qs); + + // weights are only +1 or -1, so negate y + const int8x16_t v_ysl = vec_sub(vec_xor(v_yl, v_ml), v_ml); + const int8x16_t v_ysh = vec_sub(vec_xor(v_yh, v_mh), v_mh); + + // bias to unsigned, then vec_sum4 adds each group of 4 bytes into one word + const uint32x4_t v_p = vec_add(vec_sum4(vec_xor((uint8x16_t)v_ysl, v_bias), v_zero), + vec_sum4(vec_xor((uint8x16_t)v_ysh, v_bias), v_zero)); + + // each word summed 8 biased bytes, so take back 8 * 128 + const int32x4_t v_xy = vec_sub((int32x4_t)v_p, vec_splats((int32_t)1024)); + + // apply both block scales and add into the running total + v_sumf = vec_madd(vec_float(v_xy), vec_mul(v_xd, v_yd), v_sumf); + } + } + + *s = vec_hsum_f32x4(v_sumf); +#else + UNUSED(nb); + UNUSED(x); + UNUSED(y); + ggml_vec_dot_q1_0_q8_0_generic(n, s, bs, vx, bx, vy, by, nrc); +#endif +} + void ggml_vec_dot_q4_0_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, size_t bx, const void * GGML_RESTRICT vy, size_t by, int nrc) { const int qk = QK8_0; const int nb = n / qk; @@ -349,6 +417,7 @@ void ggml_vec_dot_mxfp4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const vo sumf = vec_hsum_f32x4(v_acc); *s = sumf; #else + UNUSED(nb); UNUSED(x); UNUSED(y); UNUSED(ib); @@ -636,7 +705,7 @@ void ggml_vec_dot_q5_1_q8_1(int n, float * GGML_RESTRICT s, size_t bs, const voi const float32x4_t v_xyf = vec_float(v_xy); const float32x4_t v_d = vec_splats(GGML_CPU_FP16_TO_FP32(x0->d) * GGML_CPU_FP16_TO_FP32(y0->d)); - const float32x4_t v_acc = vec_madd(v_xyf, v_d, v_acc); + const float32x4_t v_acc = vec_madd(v_xyf, v_d, vec_splats(0.0f)); sumf += vec_hsum_f32x4(v_acc) + summs; } diff --git a/ggml/src/ggml-cpu/arch/s390/repack.cpp b/ggml/src/ggml-cpu/arch/s390/repack.cpp new file mode 100644 index 00000000..abf3433a --- /dev/null +++ b/ggml/src/ggml-cpu/arch/s390/repack.cpp @@ -0,0 +1,225 @@ +#define GGML_COMMON_IMPL_CPP +#define GGML_COMMON_DECL_CPP +#include "ggml-common.h" +#include "ggml-backend-impl.h" + +#include "ggml-impl.h" +#include "ggml-cpu.h" +#include "ggml-cpu-impl.h" +#include "simd-mappings.h" +#include "traits.h" + +#include +#include +#include + +#define GGML_CPU_CLANG_WORKAROUND +#include "../../repack.h" + +#define UNUSED GGML_UNUSED + +void ggml_quantize_mat_q8_0_4x4(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k) { + assert(QK8_0 == 32); + assert(k % QK8_0 == 0); + const int nb = k / QK8_0; + + block_q8_0x4 * GGML_RESTRICT y = (block_q8_0x4 *) vy; + +#if defined(__VXE__) || defined(__VXE2__) + float32x4_t v_src[4][8]; + float id[4]; + + for (int i = 0; i < nb; i++) { + float32x4_t v_asrc[8]; + float32x4_t v_amax[8]; + + for (int row_iter = 0; row_iter < 4; row_iter++) { + for (int j = 0; j < 8; j++) v_src[row_iter][j] = vec_xl(0, x + row_iter * k + i * 32 + 4 * j); + for (int j = 0; j < 8; j++) v_asrc[j] = vec_abs(v_src[row_iter][j]); + + for (int j = 0; j < 4; j++) v_amax[2 * j] = vec_max(v_asrc[2 * j], v_asrc[2 * j + 1]); + for (int j = 0; j < 2; j++) v_amax[4 * j] = vec_max(v_amax[4 * j], v_amax[4 * j + 2]); + for (int j = 0; j < 1; j++) v_amax[8 * j] = vec_max(v_amax[8 * j], v_amax[8 * j + 4]); + + const float amax = MAX(MAX(vec_extract(v_amax[0], 0), vec_extract(v_amax[0], 1)), + MAX(vec_extract(v_amax[0], 2), vec_extract(v_amax[0], 3))); + + const float d = amax / ((1 << 7) - 1); + id[row_iter] = d ? 1.0f / d : 0.0f; + + y[i].d[row_iter] = GGML_CPU_FP32_TO_FP16(d); + } + + for (int j = 0; j < 8; j++) { + /* Uses non-default rounding for vec_signed or vec_round */ + const int32x4_t v_qs0 = vec_signed(__builtin_s390_vfisb(vec_mul(v_src[0][j], id[0]), 4, 1)); + const int32x4_t v_qs1 = vec_signed(__builtin_s390_vfisb(vec_mul(v_src[1][j], id[1]), 4, 1)); + const int32x4_t v_qs2 = vec_signed(__builtin_s390_vfisb(vec_mul(v_src[2][j], id[2]), 4, 1)); + const int32x4_t v_qs3 = vec_signed(__builtin_s390_vfisb(vec_mul(v_src[3][j], id[3]), 4, 1)); + + const int16x8_t v_qs01 = vec_packs(v_qs0, v_qs1); + const int16x8_t v_qs23 = vec_packs(v_qs2, v_qs3); + + vec_xst(vec_packs(v_qs01, v_qs23), 0, y[i].qs + 16 * j); + } + } +#else + UNUSED(nb); + UNUSED(y); + ggml_quantize_mat_q8_0_4x4_generic(x, vy, k); +#endif +} + +#if defined(__VXE__) || defined(__VXE2__) +static inline int16x8_t vxe_dot_acc(const int8x16_t v_x, const int8x16_t v_y, const int16x8_t v_acc) { + return vec_meadd(v_x, v_y, vec_moadd(v_x, v_y, v_acc)); +} + +static inline int8x16_t vxe_splat_granule(const int8_t * qs) { + uint32_t g; + memcpy(&g, qs, sizeof(g)); + return (int8x16_t)vec_splats(g); +} + +static inline int32x4_t vxe_fold(const int16x8_t v_sumi) { + const int16x8_t v_ones = vec_splats((int16_t)1); + return vec_add(vec_mule(v_sumi, v_ones), vec_mulo(v_sumi, v_ones)); +} +#endif + +void ggml_gemv_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) { + const int qk = QK8_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(nr == 1); + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(bs); + UNUSED(nr); + +#if defined(__VXE__) || defined(__VXE2__) + const block_q8_0 * a_ptr = (const block_q8_0 *) vy; + float * res_ptr = s; + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q4_0x4 * b_ptr = (const block_q4_0x4 *) vx + (x * nb); + + float32x4_t v_sumf = vec_splats(0.0f); + + for (int l = 0; l < nb; l++) { + const int8_t * x_qs = b_ptr[l].qs; + + const int8x16_t v_x0 = vec_xl( 0, x_qs); + const int8x16_t v_x1 = vec_xl(16, x_qs); + const int8x16_t v_x2 = vec_xl(32, x_qs); + const int8x16_t v_x3 = vec_xl(48, x_qs); + + const int8x16_t v_x0l = vec_sra(vec_sl(v_x0, 4), 4); + const int8x16_t v_x1l = vec_sra(vec_sl(v_x1, 4), 4); + const int8x16_t v_x2l = vec_sra(vec_sl(v_x2, 4), 4); + const int8x16_t v_x3l = vec_sra(vec_sl(v_x3, 4), 4); + + const int8x16_t v_x0h = vec_sra(v_x0, 4); + const int8x16_t v_x1h = vec_sra(v_x1, 4); + const int8x16_t v_x2h = vec_sra(v_x2, 4); + const int8x16_t v_x3h = vec_sra(v_x3, 4); + + const int8_t * y_lo = a_ptr[l].qs; + const int8_t * y_hi = y_lo + qk / 2; + + int16x8_t v_sumi = vec_splats((int16_t)0); + + v_sumi = vxe_dot_acc(v_x0l, vxe_splat_granule(y_lo + 0), v_sumi); + v_sumi = vxe_dot_acc(v_x1l, vxe_splat_granule(y_lo + 4), v_sumi); + v_sumi = vxe_dot_acc(v_x2l, vxe_splat_granule(y_lo + 8), v_sumi); + v_sumi = vxe_dot_acc(v_x3l, vxe_splat_granule(y_lo + 12), v_sumi); + + v_sumi = vxe_dot_acc(v_x0h, vxe_splat_granule(y_hi + 0), v_sumi); + v_sumi = vxe_dot_acc(v_x1h, vxe_splat_granule(y_hi + 4), v_sumi); + v_sumi = vxe_dot_acc(v_x2h, vxe_splat_granule(y_hi + 8), v_sumi); + v_sumi = vxe_dot_acc(v_x3h, vxe_splat_granule(y_hi + 12), v_sumi); + + const float32x4_t v_yd = vec_splats(GGML_CPU_FP16_TO_FP32(a_ptr[l].d)); + const float32x4_t v_xd = __lzs_f16cx4_load(b_ptr[l].d); + const float32x4_t v_d = vec_mul(v_yd, v_xd); + + v_sumf = vec_madd(vec_float(vxe_fold(v_sumi)), v_d, v_sumf); + } + + vec_xst(v_sumf, 0, res_ptr + x * ncols_interleaved); + } +#else + UNUSED(nb); + UNUSED(ncols_interleaved); + ggml_gemv_q4_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc); +#endif +} + +void ggml_gemm_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) { + const int qk = QK8_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(nr % 4 == 0); + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + +#if defined(__VXE__) || defined(__VXE2__) + for (int y = 0; y < nr / 4; y++) { + const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (y * nb); + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q4_0x4 * b_ptr = (const block_q4_0x4 *) vx + (x * nb); + + float32x4_t v_sumf[4]; + for (int m = 0; m < 4; m++) { + v_sumf[m] = vec_splats(0.0f); + } + + for (int l = 0; l < nb; l++) { + int16x8_t v_sumi0 = vec_splats((int16_t)0); + int16x8_t v_sumi1 = vec_splats((int16_t)0); + int16x8_t v_sumi2 = vec_splats((int16_t)0); + int16x8_t v_sumi3 = vec_splats((int16_t)0); + + for (int k = 0; k < 4; k++) { + const int8x16_t v_x = vec_xl(0, b_ptr[l].qs + 16 * k); + const int8x16_t v_xl = vec_sra(vec_sl(v_x, 4), 4); + const int8x16_t v_xh = vec_sra(v_x, 4); + + const int8_t * y_lo = a_ptr[l].qs + 16 * k; + const int8_t * y_hi = y_lo + qk / 2 * 4; + + v_sumi0 = vxe_dot_acc(v_xl, vxe_splat_granule(y_lo + 0), v_sumi0); + v_sumi1 = vxe_dot_acc(v_xl, vxe_splat_granule(y_lo + 4), v_sumi1); + v_sumi2 = vxe_dot_acc(v_xl, vxe_splat_granule(y_lo + 8), v_sumi2); + v_sumi3 = vxe_dot_acc(v_xl, vxe_splat_granule(y_lo + 12), v_sumi3); + + v_sumi0 = vxe_dot_acc(v_xh, vxe_splat_granule(y_hi + 0), v_sumi0); + v_sumi1 = vxe_dot_acc(v_xh, vxe_splat_granule(y_hi + 4), v_sumi1); + v_sumi2 = vxe_dot_acc(v_xh, vxe_splat_granule(y_hi + 8), v_sumi2); + v_sumi3 = vxe_dot_acc(v_xh, vxe_splat_granule(y_hi + 12), v_sumi3); + } + + const float32x4_t v_yd = __lzs_f16cx4_load(a_ptr[l].d); + const float32x4_t v_xd = __lzs_f16cx4_load(b_ptr[l].d); + + v_sumf[0] = vec_madd(vec_float(vxe_fold(v_sumi0)), vec_mul(v_xd, vec_splat(v_yd, 0)), v_sumf[0]); + v_sumf[1] = vec_madd(vec_float(vxe_fold(v_sumi1)), vec_mul(v_xd, vec_splat(v_yd, 1)), v_sumf[1]); + v_sumf[2] = vec_madd(vec_float(vxe_fold(v_sumi2)), vec_mul(v_xd, vec_splat(v_yd, 2)), v_sumf[2]); + v_sumf[3] = vec_madd(vec_float(vxe_fold(v_sumi3)), vec_mul(v_xd, vec_splat(v_yd, 3)), v_sumf[3]); + } + + for (int m = 0; m < 4; m++) { + vec_xst(v_sumf[m], 0, s + (y * 4 + m) * bs + x * ncols_interleaved); + } + } + } +#else + UNUSED(nb); + UNUSED(ncols_interleaved); + ggml_gemm_q4_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc); +#endif +} diff --git a/ggml/src/ggml-cpu/ggml-cpu-impl.h b/ggml/src/ggml-cpu/ggml-cpu-impl.h index 5d1ca5ff..5dd9ec8e 100644 --- a/ggml/src/ggml-cpu/ggml-cpu-impl.h +++ b/ggml/src/ggml-cpu/ggml-cpu-impl.h @@ -78,7 +78,7 @@ struct ggml_compute_params { #if defined(__ARM_NEON) // ref: https://github.com/ggml-org/llama.cpp/pull/5404 -#ifdef _MSC_VER +#if defined(_MSC_VER) && !defined(__clang__) #define ggml_vld1q_u32(w,x,y,z) { ((w) + ((uint64_t)(x) << 32)), ((y) + ((uint64_t)(z) << 32)) } #else #define ggml_vld1q_u32(w,x,y,z) { (w), (x), (y), (z) } diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c index 21f57218..e8bfab1b 100644 --- a/ggml/src/ggml-cpu/ggml-cpu.c +++ b/ggml/src/ggml-cpu/ggml-cpu.c @@ -4,6 +4,7 @@ #include "ggml-backend-impl.h" #include "ggml-backend.h" #include "traits.h" +#include "iqp.h" #include "ggml-cpu-impl.h" #include "ggml-impl.h" #include "quants.h" @@ -1363,7 +1364,9 @@ UseGgmlGemm1:; const size_t nbw3 = nbw2*ne12; assert(params->wsize >= ne13*nbw3); - GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); + // the F16 path below writes plain floats into wdata, so it needs an F32 vec_dot_type + GGML_ASSERT(src1->type == GGML_TYPE_F32 || vec_dot_type == GGML_TYPE_F32); #if 0 for (int64_t i13 = 0; i13 < ne13; ++i13) { @@ -1382,9 +1385,15 @@ UseGgmlGemm1:; size_t bs = ggml_blck_size(vec_dot_type); int64_t ne10_block_start = (ith * ne10/bs) / nth; int64_t ne10_block_end = ((ith + 1) * ne10/bs) / nth; - from_float((float *)((char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10), - (void *) (wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0), - (ne10_block_end - ne10_block_start) * bs); + const char * src1_block = (const char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10; + char * dst_block = wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0; + const int64_t n_block = (ne10_block_end - ne10_block_start) * bs; + + if (src1->type == GGML_TYPE_F32) { + from_float((const float *) src1_block, dst_block, n_block); + } else { + ggml_cpu_fp16_to_fp32((const ggml_fp16_t *) src1_block, (float *) dst_block, n_block); + } } } } @@ -1398,6 +1407,13 @@ UseGgmlGemm1:; ggml_barrier(params->threadpool); + // IQ panel gemm (see iqp.h) - must come after the barrier above, it consumes the q8_K rows + // of src1 from the work buffer + if (ggml_cpu_iqp_supports_mul_mat(dst) && !params->use_ref) { + ggml_compute_forward_mul_mat_iqp(params, dst); + return; + } + #if GGML_USE_LLAMAFILE if (src1->type != vec_dot_type) { const void* wdata = (src1->type == vec_dot_type) ? src1->data : params->wdata; @@ -1615,6 +1631,16 @@ static void ggml_compute_forward_mul_mat_id( char (*atomic_current_chunk)[CACHE_LINE_SIZE] = // [n_as] incr_ptr_aligned(&wdata_cur, CACHE_LINE_SIZE * n_as, CACHE_LINE_SIZE); + // IQ panel gemm (see iqp.h); per expert eligibility is decided below, but the work buffer is + // reserved for the whole node (ggml_graph_plan sizes it without params, use_ref only skips the dispatch) + const bool iqp = ggml_cpu_iqp_supports_mul_mat_id(dst) && !params->use_ref; + + char * iqp_panels = NULL; + + if (iqp) { + iqp_panels = incr_ptr_aligned(&wdata_cur, nth * ggml_cpu_iqp_scratch_size(dst), 64); + } + GGML_ASSERT(params->wsize >= (size_t)((char *) wdata_cur - (char *) params->wdata)); if (src1->type != vec_dot_type) { @@ -1686,6 +1712,13 @@ static void ggml_compute_forward_mul_mat_id( continue; } + if (iqp && ggml_cpu_iqp_mul_mat_id_min_batch(cne1)) { + ggml_compute_forward_mul_mat_id_iqp(params, dst, cur_a, cne1, (const int32_t *) &MMID_MATRIX_ROW(cur_a, 0), + iqp_panels); + + continue; + } + const char * src0_cur = (const char *) src0->data + cur_a * nb02; const void * wdata = (src1->type == vec_dot_type) ? src1->data : params->wdata; const size_t row_size = ggml_row_size(vec_dot_type, ne10); @@ -2346,6 +2379,7 @@ static int ggml_get_n_tasks(struct ggml_tensor * node, int n_threads) { case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: { n_tasks = n_threads; } break; @@ -2892,6 +2926,11 @@ struct ggml_cplan ggml_graph_plan( if (node->src[1]->type != vec_dot_type) { cur = ggml_row_size(vec_dot_type, ggml_nelements(node->src[1])); } + + // the IQ panel path needs one scratch panel per thread past the q8_K rows + if (ggml_cpu_iqp_supports_mul_mat(node)) { + cur = GGML_PAD(cur, 64) + n_tasks * ggml_cpu_iqp_scratch_size(node); + } } break; case GGML_OP_MUL_MAT_ID: { @@ -2911,6 +2950,10 @@ struct ggml_cplan ggml_graph_plan( cur += n_as*ids->ne[0]*ids->ne[1]*sizeof(struct mmid_row_mapping) + sizeof(int64_t); // atomic_current_chunk cur += CACHE_LINE_SIZE*n_as + CACHE_LINE_SIZE; + // the IQ panel path needs one scratch panel per thread on top of that + if (ggml_cpu_iqp_supports_mul_mat_id(node)) { + cur += n_tasks * ggml_cpu_iqp_scratch_size(node) + 64; + } } break; case GGML_OP_OUT_PROD: { @@ -2971,12 +3014,13 @@ struct ggml_cplan ggml_graph_plan( const int64_t ne10 = node->src[1]->ne[0]; // W const int64_t ne11 = node->src[1]->ne[1]; // H const int64_t ne12 = node->src[1]->ne[2]; // Channels In + const int64_t ne13 = node->src[1]->ne[3]; // Batch GGML_ASSERT(node->src[0]->type == GGML_TYPE_F16 || node->src[0]->type == GGML_TYPE_F32); GGML_ASSERT(node->src[1]->type == GGML_TYPE_F32); cur += ggml_type_size(node->src[0]->type) * ne00 * ne01 * ne02 * ne03; - cur += ggml_type_size(node->src[0]->type) * ne10 * ne11 * ne12; + cur += ggml_type_size(node->src[0]->type) * ne10 * ne11 * ne12 * ne13; } break; case GGML_OP_TOP_K: diff --git a/ggml/src/ggml-cpu/ggml-cpu.cpp b/ggml/src/ggml-cpu/ggml-cpu.cpp index 8cece71f..1df0f2bb 100644 --- a/ggml/src/ggml-cpu/ggml-cpu.cpp +++ b/ggml/src/ggml-cpu/ggml-cpu.cpp @@ -451,6 +451,10 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st op->type != GGML_TYPE_IQ1_S && op->type != GGML_TYPE_IQ1_M; // missing type_traits.from_float case GGML_OP_MUL_MAT: + if (ggml_get_op_params_i32(op, 1) == GGML_HINT_SRC0_IS_HADAMARD && + src0->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32) { + return src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16; + } return src1->type == GGML_TYPE_F32 || src1->type == ggml_get_type_traits_cpu(src0->type)->vec_dot_type; case GGML_OP_SOFT_MAX_BACK: { if (op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32) { diff --git a/ggml/src/ggml-cpu/iqp.cpp b/ggml/src/ggml-cpu/iqp.cpp new file mode 100644 index 00000000..b9201db3 --- /dev/null +++ b/ggml/src/ggml-cpu/iqp.cpp @@ -0,0 +1,1253 @@ +#define GGML_COMMON_IMPL_CPP +#define GGML_COMMON_DECL_CPP +#include "ggml-common.h" + +#include "ggml-impl.h" +#include "ggml-cpu.h" +#include "ggml-cpu-impl.h" +#include "simd-mappings.h" +#include "traits.h" + +#include +#include +#include + +#include "iqp.h" + +#define UNUSED GGML_UNUSED + +// smallest src1 batch for which the decode pays for itself +#define GGML_IQP_MIN_BATCH 8 + +// same, per expert, for MUL_MAT_ID +#define GGML_IQP_MIN_BATCH_ID 8 + +bool ggml_cpu_iqp_mul_mat_id_min_batch(int64_t cne1) { + return cne1 >= GGML_IQP_MIN_BATCH_ID; +} + +// src0 rows interleaved per panel +#define IQP_NB_ROWS 8 + +#define IQP_SB_SIZE 16 // weights per sub-block +#define IQP_NSB (QK_K / IQP_SB_SIZE) // sub-blocks per super-block + +// one super-block of a grid based IQ type decoded to int8, 8 rows interleaved: +// dfac[row] * iscales[sb*8 + row] * qs is bit identical to dequantize_row_iq* +struct block_iqp_x8 { + float dfac[8]; // f32 super-block scale, d * 2^-k + int32_t bias[8]; // 128 * sum(qs * iscale), see GGML_IQP_USE_BIAS + int8_t iscales[IQP_NSB * 8]; // integer sub-block scales, in [-32, 31] + int8_t qs[QK_K * 8]; // qs[sb*128 + g*32 + row*4 + k] = column sb*16 + g*4 + k +}; + +static_assert(sizeof(block_iqp_x8) == 8 * sizeof(float) + 8 * sizeof(int32_t) + IQP_NSB * 8 + QK_K * 8, + "wrong iqp_x8 block size/padding"); + +// feed the activations to VNNI as unsigned bytes (y + 128) and correct with bias[]; without VNNI the kernels use the maddubs sign trick instead and bias[] is not filled +#if defined(__AVX2__) && ((defined(__AVX512VNNI__) && defined(__AVX512VL__)) || defined(__AVXVNNI__)) +# define GGML_IQP_USE_BIAS 1 +#else +# define GGML_IQP_USE_BIAS 0 +#endif + +static inline size_t ggml_cpu_iqp_row_size(const struct ggml_tensor * dst) { + return ggml_row_size(GGML_TYPE_Q8_K, dst->src[1]->ne[0]); +} + +// the low 7 bits of v are the first 7 signs and the 8th is their parity (cf. unpack_ksigns in the CUDA backend) +static inline uint8_t iqp_unpack_ksigns(uint32_t v) { + uint32_t p = v ^ (v >> 4); + + p ^= p >> 2; + p ^= p >> 1; + + return (uint8_t) (v ^ ((p & 1) << 7)); +} + +#if defined(__AVX2__) + +// 0xFF in every byte whose sign bit is set; sv holds each sign byte broadcast over the 8 bytes it governs +static inline __m256i iqp_sign_mask(__m256i sv) { + const __m256i sel = _mm256_set1_epi64x((int64_t) 0x8040201008040201ULL); + +# if defined(__GFNI__) + // computes the and + compare in one instruction + return _mm256_gf2p8affine_epi64_epi8(sel, sv, 0); +# else + return _mm256_cmpeq_epi8(_mm256_and_si256(sv, sel), sel); +# endif +} + +// signs holds four sign bytes, byte l governing values 8*l .. 8*l+7 - spread each over its 8 lanes +static inline __m256i iqp_sign_bytes(uint32_t signs) { + const __m256i bcast = _mm256_setr_epi8(0, 0, 0, 0, 0, 0, 0, 0, 1, 1, 1, 1, 1, 1, 1, 1, // + 2, 2, 2, 2, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3); + + return _mm256_shuffle_epi8(_mm256_set1_epi32((int32_t) signs), bcast); +} + +// x ^ m - m negates the lanes where m is 0xFF +static inline __m256i iqp_apply_signs(__m256i x, __m256i m) { + return _mm256_sub_epi8(_mm256_xor_si256(x, m), m); +} + +#endif + +// 32 values from four 8 byte grid entries, sign byte l of signs applied to group l +static inline void iqp_store_signed_x8(int8_t * GGML_RESTRICT dst, + uint64_t g0, + uint64_t g1, + uint64_t g2, + uint64_t g3, + uint32_t signs) { +#if defined(__AVX2__) + const __m256i g = _mm256_set_epi64x((int64_t) g3, (int64_t) g2, (int64_t) g1, (int64_t) g0); + const __m256i m = iqp_sign_mask(iqp_sign_bytes(signs)); + + _mm256_storeu_si256((__m256i *) dst, iqp_apply_signs(g, m)); +#else + const uint64_t g[4] = { g0, g1, g2, g3 }; + + for (int l = 0; l < 4; ++l) { + const uint8_t * grid = (const uint8_t *) &g[l]; + const uint8_t s = (uint8_t) (signs >> 8 * l); + + for (int j = 0; j < 8; ++j) { + dst[8 * l + j] = s & kmask_iq2xs[j] ? -grid[j] : grid[j]; + } + } +#endif +} + +// same, but the eight values of group l come from two 4 byte grid entries +static inline void iqp_store_signed_x4(int8_t * GGML_RESTRICT dst, + uint32_t g0a, + uint32_t g0b, + uint32_t g1a, + uint32_t g1b, + uint32_t g2a, + uint32_t g2b, + uint32_t g3a, + uint32_t g3b, + uint32_t signs) { +#if defined(__AVX2__) + const __m256i g = _mm256_setr_epi32((int32_t) g0a, (int32_t) g0b, (int32_t) g1a, (int32_t) g1b, (int32_t) g2a, + (int32_t) g2b, (int32_t) g3a, (int32_t) g3b); + const __m256i m = iqp_sign_mask(iqp_sign_bytes(signs)); + + _mm256_storeu_si256((__m256i *) dst, iqp_apply_signs(g, m)); +#else + const uint32_t ga[4] = { g0a, g1a, g2a, g3a }; + const uint32_t gb[4] = { g0b, g1b, g2b, g3b }; + + for (int l = 0; l < 4; ++l) { + const uint8_t * grid1 = (const uint8_t *) &ga[l]; + const uint8_t * grid2 = (const uint8_t *) &gb[l]; + const uint8_t s = (uint8_t) (signs >> 8 * l); + + for (int j = 0; j < 4; ++j) { + dst[8 * l + j + 0] = s & kmask_iq2xs[j + 0] ? -grid1[j] : grid1[j]; + dst[8 * l + j + 4] = s & kmask_iq2xs[j + 4] ? -grid2[j] : grid2[j]; + } + } +#endif +} + +// 32 values of 8 * grid + delta from four 8 byte grid entries (grid bytes are in {-1, 0, 1}), byte l of deltas applying to group l +static inline void iqp_store_iq1_x8(int8_t * GGML_RESTRICT dst, + uint64_t g0, + uint64_t g1, + uint64_t g2, + uint64_t g3, + uint32_t deltas) { +#if defined(__AVX2__) + __m256i g = _mm256_set_epi64x((int64_t) g3, (int64_t) g2, (int64_t) g1, (int64_t) g0); + + // no byte shift in AVX2 + g = _mm256_add_epi8(g, g); + g = _mm256_add_epi8(g, g); + g = _mm256_add_epi8(g, g); + + _mm256_storeu_si256((__m256i *) dst, _mm256_add_epi8(g, iqp_sign_bytes(deltas))); +#else + const uint64_t g[4] = { g0, g1, g2, g3 }; + + for (int l = 0; l < 4; ++l) { + const int8_t * grid = (const int8_t *) &g[l]; + const int8_t delta = (int8_t) (deltas >> 8 * l); + + for (int j = 0; j < 8; ++j) { + dst[8 * l + j] = 8 * grid[j] + delta; + } + } +#endif +} + +// 32 values from 16 packed nibbles through the kvalues_iq4nl lookup: low nibbles first, then high +static inline void iqp_store_iq4_x32(int8_t * GGML_RESTRICT dst, const uint8_t * GGML_RESTRICT qs) { +#if defined(__AVX2__) + const __m128i q = _mm_loadu_si128((const __m128i *) qs); + const __m128i lut = _mm_loadu_si128((const __m128i *) kvalues_iq4nl); + const __m128i m4 = _mm_set1_epi8(0xf); + + _mm_storeu_si128((__m128i *) (dst + 0), _mm_shuffle_epi8(lut, _mm_and_si128(q, m4))); + _mm_storeu_si128((__m128i *) (dst + 16), _mm_shuffle_epi8(lut, _mm_and_si128(_mm_srli_epi16(q, 4), m4))); +#else + for (int j = 0; j < 16; ++j) { + dst[j + 0] = kvalues_iq4nl[qs[j] & 0xf]; + dst[j + 16] = kvalues_iq4nl[qs[j] >> 4]; + } +#endif +} + +#if GGML_IQP_USE_BIAS + +// sum of qs * iscale over one super-block, at most 256 * 127 * 32 = 1.04e6 +static inline int32_t iqp_weighted_sum(const int8_t * GGML_RESTRICT vals, const int8_t * GGML_RESTRICT iscales) { +#if defined(__AVX2__) + static_assert(IQP_SB_SIZE == 16, "the vector path folds two sub-blocks per 32 byte load"); + + const __m256i ones8 = _mm256_set1_epi8(1); + const __m256i ones16 = _mm256_set1_epi16(1); + + __m256i acc = _mm256_setzero_si256(); + + for (int i = 0; i < QK_K / 32; ++i) { + // sum groups of 4 bytes into int32, the low four lanes cover sub-block 2*i and the high four 2*i + 1 + const __m256i v = _mm256_loadu_si256((const __m256i *) (vals + 32 * i)); + const __m256i p = _mm256_madd_epi16(_mm256_maddubs_epi16(ones8, v), ones16); + + const __m256i s = _mm256_set_m128i(_mm_set1_epi32(iscales[2 * i + 1]), _mm_set1_epi32(iscales[2 * i + 0])); + + acc = _mm256_add_epi32(acc, _mm256_mullo_epi32(p, s)); + } + + __m128i sum = _mm_add_epi32(_mm256_castsi256_si128(acc), _mm256_extracti128_si256(acc, 1)); + + sum = _mm_add_epi32(sum, _mm_shuffle_epi32(sum, _MM_SHUFFLE(1, 0, 3, 2))); + sum = _mm_add_epi32(sum, _mm_shuffle_epi32(sum, _MM_SHUFFLE(2, 3, 0, 1))); + + return _mm_cvtsi128_si32(sum); +#else + int32_t wsum = 0; + + for (int sb = 0; sb < IQP_NSB; ++sb) { + int32_t vsum = 0; + + for (int k = 0; k < IQP_SB_SIZE; ++k) { + vsum += vals[sb * IQP_SB_SIZE + k]; + } + + wsum += iscales[sb] * vsum; + } + + return wsum; +#endif +} + +#endif // GGML_IQP_USE_BIAS + +static void iqp_decode_iq2_xxs(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq2_xxs * x = (const block_iq2_xxs *) vx; + + // db = d * (0.5 + ls) * 0.25 = (d / 8) * (2 * ls + 1), ls 4 bit + *dfac = GGML_CPU_FP16_TO_FP32(x->d) * 0.125f; + + uint32_t aux32[2]; + const uint8_t * aux8 = (const uint8_t *) aux32; + + for (int ib32 = 0; ib32 < QK_K / 32; ++ib32) { + memcpy(aux32, x->qs + 4 * ib32, 2 * sizeof(uint32_t)); + const int8_t ls = (int8_t) (2 * (aux32[1] >> 28) + 1); + + iscales[2 * ib32 + 0] = ls; + iscales[2 * ib32 + 1] = ls; + + const uint32_t signs = (uint32_t) iqp_unpack_ksigns((aux32[1] >> 0) & 127) | + (uint32_t) iqp_unpack_ksigns((aux32[1] >> 7) & 127) << 8 | + (uint32_t) iqp_unpack_ksigns((aux32[1] >> 14) & 127) << 16 | + (uint32_t) iqp_unpack_ksigns((aux32[1] >> 21) & 127) << 24; + + iqp_store_signed_x8(vals + 32 * ib32, iq2xxs_grid[aux8[0]], iq2xxs_grid[aux8[1]], iq2xxs_grid[aux8[2]], + iq2xxs_grid[aux8[3]], signs); + } +} + +static void iqp_decode_iq2_xs(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq2_xs * x = (const block_iq2_xs *) vx; + + *dfac = GGML_CPU_FP16_TO_FP32(x->d) * 0.125f; + + for (int ib32 = 0; ib32 < QK_K / 32; ++ib32) { + iscales[2 * ib32 + 0] = (int8_t) (2 * (x->scales[ib32] & 0xf) + 1); + iscales[2 * ib32 + 1] = (int8_t) (2 * (x->scales[ib32] >> 4) + 1); + + const uint16_t * q = x->qs + 4 * ib32; + + const uint32_t signs = (uint32_t) iqp_unpack_ksigns(q[0] >> 9) | (uint32_t) iqp_unpack_ksigns(q[1] >> 9) << 8 | + (uint32_t) iqp_unpack_ksigns(q[2] >> 9) << 16 | + (uint32_t) iqp_unpack_ksigns(q[3] >> 9) << 24; + + iqp_store_signed_x8(vals + 32 * ib32, iq2xs_grid[q[0] & 511], iq2xs_grid[q[1] & 511], iq2xs_grid[q[2] & 511], + iq2xs_grid[q[3] & 511], signs); + } +} + +static void iqp_decode_iq2_s(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq2_s * x = (const block_iq2_s *) vx; + + const uint8_t * qs = x->qs; + const uint8_t * qh = x->qh; + const uint8_t * signs = qs + QK_K / 8; + + *dfac = GGML_CPU_FP16_TO_FP32(x->d) * 0.125f; + + for (int ib32 = 0; ib32 < QK_K / 32; ++ib32) { + iscales[2 * ib32 + 0] = (int8_t) (2 * (x->scales[ib32] & 0xf) + 1); + iscales[2 * ib32 + 1] = (int8_t) (2 * (x->scales[ib32] >> 4) + 1); + + const uint32_t sbits = + (uint32_t) signs[0] | (uint32_t) signs[1] << 8 | (uint32_t) signs[2] << 16 | (uint32_t) signs[3] << 24; + + iqp_store_signed_x8(vals + 32 * ib32, iq2s_grid[qs[0] | (qh[ib32] << 8 & 0x300)], + iq2s_grid[qs[1] | (qh[ib32] << 6 & 0x300)], iq2s_grid[qs[2] | (qh[ib32] << 4 & 0x300)], + iq2s_grid[qs[3] | (qh[ib32] << 2 & 0x300)], sbits); + qs += 4; + signs += 4; + } +} + +static void iqp_decode_iq3_xxs(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq3_xxs * x = (const block_iq3_xxs *) vx; + + const uint8_t * qs = x->qs; + const uint8_t * scales_and_signs = qs + QK_K / 4; + + // db = d * (0.5 + ls) * 0.5 = (d / 4) * (2 * ls + 1), ls 4 bit + *dfac = GGML_CPU_FP16_TO_FP32(x->d) * 0.25f; + + uint32_t aux32; + + for (int ib32 = 0; ib32 < QK_K / 32; ++ib32) { + memcpy(&aux32, scales_and_signs + 4 * ib32, sizeof(uint32_t)); + const int8_t ls = (int8_t) (2 * (aux32 >> 28) + 1); + + iscales[2 * ib32 + 0] = ls; + iscales[2 * ib32 + 1] = ls; + + const uint32_t signs = (uint32_t) iqp_unpack_ksigns((aux32 >> 0) & 127) | + (uint32_t) iqp_unpack_ksigns((aux32 >> 7) & 127) << 8 | + (uint32_t) iqp_unpack_ksigns((aux32 >> 14) & 127) << 16 | + (uint32_t) iqp_unpack_ksigns((aux32 >> 21) & 127) << 24; + + iqp_store_signed_x4(vals + 32 * ib32, iq3xxs_grid[qs[0]], iq3xxs_grid[qs[1]], iq3xxs_grid[qs[2]], + iq3xxs_grid[qs[3]], iq3xxs_grid[qs[4]], iq3xxs_grid[qs[5]], iq3xxs_grid[qs[6]], + iq3xxs_grid[qs[7]], signs); + qs += 8; + } +} + +static void iqp_decode_iq3_s(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq3_s * x = (const block_iq3_s *) vx; + + const uint8_t * qs = x->qs; + const uint8_t * qh = x->qh; + const uint8_t * signs = x->signs; + + // db = d * (1 + 2 * ls), ls 4 bit + *dfac = GGML_CPU_FP16_TO_FP32(x->d); + + int k = 0; + + for (int ib32 = 0; ib32 < QK_K / 32; ib32 += 2) { + const int8_t db1 = (int8_t) (1 + 2 * (x->scales[ib32 / 2] & 0xf)); + const int8_t db2 = (int8_t) (1 + 2 * (x->scales[ib32 / 2] >> 4)); + + iscales[2 * ib32 + 0] = db1; + iscales[2 * ib32 + 1] = db1; + iscales[2 * ib32 + 2] = db2; + iscales[2 * ib32 + 3] = db2; + + for (int h = 0; h < 2; ++h) { + const uint32_t sbits = + (uint32_t) signs[0] | (uint32_t) signs[1] << 8 | (uint32_t) signs[2] << 16 | (uint32_t) signs[3] << 24; + + iqp_store_signed_x4(vals + k, iq3s_grid[qs[0] | ((qh[h] << 8) & 256)], + iq3s_grid[qs[1] | ((qh[h] << 7) & 256)], iq3s_grid[qs[2] | ((qh[h] << 6) & 256)], + iq3s_grid[qs[3] | ((qh[h] << 5) & 256)], iq3s_grid[qs[4] | ((qh[h] << 4) & 256)], + iq3s_grid[qs[5] | ((qh[h] << 3) & 256)], iq3s_grid[qs[6] | ((qh[h] << 2) & 256)], + iq3s_grid[qs[7] | ((qh[h] << 1) & 256)], sbits); + + k += 32; + qs += 8; + signs += 4; + } + qh += 2; + } +} + +// dequantize_row_iq1_* computes y = dl * (grid[j] + delta) with delta = +-1/8, so the panel stores 8 * grid[j] +- 1 and folds the /8 into dfac +static void iqp_decode_iq1_s(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq1_s * x = (const block_iq1_s *) vx; + + const uint8_t * qs = x->qs; + const uint16_t * qh = x->qh; + + // dl = d * (2 * ls + 1) * 0.125, ls 3 bit + *dfac = GGML_CPU_FP16_TO_FP32(x->d) * 0.125f; + + for (int ib = 0; ib < QK_K / 32; ++ib) { + const int8_t dl = (int8_t) (2 * ((qh[ib] >> 12) & 7) + 1); + const int8_t delta = qh[ib] & 0x8000 ? -1 : 1; + + iscales[2 * ib + 0] = dl; + iscales[2 * ib + 1] = dl; + + iqp_store_iq1_x8(vals + 32 * ib, iq1s_grid[qs[0] | (((qh[ib] >> 0) & 7) << 8)], + iq1s_grid[qs[1] | (((qh[ib] >> 3) & 7) << 8)], iq1s_grid[qs[2] | (((qh[ib] >> 6) & 7) << 8)], + iq1s_grid[qs[3] | (((qh[ib] >> 9) & 7) << 8)], ((uint8_t) delta) * 0x01010101u); + qs += 4; + } +} + +static void iqp_decode_iq1_m(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq1_m * x = (const block_iq1_m *) vx; + + // block_iq1_m has no d field - the fp16 super-block scale is spread over the top nibbles of the four scale words + const uint16_t * sc = (const uint16_t *) x->scales; + + iq1m_scale_t scale; + scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000); + + *dfac = GGML_CPU_FP16_TO_FP32(scale.f16) * 0.125f; + + const uint8_t * qs = x->qs; + const uint8_t * qh = x->qh; + + for (int ib = 0; ib < QK_K / 32; ++ib) { + iscales[2 * ib + 0] = (int8_t) (2 * ((sc[ib / 2] >> (6 * (ib % 2) + 0)) & 0x7) + 1); + iscales[2 * ib + 1] = (int8_t) (2 * ((sc[ib / 2] >> (6 * (ib % 2) + 3)) & 0x7) + 1); + + const uint16_t idx[4] = { + (uint16_t) (qs[0] | ((qh[0] << 8) & 0x700)), + (uint16_t) (qs[1] | ((qh[0] << 4) & 0x700)), + (uint16_t) (qs[2] | ((qh[1] << 8) & 0x700)), + (uint16_t) (qs[3] | ((qh[1] << 4) & 0x700)), + }; + const uint32_t deltas = (uint32_t) (qh[0] & 0x08 ? 0xff : 0x01) | (uint32_t) (qh[0] & 0x80 ? 0xff : 0x01) << 8 | + (uint32_t) (qh[1] & 0x08 ? 0xff : 0x01) << 16 | + (uint32_t) (qh[1] & 0x80 ? 0xff : 0x01) << 24; + + iqp_store_iq1_x8(vals + 32 * ib, iq1s_grid[idx[0]], iq1s_grid[idx[1]], iq1s_grid[idx[2]], iq1s_grid[idx[3]], + deltas); + qs += 4; + qh += 2; + } +} + +static void iqp_decode_iq4_xs(const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + const block_iq4_xs * x = (const block_iq4_xs *) vx; + + const uint8_t * qs = x->qs; + + // dl = d * (ls - 32), ls 6 bit, so the integer scale is in [-32, 31] + *dfac = GGML_CPU_FP16_TO_FP32(x->d); + + for (int ib = 0; ib < QK_K / 32; ++ib) { + const int ls = ((x->scales_l[ib / 2] >> 4 * (ib % 2)) & 0xf) | (((x->scales_h >> 2 * ib) & 3) << 4); + const int8_t dl = (int8_t) (ls - 32); + + iscales[2 * ib + 0] = dl; + iscales[2 * ib + 1] = dl; + + iqp_store_iq4_x32(vals + 32 * ib, qs); + qs += 16; + } +} + +// expanded by the eligibility test and the decode dispatch +#define IQP_TYPE_LIST(T) \ + T(IQ2_XXS, iq2_xxs) \ + T(IQ2_XS, iq2_xs) \ + T(IQ2_S, iq2_s) \ + T(IQ3_XXS, iq3_xxs) \ + T(IQ3_S, iq3_s) \ + T(IQ1_S, iq1_s) \ + T(IQ1_M, iq1_m) \ + T(IQ4_XS, iq4_xs) + +static bool iqp_decode_superblock(enum ggml_type type, + const void * GGML_RESTRICT vx, + int8_t * GGML_RESTRICT vals, + int8_t * GGML_RESTRICT iscales, + float * GGML_RESTRICT dfac) { + switch (type) { +#define IQP_CASE(E, name) \ + case GGML_TYPE_##E: \ + iqp_decode_##name(vx, vals, iscales, dfac); \ + return true; + IQP_TYPE_LIST(IQP_CASE) +#undef IQP_CASE + default: + return false; + } +} + +#if defined(__AVX2__) + +// 8x8 int32 transpose of the 32 column group starting at column off +static inline void iqp_interleave_x8(int8_t * GGML_RESTRICT dst, const int8_t (*vals)[QK_K], int off) { + static_assert(IQP_NB_ROWS == 8, "the transpose is 8x8"); + + __m256i v[IQP_NB_ROWS]; + + for (int r = 0; r < IQP_NB_ROWS; ++r) { + v[r] = _mm256_loadu_si256((const __m256i *) (vals[r] + off)); + } + + // pair rows into dword couples, then into qword quadruples, then swap the 128 bit lanes + const __m256i a0 = _mm256_unpacklo_epi32(v[0], v[1]); + const __m256i a1 = _mm256_unpackhi_epi32(v[0], v[1]); + const __m256i a2 = _mm256_unpacklo_epi32(v[2], v[3]); + const __m256i a3 = _mm256_unpackhi_epi32(v[2], v[3]); + const __m256i a4 = _mm256_unpacklo_epi32(v[4], v[5]); + const __m256i a5 = _mm256_unpackhi_epi32(v[4], v[5]); + const __m256i a6 = _mm256_unpacklo_epi32(v[6], v[7]); + const __m256i a7 = _mm256_unpackhi_epi32(v[6], v[7]); + + const __m256i b0 = _mm256_unpacklo_epi64(a0, a2); + const __m256i b1 = _mm256_unpackhi_epi64(a0, a2); + const __m256i b2 = _mm256_unpacklo_epi64(a1, a3); + const __m256i b3 = _mm256_unpackhi_epi64(a1, a3); + const __m256i b4 = _mm256_unpacklo_epi64(a4, a6); + const __m256i b5 = _mm256_unpackhi_epi64(a4, a6); + const __m256i b6 = _mm256_unpacklo_epi64(a5, a7); + const __m256i b7 = _mm256_unpackhi_epi64(a5, a7); + + _mm256_storeu_si256((__m256i *) (dst + 0 * 32), _mm256_permute2x128_si256(b0, b4, 0x20)); + _mm256_storeu_si256((__m256i *) (dst + 1 * 32), _mm256_permute2x128_si256(b1, b5, 0x20)); + _mm256_storeu_si256((__m256i *) (dst + 2 * 32), _mm256_permute2x128_si256(b2, b6, 0x20)); + _mm256_storeu_si256((__m256i *) (dst + 3 * 32), _mm256_permute2x128_si256(b3, b7, 0x20)); + _mm256_storeu_si256((__m256i *) (dst + 4 * 32), _mm256_permute2x128_si256(b0, b4, 0x31)); + _mm256_storeu_si256((__m256i *) (dst + 5 * 32), _mm256_permute2x128_si256(b1, b5, 0x31)); + _mm256_storeu_si256((__m256i *) (dst + 6 * 32), _mm256_permute2x128_si256(b2, b6, 0x31)); + _mm256_storeu_si256((__m256i *) (dst + 7 * 32), _mm256_permute2x128_si256(b3, b7, 0x31)); +} + +#endif + +// decode IQP_NB_ROWS consecutive source rows (starting at src, row stride nb01) into a panel of nblocks block_iqp_x8 +static void iqp_decode_panel_8(enum ggml_type type, + const char * GGML_RESTRICT src, + size_t nb01, + int64_t nblocks, + block_iqp_x8 * GGML_RESTRICT dst) { + const size_t bsize = ggml_type_size(type); + + int8_t vals[IQP_NB_ROWS][QK_K]; + int8_t iscales[IQP_NB_ROWS][IQP_NSB]; + float dfac[IQP_NB_ROWS]; + + for (int64_t x = 0; x < nblocks; x++) { + for (int r = 0; r < IQP_NB_ROWS; r++) { + const char * blk = src + r * nb01 + x * bsize; + + const bool ok = iqp_decode_superblock(type, blk, vals[r], iscales[r], &dfac[r]); + GGML_ASSERT(ok); + +#ifdef GGML_IQP_VERIFY + // check that the panel reproduces the reference dequantization bit exactly + float ref[QK_K]; + ggml_get_type_traits(type)->to_float(blk, ref, QK_K); + for (int j = 0; j < QK_K; j++) { + const float scale = dfac[r] * iscales[r][j / IQP_SB_SIZE]; + GGML_ASSERT(scale * vals[r][j] == ref[j]); + } +#endif + } + + for (int r = 0; r < IQP_NB_ROWS; r++) { + dst->dfac[r] = dfac[r]; + + for (int sb = 0; sb < IQP_NSB; sb++) { + dst->iscales[sb * IQP_NB_ROWS + r] = iscales[r][sb]; + } + +#if GGML_IQP_USE_BIAS + dst->bias[r] = 128 * iqp_weighted_sum(vals[r], iscales[r]); +#endif + } + +#if defined(__AVX2__) + for (int grp = 0; grp < QK_K / 32; grp++) { + iqp_interleave_x8(dst->qs + grp * 256, vals, grp * 32); + } +#else + for (int r = 0; r < IQP_NB_ROWS; r++) { + for (int sb = 0; sb < IQP_NSB; sb++) { + for (int g = 0; g < IQP_SB_SIZE / 4; g++) { + memcpy(dst->qs + sb * 128 + g * 32 + r * 4, vals[r] + sb * IQP_SB_SIZE + g * 4, 4); + } + } + } +#endif + + dst++; + } +} + +// gemm/gemv kernels: vx points at block_iqp_x8, vy at plain (non interleaved) block_q8_K rows + +static void iqp_gemv_8x8_q8_K_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int nb = n / QK_K; + const int ncols_interleaved = 8; + + assert(n % QK_K == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(bs); + UNUSED(nr); + + const block_iqp_x8 * b_ptr_start = (const block_iqp_x8 *) vx; + const block_q8_K * a_ptr = (const block_q8_K *) vy; + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_iqp_x8 * b_ptr = b_ptr_start + x * nb; + + float sumf[8] = { 0 }; + + for (int l = 0; l < nb; l++) { + int32_t sumi[8] = { 0 }; + + for (int sb = 0; sb < IQP_NSB; sb++) { + int32_t isum[8] = { 0 }; + + for (int g = 0; g < 4; g++) { + for (int j = 0; j < ncols_interleaved; j++) { + for (int k = 0; k < 4; k++) { + isum[j] += b_ptr[l].qs[sb * 128 + g * 32 + j * 4 + k] * a_ptr[l].qs[sb * 16 + g * 4 + k]; + } + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + sumi[j] += isum[j] * b_ptr[l].iscales[sb * 8 + j]; + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + sumf[j] += (float) sumi[j] * (b_ptr[l].dfac[j] * a_ptr[l].d); + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + s[x * ncols_interleaved + j] = sumf[j]; + } + } +} + +// one 4 row x nc column tile; s points at the first of the four output rows, bs floats apart +static void iqp_gemm_tile_4_generic(int nb, + float * GGML_RESTRICT s, + size_t bs, + const block_iqp_x8 * GGML_RESTRICT b_ptr_start, + const block_q8_K * const a_ptr[4], + int nc) { + const int ncols_interleaved = 8; + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_iqp_x8 * b_ptr = b_ptr_start + x * nb; + + float sumf[4][8]; + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + sumf[m][j] = 0.0f; + } + } + + for (int l = 0; l < nb; l++) { + for (int m = 0; m < 4; m++) { + int32_t sumi[8] = { 0 }; + + for (int sb = 0; sb < IQP_NSB; sb++) { + int32_t isum[8] = { 0 }; + + for (int g = 0; g < 4; g++) { + for (int j = 0; j < ncols_interleaved; j++) { + for (int k = 0; k < 4; k++) { + isum[j] += + b_ptr[l].qs[sb * 128 + g * 32 + j * 4 + k] * a_ptr[m][l].qs[sb * 16 + g * 4 + k]; + } + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + sumi[j] += isum[j] * b_ptr[l].iscales[sb * 8 + j]; + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + sumf[m][j] += (float) sumi[j] * (b_ptr[l].dfac[j] * a_ptr[m][l].d); + } + } + } + + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + s[m * bs + x * ncols_interleaved + j] = sumf[m][j]; + } + } + } +} + +static void iqp_gemm_8x8_q8_K_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int nb = n / QK_K; + + assert(n % QK_K == 0); + assert(nr % 4 == 0); + assert(nc % 8 == 0); + + const block_iqp_x8 * b_ptr_start = (const block_iqp_x8 *) vx; + const block_q8_K * a_ptr_start = (const block_q8_K *) vy; + + for (int y = 0; y < nr / 4; y++) { + const block_q8_K * a_ptr[4]; + for (int m = 0; m < 4; m++) { + a_ptr[m] = a_ptr_start + (y * 4 + m) * nb; + } + + iqp_gemm_tile_4_generic(nb, s + y * 4 * bs, bs, b_ptr_start, a_ptr, nc); + } +} + +static void iqp_gemm_8x8_q8_K_p4_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * const * GGML_RESTRICT vy, + int nc) { + const int nb = n / QK_K; + + assert(n % QK_K == 0); + assert(nc % 8 == 0); + + const block_q8_K * a_ptr[4]; + for (int m = 0; m < 4; m++) { + a_ptr[m] = (const block_q8_K *) vy[m]; + } + + iqp_gemm_tile_4_generic(nb, s, bs, (const block_iqp_x8 *) vx, a_ptr, nc); +} + +#if defined(__AVX2__) + +// add int16_t pairwise and return as 256 bit int vector, then add the accumulator +static inline __m256i sum_i16_pairs_acc_int32x8(const __m256i acc, const __m256i x) { + const __m256i ones = _mm256_set1_epi16(1); + return _mm256_add_epi32(acc, _mm256_madd_epi16(ones, x)); +} + +static inline __m256i mul_sum_us8_pairs_acc_int32x8(const __m256i acc, const __m256i ax, const __m256i sy) { +# if defined(__AVX512VNNI__) && defined(__AVX512VL__) + return _mm256_dpbusd_epi32(acc, ax, sy); +# elif defined(__AVXVNNI__) + return _mm256_dpbusd_avx_epi32(acc, ax, sy); +# else + // Perform multiplication and create 16-bit values + const __m256i dot = _mm256_maddubs_epi16(ax, sy); + return sum_i16_pairs_acc_int32x8(acc, dot); +# endif +} + +// Integer variant of the function defined in ggml-quants.c +// multiply int8_t, add results pairwise twice and return as 256 bit int vector, then add the accumulator +static inline __m256i mul_sum_i8_pairs_acc_int32x8(const __m256i acc, const __m256i x, const __m256i y) { +# if defined(__AVXVNNIINT8__) + return _mm256_dpbssd_epi32(acc, x, y); +# else + // Get absolute values of x vectors + const __m256i ax = _mm256_sign_epi8(x, x); + // Sign the values of the y vectors + const __m256i sy = _mm256_sign_epi8(y, x); + return mul_sum_us8_pairs_acc_int32x8(acc, ax, sy); +# endif +} + +// load the 16 activations of one sub-block, offset by 128 when they are fed to dpbusd as unsigned bytes +static inline __m256i iqp_load_y(const int8_t * GGML_RESTRICT qs) { + __m128i y = _mm_loadu_si128((const __m128i *) qs); +# if GGML_IQP_USE_BIAS + y = _mm_xor_si128(y, _mm_set1_epi8((char) 0x80)); +# endif + return _mm256_broadcastsi128_si256(y); +} + +// xv: 8 rows x 4 signed weights, yb: the matching 4 activation bytes broadcast to all 8 lanes +static inline __m256i iqp_dot4(const __m256i acc, const __m256i xv, const __m256i yb) { +# if GGML_IQP_USE_BIAS + return mul_sum_us8_pairs_acc_int32x8(acc, yb, xv); +# else + return mul_sum_i8_pairs_acc_int32x8(acc, xv, yb); +# endif +} + +static inline __m256i iqp_load_iscales(const int8_t * GGML_RESTRICT iscales) { + return _mm256_cvtepi8_epi32(_mm_loadl_epi64((const __m128i *) iscales)); +} + +// accumulate one super-block of 8 interleaved rows against one q8_K row in int32; worst case 16 * 32 * 16 * 255 * 127 = 2.65e8 plus a bias of at most 1.33e8 does not overflow +static inline __m256i iqp_acc_block(const block_iqp_x8 * GGML_RESTRICT b, const block_q8_K * GGML_RESTRICT a) { + __m256i sumi = _mm256_setzero_si256(); + + for (int sb = 0; sb < IQP_NSB; sb++) { + const int8_t * qs = b->qs + sb * 128; + + const __m256i yv = iqp_load_y(a->qs + sb * 16); + + __m256i isum = _mm256_setzero_si256(); + + isum = iqp_dot4(isum, _mm256_loadu_si256((const __m256i *) (qs + 0)), _mm256_shuffle_epi32(yv, 0x00)); + isum = iqp_dot4(isum, _mm256_loadu_si256((const __m256i *) (qs + 32)), _mm256_shuffle_epi32(yv, 0x55)); + isum = iqp_dot4(isum, _mm256_loadu_si256((const __m256i *) (qs + 64)), _mm256_shuffle_epi32(yv, 0xAA)); + isum = iqp_dot4(isum, _mm256_loadu_si256((const __m256i *) (qs + 96)), _mm256_shuffle_epi32(yv, 0xFF)); + + sumi = _mm256_add_epi32(sumi, _mm256_mullo_epi32(isum, iqp_load_iscales(b->iscales + sb * 8))); + } + +# if GGML_IQP_USE_BIAS + sumi = _mm256_sub_epi32(sumi, _mm256_loadu_si256((const __m256i *) b->bias)); +# endif + + return sumi; +} + +// one 4 row x nc column tile; s points at the first of the four output rows, bs floats apart +static inline void iqp_gemm_tile_4(int nb, + float * GGML_RESTRICT s, + size_t bs, + const block_iqp_x8 * GGML_RESTRICT b_ptr_start, + const block_q8_K * const a_ptr[4], + int nc) { + const int ncols_interleaved = 8; + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_iqp_x8 * b_ptr = b_ptr_start + x * nb; + + __m256 sumf[4]; + for (int m = 0; m < 4; m++) { + sumf[m] = _mm256_setzero_ps(); + } + + for (int l = 0; l < nb; l++) { + __m256i sumi[4]; + for (int m = 0; m < 4; m++) { + sumi[m] = _mm256_setzero_si256(); + } + + for (int sb = 0; sb < IQP_NSB; sb++) { + const int8_t * qs = b_ptr[l].qs + sb * 128; + + __m256i yv[4]; + __m256i isum[4]; + for (int m = 0; m < 4; m++) { + yv[m] = iqp_load_y(a_ptr[m][l].qs + sb * 16); + isum[m] = _mm256_setzero_si256(); + } + + const __m256i xv0 = _mm256_loadu_si256((const __m256i *) (qs + 0)); + const __m256i xv1 = _mm256_loadu_si256((const __m256i *) (qs + 32)); + const __m256i xv2 = _mm256_loadu_si256((const __m256i *) (qs + 64)); + const __m256i xv3 = _mm256_loadu_si256((const __m256i *) (qs + 96)); + + for (int m = 0; m < 4; m++) { + isum[m] = iqp_dot4(isum[m], xv0, _mm256_shuffle_epi32(yv[m], 0x00)); + isum[m] = iqp_dot4(isum[m], xv1, _mm256_shuffle_epi32(yv[m], 0x55)); + isum[m] = iqp_dot4(isum[m], xv2, _mm256_shuffle_epi32(yv[m], 0xAA)); + isum[m] = iqp_dot4(isum[m], xv3, _mm256_shuffle_epi32(yv[m], 0xFF)); + } + + const __m256i isc = iqp_load_iscales(b_ptr[l].iscales + sb * 8); + for (int m = 0; m < 4; m++) { + sumi[m] = _mm256_add_epi32(sumi[m], _mm256_mullo_epi32(isum[m], isc)); + } + } + +# if GGML_IQP_USE_BIAS + const __m256i bias = _mm256_loadu_si256((const __m256i *) b_ptr[l].bias); + for (int m = 0; m < 4; m++) { + sumi[m] = _mm256_sub_epi32(sumi[m], bias); + } +# endif + + const __m256 dfac = _mm256_loadu_ps(b_ptr[l].dfac); + for (int m = 0; m < 4; m++) { + sumf[m] = _mm256_fmadd_ps(_mm256_cvtepi32_ps(sumi[m]), + _mm256_mul_ps(dfac, _mm256_set1_ps(a_ptr[m][l].d)), sumf[m]); + } + } + + for (int m = 0; m < 4; m++) { + _mm256_storeu_ps(s + m * bs + x * ncols_interleaved, sumf[m]); + } + } +} + +#endif // __AVX2__ + +static void iqp_gemv_8x8_q8_K(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int nb = n / QK_K; + const int ncols_interleaved = 8; + + assert(n % QK_K == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(bs); + UNUSED(nr); + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__AVX2__) + const block_iqp_x8 * b_ptr_start = (const block_iqp_x8 *) vx; + const block_q8_K * a_ptr = (const block_q8_K *) vy; + + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_iqp_x8 * b_ptr = b_ptr_start + x * nb; + + __m256 sumf = _mm256_setzero_ps(); + + for (int l = 0; l < nb; l++) { + const __m256 dv = _mm256_mul_ps(_mm256_loadu_ps(b_ptr[l].dfac), _mm256_set1_ps(a_ptr[l].d)); + + sumf = _mm256_fmadd_ps(_mm256_cvtepi32_ps(iqp_acc_block(b_ptr + l, a_ptr + l)), dv, sumf); + } + + _mm256_storeu_ps(s + x * ncols_interleaved, sumf); + } + + return; +#endif + + iqp_gemv_8x8_q8_K_generic(n, s, bs, vx, vy, nr, nc); +} + +static void iqp_gemm_8x8_q8_K(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int nb = n / QK_K; + const int ncols_interleaved = 8; + + assert(n % QK_K == 0); + assert(nr % 4 == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__AVX2__) + const block_iqp_x8 * b_ptr_start = (const block_iqp_x8 *) vx; + const block_q8_K * a_ptr_start = (const block_q8_K *) vy; + + for (int y = 0; y < nr / 4; y++) { + const block_q8_K * a_ptr[4]; + for (int m = 0; m < 4; m++) { + a_ptr[m] = a_ptr_start + (y * 4 + m) * nb; + } + + iqp_gemm_tile_4(nb, s + y * 4 * bs, bs, b_ptr_start, a_ptr, nc); + } + + return; +#endif + + iqp_gemm_8x8_q8_K_generic(n, s, bs, vx, vy, nr, nc); +} + +// same as iqp_gemm_8x8_q8_K with nr = 4, but the activation rows are passed as separate pointers (for the scattered rows of MUL_MAT_ID) +static void iqp_gemm_8x8_q8_K_p4(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * const * GGML_RESTRICT vy, + int nc) { + const int nb = n / QK_K; + const int ncols_interleaved = 8; + + assert(n % QK_K == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(nb); + UNUSED(ncols_interleaved); + +#if defined(__AVX2__) + const block_q8_K * a_ptr[4]; + for (int m = 0; m < 4; m++) { + a_ptr[m] = (const block_q8_K *) vy[m]; + } + + iqp_gemm_tile_4(nb, s, bs, (const block_iqp_x8 *) vx, a_ptr, nc); + + return; +#endif + + iqp_gemm_8x8_q8_K_p4_generic(n, s, bs, vx, vy, nc); +} + +static bool iqp_type_supported(enum ggml_type type) { + switch (type) { +#define IQP_CASE(E, name) case GGML_TYPE_##E: + IQP_TYPE_LIST(IQP_CASE) +#undef IQP_CASE + return true; + default: + return false; + } +} + +static bool iqp_supported_common(const struct ggml_tensor * dst) { + const struct ggml_tensor * src0 = dst->src[0]; + const struct ggml_tensor * src1 = dst->src[1]; + + if (!iqp_type_supported(src0->type)) { + return false; + } + + // the path assumes the src1 conversion type is q8_K + if (ggml_get_type_traits_cpu(src0->type)->vec_dot_type != GGML_TYPE_Q8_K) { + return false; + } + + // escape hatch to A/B the panel against the plain vec_dot path without rebuilding (--no-repack does not cover this path) + static const bool disabled = getenv("GGML_NO_IQ_PANEL") != nullptr; + if (disabled) { + return false; + } + + if (!ggml_cpu_has_avx2()) { + return false; + } + + if (src1->type != GGML_TYPE_F32) { + return false; + } + + if (src0->ne[0] % QK_K != 0 || src0->ne[1] % IQP_NB_ROWS != 0) { + return false; + } + + if (src0->ne[3] != 1 || src1->ne[3] != 1 || !ggml_is_contiguous(src0)) { + return false; + } + + if (dst->type != GGML_TYPE_F32 || dst->nb[0] != sizeof(float)) { + return false; + } + + return true; +} + +bool ggml_cpu_iqp_supports_mul_mat(const struct ggml_tensor * dst) { + const struct ggml_tensor * src0 = dst->src[0]; + const struct ggml_tensor * src1 = dst->src[1]; + + if (!iqp_supported_common(dst)) { + return false; + } + + if (src1->ne[1] < GGML_IQP_MIN_BATCH) { + return false; + } + + // plain 2D weight matmuls only (src1 may still be batched over ne12) + if (src0->ne[2] != 1) { + return false; + } + + return true; +} + +bool ggml_cpu_iqp_supports_mul_mat_id(const struct ggml_tensor * dst) { + const struct ggml_tensor * ids = dst->src[2]; + + if (!iqp_supported_common(dst)) { + return false; + } + + // skip the node entirely (work buffer included) if no expert can reach the per expert threshold + if (!ggml_cpu_iqp_mul_mat_id_min_batch(ids->ne[0] * ids->ne[1])) { + return false; + } + + return true; +} + +void ggml_compute_forward_mul_mat_id_iqp(const struct ggml_compute_params * params, + struct ggml_tensor * dst, + int64_t cur_a, + int64_t cne1, + const int32_t * expert_rows, + void * panels) { + const struct ggml_tensor * src0 = dst->src[0]; + const struct ggml_tensor * src1 = dst->src[1]; + + GGML_TENSOR_BINARY_OP_LOCALS + + const int ith = params->ith; + const int nth = params->nth; + + const int64_t nblocks = ne00 / QK_K; + + const size_t nbw1 = ggml_cpu_iqp_row_size(dst); + + block_iqp_x8 * panel = (block_iqp_x8 *) ((char *) panels + (size_t) ith * ggml_cpu_iqp_scratch_size(dst)); + + const char * src0_cur = (const char *) src0->data + cur_a * nb02; + + const int64_t ngroups = ne01 / IQP_NB_ROWS; + + const int64_t g0 = (ngroups * ith) / nth; + const int64_t g1 = (ngroups * (ith + 1)) / nth; + + for (int64_t g = g0; g < g1; g++) { + const int64_t r = g * IQP_NB_ROWS; + + iqp_decode_panel_8(src0->type, src0_cur + r * nb01, nb01, nblocks, panel); + + // the dst rows are scattered, so the gemm writes into tmp and it is copied out row by row + float tmp[4 * IQP_NB_ROWS]; + + for (int64_t k = 0; k < cne1; k += 4) { + const int64_t nrows = MIN(4, cne1 - k); + + // a short tail tile duplicates its last row into the unused slots; the padding is never copied out + const void * rows[4]; + + for (int64_t m = 0; m < 4; m++) { + const int64_t kk = k + MIN(m, nrows - 1); + + rows[m] = (const char *) params->wdata + + ((expert_rows[2 * kk + 0] % ne11) + expert_rows[2 * kk + 1] * ne11) * nbw1; + } + + iqp_gemm_8x8_q8_K_p4(ne00, tmp, IQP_NB_ROWS, panel, rows, IQP_NB_ROWS); + + for (int64_t m = 0; m < nrows; m++) { + float * dst_col = (float *) ((char *) dst->data + expert_rows[2 * (k + m) + 0] * nb1 + + expert_rows[2 * (k + m) + 1] * nb2); + memcpy(dst_col + r, tmp + m * IQP_NB_ROWS, IQP_NB_ROWS * sizeof(float)); + } + } + } +} + +size_t ggml_cpu_iqp_scratch_size(const struct ggml_tensor * dst) { + return GGML_PAD((dst->src[0]->ne[0] / QK_K) * sizeof(block_iqp_x8), 64); +} + +void ggml_compute_forward_mul_mat_iqp(const struct ggml_compute_params * params, struct ggml_tensor * dst) { + const struct ggml_tensor * src0 = dst->src[0]; + const struct ggml_tensor * src1 = dst->src[1]; + + GGML_TENSOR_BINARY_OP_LOCALS + + const int ith = params->ith; + const int nth = params->nth; + + const int64_t nblocks = ne00 / QK_K; + + const size_t nbw1 = ggml_row_size(GGML_TYPE_Q8_K, ne10); + const size_t nbw2 = nbw1 * ne11; + + const size_t scratch_size = ggml_cpu_iqp_scratch_size(dst); + + const size_t scratch_offset = GGML_PAD(nbw2 * ne12, 64); + + GGML_ASSERT(scratch_offset + (size_t) nth * scratch_size <= params->wsize); + + block_iqp_x8 * panel = (block_iqp_x8 *) ((char *) params->wdata + scratch_offset + (size_t) ith * scratch_size); + + const int64_t nrows = ne11; + + const int64_t ngroups = ne01 / IQP_NB_ROWS; + + // aim for 4 chunks per thread; the caller has already reset the chunk counter + // on NUMA systems fall back to one chunk per thread + const int64_t chunks_per_thread = ggml_is_numa() ? 1 : 4; + const int64_t groups_per_chunk = MAX(1, (ngroups + nth * chunks_per_thread - 1) / (nth * chunks_per_thread)); + const int64_t nchunk = (ngroups + groups_per_chunk - 1) / groups_per_chunk; + + int current_chunk = ith; + + while (current_chunk < nchunk) { + const int64_t g0 = current_chunk * groups_per_chunk; + const int64_t g1 = MIN(g0 + groups_per_chunk, ngroups); + + for (int64_t g = g0; g < g1; g++) { + const int64_t r = g * IQP_NB_ROWS; + + iqp_decode_panel_8(src0->type, (const char *) src0->data + r * nb01, nb01, nblocks, panel); + + for (int64_t i12 = 0; i12 < ne12; i12++) { + const char * src1_ptr = (const char *) params->wdata + i12 * nbw2; + char * dst_ptr = (char *) dst->data + i12 * nb2; + + if (nrows > 3) { + iqp_gemm_8x8_q8_K(ne00, (float *) dst_ptr + r, nb1 / nb0, panel, src1_ptr, nrows - (nrows % 4), + IQP_NB_ROWS); + } + for (int64_t iter = nrows - (nrows % 4); iter < nrows; iter++) { + iqp_gemv_8x8_q8_K(ne00, (float *) (dst_ptr + iter * nb1) + r, ne01, panel, src1_ptr + nbw1 * iter, + 1 /* nrows */, IQP_NB_ROWS); + } + } + } + + current_chunk = ggml_threadpool_chunk_add(params->threadpool, 1); + } +} diff --git a/ggml/src/ggml-cpu/iqp.h b/ggml/src/ggml-cpu/iqp.h new file mode 100644 index 00000000..017b03fb --- /dev/null +++ b/ggml/src/ggml-cpu/iqp.h @@ -0,0 +1,39 @@ +#pragma once + +#include "ggml-cpu-impl.h" +#include "ggml.h" + +// GGML internal header + +// batched mul_mat path for the grid based IQ types: decode 8 src0 rows at a time into per thread scratch +// (block_iqp_x8, see iqp.cpp) and run an integer gemm over them against all src1 columns + +#ifdef __cplusplus +extern "C" { +#endif + +// whether cne1 rows of src1 are enough for the decode to pay for itself, per expert, for MUL_MAT_ID +bool ggml_cpu_iqp_mul_mat_id_min_batch(int64_t cne1); + +bool ggml_cpu_iqp_supports_mul_mat(const struct ggml_tensor * dst); + +// node level test only - per expert eligibility is decided with ggml_cpu_iqp_mul_mat_id_min_batch +bool ggml_cpu_iqp_supports_mul_mat_id(const struct ggml_tensor * dst); + +// per thread panel scratch bytes, padded +size_t ggml_cpu_iqp_scratch_size(const struct ggml_tensor * dst); + +// must be called after src1 has been converted to q8_K into params->wdata and the threads have synchronized on it +void ggml_compute_forward_mul_mat_iqp(const struct ggml_compute_params * params, struct ggml_tensor * dst); + +// one expert: expert_rows points at its row of the matrix_rows table of (i1, i2) int32 pairs, panels at the base of the per thread panel scratches +void ggml_compute_forward_mul_mat_id_iqp(const struct ggml_compute_params * params, + struct ggml_tensor * dst, + int64_t cur_a, + int64_t cne1, + const int32_t * expert_rows, + void * panels); + +#ifdef __cplusplus +} +#endif diff --git a/ggml/src/ggml-cpu/kleidiai/CMakeLists.txt b/ggml/src/ggml-cpu/kleidiai/CMakeLists.txt new file mode 100644 index 00000000..b36cb6d3 --- /dev/null +++ b/ggml/src/ggml-cpu/kleidiai/CMakeLists.txt @@ -0,0 +1,14 @@ +set(BUILD_SHARED_LIBS OFF) +set(CMAKE_SKIP_INSTALL_RULES TRUE) + +add_subdirectory("${KLEIDIAI_SRC}" "${KLEIDIAI_BIN}" EXCLUDE_FROM_ALL) + +if (NOT TARGET kleidiai) + message(FATAL_ERROR "KleidiAI target was not created") +endif() + +if (MSVC) + target_compile_options(kleidiai PRIVATE $<$:/WX->) +else() + target_compile_options(kleidiai PRIVATE $<$:-Wno-error>) +endif() diff --git a/ggml/src/ggml-cpu/kleidiai/kernels.cpp b/ggml/src/ggml-cpu/kleidiai/kernels.cpp index 3c31ab9d..d4551298 100644 --- a/ggml/src/ggml-cpu/kleidiai/kernels.cpp +++ b/ggml/src/ggml-cpu/kleidiai/kernels.cpp @@ -3,43 +3,44 @@ // // KleidiAI micro-kernels -#include "kai_matmul_clamp_f32_qsi8d32p_qsi4c32p_interface.h" -#include "kai_matmul_clamp_f32_qai8dxp_qsi8cxp_interface.h" -#include "kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4x4_1x4_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qsi8d32p4x4_qsi4c32p4x4_16x4_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p4x8_16x4_neon_i8mm.h" -#include "kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4vlx4_1x4vl_sme2_sdot.h" -#include "kai_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa.h" -#include "kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa.h" -#include "kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot.h" -#include "kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme_mopa.h" -#include "kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme_dot.h" -#include "kai_matmul_clamp_f32_qai8dxp1x8_qsi8cxp4x8_1x4_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4x4_1x4_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qai8dxp4x4_qsi8cxp4x4_16x4_neon_dotprod.h" -#include "kai_matmul_clamp_f32_qai8dxp4x8_qsi8cxp4x8_16x4_neon_i8mm.h" -#include "kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p8x8_16x8_sve_i8mm.h" -#include "kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p8x8_1x8_sve_dotprod.h" -#include "kai_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa.h" -#include "kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa.h" -#include "kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1b_2vlx2vl_sme_mopa.h" - -#include "kai_lhs_pack_bf16p2vlx2_f32_sme.h" -#include "kai_lhs_pack_f32p2vlx1_f32_sme.h" -#include "kai_lhs_quant_pack_qsi8d32p_f32.h" -#include "kai_lhs_quant_pack_qsi8d32p4x8sb_f32_neon.h" -#include "kai_lhs_quant_pack_qsi8d32p_f32_neon.h" -#include "kai_lhs_quant_pack_qai8dxp_f32.h" - -#include "kai_rhs_pack_kxn_bf16p2vlx2b_f32_x32_sme.h" -#include "kai_rhs_pack_nxk_f32p2vlx1biasf32_f32_f32_sme.h" -#include "kai_rhs_pack_nxk_qsi4c32pscalef16_qsu4c32s16s0.h" -#include "kai_rhs_pack_nxk_qsi4c32ps1s0scalef16_qsu4c32s16s0_neon.h" -#include "kai_rhs_pack_nxk_qsi8cxp_qsi8cx_neon.h" -#include "kai_lhs_pack_f16pmrx2_f32_neon.h" - -#include "kai_common.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p_qsi4c32p_interface.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp_qsi8cxp_interface.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4x4_1x4_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x4_qsi4c32p4x4_16x4_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p4x8_16x4_neon_i8mm.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4vlx4_1x4vl_sme2_sdot.h" +#include "kai/ukernels/matmul/matmul_clamp_fp32_bf16p_bf16p/kai_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme_mopa.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme_dot.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x8_qsi8cxp4x8_1x4_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4x4_1x4_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp4x4_qsi8cxp4x4_16x4_neon_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qai8dxp_qsi8cxp/kai_matmul_clamp_f32_qai8dxp4x8_qsi8cxp4x8_16x4_neon_i8mm.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p4x8_qsi4c32p8x8_16x8_sve_i8mm.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_qsi8d32p_qsi4c32p/kai_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p8x8_1x8_sve_dotprod.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_f16p_qsi4c32p/kai_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_f32_f32p/kai_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla.h" +#include "kai/ukernels/matmul/matmul_clamp_f32_f32p_f32p/kai_matmul_clamp_f32_f32p2vlx1_f32p2vlx1b_2vlx2vl_sme_mopa.h" + +#include "kai/ukernels/matmul/pack/kai_lhs_pack_bf16p2vlx2_f32_sme.h" +#include "kai/ukernels/matmul/pack/kai_lhs_pack_f32p2vlx1_f32_sme.h" +#include "kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p_f32.h" +#include "kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p4x8sb_f32_neon.h" +#include "kai/ukernels/matmul/pack/kai_lhs_quant_pack_qsi8d32p_f32_neon.h" +#include "kai/ukernels/matmul/pack/kai_lhs_quant_pack_qai8dxp_f32.h" + +#include "kai/ukernels/matmul/pack/kai_rhs_pack_kxn_bf16p2vlx2b_f32_x32_sme.h" +#include "kai/ukernels/matmul/pack/kai_rhs_pack_nxk_f32p2vlx1biasf32_f32_f32_sme.h" +#include "kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi4c32pscalef16_qsu4c32s16s0.h" +#include "kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi4c32ps1s0scalef16_qsu4c32s16s0_neon.h" +#include "kai/ukernels/matmul/pack/kai_rhs_pack_nxk_qsi8cxp_qsi8cx_neon.h" +#include "kai/ukernels/matmul/pack/kai_lhs_pack_f16pmrx2_f32_neon.h" + +#include "kai/kai_common.h" #include "simd-mappings.h" @@ -76,6 +77,21 @@ static inline void kernel_run_fn10(size_t m, size_t n, size_t k, size_t /*bl*/, Fn(m, n, k, lhs, rhs, dst, dst_stride_row, dst_stride_col, clamp_min, clamp_max); } +template +static inline void kernel_run_lhs_stride_fn10(size_t m, + size_t n, + size_t k, + size_t lhs_stride, + const void * lhs, + const void * rhs, + void * dst, + size_t dst_stride_row, + size_t dst_stride_col, + float clamp_min, + float clamp_max) { + Fn(m, n, k, lhs, lhs_stride, rhs, dst, dst_stride_row, dst_stride_col, clamp_min, clamp_max); +} + template static inline void kernel_run_float_fn10(size_t m, size_t n, size_t k, size_t /*bl*/, const void* lhs, const void* rhs, void* dst, @@ -312,9 +328,8 @@ static void dequantize_row_qsi8cxp( } static ggml_kleidiai_kernels gemm_gemv_kernels[] = { -#if defined(__ARM_FEATURE_SME) { - /* SME GEMM */ + /* SME2 GEMM */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_f16p1vlx2_qsi4c32p4vlx2_1vlx4vl_sme2_mopa, @@ -335,7 +350,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_size_ex = */ &lhs_ps_fn6, /* .pack_func_ex = */ &lhs_pack_void_fn10, }, - /* SME GEMV */ + /* SME2 GEMV */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4vlx4_1x4vl_sme2_sdot, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_qsi8d32p1x4_qsi4c32p4vlx4_1x4vl_sme2_sdot, @@ -362,13 +377,13 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_stride_ex = */ &rhs_stride_fn4, /* .pack_func_ex = */ &rhs_pack_fn12, }, - /* .required_cpu = */ CPU_FEATURE_SME2, + /* .required_cpu = */ CPU_FEATURE_SME2 | CPU_FEATURE_FP16, /* .lhs_type = */ GGML_TYPE_F32, /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, { - /* SME GEMM */ + /* SME2 GEMM */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa, @@ -388,7 +403,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_size_ex = */ &lhs_ps_fn5, /* .pack_func_ex = */ &lhs_pack_void_fn9, }, - /* SME GEMV */ + /* SME2 GEMV */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_bf16p2vlx2_bf16p2vlx2_2vlx2vl_sme2_mopa, @@ -420,9 +435,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .rhs_type = */ GGML_TYPE_F16, /* .op_type = */ GGML_TYPE_F32, }, -#endif #if defined(__APPLE__) -#if defined(__ARM_FEATURE_DOTPROD) { /* DOTPROD GEMM */ /* .kern_info = */ { @@ -476,8 +489,6 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif -#if defined(__ARM_FEATURE_MATMUL_INT8) { /* i8mm GEMM */ /* .kern_info = */ { @@ -499,7 +510,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_size_ex = */ &lhs_ps_fn6, /* .pack_func_ex = */ &lhs_pack_float_fn10, }, - /* i8mm GEMV */ + /* DOTPROD GEMV */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod, @@ -526,14 +537,12 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_stride_ex = */ &rhs_stride_fn4, /* .pack_func_ex = */ &rhs_pack_fn12, }, - /* .required_cpu = */ CPU_FEATURE_I8MM, + /* .required_cpu = */ CPU_FEATURE_I8MM | CPU_FEATURE_DOTPROD, /* .lhs_type = */ GGML_TYPE_F32, /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif #else -#if defined(__ARM_FEATURE_SVE) { /* SVE i8mm GEMM */ /* .kern_info = */ { @@ -587,8 +596,6 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif -#if defined(__ARM_FEATURE_MATMUL_INT8) { /* i8mm GEMM */ /* .kern_info = */ { @@ -610,7 +617,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_size_ex = */ &lhs_ps_fn6, /* .pack_func_ex = */ &lhs_pack_float_fn10, }, - /* i8mm GEMV */ + /* DOTPROD GEMV */ /* .kern_info = */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_qsi8d32p1x8_qsi4c32p4x8_1x4x32_neon_dotprod, @@ -637,13 +644,11 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .packed_stride_ex = */ &rhs_stride_fn4, /* .pack_func_ex = */ &rhs_pack_fn12, }, - /* .required_cpu = */ CPU_FEATURE_I8MM, + /* .required_cpu = */ CPU_FEATURE_I8MM | CPU_FEATURE_DOTPROD, /* .lhs_type = */ GGML_TYPE_F32, /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif // __ARM_FEATURE_MATMUL_INT8 -#if defined(__ARM_FEATURE_DOTPROD) { /* DOTPROD GEMM */ /* .kern_info = */ { @@ -697,15 +702,13 @@ static ggml_kleidiai_kernels gemm_gemv_kernels[] = { /* .rhs_type = */ GGML_TYPE_Q4_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif #endif { /* Sentinel */ } }; static ggml_kleidiai_kernels gemm_gemv_kernels_q8[] = { -#if defined(__ARM_FEATURE_SME) { - /* SME GEMM */ + /* SME2 GEMM */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_qai8dxp1vlx4_qsi8cxp4vlx4_1vlx4vl_sme2_mopa, @@ -725,7 +728,7 @@ static ggml_kleidiai_kernels gemm_gemv_kernels_q8[] = { /* .packed_size_ex = */ &lhs_ps_fn5, /* .pack_func_ex = */ &lhs_pack_float_fn9_no_bl, }, - /* SME GEMV */ + /* SME2 GEMV */ { /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot, /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_qai8dxp1x4_qsi8cxp4vlx4_1x4vl_sme2_dot, @@ -810,8 +813,6 @@ static ggml_kleidiai_kernels gemm_gemv_kernels_q8[] = { /* .rhs_type = */ GGML_TYPE_Q8_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif -#if defined(__ARM_FEATURE_MATMUL_INT8) { /* I8MM GEMM */ { @@ -860,13 +861,11 @@ static ggml_kleidiai_kernels gemm_gemv_kernels_q8[] = { /* .packed_stride_ex = */ &rhs_stride_fn4, /* .pack_func_ex = */ &rhs_pack_scale_fn12, }, - /* .required_cpu = */ CPU_FEATURE_I8MM, + /* .required_cpu = */ CPU_FEATURE_I8MM | CPU_FEATURE_DOTPROD, /* .lhs_type = */ GGML_TYPE_F32, /* .rhs_type = */ GGML_TYPE_Q8_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif -#if defined(__ARM_FEATURE_DOTPROD) { /* DOTPROD GEMM */ { @@ -920,12 +919,10 @@ static ggml_kleidiai_kernels gemm_gemv_kernels_q8[] = { /* .rhs_type = */ GGML_TYPE_Q8_0, /* .op_type = */ GGML_TYPE_F32, }, -#endif { /* Sentinel */ } }; static ggml_kleidiai_kernels ggml_kleidiai_kernels_f32[] = { -#if defined(__ARM_FEATURE_SME) { /* SME2 GEMM */ { @@ -947,25 +944,25 @@ static ggml_kleidiai_kernels ggml_kleidiai_kernels_f32[] = { /* .packed_size_ex = */ &lhs_ps_fn5, /* .pack_func_ex = */ &lhs_pack_void_fn9, }, - /* SME GEMV */ + /* SME2 GEMV */ { - /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_mr = */ kai_get_mr_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_nr = */ kai_get_nr_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_kr = */ kai_get_kr_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_sr = */ kai_get_sr_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_dst_offset = */ kai_get_dst_offset_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_dst_size = */ kai_get_dst_size_matmul_clamp_f32_f32p2vlx1_f32p2vlx1biasf32_sme2_mopa, - /* .get_lhs_offset_ex = */ nullptr, - /* .get_rhs_packed_offset_ex = */ nullptr, - /* .run_kernel_ex = */ nullptr, + /* .get_m_step = */ kai_get_m_step_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_n_step = */ kai_get_n_step_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_mr = */ kai_get_m_step_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_nr = */ kai_get_nr_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_kr = */ kai_get_kr_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_sr = */ kai_get_sr_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_dst_offset = */ kai_get_dst_offset_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_dst_size = */ kai_get_dst_size_matmul_clamp_f32_f32_f32p2vlx1b_1x16vl_sme2_mla, + /* .get_lhs_offset_ex = */ &kernel_offs_fn2, + /* .get_rhs_packed_offset_ex = */ &kernel_offs_fn2, + /* .run_kernel_ex = */ &kernel_run_lhs_stride_fn10, }, /* .gemv_lhs_info = */ { - /* .get_offset = */ kai_get_lhs_offset_lhs_pack_f32p2vlx1_f32_sme, - /* .get_packed_offset_ex = */ &lhs_offs_fn5, - /* .packed_size_ex = */ &lhs_ps_fn5, - /* .pack_func_ex = */ &lhs_pack_void_fn9, + /* .get_offset = */ nullptr, + /* .get_packed_offset_ex = */ nullptr, + /* .packed_size_ex = */ nullptr, + /* .pack_func_ex = */ nullptr, }, /* .rhs_info = */ { /* .packed_stride = */ nullptr, @@ -1032,7 +1029,6 @@ static ggml_kleidiai_kernels ggml_kleidiai_kernels_f32[] = { /* .rhs_type = */ GGML_TYPE_F32, /* .op_type = */ GGML_TYPE_F32, }, -#endif { /* Sentinel */ } }; @@ -1040,10 +1036,6 @@ ggml_kleidiai_kernels * ggml_kleidiai_select_kernels(cpu_feature cpu_features, c ggml_kleidiai_kernels * kernel = nullptr; if (tensor->op == GGML_OP_MUL_MAT && tensor->src[0] != nullptr && tensor->src[1] != nullptr) { -#if defined(__ARM_FEATURE_SME) || \ - defined(__ARM_FEATURE_DOTPROD) || \ - defined(__ARM_FEATURE_MATMUL_INT8) || \ - defined(__ARM_FEATURE_SVE) auto try_table = [&](auto & table) { for (size_t i = 0; i < NELEMS(table) - 1; ++i) { if ((cpu_features & table[i].required_cpu) == table[i].required_cpu && @@ -1064,12 +1056,6 @@ ggml_kleidiai_kernels * ggml_kleidiai_select_kernels(cpu_feature cpu_features, c } else { try_table(gemm_gemv_kernels); } -#else - GGML_UNUSED(gemm_gemv_kernels); - GGML_UNUSED(gemm_gemv_kernels_q8); - GGML_UNUSED(ggml_kleidiai_kernels_f32); - GGML_UNUSED(cpu_features); -#endif } return kernel; @@ -1078,19 +1064,13 @@ ggml_kleidiai_kernels * ggml_kleidiai_select_kernels(cpu_feature cpu_features, c ggml_kleidiai_kernels * ggml_kleidiai_select_kernels_q4_0(cpu_feature features) { ggml_kleidiai_kernels * kernels = nullptr; -#if defined(__ARM_FEATURE_SME) || \ - defined(__ARM_FEATURE_DOTPROD) || \ - defined(__ARM_FEATURE_MATMUL_INT8) || \ - defined(__ARM_FEATURE_SVE) for (size_t i = 0; i < NELEMS(gemm_gemv_kernels) - 1; ++i) { - if ((features & gemm_gemv_kernels[i].required_cpu) == gemm_gemv_kernels[i].required_cpu) { + if ((features & gemm_gemv_kernels[i].required_cpu) == gemm_gemv_kernels[i].required_cpu && + gemm_gemv_kernels[i].rhs_type == GGML_TYPE_Q4_0) { kernels = &gemm_gemv_kernels[i]; break; } } -#else - GGML_UNUSED(features); -#endif return kernels; } @@ -1098,16 +1078,12 @@ ggml_kleidiai_kernels * ggml_kleidiai_select_kernels_q4_0(cpu_feature features) ggml_kleidiai_kernels * ggml_kleidiai_select_kernels_q8_0(cpu_feature features) { ggml_kleidiai_kernels * kernels = nullptr; -#if defined(__ARM_FEATURE_SME) || defined(__ARM_FEATURE_DOTPROD) || defined(__ARM_FEATURE_MATMUL_INT8) for (size_t i = 0; i < NELEMS(gemm_gemv_kernels_q8) - 1; ++i) { if ((features & gemm_gemv_kernels_q8[i].required_cpu) == gemm_gemv_kernels_q8[i].required_cpu) { kernels = &gemm_gemv_kernels_q8[i]; break; } } -#else - GGML_UNUSED(features); -#endif return kernels; } @@ -1115,16 +1091,11 @@ ggml_kleidiai_kernels * ggml_kleidiai_select_kernels_q8_0(cpu_feature features) ggml_kleidiai_kernels * ggml_kleidiai_select_kernels_f32(cpu_feature features) { ggml_kleidiai_kernels * kernels = nullptr; -#if defined(__ARM_FEATURE_SME) for (size_t i = 0; i < NELEMS(ggml_kleidiai_kernels_f32) - 1; ++i) { if ((features & ggml_kleidiai_kernels_f32[i].required_cpu) == ggml_kleidiai_kernels_f32[i].required_cpu) { kernels = &ggml_kleidiai_kernels_f32[i]; break; } } -#else - GGML_UNUSED(features); -#endif - return kernels; } diff --git a/ggml/src/ggml-cpu/kleidiai/kernels.h b/ggml/src/ggml-cpu/kleidiai/kernels.h index 0da5e65a..1da8610e 100644 --- a/ggml/src/ggml-cpu/kleidiai/kernels.h +++ b/ggml/src/ggml-cpu/kleidiai/kernels.h @@ -1,4 +1,4 @@ -// SPDX-FileCopyrightText: Copyright 2025 Arm Limited and/or its affiliates +// SPDX-FileCopyrightText: Copyright 2025-2026 Arm Limited and/or its affiliates // SPDX-License-Identifier: MIT // @@ -12,7 +12,8 @@ enum cpu_feature { CPU_FEATURE_I8MM = 2, CPU_FEATURE_SVE = 4, CPU_FEATURE_SME = 8, - CPU_FEATURE_SME2 = 16 + CPU_FEATURE_SME2 = 16, + CPU_FEATURE_FP16 = 32 }; inline cpu_feature& operator|=(cpu_feature& lhs, cpu_feature rhs) { diff --git a/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp b/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp index 2266c168..dbd19878 100644 --- a/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp +++ b/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp @@ -48,7 +48,7 @@ #include "kernels.h" -#include "kai_common.h" +#include "kai/kai_common.h" #define GGML_COMMON_DECL_CPP #include "ggml-common.h" @@ -316,6 +316,7 @@ static void init_kleidiai_context(void) { ctx.features = (runtime_feat.has_dotprod ? CPU_FEATURE_DOTPROD : CPU_FEATURE_NONE) | (runtime_feat.has_i8mm ? CPU_FEATURE_I8MM : CPU_FEATURE_NONE) | + (runtime_feat.has_fp16 ? CPU_FEATURE_FP16 : CPU_FEATURE_NONE) | (runtime_feat.sve_cnt == QK8_0 ? CPU_FEATURE_SVE : CPU_FEATURE_NONE); if (env_threads) { @@ -696,6 +697,15 @@ class tensor_traits : public ggml::cpu::tensor_traits { } if (op->src[0]->type == GGML_TYPE_F32) { + ggml_kleidiai_kernels * primary = kernel_chain[0]; + kernel_info * gemv_kernel = primary ? &primary->gemv : nullptr; + if (is_gemv && op->src[1]->nb[0] == (int64_t) sizeof(float) && gemv_kernel && + gemv_kernel->get_lhs_offset_ex && gemv_kernel->get_rhs_packed_offset_ex && + gemv_kernel->run_kernel_ex && gemv_kernel->get_dst_offset) { + size = 0; + return true; + } + size_t cursor = 0; bool any_slot = false; @@ -811,15 +821,28 @@ class tensor_traits : public ggml::cpu::tensor_traits { return false; } - kernel_info * kernel = &kernels->gemm; + const size_t k = ne00; + const size_t m = ne11; + const size_t n = ne01; + const bool use_gemv = m == 1 && src1->nb[0] == (int64_t) sizeof(float) && + kernels->gemv.get_lhs_offset_ex && + kernels->gemv.get_rhs_packed_offset_ex && + kernels->gemv.run_kernel_ex && + kernels->gemv.get_dst_offset; + + kernel_info * kernel = use_gemv ? &kernels->gemv : &kernels->gemm; lhs_packing_info * lhs_info = &kernels->gemm_lhs_info; - if (!kernel || !lhs_info || !lhs_info->get_offset || !lhs_info->get_packed_offset_ex || - !lhs_info->packed_size_ex || !lhs_info->pack_func_ex || + if (!kernel || !kernel->get_lhs_offset_ex || !kernel->get_rhs_packed_offset_ex || !kernel->run_kernel_ex || !kernel->get_dst_offset) { return false; } + if (!use_gemv && (!lhs_info || !lhs_info->get_offset || !lhs_info->get_packed_offset_ex || + !lhs_info->packed_size_ex || !lhs_info->pack_func_ex)) { + return false; + } + const kleidiai_weight_header * header = kleidiai_weight_header_from_ptr(src0->data); const bool has_header = kleidiai_is_weight_header_valid(header); @@ -832,16 +855,14 @@ class tensor_traits : public ggml::cpu::tensor_traits { const int nth = params->nth > 0 ? params->nth : 1; const int ith = params->ith; - const size_t k = ne00; - const size_t m = ne11; - const size_t n = ne01; - const size_t mr = kernel->get_mr(); const size_t kr = kernel->get_kr(); const size_t sr = kernel->get_sr(); - const size_t lhs_packed_size = lhs_info->packed_size_ex(m, k, 0, mr, kr, sr); - GGML_ASSERT(lhs_packed_size <= params->wsize); + const size_t lhs_packed_size = use_gemv ? 0 : lhs_info->packed_size_ex(m, k, 0, mr, kr, sr); + if (!use_gemv) { + GGML_ASSERT(lhs_packed_size <= params->wsize); + } uint8_t * lhs_packed = static_cast(params->wdata); const size_t dst_stride = dst->nb[1]; @@ -853,7 +874,7 @@ class tensor_traits : public ggml::cpu::tensor_traits { const uint8_t * lhs_batch_base = static_cast(src1->data) + batch_idx * src1->nb[2]; uint8_t * dst_batch_base = static_cast(dst->data) + batch_idx * dst->nb[2]; - { + if (!use_gemv) { const int64_t m_roundup_mr = kai_roundup((int64_t)m, (int64_t)mr); int64_t max_threads = mr ? (m_roundup_mr / (int64_t)mr) : nth; max_threads = std::max(1, max_threads); @@ -903,15 +924,17 @@ class tensor_traits : public ggml::cpu::tensor_traits { const size_t n_to_process = std::min(chunk_cols, n - n_start); if (n_to_process > 0) { - const size_t lhs_packed_offset = lhs_info->get_packed_offset_ex(0, k, 0, mr, kr, sr); + const size_t lhs_offset = use_gemv ? kernel->get_lhs_offset_ex(0, k, 0) + : lhs_info->get_packed_offset_ex(0, k, 0, mr, kr, sr); const size_t rhs_packed_offset = kernel->get_rhs_packed_offset_ex(n_start, k, 0); const size_t dst_offset = kernel->get_dst_offset(0, n_start, dst_stride); - const void * lhs_ptr = lhs_packed + lhs_packed_offset; + const void * lhs_ptr = use_gemv ? lhs_batch_base + lhs_offset + : lhs_packed + lhs_offset; const void * rhs_ptr = rhs_base + rhs_packed_offset; float * dst_ptr = reinterpret_cast(dst_batch_base + dst_offset); - kernel->run_kernel_ex(m, n_to_process, k, 0, + kernel->run_kernel_ex(m, n_to_process, k, use_gemv ? src1->nb[1] : 0, lhs_ptr, rhs_ptr, dst_ptr, @@ -1800,7 +1823,7 @@ class extra_buffer_type : ggml::cpu::extra_buffer_type { const bool src0_is_kleidiai = op->src[0]->buffer && (ggml_n_dims(op->src[0]) == 2) && - op->src[0]->buffer->buft == ggml_backend_cpu_kleidiai_buffer_type() && + op->src[0]->buffer->buft->context == this && slot_total > 0; if ((op->op == GGML_OP_MUL_MAT || op->op == GGML_OP_GET_ROWS) && @@ -1839,7 +1862,7 @@ class extra_buffer_type : ggml::cpu::extra_buffer_type { ggml::cpu::tensor_traits * get_tensor_traits(const struct ggml_tensor * op) override { if (op->op == GGML_OP_MUL_MAT || op->op == GGML_OP_GET_ROWS) { - if (op->src[0]->buffer && op->src[0]->buffer->buft == ggml_backend_cpu_kleidiai_buffer_type()) { + if (op->src[0]->buffer && op->src[0]->buffer->buft->context == this) { return (ggml::cpu::tensor_traits *) op->src[0]->extra; } else { // KleidiAI only has kernels for Q4_0 and Q8_0. For a quantized weight of any diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp index 001e1ae8..ba00a0a7 100644 --- a/ggml/src/ggml-cpu/ops.cpp +++ b/ggml/src/ggml-cpu/ops.cpp @@ -1896,7 +1896,6 @@ void ggml_compute_forward_repeat_back( } // ggml_compute_forward_concat - static void ggml_compute_forward_concat_any( const ggml_compute_params * params, ggml_tensor * dst) { @@ -1904,8 +1903,6 @@ static void ggml_compute_forward_concat_any( const ggml_tensor * src0 = dst->src[0]; const ggml_tensor * src1 = dst->src[1]; - const size_t len = ggml_type_size(src0->type); - const int ith = params->ith; const int nth = params->nth; @@ -1914,31 +1911,38 @@ static void ggml_compute_forward_concat_any( const int32_t dim = ggml_get_op_params_i32(dst, 0); GGML_ASSERT(dim >= 0 && dim < 4); + GGML_ASSERT(ggml_is_contiguous_rows(src0)); + GGML_ASSERT(ggml_is_contiguous_rows(src1)); int64_t o[4] = {0, 0, 0, 0}; + if (dim == 0) { + GGML_ASSERT(src0->ne[0] % ggml_blck_size(src0->type) == 0); + GGML_ASSERT(src1->ne[0] % ggml_blck_size(src1->type) == 0); + o[dim] = src0->ne[dim]/ggml_blck_size(src0->type); } else { o[dim] = src0->ne[dim]; } - const char * x; - - // TODO: smarter multi-theading - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = ith; i2 < ne2; i2 += nth) { - for (int i1 = 0; i1 < ne1; i1++) { - for (int i0 = 0; i0 < ne0/ggml_blck_size(dst->type); i0++) { - if (i0 < ne00/ggml_blck_size(src0->type) && i1 < ne01 && i2 < ne02 && i3 < ne03) { - x = (const char *)src0->data + (i0 )*nb00 + (i1 )*nb01 + (i2 )*nb02 + (i3 )*nb03; - } else { - x = (const char *)src1->data + (i0 - o[0])*nb10 + (i1 - o[1])*nb11 + (i2 - o[2])*nb12 + (i3 - o[3])*nb13; - } - - char * y = (char *)dst->data + i0*nb0 + i1*nb1 + i2*nb2 + i3*nb3; + // Region 1: copy rows from src0 + for (int i3 = 0; i3 < ne03; i3++) { + for (int i2 = ith; i2 < ne02; i2 += nth) { + for (int i1 = 0; i1 < ne01; i1++) { + const char * x = (const char *) src0->data + i1*nb01 + i2*nb02 + i3*nb03; + char * y = ( char *) dst->data + i1*nb1 + i2*nb2 + i3*nb3; + memcpy(y, x, ggml_row_size(src0->type, ne00)); + } + } + } - memcpy(y, x, len); - } + // Region 2: copy rows from src1, offset into dst by o[] + for (int i3 = 0; i3 < ne13; i3++) { + for (int i2 = ith; i2 < ne12; i2 += nth) { + for (int i1 = 0; i1 < ne11; i1++) { + const char * x = (const char *) src1->data + i1*nb11 + i2*nb12 + i3*nb13; + char * y = ( char *) dst->data + (i1 + o[1])*nb1 + (i2 + o[2])*nb2 + (i3 + o[3])*nb3 + o[0]*nb0; + memcpy(y, x, ggml_row_size(src1->type, ne10)); } } } @@ -2078,14 +2082,6 @@ void ggml_compute_forward_concat( ggml_tensor * dst) { const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - - if (ggml_is_quantized(src0->type)) { - GGML_ASSERT(ggml_is_contiguous_rows(src0)); - GGML_ASSERT(ggml_is_contiguous_rows(src1)); - GGML_ASSERT(src0->ne[0] % ggml_blck_size(src0->type) == 0); - GGML_ASSERT(src1->ne[0] % ggml_blck_size(src1->type) == 0); - } switch (src0->type) { case GGML_TYPE_F16: @@ -3407,6 +3403,139 @@ static void ggml_compute_forward_swiglu_oai( } } +// ggml_compute_forward_swiglu_clamp + +static void ggml_compute_forward_swiglu_clamp_f32(const ggml_compute_params * params, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + char * src0_d = (char *) src0->data; + char * src1_d = (char *) (src1 ? src1->data : src0->data); + const size_t src0_o = src0->nb[1]; + const size_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; + + GGML_ASSERT(ggml_is_contiguous_1(src0)); + GGML_ASSERT(ggml_is_contiguous_1(dst)); + + if (src1) { + GGML_ASSERT(ggml_is_contiguous_1(src1)); + GGML_ASSERT(src0->type == src1->type); + } + + const int ith = params->ith; + const int nth = params->nth; + + const int nc = src1 ? src0->ne[0] : src0->ne[0] / 2; + const int nr = ggml_nrows(src0); + + GGML_ASSERT(dst->ne[0] == nc); + GGML_ASSERT(ggml_nrows(dst) == nr); + + const int32_t swapped = ggml_get_op_params_i32(dst, 1); + const float limit = ggml_get_op_params_f32(dst, 3); + + const int dr = (nr + nth - 1) / nth; + const int ir0 = dr * ith; + const int ir1 = MIN(ir0 + dr, nr); + + for (int i1 = ir0; i1 < ir1; i1++) { + float * src0_p = (float *) (src0_d + i1 * src0_o); + float * src1_p = (float *) (src1_d + i1 * src1_o); + float * dst_p = (float *) ((char *) dst->data + i1 * (dst->nb[1])); + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + for (int k = 0; k < nc; k++) { + const float gate = std::min(src0_p[k], limit); + const float up = std::clamp(src1_p[k], -limit, limit); + dst_p[k] = gate / (1.f + expf(-gate)) * up; + } + +#ifndef NDEBUG + for (int k = 0; k < nc; k++) { + const float x = dst_p[k]; + GGML_UNUSED(x); + assert(!isnan(x)); + assert(!isinf(x)); + } +#endif // NDEBUG + } +} + +static void ggml_compute_forward_swiglu_clamp_f16(const ggml_compute_params * params, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + char * src0_d = (char *) src0->data; + char * src1_d = (char *) (src1 ? src1->data : src0->data); + const size_t src0_o = src0->nb[1]; + const size_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; + + GGML_ASSERT(ggml_is_contiguous_1(src0)); + GGML_ASSERT(ggml_is_contiguous_1(dst)); + + if (src1) { + GGML_ASSERT(ggml_is_contiguous_1(src1)); + GGML_ASSERT(src0->type == src1->type); + } + + const int ith = params->ith; + const int nth = params->nth; + + const int nc = src1 ? src0->ne[0] : src0->ne[0] / 2; + const int nr = ggml_nrows(src0); + + GGML_ASSERT(dst->ne[0] == nc); + GGML_ASSERT(ggml_nrows(dst) == nr); + + const int32_t swapped = ggml_get_op_params_i32(dst, 1); + const float limit = ggml_get_op_params_f32(dst, 3); + + const int dr = (nr + nth - 1) / nth; + const int ir0 = dr * ith; + const int ir1 = MIN(ir0 + dr, nr); + + for (int i1 = ir0; i1 < ir1; i1++) { + ggml_fp16_t * src0_p = (ggml_fp16_t *) (src0_d + i1 * src0_o); + ggml_fp16_t * src1_p = (ggml_fp16_t *) (src1_d + i1 * src1_o); + ggml_fp16_t * dst_p = (ggml_fp16_t *) ((char *) dst->data + i1 * (dst->nb[1])); + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + for (int k = 0; k < nc; k++) { + const float gate = std::min(GGML_FP16_TO_FP32(src0_p[k]), limit); + const float up = std::clamp(GGML_FP16_TO_FP32(src1_p[k]), -limit, limit); + dst_p[k] = GGML_FP32_TO_FP16(gate / (1.f + expf(-gate)) * up); + } + +#ifndef NDEBUG + for (int k = 0; k < nc; k++) { + const float x = GGML_FP16_TO_FP32(dst_p[k]); + GGML_UNUSED(x); + assert(!isnan(x)); + assert(!isinf(x)); + } +#endif // NDEBUG + } +} + +static void ggml_compute_forward_swiglu_clamp(const ggml_compute_params * params, ggml_tensor * dst) { + switch (dst->src[0]->type) { + case GGML_TYPE_F32: + ggml_compute_forward_swiglu_clamp_f32(params, dst); + break; + case GGML_TYPE_F16: + ggml_compute_forward_swiglu_clamp_f16(params, dst); + break; + default: + GGML_ABORT("fatal error"); + } +} + // ggml_compute_forward_geglu_erf static void ggml_compute_forward_geglu_erf_f32( @@ -5979,6 +6108,8 @@ static void ggml_compute_forward_rope_flt( memcpy(&beta_slow, (int32_t *) dst->op_params + 10, sizeof(float)); memcpy(§ions, (int32_t *) dst->op_params + 11, sizeof(int)*4); + const int n_offs = ((int32_t *) dst->op_params)[15]; + GGML_TENSOR_UNARY_OP_LOCALS //printf("ne0: %d, ne1: %d, ne2: %d, ne3: %d\n", ne0, ne1, ne2, ne3); @@ -5995,6 +6126,10 @@ static void ggml_compute_forward_rope_flt( GGML_ASSERT(n_dims <= ne0); GGML_ASSERT(n_dims % 2 == 0); + GGML_ASSERT(n_offs >= 0); + GGML_ASSERT(n_offs % 2 == 0); + GGML_ASSERT(n_offs + n_dims <= ne0); + // rows per thread const int dr = (nr + nth - 1)/nth; @@ -6020,6 +6155,7 @@ static void ggml_compute_forward_rope_flt( if (is_vision) { GGML_ASSERT(n_dims == ne0/2); + GGML_ASSERT(n_offs == 0); } const float * freq_factors = NULL; @@ -6068,12 +6204,12 @@ static void ggml_compute_forward_rope_flt( switch (mode) { case GGML_ROPE_TYPE_NORMAL: - rotate_pairs(n_dims, 1, cache, src, dst_data, 1); + rotate_pairs(n_dims, 1, cache, src + n_offs, dst_data + n_offs, 1); break; case GGML_ROPE_TYPE_NEOX: case GGML_ROPE_TYPE_MROPE: case GGML_ROPE_TYPE_IMROPE: - rotate_pairs(n_dims, n_dims/2, cache, src, dst_data); + rotate_pairs(n_dims, n_dims/2, cache, src + n_offs, dst_data + n_offs); break; case GGML_ROPE_TYPE_VISION: rotate_pairs(ne0, n_dims, cache, src, dst_data); @@ -6084,7 +6220,11 @@ static void ggml_compute_forward_rope_flt( if (!is_vision) { // fill the remain channels with data from src tensor - for (int64_t i0 = n_dims; i0 < ne0; i0 += 2) { + for (int64_t i0 = 0; i0 < ne0; i0 += 2) { + if (i0 == n_offs) { + i0 += n_dims - 2; // skip the rotated channels + continue; + } const T * const src = (T *)((char *) src0->data + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00); T * dst_data = (T *)((char *) dst->data + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); @@ -7260,18 +7400,21 @@ static void ggml_compute_forward_conv_transpose_2d_impl( } } - // permute source data (src1) from (Sw x Sh x Cin) to (Cin x Sw x Sh) + // permute source data (src1) from (Sw x Sh x Cin) to (Cin x Sw x Sh), for all batches { kernel_t * const wdata = (kernel_t *) params->wdata + nk; - for (int i12 = 0; i12 < ne12; i12++) { - for (int i11 = 0; i11 < ne11; i11++) { - const float * const src = (float *)((char *) src1->data + i12*nb12 + i11*nb11); - kernel_t * dst_data = wdata + i11*ne10*ne12; - for (int i10 = 0; i10 < ne10; i10++) { - if constexpr (std::is_same_v) { - dst_data[i10*ne12 + i12] = GGML_CPU_FP32_TO_FP16(src[i10]); - } else { - dst_data[i10*ne12 + i12] = src[i10]; + for (int i13 = 0; i13 < ne13; i13++) { + kernel_t * const wdata_b = wdata + i13*ne10*ne11*ne12; + for (int i12 = 0; i12 < ne12; i12++) { + for (int i11 = 0; i11 < ne11; i11++) { + const float * const src = (float *)((char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11); + kernel_t * dst_data = wdata_b + i11*ne10*ne12; + for (int i10 = 0; i10 < ne10; i10++) { + if constexpr (std::is_same_v) { + dst_data[i10*ne12 + i12] = GGML_CPU_FP32_TO_FP16(src[i10]); + } else { + dst_data[i10*ne12 + i12] = src[i10]; + } } } } @@ -7298,24 +7441,27 @@ static void ggml_compute_forward_conv_transpose_2d_impl( kernel_t * const wdata_src = wdata + nk; for (int i2 = ip0; i2 < ip1; i2++) { // Cout - float * dst_data = (float *)((char *) dst->data + i2*nb2); kernel_t * wdata_kernel = wdata + i2*ne01*ne00*ne03; - for (int i11 = 0; i11 < ne11; i11++) { - for (int i10 = 0; i10 < ne10; i10++) { - const int i1n = i11*ne10*ne12 + i10*ne12; - for (int i01 = 0; i01 < ne01; i01++) { - for (int i00 = 0; i00 < ne00; i00++) { - float v = 0; - if constexpr (std::is_same_v) { - ggml_vec_dot_f16(ne03, &v, 0, - wdata_src + i1n, 0, - wdata_kernel + i01*ne00*ne03 + i00*ne03, 0, 1); - } else { - ggml_vec_dot_f32(ne03, &v, 0, - wdata_src + i1n, 0, - wdata_kernel + i01*ne00*ne03 + i00*ne03, 0, 1); + for (int i3 = 0; i3 < ne3; i3++) { // batch + float * dst_data = (float *)((char *) dst->data + i3*nb3 + i2*nb2); + kernel_t * wdata_src_b = wdata_src + i3*ne10*ne11*ne12; + for (int i11 = 0; i11 < ne11; i11++) { + for (int i10 = 0; i10 < ne10; i10++) { + const int i1n = i11*ne10*ne12 + i10*ne12; + for (int i01 = 0; i01 < ne01; i01++) { + for (int i00 = 0; i00 < ne00; i00++) { + float v = 0; + if constexpr (std::is_same_v) { + ggml_vec_dot_f16(ne03, &v, 0, + wdata_src_b + i1n, 0, + wdata_kernel + i01*ne00*ne03 + i00*ne03, 0, 1); + } else { + ggml_vec_dot_f32(ne03, &v, 0, + wdata_src_b + i1n, 0, + wdata_kernel + i01*ne00*ne03 + i00*ne03, 0, 1); + } + dst_data[(i11*stride + i01)*ne0 + i10*stride + i00] += v; } - dst_data[(i11*stride + i01)*ne0 + i10*stride + i00] += v; } } } @@ -10123,6 +10269,10 @@ void ggml_compute_forward_glu( { ggml_compute_forward_geglu_quick(params, dst); } break; + case GGML_GLU_OP_SWIGLU_CLAMP: + { + ggml_compute_forward_swiglu_clamp(params, dst); + } break; default: { GGML_ABORT("fatal error"); @@ -11109,10 +11259,19 @@ static void ggml_compute_forward_dsv4_hc_pre_f32( const int64_t hc = x->ne[1]; const int64_t n_tokens = x->ne[2]; + const float scale = ggml_get_op_params_f32(dst, 0); + const bool gated = ggml_get_op_params_i32(dst, 1) != 0; + GGML_ASSERT(dst->ne[0] == n_embd); GGML_ASSERT(dst->ne[1] == n_tokens); - GGML_ASSERT(weights->ne[0] == hc); - GGML_ASSERT(weights->ne[1] == n_tokens); + if (gated) { + GGML_ASSERT(weights->ne[0] == n_embd); + GGML_ASSERT(weights->ne[1] == hc); + GGML_ASSERT(weights->ne[2] == n_tokens); + } else { + GGML_ASSERT(weights->ne[0] == hc); + GGML_ASSERT(weights->ne[1] == n_tokens); + } GGML_TENSOR_LOCALS(size_t, nbx, x, nb); GGML_TENSOR_LOCALS(size_t, nbw, weights, nb); @@ -11132,12 +11291,18 @@ static void ggml_compute_forward_dsv4_hc_pre_f32( float sum = 0.0f; for (int64_t ih = 0; ih < hc; ++ih) { - const float xv = *(const float *) ((const char *) x->data + i0*nbx0 + ih*nbx1 + it*nbx2); - const float wv = *(const float *) ((const char *) weights->data + ih*nbw0 + it*nbw1); + const float xv = *(const float *) ((const char *) x->data + i0*nbx0 + ih*nbx1 + it*nbx2); + float wv; + if (gated) { + const float gv = *(const float *) ((const char *) weights->data + i0*nbw0 + ih*nbw1 + it*nbw2); + wv = 1.0f / (1.0f + expf(-gv)); + } else { + wv = *(const float *) ((const char *) weights->data + ih*nbw0 + it*nbw1); + } sum += xv * wv; } - *(float *) ((char *) dst->data + i0*nbd0 + it*nbd1) = sum; + *(float *) ((char *) dst->data + i0*nbd0 + it*nbd1) = scale * sum; } } @@ -11171,7 +11336,6 @@ static void ggml_compute_forward_dsv4_hc_post_f32( GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(residual->type == GGML_TYPE_F32); GGML_ASSERT(post->type == GGML_TYPE_F32); - GGML_ASSERT(comb->type == GGML_TYPE_F32); GGML_ASSERT(dst->type == GGML_TYPE_F32); const int64_t n_embd = x->ne[0]; @@ -11185,14 +11349,24 @@ static void ggml_compute_forward_dsv4_hc_post_f32( GGML_ASSERT(residual->ne[2] == n_tokens); GGML_ASSERT(post->ne[0] == hc); GGML_ASSERT(post->ne[1] == n_tokens); - GGML_ASSERT(comb->ne[0] == hc); - GGML_ASSERT(comb->ne[1] == hc); - GGML_ASSERT(comb->ne[2] == n_tokens); + + // comb == NULL: identity mixing, each stream keeps its own residual + size_t nbc0 = 0; + size_t nbc1 = 0; + size_t nbc2 = 0; + if (comb) { + GGML_ASSERT(comb->type == GGML_TYPE_F32); + GGML_ASSERT(comb->ne[0] == hc); + GGML_ASSERT(comb->ne[1] == hc); + GGML_ASSERT(comb->ne[2] == n_tokens); + nbc0 = comb->nb[0]; + nbc1 = comb->nb[1]; + nbc2 = comb->nb[2]; + } GGML_TENSOR_LOCALS(size_t, nbx, x, nb); GGML_TENSOR_LOCALS(size_t, nbr, residual, nb); GGML_TENSOR_LOCALS(size_t, nbp, post, nb); - GGML_TENSOR_LOCALS(size_t, nbc, comb, nb); GGML_TENSOR_LOCALS(size_t, nbd, dst, nb); const int ith = params->ith; @@ -11212,10 +11386,14 @@ static void ggml_compute_forward_dsv4_hc_post_f32( const float pv = *(const float *) ((const char *) post->data + idst*nbp0 + it*nbp1); float sum = xv * pv; - for (int64_t isrc = 0; isrc < hc; ++isrc) { - const float rv = *(const float *) ((const char *) residual->data + i0*nbr0 + isrc*nbr1 + it*nbr2); - const float cv = *(const float *) ((const char *) comb->data + idst*nbc0 + isrc*nbc1 + it*nbc2); - sum += rv * cv; + if (comb) { + for (int64_t isrc = 0; isrc < hc; ++isrc) { + const float rv = *(const float *) ((const char *) residual->data + i0*nbr0 + isrc*nbr1 + it*nbr2); + const float cv = *(const float *) ((const char *) comb->data + idst*nbc0 + isrc*nbc1 + it*nbc2); + sum += rv * cv; + } + } else { + sum += *(const float *) ((const char *) residual->data + i0*nbr0 + idst*nbr1 + it*nbr2); } *(float *) ((char *) dst->data + i0*nbd0 + idst*nbd1 + it*nbd2) = sum; @@ -11837,11 +12015,20 @@ void ggml_compute_forward_opt_step_sgd(const ggml_compute_params * params, ggml_ } } -static void ggml_compute_forward_fwht_f32(const ggml_compute_params * params, ggml_tensor * dst) { +static inline float ggml_fwht_load(const float value) { + return value; +} + +static inline float ggml_fwht_load(const ggml_fp16_t value) { + return ggml_fp16_to_fp32(value); +} + +template +static void ggml_compute_forward_fwht_impl(const ggml_compute_params * params, ggml_tensor * dst) { const ggml_tensor * src0 = dst->src[0]; const ggml_tensor * src1 = dst->src[1]; - GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT(src1->type == (std::is_same_v ? GGML_TYPE_F32 : GGML_TYPE_F16)); GGML_ASSERT(dst->type == GGML_TYPE_F32); GGML_TENSOR_BINARY_OP_LOCALS @@ -11868,11 +12055,11 @@ static void ggml_compute_forward_fwht_f32(const ggml_compute_params * params, gg const int64_t i12 = (r - i13 * ne11 * ne12) / ne11; const int64_t i11 = r - i13 * ne11 * ne12 - i12 * ne11; - const float * src_row = (const float *) ((const char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13); + const src_t * src_row = (const src_t *) ((const char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13); float * dst_row = (float *) ((char *) dst->data + i11 * nb1 + i12 * nb2 + i13 * nb3); for (int64_t j = 0; j < n; j++) { - dst_row[j] = src_row[j] * scale; + dst_row[j] = ggml_fwht_load(src_row[j]) * scale; } // Scalar passes @@ -11919,12 +12106,17 @@ void ggml_compute_forward_fwht(const ggml_compute_params * params, ggml_tensor * switch (src1->type) { case GGML_TYPE_F32: { - ggml_compute_forward_fwht_f32(params, dst); + ggml_compute_forward_fwht_impl(params, dst); + } + break; + case GGML_TYPE_F16: + { + ggml_compute_forward_fwht_impl(params, dst); } break; default: { - GGML_ABORT("fatal error - fwht is F32 only"); + GGML_ABORT("fatal error - fwht supports F32 and F16 input"); } } } diff --git a/ggml/src/ggml-cpu/ops.h b/ggml/src/ggml-cpu/ops.h index 4c1642a6..2728b08b 100644 --- a/ggml/src/ggml-cpu/ops.h +++ b/ggml/src/ggml-cpu/ops.h @@ -5,10 +5,10 @@ // // cache line // - -#if defined(__cpp_lib_hardware_interference_size) -#define CACHE_LINE_SIZE std::hardware_destructive_interference_size -#else +// TODO: rework CACHE_LINE_SIZE so std::hardware_destructive_interference_size +// can be used consistently between C and C++ TUs; the previous macro form +// diverged based on include order and undersized the work buffer. +// ref: https://github.com/ggml-org/llama.cpp/pull/28882 #if defined(__POWER9_VECTOR__) #define CACHE_LINE_SIZE 128 #elif defined(__VXE__) || defined(__VXE2__) @@ -16,7 +16,6 @@ #else #define CACHE_LINE_SIZE 64 #endif -#endif static const size_t CACHE_LINE_SIZE_F32 = CACHE_LINE_SIZE/sizeof(float); diff --git a/ggml/src/ggml-cpu/repack.cpp b/ggml/src/ggml-cpu/repack.cpp index 9689ca3c..d56db980 100644 --- a/ggml/src/ggml-cpu/repack.cpp +++ b/ggml/src/ggml-cpu/repack.cpp @@ -1365,6 +1365,133 @@ void ggml_gemv_q8_0_4x8_q8_0_generic(int n, } } +void ggml_gemv_q1_0_4x4_q8_0_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(nr == 1); + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(bs); + UNUSED(nr); + + float sumf[4]; + + const block_q8_0 * a_ptr = (const block_q8_0 *) vy; + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + for (int j = 0; j < ncols_interleaved; j++) { + sumf[j] = 0.0; + } + + for (int l = 0; l < nb; l++) { + const float d0[4] = { + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]), + }; + + for (int k = 0; k < QK1_0 / QK8_0; ++k) { + const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * (QK1_0 / QK8_0) + k; + const float d1 = GGML_CPU_FP16_TO_FP32(a_blk->d); + const float scale[4] = { d0[0] * d1, d0[1] * d1, d0[2] * d1, d0[3] * d1 }; + + for (int tile = 0; tile < QK8_0 / 4; ++tile) { + const uint8_t bits_lo = b_ptr[l].qs[k * 16 + 2 * tile + 0]; + const uint8_t bits_hi = b_ptr[l].qs[k * 16 + 2 * tile + 1]; + + for (int p = 0; p < 4; ++p) { + const float q = (float) a_blk->qs[tile * 4 + p]; + + sumf[0] += ((bits_lo & (1u << p)) ? scale[0] : -scale[0]) * q; + sumf[1] += ((bits_lo & (1u << (4 + p))) ? scale[1] : -scale[1]) * q; + sumf[2] += ((bits_hi & (1u << p)) ? scale[2] : -scale[2]) * q; + sumf[3] += ((bits_hi & (1u << (4 + p))) ? scale[3] : -scale[3]) * q; + } + } + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + s[x * ncols_interleaved + j] = sumf[j]; + } + } +} + +void ggml_gemv_q1_0_4x8_q8_0_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + const int blocklen = 8; + + assert(nr == 1); + assert(n % qk == 0); + assert(nc % ncols_interleaved == 0); + + UNUSED(bs); + UNUSED(nr); + + float sumf[4]; + + const block_q8_0 * a_ptr = (const block_q8_0 *) vy; + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + for (int j = 0; j < ncols_interleaved; j++) { + sumf[j] = 0.0f; + } + + for (int l = 0; l < nb; l++) { + const float d0[4] = { + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]), + }; + + for (int k = 0; k < qk / blocklen; ++k) { + const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * (qk / QK8_0) + k / (QK8_0 / blocklen); + const float d1 = GGML_CPU_FP16_TO_FP32(a_blk->d); + const float scale[4] = { d0[0] * d1, d0[1] * d1, d0[2] * d1, d0[3] * d1 }; + const uint8_t bits0 = b_ptr[l].qs[k * ncols_interleaved + 0]; + const uint8_t bits1 = b_ptr[l].qs[k * ncols_interleaved + 1]; + const uint8_t bits2 = b_ptr[l].qs[k * ncols_interleaved + 2]; + const uint8_t bits3 = b_ptr[l].qs[k * ncols_interleaved + 3]; + const int q_offset = (k % (QK8_0 / blocklen)) * blocklen; + + for (int p = 0; p < blocklen; ++p) { + const float q = (float) a_blk->qs[q_offset + p]; + + sumf[0] += ((bits0 & (1u << p)) ? scale[0] : -scale[0]) * q; + sumf[1] += ((bits1 & (1u << p)) ? scale[1] : -scale[1]) * q; + sumf[2] += ((bits2 & (1u << p)) ? scale[2] : -scale[2]) * q; + sumf[3] += ((bits3 & (1u << p)) ? scale[3] : -scale[3]) * q; + } + } + } + + for (int j = 0; j < ncols_interleaved; j++) { + s[x * ncols_interleaved + j] = sumf[j]; + } + } +} + // Only enable these for RISC-V. #if defined __riscv_zvfh void ggml_gemv_q4_0_16x1_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) { @@ -2383,6 +2510,176 @@ void ggml_gemm_q8_0_4x8_q8_0_generic(int n, } } +void ggml_gemm_q1_0_4x4_q8_0_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + + assert(n % qk == 0); + assert(nr % 4 == 0); + assert(nc % ncols_interleaved == 0); + + float sumf[4][4]; + + for (int y = 0; y < nr / 4; y++) { + const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb); + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + sumf[m][j] = 0.0; + } + } + + for (int l = 0; l < nb; l++) { + const float d0[4] = { + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]), + }; + + for (int k = 0; k < QK1_0 / QK8_0; ++k) { + const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k; + const float a_d[4] = { + GGML_CPU_FP16_TO_FP32(a_blk->d[0]), + GGML_CPU_FP16_TO_FP32(a_blk->d[1]), + GGML_CPU_FP16_TO_FP32(a_blk->d[2]), + GGML_CPU_FP16_TO_FP32(a_blk->d[3]), + }; + + for (int tile = 0; tile < QK8_0 / 4; ++tile) { + const uint8_t bits_lo = b_ptr[l].qs[k * 16 + 2 * tile + 0]; + const uint8_t bits_hi = b_ptr[l].qs[k * 16 + 2 * tile + 1]; + const int tile_offset = tile * 16; + + for (int p = 0; p < 4; ++p) { + const int8_t q_row[4] = { + a_blk->qs[tile_offset + 0 * 4 + p], + a_blk->qs[tile_offset + 1 * 4 + p], + a_blk->qs[tile_offset + 2 * 4 + p], + a_blk->qs[tile_offset + 3 * 4 + p], + }; + const int sign[4] = { + (bits_lo & (1u << p)) ? 1 : -1, + (bits_lo & (1u << (4 + p))) ? 1 : -1, + (bits_hi & (1u << p)) ? 1 : -1, + (bits_hi & (1u << (4 + p))) ? 1 : -1, + }; + + for (int m = 0; m < 4; ++m) { + const float row_scale = a_d[m]; + sumf[m][0] += sign[0] * q_row[m] * d0[0] * row_scale; + sumf[m][1] += sign[1] * q_row[m] * d0[1] * row_scale; + sumf[m][2] += sign[2] * q_row[m] * d0[2] * row_scale; + sumf[m][3] += sign[3] * q_row[m] * d0[3] * row_scale; + } + } + } + } + } + + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + s[(y * 4 + m) * bs + x * ncols_interleaved + j] = sumf[m][j]; + } + } + } + } +} + +void ggml_gemm_q1_0_4x8_q8_0_generic(int n, + float * GGML_RESTRICT s, + size_t bs, + const void * GGML_RESTRICT vx, + const void * GGML_RESTRICT vy, + int nr, + int nc) { + const int qk = QK1_0; + const int nb = n / qk; + const int ncols_interleaved = 4; + const int blocklen = 8; + + assert(n % qk == 0); + assert(nr % 4 == 0); + assert(nc % ncols_interleaved == 0); + + float sumf[4][4]; + + for (int y = 0; y < nr / 4; y++) { + const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb); + for (int x = 0; x < nc / ncols_interleaved; x++) { + const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb); + + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + sumf[m][j] = 0.0f; + } + } + + for (int l = 0; l < nb; l++) { + const float d0[4] = { + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]), + GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]), + }; + + for (int k = 0; k < qk / blocklen; ++k) { + const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k / (QK8_0 / blocklen); + const float a_d[4] = { + GGML_CPU_FP16_TO_FP32(a_blk->d[0]), + GGML_CPU_FP16_TO_FP32(a_blk->d[1]), + GGML_CPU_FP16_TO_FP32(a_blk->d[2]), + GGML_CPU_FP16_TO_FP32(a_blk->d[3]), + }; + const uint8_t bits0 = b_ptr[l].qs[k * ncols_interleaved + 0]; + const uint8_t bits1 = b_ptr[l].qs[k * ncols_interleaved + 1]; + const uint8_t bits2 = b_ptr[l].qs[k * ncols_interleaved + 2]; + const uint8_t bits3 = b_ptr[l].qs[k * ncols_interleaved + 3]; + const int q_offset = (k % (QK8_0 / blocklen)) * 4 * blocklen; + + for (int p = 0; p < blocklen; ++p) { + const int8_t q_row[4] = { + a_blk->qs[q_offset + 0 * blocklen + p], + a_blk->qs[q_offset + 1 * blocklen + p], + a_blk->qs[q_offset + 2 * blocklen + p], + a_blk->qs[q_offset + 3 * blocklen + p], + }; + const int sign[4] = { + (bits0 & (1u << p)) ? 1 : -1, + (bits1 & (1u << p)) ? 1 : -1, + (bits2 & (1u << p)) ? 1 : -1, + (bits3 & (1u << p)) ? 1 : -1, + }; + + for (int m = 0; m < 4; ++m) { + const float row_scale = a_d[m]; + sumf[m][0] += sign[0] * q_row[m] * d0[0] * row_scale; + sumf[m][1] += sign[1] * q_row[m] * d0[1] * row_scale; + sumf[m][2] += sign[2] * q_row[m] * d0[2] * row_scale; + sumf[m][3] += sign[3] * q_row[m] * d0[3] * row_scale; + } + } + } + } + + for (int m = 0; m < 4; m++) { + for (int j = 0; j < ncols_interleaved; j++) { + s[(y * 4 + m) * bs + x * ncols_interleaved + j] = sumf[m][j]; + } + } + } + } +} + // Only enable these for RISC-V. #if defined __riscv_zvfh void ggml_gemm_q4_0_16x1_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) { @@ -2739,6 +3036,50 @@ static block_q8_0x4 make_block_q8_0x4(block_q8_0 * in, unsigned int blck_size_in return out; } +static block_q1_0x4 make_block_q1_0x4(block_q1_0 * in, unsigned int blck_size_interleave) { + block_q1_0x4 out; + + for (int i = 0; i < 4; i++) { + out.d[i] = in[i].d; + } + + GGML_ASSERT(blck_size_interleave == 4 || blck_size_interleave == 8); + + if (blck_size_interleave == 4) { + for (int k = 0; k < QK1_0 / QK8_0; ++k) { + for (int tile = 0; tile < QK8_0 / 4; ++tile) { + uint8_t packed_lo = 0; + uint8_t packed_hi = 0; + + const int weight_base = k * QK8_0 + tile * 4; + for (int pos = 0; pos < 4; ++pos) { + const int weight_idx = weight_base + pos; + const int byte_idx = weight_idx / 8; + const int bit_idx = weight_idx % 8; + + packed_lo |= ((in[0].qs[byte_idx] >> bit_idx) & 1u) << pos; + packed_lo |= ((in[1].qs[byte_idx] >> bit_idx) & 1u) << (4 + pos); + packed_hi |= ((in[2].qs[byte_idx] >> bit_idx) & 1u) << pos; + packed_hi |= ((in[3].qs[byte_idx] >> bit_idx) & 1u) << (4 + pos); + } + + out.qs[k * 16 + 2 * tile + 0] = packed_lo; + out.qs[k * 16 + 2 * tile + 1] = packed_hi; + } + } + return out; + } + + for (int byte_idx = 0; byte_idx < QK1_0 / 8; ++byte_idx) { + out.qs[byte_idx * 4 + 0] = in[0].qs[byte_idx]; + out.qs[byte_idx * 4 + 1] = in[1].qs[byte_idx]; + out.qs[byte_idx * 4 + 2] = in[2].qs[byte_idx]; + out.qs[byte_idx * 4 + 3] = in[3].qs[byte_idx]; + } + + return out; +} + static block_q4_0x4 make_block_q4_0x4(block_q4_0 * in, int blck_size_interleave) { block_q4_0x4 out; @@ -3509,6 +3850,38 @@ static int repack_q8_0_to_q8_0_4_bl(struct ggml_tensor * t, return 0; } +static int repack_q1_0_to_q1_0_4_bl(struct ggml_tensor * t, + int interleave_block, + const void * GGML_RESTRICT data, + size_t data_size) { + GGML_ASSERT(t->type == GGML_TYPE_Q1_0); + GGML_ASSERT(interleave_block == 4 || interleave_block == 8); + constexpr int nrows_interleaved = 4; + + block_q1_0x4 * dst = (block_q1_0x4 *) t->data; + const block_q1_0 * src = (const block_q1_0 *) data; + block_q1_0 dst_tmp[4]; + int nrow = ggml_nrows(t); + int nblocks = t->ne[0] / QK1_0; + + GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q1_0)); + + if (t->ne[1] % nrows_interleaved != 0) { + return -1; + } + + for (int b = 0; b < nrow; b += nrows_interleaved) { + for (int64_t x = 0; x < nblocks; x++) { + for (int i = 0; i < nrows_interleaved; i++) { + dst_tmp[i] = src[x + i * nblocks]; + } + *dst++ = make_block_q1_0x4(dst_tmp, interleave_block); + } + src += nrows_interleaved * nblocks; + } + return 0; +} + static block_q8_0x16 make_block_q8_0x16(block_q8_0 * in, unsigned int blck_size_interleave) { block_q8_0x16 out; @@ -3865,6 +4238,14 @@ template int repack(struct ggml_tensor *, const void *, size_t); // TODO: generalise. +template <> int repack(struct ggml_tensor * t, const void * data, size_t data_size) { + return repack_q1_0_to_q1_0_4_bl(t, 4, data, data_size); +} + +template <> int repack(struct ggml_tensor * t, const void * data, size_t data_size) { + return repack_q1_0_to_q1_0_4_bl(t, 8, data, data_size); +} + template <> int repack(struct ggml_tensor * t, const void * data, size_t data_size) { return repack_q4_0_to_q4_0_4_bl(t, 4, data, data_size); } @@ -3960,6 +4341,14 @@ template <> int repack(struct ggml_tensor * t, const void * d template void gemv(int, float *, size_t, const void *, const void *, int, int); +template <> void gemv(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { + ggml_gemv_q1_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc); +} + +template <> void gemv(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { + ggml_gemv_q1_0_4x8_q8_0(n, s, bs, vx, vy, nr, nc); +} + template <> void gemv(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { ggml_gemv_q4_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc); } @@ -4057,6 +4446,14 @@ template <> void gemv(int n, float * s, size_ template void gemm(int, float *, size_t, const void *, const void *, int, int); +template <> void gemm(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { + ggml_gemm_q1_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc); +} + +template <> void gemm(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { + ggml_gemm_q1_0_4x8_q8_0(n, s, bs, vx, vy, nr, nc); +} + template <> void gemm(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) { ggml_gemm_q4_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc); } @@ -4526,6 +4923,10 @@ template q1_0_4x4_q8_0; + static const ggml::cpu::repack::tensor_traits q1_0_4x8_q8_0; + // instance for Q4 static const ggml::cpu::repack::tensor_traits q4_0_4x4_q8_0; static const ggml::cpu::repack::tensor_traits q4_0_4x8_q8_0; @@ -4586,6 +4987,11 @@ static const ggml::cpu::tensor_traits * ggml_repack_get_optimal_repack_type(cons return &q4_0_4x4_q8_0; } } + if (ggml_cpu_has_vxe()) { + if (cur->ne[1] % 4 == 0) { + return &q4_0_4x4_q8_0; + } + } if (ggml_cpu_has_riscv_v()) { #if defined __riscv_zvfh switch (__riscv_vlenb() * 8) { @@ -4718,6 +5124,17 @@ static const ggml::cpu::tensor_traits * ggml_repack_get_optimal_repack_type(cons } #endif } + } else if (cur->type == GGML_TYPE_Q1_0) { + if (ggml_cpu_has_neon() && ggml_cpu_has_matmul_int8()) { + if (cur->ne[1] % 4 == 0) { + return &q1_0_4x8_q8_0; + } + } + if (ggml_cpu_has_neon() && ggml_cpu_has_dotprod()) { + if (cur->ne[1] % 4 == 0) { + return &q1_0_4x4_q8_0; + } + } } return nullptr; diff --git a/ggml/src/ggml-cpu/repack.h b/ggml/src/ggml-cpu/repack.h index cb21edf6..fc6715c3 100644 --- a/ggml/src/ggml-cpu/repack.h +++ b/ggml/src/ggml-cpu/repack.h @@ -11,6 +11,9 @@ ggml_backend_buffer_type_t ggml_backend_cpu_repack_buffer_type(void); template constexpr int QK_0() { + if constexpr (K == 1) { + return QK1_0; + } if constexpr (K == 4) { return QK4_0; } @@ -26,6 +29,7 @@ template struct block { }; // control size +static_assert(sizeof(block<1, 4>) == 4 * sizeof(ggml_half) + QK1_0 / 2, "wrong block<1,4> size/padding"); static_assert(sizeof(block<4, 4>) == 4 * sizeof(ggml_half) + QK8_0 * 2, "wrong block<4,4> size/padding"); static_assert(sizeof(block<4, 8>) == 8 * sizeof(ggml_half) + QK8_0 * 4, "wrong block<4,8> size/padding"); static_assert(sizeof(block<4, 16>) == 16 * sizeof(ggml_half) + QK8_0 * 8, "wrong block<4,16> size/padding"); @@ -33,6 +37,7 @@ static_assert(sizeof(block<8, 4>) == 4 * sizeof(ggml_half) + QK8_0 * 4, "wrong b static_assert(sizeof(block<8, 8>) == 8 * sizeof(ggml_half) + QK8_0 * 8, "wrong block<8,8> size/padding"); static_assert(sizeof(block<8, 16>) == 16 * sizeof(ggml_half) + QK8_0 * 16, "wrong block<8,16> size/padding"); +using block_q1_0x4 = block<1, 4>; using block_q4_0x4 = block<4, 4>; using block_q4_0x8 = block<4, 8>; using block_q4_0x16 = block<4, 16>; @@ -141,6 +146,8 @@ void ggml_quantize_mat_q8_0_4x4(const float * GGML_RESTRICT x, void * GGML_RESTR void ggml_quantize_mat_q8_0_4x8(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); void ggml_quantize_mat_q8_K_4x4(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); void ggml_quantize_mat_q8_K_4x8(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); +void ggml_gemv_q1_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemv_q1_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); @@ -157,6 +164,8 @@ void ggml_gemv_mxfp4_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const v void ggml_gemv_mxfp4_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q8_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q8_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemm_q1_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemm_q1_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); @@ -193,6 +202,8 @@ void ggml_quantize_mat_q8_0_4x4_generic(const float * GGML_RESTRICT x, void * GG void ggml_quantize_mat_q8_0_4x8_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); void ggml_quantize_mat_q8_K_4x4_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); void ggml_quantize_mat_q8_K_4x8_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k); +void ggml_gemv_q1_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemv_q1_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q4_0_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); @@ -209,6 +220,8 @@ void ggml_gemv_mxfp4_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, void ggml_gemv_mxfp4_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q8_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemv_q8_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemm_q1_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); +void ggml_gemm_q1_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); void ggml_gemm_q4_0_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc); diff --git a/ggml/src/ggml-cpu/simd-mappings.h b/ggml/src/ggml-cpu/simd-mappings.h index fca5119e..10ce4bfc 100644 --- a/ggml/src/ggml-cpu/simd-mappings.h +++ b/ggml/src/ggml-cpu/simd-mappings.h @@ -29,13 +29,15 @@ extern "C" { // FP16 to FP32 conversion // 16-bit float -// on Arm, we use __fp16 +// on Arm, we use __fp16, which requires the IEEE fp16 format: implied on +// AArch64, selected by -mfp16-format=ieee on 32 bit Arm, where the compiler +// may otherwise reject the type // on x86, we use uint16_t // // for old CUDA compilers (<= 11), we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/10616 // for MUSA compilers , we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/11843 // -#if defined(__ARM_NEON) && !(defined(__CUDACC__) && __CUDACC_VER_MAJOR__ <= 11) && !defined(__MUSACC__) +#if defined(__ARM_NEON) && defined(__ARM_FP16_FORMAT_IEEE) && !(defined(__CUDACC__) && __CUDACC_VER_MAJOR__ <= 11) && !defined(__MUSACC__) #define GGML_CPU_COMPUTE_FP16_TO_FP32(x) neon_compute_fp16_to_fp32(x) #define GGML_CPU_COMPUTE_FP32_TO_FP16(x) neon_compute_fp32_to_fp16(x) @@ -326,7 +328,7 @@ inline static float ggml_lookup_fp16_to_fp32(ggml_fp16_t f) { #define GGML_F16_VEC_REDUCE GGML_F32Cx4_REDUCE #endif -#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_FMA) +#elif defined(__ARM_NEON) && defined(__ARM_FEATURE_FMA) && defined(__ARM_FP16_FORMAT_IEEE) #define GGML_SIMD diff --git a/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp b/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp index d2f89743..13b84dcb 100644 --- a/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp +++ b/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp @@ -639,7 +639,7 @@ static void permute_transpose_impl(const ggml_tensor * src0, } } else if (n_src_stride == sizeof(int16_t)) { for (int64_t bi = ith; bi < batch; bi += nth) { - rvv_transposed_s32_mn_to_nm((int8_t *) ((char *) dst->data + bi * batch_stride), n_dst_stride, + rvv_transposed_s16_mn_to_nm((int8_t *) ((char *) dst->data + bi * batch_stride), n_dst_stride, (int8_t *) ((char *) src0->data + bi * batch_stride), m_src_stride, m, n); } } else { diff --git a/ggml/src/ggml-cuda/CMakeLists.txt b/ggml/src/ggml-cuda/CMakeLists.txt index d3953eee..2254090c 100644 --- a/ggml/src/ggml-cuda/CMakeLists.txt +++ b/ggml/src/ggml-cuda/CMakeLists.txt @@ -112,25 +112,14 @@ if (CUDAToolkit_FOUND) file(GLOB SRCS "template-instances/mmf*.cu") list(APPEND GGML_SOURCES_CUDA ${SRCS}) - if (GGML_CUDA_FA_ALL_QUANTS) - file(GLOB SRCS "template-instances/fattn-vec*.cu") - list(APPEND GGML_SOURCES_CUDA ${SRCS}) - add_compile_definitions(GGML_CUDA_FA_ALL_QUANTS) - else() - list(APPEND GGML_SOURCES_CUDA - template-instances/fattn-vec-instance-f16-f16.cu - template-instances/fattn-vec-instance-q4_0-q4_0.cu - template-instances/fattn-vec-instance-q8_0-q8_0.cu - template-instances/fattn-vec-instance-bf16-bf16.cu) - endif() + ggml_cuda_fattn_vec_instances(${CMAKE_CURRENT_SOURCE_DIR} SRCS) + list(APPEND GGML_SOURCES_CUDA ${SRCS}) ggml_add_backend_library(ggml-cuda ${GGML_HEADERS_CUDA} ${GGML_SOURCES_CUDA} ) - add_compile_definitions(GGML_CUDA_PEER_MAX_BATCH_SIZE=${GGML_CUDA_PEER_MAX_BATCH_SIZE}) - if (GGML_CUDA_GRAPHS) add_compile_definitions(GGML_CUDA_USE_GRAPHS) endif() diff --git a/ggml/src/ggml-cuda/allreduce.cu b/ggml/src/ggml-cuda/allreduce.cu index d56129a2..39b23bed 100644 --- a/ggml/src/ggml-cuda/allreduce.cu +++ b/ggml/src/ggml-cuda/allreduce.cu @@ -1,6 +1,6 @@ #include "allreduce.cuh" -#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) +#if !defined(GGML_USE_MUSA) #include "convert.cuh" #include "ggml-impl.h" @@ -11,11 +11,12 @@ #include // --------------------------------------------------------------------------- -// CUDA AllReduce for tensor-parallel inference across two GPUs. +// AllReduce for tensor-parallel inference across two GPUs (CUDA or +// ROCm/HIP). // -// Provides an in-place sum reduction over matching tensors on two CUDA -// devices in the same process. Used by the tensor-split path alongside -// NCCL; targets setups without NVLink, where data is exchanged between the +// Provides an in-place sum reduction over matching tensors on two GPUs +// in the same process. Used by the tensor-split path alongside NCCL; +// targets setups without NVLink/xGMI, where data is exchanged between the // GPUs by staging it through pinned host memory over PCIe. // // Two reduction strategies are selected per call by tensor size: @@ -161,11 +162,14 @@ static __global__ void ggml_cuda_ar_kernel( __threadfence_system(); // make our signal visible system-wide while (ggml_cuda_ar_signal_get(other_slot) != token) { -#if __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA +#ifdef GGML_USE_HIP + // Equals ~100ns at 2500 MHz (sleeps for n * [1,64] clock cycles) + __builtin_amdgcn_s_sleep(4); +#elif __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA __nanosleep(100); #else NO_DEVICE_CODE; -#endif // __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA +#endif // GGML_USE_HIP } } @@ -280,7 +284,7 @@ struct ggml_cuda_ar_host_mapping { } rc = cudaHostGetDevicePointer(reinterpret_cast(&dev), host, 0); if (rc != cudaSuccess) { - cudaFreeHost(host); + CUDA_CHECK(cudaFreeHost(host)); host = nullptr; dev = nullptr; } @@ -289,7 +293,7 @@ struct ggml_cuda_ar_host_mapping { void free() { if (host) { - cudaFreeHost(host); + CUDA_CHECK(cudaFreeHost(host)); host = nullptr; dev = nullptr; } @@ -401,7 +405,8 @@ ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init(const int * devices, size_t n return nullptr; } - // The chunked kernel uses __nanosleep, which is sm70+ (Volta+). + // The chunked kernel uses __nanosleep (NVIDIA, sm70+) or + // __builtin_amdgcn_s_sleep (AMD). for (size_t i = 0; i < n_devices; ++i) { const int cc = ggml_cuda_info().devices[devices[i]].cc; if (cc < GGML_CUDA_CC_VOLTA) { @@ -543,7 +548,7 @@ void ggml_cuda_ar_pipeline_free(ggml_cuda_ar_pipeline * p) { for (int i = 0; i < p->n_devices; ++i) { if (p->streams[i]) { ggml_cuda_set_device(p->devices[i]); - cudaStreamSynchronize(p->streams[i]); + CUDA_CHECK(cudaStreamSynchronize(p->streams[i])); } } @@ -552,28 +557,28 @@ void ggml_cuda_ar_pipeline_free(ggml_cuda_ar_pipeline * p) { p->host_large[i].free(); if (p->dev_tmp[i]) { ggml_cuda_set_device(p->devices[i]); - cudaFree(p->dev_tmp[i]); + CUDA_CHECK(cudaFree(p->dev_tmp[i])); } ggml_cuda_set_device(p->devices[i]); for (int s = 0; s < GGML_CUDA_AR_POOL_SIZE; ++s) { - if (p->ev_pool[i][s].app) { cudaEventDestroy(p->ev_pool[i][s].app); } + if (p->ev_pool[i][s].app) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].app)); } for (int c = 0; c < GGML_CUDA_AR_COPY_MAX_CHUNKS; ++c) { - if (p->ev_pool[i][s].cpy[c]) { cudaEventDestroy(p->ev_pool[i][s].cpy[c]); } + if (p->ev_pool[i][s].cpy[c]) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].cpy[c])); } } - if (p->ev_pool[i][s].h2d) { cudaEventDestroy(p->ev_pool[i][s].h2d); } - if (p->ev_pool[i][s].ker) { cudaEventDestroy(p->ev_pool[i][s].ker); } + if (p->ev_pool[i][s].h2d) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].h2d)); } + if (p->ev_pool[i][s].ker) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].ker)); } } if (p->host_large_read_done[i]) { ggml_cuda_set_device(p->devices[i]); - cudaEventDestroy(p->host_large_read_done[i]); + CUDA_CHECK(cudaEventDestroy(p->host_large_read_done[i])); } if (p->dev_tmp_kernel_done[i]) { ggml_cuda_set_device(p->devices[i]); - cudaEventDestroy(p->dev_tmp_kernel_done[i]); + CUDA_CHECK(cudaEventDestroy(p->dev_tmp_kernel_done[i])); } if (p->streams[i]) { ggml_cuda_set_device(p->devices[i]); - cudaStreamDestroy(p->streams[i]); + CUDA_CHECK(cudaStreamDestroy(p->streams[i])); } } p->arrival.free(); @@ -952,13 +957,14 @@ bool ggml_cuda_ar_allreduce( return ok; } -#else // defined(GGML_USE_HIP) || defined(GGML_USE_MUSA) +#else // defined(GGML_USE_MUSA) -// HIP and MUSA lack the host-mapped pinned-memory APIs (cudaHostAllocPortable -// / cudaHostAllocMapped / cudaHostGetDevicePointer) and __nanosleep that this -// implementation relies on, so the internal AllReduce is a CUDA-only feature. -// The dispatcher in ggml-cuda.cu treats a nullptr pipeline as "init failed" -// and silently falls back to the meta backend's generic AllReduce. +// MUSA lacks the host-mapped pinned-memory APIs (cudaHostAllocPortable +// / cudaHostAllocMapped / cudaHostGetDevicePointer) and a device-side +// sleep intrinsic that this implementation relies on, so the internal +// AllReduce is unavailable there. The dispatcher in ggml-cuda.cu treats +// a nullptr pipeline as "init failed" and silently falls back to the meta +// backend's generic AllReduce. ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init(const int *, size_t) { return nullptr; } @@ -968,4 +974,4 @@ bool ggml_cuda_ar_allreduce(ggml_cuda_ar_pipeline *, ggml_backend_t *, ggml_tens return false; } -#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) +#endif // !defined(GGML_USE_MUSA) diff --git a/ggml/src/ggml-cuda/allreduce.cuh b/ggml/src/ggml-cuda/allreduce.cuh index 0f2c9518..76205d32 100644 --- a/ggml/src/ggml-cuda/allreduce.cuh +++ b/ggml/src/ggml-cuda/allreduce.cuh @@ -9,7 +9,7 @@ struct ggml_cuda_ar_pipeline; // Allocate a pipeline for n_devices GPUs. -// devices[] holds the CUDA device IDs in rank order. +// devices[] holds the GPU device IDs in rank order. // Returns nullptr on allocation failure. ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init( const int * devices, size_t n_devices); diff --git a/ggml/src/ggml-cuda/argsort.cu b/ggml/src/ggml-cuda/argsort.cu index 26af9002..24115da0 100644 --- a/ggml/src/ggml-cuda/argsort.cu +++ b/ggml/src/ggml-cuda/argsort.cu @@ -51,9 +51,12 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool, cudaStream_t stream) { ggml_cuda_pool_alloc temp_indices_alloc(pool, ncols * nrows); ggml_cuda_pool_alloc temp_keys_alloc(pool, ncols * nrows); + // Device*Sort algorithms currently do not allow for in-place sorting/aliasing of input/outputs + ggml_cuda_pool_alloc temp_keys_out_alloc(pool, ncols * nrows); int * temp_indices = temp_indices_alloc.get(); float * temp_keys = temp_keys_alloc.get(); + float * temp_keys_out = temp_keys_out_alloc.get(); static const int block_size = 256; const dim3 grid_size((ncols + block_size - 1) / block_size, nrows); @@ -85,18 +88,18 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool, if (order == GGML_SORT_ORDER_ASC) { if (nrows == 1) { - CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys, // keys (in-place) + CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys_out, // keys in, keys out temp_indices, dst, // values (indices) ncols, 0, sizeof(float) * 8, stream)); } else if (is_capturing) { CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs( - nullptr, temp_storage_bytes, temp_keys, temp_keys, // keys (in-place) + nullptr, temp_storage_bytes, temp_keys, temp_keys_out, // keys in, keys out temp_indices, dst, // values (indices) ncols * nrows, nrows, // num items, num segments offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream)); } else { CUDA_CHECK(DeviceSegmentedSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, - temp_keys, // keys (in-place) + temp_keys_out, // keys out temp_indices, dst, // values (indices) ncols * nrows, nrows, // num items, num segments offset_iterator, offset_iterator + 1, stream)); @@ -104,15 +107,15 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool, } else { if (nrows == 1) { CUDA_CHECK(DeviceRadixSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, - temp_keys, // keys (in-place) + temp_keys_out, // keys out temp_indices, dst, // values (indices) ncols, 0, sizeof(float) * 8, stream)); } else if (is_capturing) { CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending( - nullptr, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows, + nullptr, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream)); } else { - CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys, + CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, stream)); } @@ -124,31 +127,31 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool, if (order == GGML_SORT_ORDER_ASC) { if (nrows == 1) { CUDA_CHECK(DeviceRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, - temp_keys, // keys (in-place) + temp_keys_out, // keys out temp_indices, dst, // values (indices) ncols, 0, sizeof(float) * 8, stream)); } else if (is_capturing) { - CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys, + CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream)); } else { - CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys, + CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, stream)); } } else { if (nrows == 1) { CUDA_CHECK(DeviceRadixSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys, - temp_keys, // keys (in-place) + temp_keys_out, // keys out temp_indices, dst, // values (indices) ncols, 0, sizeof(float) * 8, stream)); } else if (is_capturing) { CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending( - d_temp_storage, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows, + d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream)); } else { CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys, - temp_keys, temp_indices, dst, ncols * nrows, nrows, + temp_keys_out, temp_indices, dst, ncols * nrows, nrows, offset_iterator, offset_iterator + 1, stream)); } } diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh index d27d8acb..2e78ae4f 100644 --- a/ggml/src/ggml-cuda/common.cuh +++ b/ggml/src/ggml-cuda/common.cuh @@ -52,6 +52,7 @@ #define GGML_CUDA_CC_VOLTA 700 #define GGML_CUDA_CC_TURING 750 #define GGML_CUDA_CC_AMPERE 800 +#define GGML_CUDA_CC_ORIN 870 #define GGML_CUDA_CC_ADA_LOVELACE 890 #define GGML_CUDA_CC_HOPPER 900 // While BW spans CC 1000, 1100 & 1200, we are integrating Tensor Core instructions available to 1200 family, see @@ -68,6 +69,8 @@ #define GGML_CUDA_CC_GCN4 (GGML_CUDA_CC_OFFSET_AMD + 0x803) // Tonga, Fiji, Polaris, minimum for fast fp16 #define GGML_CUDA_CC_VEGA (GGML_CUDA_CC_OFFSET_AMD + 0x900) // Vega56/64, minimum for fp16 dual issue #define GGML_CUDA_CC_VEGA20 (GGML_CUDA_CC_OFFSET_AMD + 0x906) // MI50/Radeon VII, minimum for dp4a +#define GGML_CUDA_CC_GFX909 (GGML_CUDA_CC_OFFSET_AMD + 0x909) // GCN APU +#define GGML_CUDA_CC_GFX90C (GGML_CUDA_CC_OFFSET_AMD + 0x90c) // GCN APU #define GGML_CUDA_CC_CDNA1 (GGML_CUDA_CC_OFFSET_AMD + 0x908) // MI100, minimum for MFMA, acc registers #define GGML_CUDA_CC_CDNA2 (GGML_CUDA_CC_OFFSET_AMD + 0x90a) // MI210 (gfx90a), minimum acc register renaming #define GGML_CUDA_CC_CDNA3 (GGML_CUDA_CC_OFFSET_AMD + 0x942) // MI300 @@ -88,12 +91,13 @@ #define GGML_CUDA_CC_IS_RDNA3_5(cc) (cc >= GGML_CUDA_CC_RDNA3_5 && cc < GGML_CUDA_CC_RDNA4) #define GGML_CUDA_CC_IS_RDNA3(cc) (GGML_CUDA_CC_IS_RDNA3_0(cc) || GGML_CUDA_CC_IS_RDNA3_5(cc)) #define GGML_CUDA_CC_IS_RDNA4(cc) (cc >= GGML_CUDA_CC_RDNA4) -#define GGML_CUDA_CC_IS_GCN(cc) (cc > GGML_CUDA_CC_OFFSET_AMD && cc < GGML_CUDA_CC_CDNA1) -#define GGML_CUDA_CC_IS_CDNA(cc) (cc >= GGML_CUDA_CC_CDNA1 && cc < GGML_CUDA_CC_RDNA1) -#define GGML_CUDA_CC_IS_CDNA1(cc) (cc >= GGML_CUDA_CC_CDNA1 && cc < GGML_CUDA_CC_CDNA2) -#define GGML_CUDA_CC_IS_CDNA2(cc) (cc >= GGML_CUDA_CC_CDNA2 && cc < GGML_CUDA_CC_CDNA3) -#define GGML_CUDA_CC_IS_CDNA3(cc) (cc >= GGML_CUDA_CC_CDNA3 && cc < GGML_CUDA_CC_CDNA4) -#define GGML_CUDA_CC_IS_CDNA4(cc) (cc >= GGML_CUDA_CC_CDNA4 && cc < GGML_CUDA_CC_RDNA1) +#define GGML_CUDA_CC_IS_GCN_APU(cc) ((cc) == GGML_CUDA_CC_GFX909 || (cc) == GGML_CUDA_CC_GFX90C) +#define GGML_CUDA_CC_IS_GCN(cc) ((cc > GGML_CUDA_CC_OFFSET_AMD && cc < GGML_CUDA_CC_CDNA1) || GGML_CUDA_CC_IS_GCN_APU(cc)) +#define GGML_CUDA_CC_IS_CDNA(cc) (!GGML_CUDA_CC_IS_GCN_APU(cc) && cc >= GGML_CUDA_CC_CDNA1 && cc < GGML_CUDA_CC_RDNA1) +#define GGML_CUDA_CC_IS_CDNA1(cc) (GGML_CUDA_CC_IS_CDNA(cc) && cc >= GGML_CUDA_CC_CDNA1 && cc < GGML_CUDA_CC_CDNA2) +#define GGML_CUDA_CC_IS_CDNA2(cc) (GGML_CUDA_CC_IS_CDNA(cc) && cc >= GGML_CUDA_CC_CDNA2 && cc < GGML_CUDA_CC_CDNA3) +#define GGML_CUDA_CC_IS_CDNA3(cc) (GGML_CUDA_CC_IS_CDNA(cc) && cc >= GGML_CUDA_CC_CDNA3 && cc < GGML_CUDA_CC_CDNA4) +#define GGML_CUDA_CC_IS_CDNA4(cc) (GGML_CUDA_CC_IS_CDNA(cc) && cc >= GGML_CUDA_CC_CDNA4 && cc < GGML_CUDA_CC_RDNA1) // Moore Threads #define MUSART_HMASK 40300 // MUSA rc4.3, min. ver. for half2 -> uint mask comparisons @@ -120,6 +124,12 @@ # define GGML_CUDA_USE_PDL #endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) && (CUDART_VERSION >= 12030 || (!(defined(_MSC_VER) && !defined(__clang__)) && CUDART_VERSION >= 11080)) +static __device__ __forceinline__ void ggml_cuda_syncwarp() { +#ifndef GGML_USE_HIP + __syncwarp(); +#endif // GGML_USE_HIP +} + static __device__ __forceinline__ void ggml_cuda_pdl_sync() { #if defined(GGML_CUDA_USE_PDL) && defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_HOPPER cudaGridDependencySynchronize(); @@ -319,6 +329,12 @@ static bool fp16_mma_hardware_available(const int cc) { (GGML_CUDA_CC_IS_MTHREADS(cc) && cc >= GGML_CUDA_CC_QY2); } +// To be used for feature selection of external libraries, e.g. cuBLAS. +static bool fast_bf16_hardware_available(const int cc) { + return (GGML_CUDA_CC_IS_AMD(cc) && (cc >= GGML_CUDA_CC_RDNA3 || GGML_CUDA_CC_IS_CDNA(cc))) + || (GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE); +} + static bool bf16_mma_hardware_available(const int cc) { return (GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE) || GGML_CUDA_CC_IS_CDNA(cc) || cc >= GGML_CUDA_CC_RDNA3 || @@ -969,6 +985,7 @@ template<> struct ggml_cuda_type_traits { static constexpr int qk = 1; static constexpr int qr = 1; + static constexpr int bs = sizeof(ggml_half); }; template<> @@ -976,6 +993,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK1_0; static constexpr int qr = QR1_0; static constexpr int qi = QI1_0; + static constexpr int bs = sizeof(block_q1_0); }; template<> @@ -983,6 +1001,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK2_0; static constexpr int qr = QR2_0; static constexpr int qi = QI2_0; + static constexpr int bs = sizeof(block_q2_0); }; template<> @@ -990,6 +1009,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK4_0; static constexpr int qr = QR4_0; static constexpr int qi = QI4_0; + static constexpr int bs = sizeof(block_q4_0); }; template<> @@ -997,6 +1017,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK4_1; static constexpr int qr = QR4_1; static constexpr int qi = QI4_1; + static constexpr int bs = sizeof(block_q4_1); }; template<> @@ -1004,6 +1025,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK5_0; static constexpr int qr = QR5_0; static constexpr int qi = QI5_0; + static constexpr int bs = sizeof(block_q5_0); }; template<> @@ -1011,6 +1033,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK5_1; static constexpr int qr = QR5_1; static constexpr int qi = QI5_1; + static constexpr int bs = sizeof(block_q5_1); }; template<> @@ -1018,6 +1041,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK8_0; static constexpr int qr = QR8_0; static constexpr int qi = QI8_0; + static constexpr int bs = sizeof(block_q8_0); }; template<> @@ -1025,6 +1049,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_MXFP4; static constexpr int qr = QR_MXFP4; static constexpr int qi = QI_MXFP4; + static constexpr int bs = sizeof(block_mxfp4); }; template<> @@ -1032,6 +1057,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_NVFP4; static constexpr int qr = QR_NVFP4; static constexpr int qi = QI_NVFP4; + static constexpr int bs = sizeof(block_nvfp4); }; template<> @@ -1039,6 +1065,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR2_K; static constexpr int qi = QI2_K; + static constexpr int bs = sizeof(block_q2_K); }; template<> @@ -1046,6 +1073,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR3_K; static constexpr int qi = QI3_K; + static constexpr int bs = sizeof(block_q3_K); }; template<> @@ -1053,6 +1081,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR4_K; static constexpr int qi = QI4_K; + static constexpr int bs = sizeof(block_q4_K); }; template<> @@ -1060,6 +1089,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR5_K; static constexpr int qi = QI5_K; + static constexpr int bs = sizeof(block_q5_K); }; template<> @@ -1067,6 +1097,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR6_K; static constexpr int qi = QI6_K; + static constexpr int bs = sizeof(block_q6_K); }; template<> @@ -1074,6 +1105,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR2_XXS; static constexpr int qi = QI2_XXS; + static constexpr int bs = sizeof(block_iq2_xxs); }; template<> @@ -1081,6 +1113,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR2_XS; static constexpr int qi = QI2_XS; + static constexpr int bs = sizeof(block_iq2_xs); }; template<> @@ -1088,6 +1121,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR2_S; static constexpr int qi = QI2_S; + static constexpr int bs = sizeof(block_iq2_s); }; template<> @@ -1095,6 +1129,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR3_XXS; static constexpr int qi = QI3_XXS; + static constexpr int bs = sizeof(block_iq3_xxs); }; template<> @@ -1102,6 +1137,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR1_S; static constexpr int qi = QI1_S; + static constexpr int bs = sizeof(block_iq1_s); }; template<> @@ -1109,6 +1145,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR1_M; static constexpr int qi = QI1_M; + static constexpr int bs = sizeof(block_iq1_m); }; template<> @@ -1116,6 +1153,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK4_NL; static constexpr int qr = QR4_NL; static constexpr int qi = QI4_NL; + static constexpr int bs = sizeof(block_iq4_nl); }; template<> @@ -1123,6 +1161,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR4_XS; static constexpr int qi = QI4_XS; + static constexpr int bs = sizeof(block_iq4_xs); }; template<> @@ -1130,6 +1169,7 @@ struct ggml_cuda_type_traits { static constexpr int qk = QK_K; static constexpr int qr = QR3_S; static constexpr int qi = QI3_S; + static constexpr int bs = sizeof(block_iq3_s); }; ////////////////////// @@ -1418,7 +1458,9 @@ struct ggml_backend_cuda_context { cudaEvent_t copy_event = nullptr; cudaStream_t streams[GGML_CUDA_MAX_DEVICES][GGML_CUDA_MAX_STREAMS] = { { nullptr } }; - cublasHandle_t cublas_handles[GGML_CUDA_MAX_DEVICES] = {nullptr}; + cublasHandle_t cublas_handles[GGML_CUDA_MAX_DEVICES][GGML_CUDA_MAX_STREAMS] = {nullptr}; + void * cublas_workspaces[GGML_CUDA_MAX_DEVICES][GGML_CUDA_MAX_STREAMS] = {nullptr}; + size_t cublas_workspace_sizes[GGML_CUDA_MAX_DEVICES] = {0}; int curr_stream_no = 0; @@ -1495,17 +1537,22 @@ struct ggml_backend_cuda_context { ggml_cuda_stream_context & stream_context() { return concurrent_stream_context; } - cublasHandle_t cublas_handle(int device) { - if (cublas_handles[device] == nullptr) { + cublasHandle_t cublas_handle() { + if (cublas_handles[device][curr_stream_no] == nullptr) { ggml_cuda_set_device(device); - CUBLAS_CHECK(cublasCreate(&cublas_handles[device])); - CUBLAS_CHECK(cublasSetMathMode(cublas_handles[device], CUBLAS_TF32_TENSOR_OP_MATH)); + CUBLAS_CHECK(cublasCreate(&cublas_handles[device][curr_stream_no])); + CUBLAS_CHECK(cublasSetMathMode(cublas_handles[device][curr_stream_no], CUBLAS_TF32_TENSOR_OP_MATH)); + CUBLAS_CHECK(cublasSetStream(cublas_handles[device][curr_stream_no], stream())); +#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) && (CUBLAS_VER_MAJOR > 11 || (CUBLAS_VER_MAJOR == 11 && CUBLAS_VER_MINOR >= 2)) + if (cublas_workspace_sizes[device] == 0) { + const int cc = ggml_cuda_info().devices[device].cc; + cublas_workspace_sizes[device] = (cc >= GGML_CUDA_CC_HOPPER) ? 32 * 1024 * 1024 : 4 * 1024 * 1024; + } + CUDA_CHECK(cudaMalloc(&cublas_workspaces[device][curr_stream_no], cublas_workspace_sizes[device])); + CUBLAS_CHECK(cublasSetWorkspace(cublas_handles[device][curr_stream_no], cublas_workspaces[device][curr_stream_no], cublas_workspace_sizes[device])); +#endif } - return cublas_handles[device]; - } - - cublasHandle_t cublas_handle() { - return cublas_handle(device); + return cublas_handles[device][curr_stream_no]; } // pool @@ -1532,6 +1579,7 @@ struct ggml_cuda_mm_fusion_args_host { const ggml_tensor * x_scale = nullptr; const ggml_tensor * gate_scale = nullptr; ggml_glu_op glu_op; + float glu_limit = 0.0f; }; struct ggml_cuda_mm_fusion_args_device { const void * x_bias = nullptr; @@ -1540,6 +1588,7 @@ struct ggml_cuda_mm_fusion_args_device { const void * x_scale = nullptr; const void * gate_scale = nullptr; ggml_glu_op glu_op; + float glu_limit = 0.0f; }; struct ggml_cuda_kernel_launch_params { @@ -1666,4 +1715,3 @@ static __inline__ void ggml_cuda_kernel_launch(Kernel kernel, const ggml_cuda_ke kernel<<>>(std::forward(args)... ); CUDA_CHECK(cudaGetLastError()); } - diff --git a/ggml/src/ggml-cuda/conv2d.cu b/ggml/src/ggml-cuda/conv2d.cu index 14774d4a..10109ad3 100644 --- a/ggml/src/ggml-cuda/conv2d.cu +++ b/ggml/src/ggml-cuda/conv2d.cu @@ -1,5 +1,6 @@ #include "conv2d.cuh" #include "convert.cuh" +#include "mma.cuh" struct conv_params { const int64_t IW, IH; @@ -111,6 +112,220 @@ static void conv2d_cuda(const float * X_D, const T * K_D, float * Y_D, const con conv2d_kernel<<>>(X_D, K_D, Y_D, P); } +static __global__ void +conv2d_pad_f16(const float * input, half * output, int iw, int ih, int pw, int ph, int px, int py, int total) { + const int i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= total) { + return; + } + const int x = i % pw - px, y = i / pw % ph - py, nc = i / (pw * ph); + output[i] = __float2half( + (unsigned) x < (unsigned) iw && (unsigned) y < (unsigned) ih ? input[(nc * ih + y) * iw + x] : 0.0f); +} + +template +static __global__ void conv2d_implicit_gemm_f16(const half * __restrict__ input, + const half * __restrict__ weight, + float * __restrict__ output, + const conv_params P, + const int split_k) { + using namespace ggml_cuda_mma; + constexpr int warp_size = ggml_cuda_get_physical_warp_size(); + constexpr int nthreads = 4 * warp_size; + constexpr int BM = 64, BN = 64, BK = 64; + constexpr int AS = BK / 2 + 4; + constexpr int BS = BN / 2 + 4; + __shared__ __align__(16) half2 a_s[BM][AS]; + __shared__ __align__(16) half2 b_s[BK][BS]; + + const int tid = threadIdx.y * warp_size + threadIdx.x; + const int iw = int(P.IW), ih = int(P.IH), ow = int(P.OW), oh = int(P.OH); + const int kw = KW ? KW : int(P.KW), kh = KH ? KH : int(P.KH); + const int ic = int(P.IC), oc = int(P.OC); + const int sx = int(P.ST_X), sy = int(P.ST_Y); + const int dx = int(P.DL_X), dy = int(P.DL_Y); + const int n = blockIdx.z / split_k, split = blockIdx.z % split_k; + const int m0 = blockIdx.y * BM, n0 = blockIdx.x * BN; + + const int k_total = ic * kw * kh; + const int load_lane = warp_size == 32 ? threadIdx.x : threadIdx.x % (BN / 2); + const int load_row = threadIdx.y * (warp_size / (BN / 2)) + (warp_size == 32 ? 0 : threadIdx.x / (BN / 2)); + const int spatial = n0 + 2 * load_lane; + const int spatial0 = min(spatial, ow * oh - 1), spatial1 = min(spatial + 1, ow * oh - 1); + const int y0 = spatial0 / ow, x0 = spatial0 % ow; + const int y1 = spatial1 / ow, x1 = spatial1 % ow; + const int pos0 = y0 * sy * iw + x0 * sx, pos1 = y1 * sy * iw + x1 * sx; + + [[maybe_unused]] const int wm = threadIdx.y / 2 * 32, wn = threadIdx.y % 2 * 32; +#if defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) + using tile_ab = tile<16, 8, half2, get_input_data_layout()>; +# if defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) + // AMD accumulator fragments transpose the input fragment's row/column mapping. + using tile_c = tile<16, 16, float, DATA_LAYOUT_J_MAJOR>; +# else + using tile_c = tile<16, 16, float>; +# endif + [[maybe_unused]] tile_c c[2][2]; +#else + if constexpr (use_mma) { + NO_DEVICE_CODE; + return; + } +#endif + constexpr int RM = 4, RN = BM * BN / (nthreads * RM); + [[maybe_unused]] const int simt_m = tid / (BN / RN) * RM, simt_n = tid % (BN / RN) * RN; + [[maybe_unused]] float c_simt[RM][RN] = {}; + const int tiles = (k_total + BK - 1) / BK; + const int begin = int(int64_t(tiles) * split / split_k) * BK; + const int end = int(int64_t(tiles) * (split + 1) / split_k) * BK; + for (int k0 = begin; k0 < end; k0 += BK) { + if (k_total % 8 == 0 && uintptr_t(weight) % 16 == 0) { +#pragma unroll + for (int i = tid; i < BM * BK / 8; i += nthreads) { + const int row = i / (BK / 8), col = 8 * (i % (BK / 8)); + const int4 v = m0 + row < oc && k0 + col < k_total ? + ((const int4 *) weight)[((m0 + row) * k_total + k0 + col) / 8] : + make_int4(0, 0, 0, 0); + *(int4 *) &a_s[row][col / 2] = v; + } + } else { +#pragma unroll + for (int i = tid; i < BM * BK / 2; i += nthreads) { + const int row = i / (BK / 2), col = 2 * (i % (BK / 2)); + half lo = __float2half(0.0f), hi = lo; + if (m0 + row < oc && k0 + col < k_total) { + lo = weight[(m0 + row) * k_total + k0 + col]; + if (k0 + col + 1 < k_total) { + hi = weight[(m0 + row) * k_total + k0 + col + 1]; + } + } + a_s[row][col / 2] = __halves2half2(lo, hi); + } + } +#pragma unroll + for (int k = load_row; k < BK; k += nthreads / (BN / 2)) { + const int ki = k0 + k; + const int ci = ki / (kw * kh), ky = ki / kw % kh, kx = ki % kw; + const int offset = ki < k_total ? (n * ic + ci) * ih * iw + ky * dy * iw + kx * dx : 0; + half lo = __float2half(0.0f), hi = lo; + if (ki < k_total && spatial < ow * oh) { + lo = input[offset + pos0]; + } + if (ki < k_total && spatial + 1 < ow * oh) { + hi = input[offset + pos1]; + } + b_s[k][load_lane] = __halves2half2(lo, hi); + } + __syncthreads(); + if constexpr (use_mma) { +#if defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) +# pragma unroll + for (int k = 0; k < BK; k += 16) { + tile_ab a[2], b[2]; +# pragma unroll + for (int i = 0; i < 2; ++i) { + load_ldmatrix(a[i], &a_s[wm + 16 * i][k / 2], AS); + load_ldmatrix_trans(b[i], &b_s[k][(wn + 16 * i) / 2], BS); + } +# pragma unroll + for (int i = 0; i < 2; ++i) { +# pragma unroll + for (int j = 0; j < 2; ++j) { + mma(c[i][j], a[i], b[j]); + } + } + } +#endif + } else { +#pragma unroll 4 + for (int k = 0; k < BK; ++k) { + float a[RM], b[RN]; +#pragma unroll + for (int i = 0; i < RM; ++i) { + a[i] = __half2float(((const half *) a_s[simt_m + i])[k]); + } +#pragma unroll + for (int j = 0; j < RN; ++j) { + b[j] = __half2float(((const half *) b_s[k])[simt_n + j]); + } +#pragma unroll + for (int i = 0; i < RM; ++i) { +#pragma unroll + for (int j = 0; j < RN; ++j) { + c_simt[i][j] += a[i] * b[j]; + } + } + } + } + __syncthreads(); + } + if constexpr (use_mma) { +#if defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) +# pragma unroll + for (int i = 0; i < 2; ++i) { +# pragma unroll + for (int j = 0; j < 2; ++j) { +# pragma unroll + for (int l = 0; l < c[i][j].ne; ++l) { + const int co = m0 + wm + 16 * i + c[i][j].get_i(l); + const int pos = n0 + wn + 16 * j + c[i][j].get_j(l); + if (co < oc && pos < ow * oh) { + output[(int64_t(blockIdx.z) * oc + co) * ow * oh + pos] = c[i][j].x[l]; + } + } + } + } +#endif + } else { +#pragma unroll + for (int i = 0; i < RM; ++i) { +#pragma unroll + for (int j = 0; j < RN; ++j) { + const int co = m0 + simt_m + i, pos = n0 + simt_n + j; + if (co < oc && pos < ow * oh) { + output[(int64_t(blockIdx.z) * oc + co) * ow * oh + pos] = c_simt[i][j]; + } + } + } + } +} + +static __global__ void conv2d_reduce_split_k(const float * __restrict__ partial, + float * __restrict__ output, + const int total, + const int per_batch, + const int split_k) { + const int i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= total) { + return; + } + const int n = i / per_batch; + const float * src = partial + int64_t(n) * (split_k - 1) * per_batch + i; + float sum = 0.0f; + for (int k = 0; k < split_k; ++k) { + sum += src[int64_t(k) * per_batch]; + } + output[i] = sum; +} + +template +static void conv2d_launch_implicit_gemm(const half * input, + const half * weight, + float * output, + const conv_params & params, + int split_k, + dim3 grid, + dim3 block, + cudaStream_t stream) { + if (params.KW == 3 && params.KH == 3) { + conv2d_implicit_gemm_f16<3, 3, use_mma><<>>(input, weight, output, params, split_k); + } else if (params.KW == 1 && params.KH == 1) { + conv2d_implicit_gemm_f16<1, 1, use_mma><<>>(input, weight, output, params, split_k); + } else { + conv2d_implicit_gemm_f16<0, 0, use_mma><<>>(input, weight, output, params, split_k); + } +} + static void conv2d_cuda_f16(const float * X_D, const half * K_D, float * Y_D, const conv_params P, cudaStream_t st) { conv2d_cuda(X_D, K_D, Y_D, P, st); } @@ -126,6 +341,7 @@ void ggml_cuda_op_conv2d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const float * X_D = (const float *) input->data; float * Y_D = (float *) dst->data; + GGML_ASSERT(input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32); GGML_ASSERT(ggml_is_contiguous(input)); GGML_ASSERT(ggml_is_contiguous(kernel)); GGML_ASSERT(kernel->type == GGML_TYPE_F16 || kernel->type == GGML_TYPE_F32); @@ -146,19 +362,86 @@ void ggml_cuda_op_conv2d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { // No cwhn GGML_ASSERT(p[6] == false); - const int IW = input->ne[0]; // input_w - const int IH = input->ne[1]; // input_h - const int OW = dst->ne[0]; // output_w - const int OH = dst->ne[1]; // output_h - const int KW = kernel->ne[0]; // kernel_w - const int KH = kernel->ne[1]; // kernel_h - const int IC = input->ne[2]; // input_channels - const int OC = kernel->ne[3]; // ouptut_chanles - const int B = input->ne[3]; // n_batches + const int64_t IW = input->ne[0]; // input_w + const int64_t IH = input->ne[1]; // input_h + const int64_t OW = dst->ne[0]; // output_w + const int64_t OH = dst->ne[1]; // output_h + const int64_t KW = kernel->ne[0]; // kernel_w + const int64_t KH = kernel->ne[1]; // kernel_h + const int64_t IC = input->ne[2]; // input_channels + const int64_t OC = kernel->ne[3]; // ouptut_chanles + const int64_t B = input->ne[3]; // n_batches const int64_t total = B * OC * OH * OW; conv_params params = { IW, IH, OW, OH, KW, KH, ST_X, ST_Y, PD_X, PD_Y, DL_X, DL_Y, IC, OC, B, total }; + const auto & device = ggml_cuda_info().devices[ctx.device]; + const bool use_mma = + turing_mma_available(device.cc) || amd_wmma_available(device.cc) || amd_mfma_available(device.cc); + // MUSA can share the tiling without a native fragment implementation in mma.cuh. + const bool use_simt = GGML_CUDA_CC_IS_MTHREADS(device.cc); + const bool pointwise = KW == 1 && KH == 1 && ST_X == 1 && ST_Y == 1 && PD_X == 0 && PD_Y == 0; + const bool use_blas = pointwise && fast_fp16_hardware_available(device.cc); + // Short reductions on small maps do not amortize conversion and launch costs. + const bool small_conv = IC * KW * KH < 64 && OW * OH < 512; + + const int64_t limit = INT_MAX - 256; + const int64_t padded_w = IW + 2 * int64_t(PD_X), padded_h = IH + 2 * int64_t(PD_Y); + const bool padded_fits = padded_w > 0 && padded_w <= limit && padded_h > 0 && padded_h <= limit && + padded_w * padded_h <= limit && IC * B <= limit / (padded_w * padded_h); + if (kernel->type == GGML_TYPE_F16 && (use_mma || use_blas || use_simt) && (use_blas || !small_conv) && + ggml_nelements(input) <= limit && ggml_nelements(kernel) <= limit && total <= limit && padded_fits && + PD_X >= 0 && PD_Y >= 0 && ST_X > 0 && ST_Y > 0 && DL_X > 0 && DL_Y > 0 && + (OW - 1) * ST_X + (KW - 1) * DL_X < padded_w && (OH - 1) * ST_Y + (KH - 1) * DL_Y < padded_h && + (OC + 63) / 64 <= 65535 && B <= 65535) { + const int pw = int(padded_w), ph = int(padded_h); + const int padded_total = int(padded_w * padded_h * IC * B); + + ggml_cuda_pool_alloc x_half(ctx.pool(), padded_total); + // Match im2col's F16 input precision, but expand patches only in shared memory and accumulate in F32. + if (PD_X == 0 && PD_Y == 0) { + ggml_get_to_fp16_cuda(input->type)(X_D, x_half.get(), padded_total, st); + } else { + conv2d_pad_f16<<<(padded_total + 255) / 256, 256, 0, st>>>(X_D, x_half.get(), int(IW), int(IH), pw, ph, + PD_X, PD_Y, padded_total); + } + const conv_params padded_params = { pw, ph, OW, OH, KW, KH, ST_X, ST_Y, 0, 0, DL_X, DL_Y, IC, OC, B, total }; + if (use_blas) { + const float alpha = 1.0f, beta = 0.0f; + const int positions = int(OW * OH); + cublasHandle_t cublas_h = ctx.cublas_handle(); + for (int n = 0; n < B; ++n) { + CUBLAS_CHECK(cublasGemmEx(cublas_h, CUBLAS_OP_N, CUBLAS_OP_N, positions, int(OC), int(IC), &alpha, + x_half.get() + int64_t(n) * IC * positions, CUDA_R_16F, positions, K_D, + CUDA_R_16F, int(IC), &beta, Y_D + int64_t(n) * OC * positions, CUDA_R_32F, + positions, CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP)); + } + return; + } + const int64_t blocks = ((OW * OH + 63) / 64) * ((OC + 63) / 64) * B; + const int target = 8 * ggml_cuda_info().devices[ctx.device].nsm; + // Split long reductions so small spatial maps still occupy the GPU. + const int split_k = int(std::min({ int64_t(32), int64_t(65535) / B, (IC * KW * KH + 63) / 64, + std::max(int64_t(1), (target + blocks - 1) / blocks) })); + + ggml_cuda_pool_alloc partial(ctx.pool()); + float * result = split_k == 1 ? Y_D : partial.alloc(total * split_k); + const dim3 block(device.warp_size, 4); + const dim3 grid(unsigned((OW * OH + 63) / 64), unsigned((OC + 63) / 64), unsigned(B * split_k)); + if (use_mma) { + conv2d_launch_implicit_gemm(x_half.get(), (const half *) K_D, result, padded_params, split_k, grid, + block, st); + } else { + conv2d_launch_implicit_gemm(x_half.get(), (const half *) K_D, result, padded_params, split_k, grid, + block, st); + } + if (split_k > 1) { + conv2d_reduce_split_k<<<(total + 255) / 256, 256, 0, st>>>(result, Y_D, int(total), int(OC * OW * OH), + split_k); + } + return; + } + if (kernel->type == GGML_TYPE_F16) { conv2d_cuda_f16(X_D, (half *) K_D, Y_D, params, st); } else { diff --git a/ggml/src/ggml-cuda/convert.cu b/ggml/src/ggml-cuda/convert.cu index 360c614a..0619f476 100644 --- a/ggml/src/ggml-cuda/convert.cu +++ b/ggml/src/ggml-cuda/convert.cu @@ -439,6 +439,29 @@ static __global__ void convert_unary( } } +template struct alignas(sizeof(T)*4) cvt_vec4 { T v[4]; }; + +// four elements per thread, so a warp moves 512B (RDNA) / 1k (CDNA) per load +template +static __global__ void convert_unary_cont_vec4( + const void * __restrict__ vx, dst_t * __restrict__ y, const int64_t k4) { + const int64_t i = (int64_t)blockDim.x*blockIdx.x + threadIdx.x; + + if (i >= k4) { + return; + } + + const cvt_vec4 xv = ((const cvt_vec4 *) vx)[i]; + + cvt_vec4 yv; +#pragma unroll + for (int j = 0; j < 4; ++j) { + yv.v[j] = ggml_cuda_cast(xv.v[j]); + } + + ((cvt_vec4 *) y)[i] = yv; +} + template static void convert_unary_cuda(const void * vx, dst_t * y, const int64_t ne00, const int64_t ne01, const int64_t ne02, const int64_t ne03, @@ -452,6 +475,15 @@ static void convert_unary_cuda(const void * vx, dst_t * y, template static void convert_unary_cont_cuda(const void * vx, dst_t * y, const int64_t k, cudaStream_t stream) { + if (k % 4 == 0 && + (uintptr_t) vx % alignof(cvt_vec4) == 0 && + (uintptr_t) y % alignof(cvt_vec4) == 0) { + const int64_t k4 = k/4; + const int64_t num_blocks = (k4 + CUDA_DEQUANTIZE_BLOCK_SIZE - 1) / CUDA_DEQUANTIZE_BLOCK_SIZE; + convert_unary_cont_vec4<<>>(vx, y, k4); + return; + } + convert_unary_cuda(vx, y, k, 1, 1, 1, k, k, k, stream); } diff --git a/ggml/src/ggml-cuda/cpy.cu b/ggml/src/ggml-cuda/cpy.cu index fd7ffc0b..7a998458 100644 --- a/ggml/src/ggml-cuda/cpy.cu +++ b/ggml/src/ggml-cuda/cpy.cu @@ -589,6 +589,14 @@ void ggml_cuda_cpy(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, gg ggml_cpy_scalar_cuda (src0_ddc, src1_ddc, ne, ne00, ne01, ne02, nb00, nb01, nb02, nb03, ne10, ne11, ne12, nb10, nb11, nb12, nb13, main_stream); } + } else if (src0->type == GGML_TYPE_I16 && src1->type == GGML_TYPE_I16) { + if (can_be_transposed) { + ggml_cpy_scalar_cuda + (src0_ddc, src1_ddc, ne, ne00, ne01, ne02, nb00, nb01, nb02, nb03, ne10, ne11, ne12, nb10, nb11, nb12, nb13, main_stream); + } else { + ggml_cpy_scalar_cuda + (src0_ddc, src1_ddc, ne, ne00, ne01, ne02, nb00, nb01, nb02, nb03, ne10, ne11, ne12, nb10, nb11, nb12, nb13, main_stream); + } } else if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_I32) { if (contiguous_srcs) { ggml_cpy_scalar_contiguous_cuda diff --git a/ggml/src/ggml-cuda/dsv4-hc.cu b/ggml/src/ggml-cuda/dsv4-hc.cu index c4b19a78..ca1d2dc8 100644 --- a/ggml/src/ggml-cuda/dsv4-hc.cu +++ b/ggml/src/ggml-cuda/dsv4-hc.cu @@ -100,6 +100,7 @@ static __global__ void dsv4_hc_comb_f32( } } +template static __global__ void dsv4_hc_pre_f32( const float * x, const float * weights, @@ -112,8 +113,10 @@ static __global__ void dsv4_hc_pre_f32( int64_t sx2, int64_t sw0, int64_t sw1, + int64_t sw2, int64_t sd0, - int64_t sd1) { + int64_t sd1, + float scale) { ggml_cuda_pdl_lc(); const int64_t ir = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; const int64_t nr = n_embd * n_tokens; @@ -127,16 +130,22 @@ static __global__ void dsv4_hc_pre_f32( const int64_t i0 = ir % n_embd; const int64_t it = ir / n_embd; - float sum = x[i0*sx0 + it*sx2] * weights[it*sw1]; - for (int64_t ih = 1; ih < hc; ++ih) { + float sum = 0.0f; + for (int64_t ih = 0; ih < hc; ++ih) { const float xv = x[i0*sx0 + ih*sx1 + it*sx2]; - const float wv = weights[ih*sw0 + it*sw1]; + float wv; + if constexpr (gated) { + wv = 1.0f / (1.0f + expf(-weights[i0*sw0 + ih*sw1 + it*sw2])); + } else { + wv = weights[ih*sw0 + it*sw1]; + } sum += xv * wv; } - dst[i0*sd0 + it*sd1] = sum; + dst[i0*sd0 + it*sd1] = scale * sum; } +template static __global__ void dsv4_hc_post_f32( const float * x, const float * residual, @@ -174,8 +183,12 @@ static __global__ void dsv4_hc_post_f32( const int64_t it = ir / (n_embd * hc); float sum = x[i0*sx0 + it*sx1] * post[idst*sp0 + it*sp1]; - for (int64_t isrc = 0; isrc < hc; ++isrc) { - sum += residual[i0*sr0 + isrc*sr1 + it*sr2] * comb[idst*sc0 + isrc*sc1 + it*sc2]; + if constexpr (has_comb) { + for (int64_t isrc = 0; isrc < hc; ++isrc) { + sum += residual[i0*sr0 + isrc*sr1 + it*sr2] * comb[idst*sc0 + isrc*sc1 + it*sc2]; + } + } else { + sum += residual[i0*sr0 + idst*sr1 + it*sr2]; } dst[i0*sd0 + idst*sd1 + it*sd2] = sum; @@ -240,18 +253,23 @@ void ggml_cuda_op_dsv4_hc_pre(ggml_backend_cuda_context & ctx, ggml_tensor * dst const int64_t hc = x->ne[1]; const int64_t n_tokens = x->ne[2]; + const float scale = ggml_get_op_params_f32(dst, 0); + const bool gated = ggml_get_op_params_i32(dst, 1) != 0; + const int block_size = 256; const int64_t nr = n_embd * n_tokens; const dim3 block_dims(block_size, 1, 1); const dim3 grid_dims((nr + block_size - 1) / block_size, 1, 1); const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(grid_dims, block_dims, 0, ctx.stream()); - ggml_cuda_kernel_launch(dsv4_hc_pre_f32, launch_params, + auto kernel = gated ? dsv4_hc_pre_f32 : dsv4_hc_pre_f32; + ggml_cuda_kernel_launch(kernel, launch_params, (const float *) x->data, (const float *) weights->data, (float *) dst->data, n_embd, hc, n_tokens, nbx0 / sizeof(float), nbx1 / sizeof(float), nbx2 / sizeof(float), - nbw0 / sizeof(float), nbw1 / sizeof(float), - nbd0 / sizeof(float), nbd1 / sizeof(float)); + nbw0 / sizeof(float), nbw1 / sizeof(float), nbw2 / sizeof(float), + nbd0 / sizeof(float), nbd1 / sizeof(float), + scale); } void ggml_cuda_op_dsv4_hc_post(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { @@ -263,15 +281,18 @@ void ggml_cuda_op_dsv4_hc_post(ggml_backend_cuda_context & ctx, ggml_tensor * ds GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(residual->type == GGML_TYPE_F32); GGML_ASSERT(post->type == GGML_TYPE_F32); - GGML_ASSERT(comb->type == GGML_TYPE_F32); + GGML_ASSERT(comb == nullptr || comb->type == GGML_TYPE_F32); GGML_ASSERT(dst->type == GGML_TYPE_F32); GGML_TENSOR_LOCALS(size_t, nbx, x, nb); GGML_TENSOR_LOCALS(size_t, nbr, residual, nb); GGML_TENSOR_LOCALS(size_t, nbp, post, nb); - GGML_TENSOR_LOCALS(size_t, nbc, comb, nb); GGML_TENSOR_LOCALS(size_t, nbd, dst, nb); + const size_t nbc0 = comb ? comb->nb[0] : 0; + const size_t nbc1 = comb ? comb->nb[1] : 0; + const size_t nbc2 = comb ? comb->nb[2] : 0; + const int64_t n_embd = x->ne[0]; const int64_t n_tokens = x->ne[1]; const int64_t hc = residual->ne[1]; @@ -282,9 +303,10 @@ void ggml_cuda_op_dsv4_hc_post(ggml_backend_cuda_context & ctx, ggml_tensor * ds const dim3 grid_dims((nr + block_size - 1) / block_size, 1, 1); const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(grid_dims, block_dims, 0, ctx.stream()); - ggml_cuda_kernel_launch(dsv4_hc_post_f32, launch_params, + auto kernel = comb ? dsv4_hc_post_f32 : dsv4_hc_post_f32; + ggml_cuda_kernel_launch(kernel, launch_params, (const float *) x->data, (const float *) residual->data, - (const float *) post->data, (const float *) comb->data, (float *) dst->data, + (const float *) post->data, comb ? (const float *) comb->data : nullptr, (float *) dst->data, n_embd, hc, n_tokens, nbx0 / sizeof(float), nbx1 / sizeof(float), nbr0 / sizeof(float), nbr1 / sizeof(float), nbr2 / sizeof(float), diff --git a/ggml/src/ggml-cuda/fattn-common.cuh b/ggml/src/ggml-cuda/fattn-common.cuh index e67cc7fd..6d1ce52d 100644 --- a/ggml/src/ggml-cuda/fattn-common.cuh +++ b/ggml/src/ggml-cuda/fattn-common.cuh @@ -718,6 +718,9 @@ static __global__ void flash_attn_mask_to_KV_max( KV_max[sequence*ne31 + jt] = KV_max_sj; } +void ggml_cuda_flash_attn_ext_compact_mask( + const ggml_tensor * mask, int32_t * indices, int32_t * counts, int32_t n_queries, int32_t ncols1, int32_t n_kv_max, cudaStream_t stream); + template // D == head size __launch_bounds__(D, 1) static __global__ void flash_attn_stream_k_fixup_uniform( @@ -972,7 +975,8 @@ static __global__ void flash_attn_combine_results( template void launch_fattn( ggml_backend_cuda_context & ctx, ggml_tensor * dst, fattn_kernel_t fattn_kernel, const int nwarps, const size_t nbytes_shared, - const int nbatch_fa, const bool need_f16_K, const bool need_f16_V, const bool stream_k, const int warp_size = WARP_SIZE + const int nbatch_fa, const bool need_f16_K, const bool need_f16_V, const bool stream_k, const bool use_sparse, + const int warp_size = WARP_SIZE ) { constexpr int ncols = ncols1 * ncols2; @@ -1088,10 +1092,24 @@ void launch_fattn( const int ntiles_z_gqa = ((gqa_ratio + ncols2 - 1) / ncols2); const int ntiles_dst = ntiles_x * ntiles_z_gqa * K->ne[2] * Q->ne[3]; + // sparse: a query tile of ncols1 queries shares one index list, the union of the queries' visible columns + int32_t n_kv_max = 0; + if (use_sparse) { + GGML_ASSERT(mask != nullptr); + const int32_t n_kv_max_query = ggml_get_op_params_i32(KQV, 4); + GGML_ASSERT(n_kv_max_query > 0); + n_kv_max = std::min(K->ne[1], int64_t(ncols1)*n_kv_max_query); + + const size_t n_lists = size_t(ntiles_x) * mask->ne[3]; + + KV_max.alloc(size_t(n_kv_max)*n_lists + n_lists); + ggml_cuda_flash_attn_ext_compact_mask(mask, KV_max.ptr, KV_max.ptr + size_t(n_kv_max)*n_lists, Q->ne[1], ncols1, n_kv_max, main_stream); + } + // Optional optimization where the mask is scanned to determine whether part of the calculation can be skipped. // Only worth the overhead if there is at lease one FATTN_KQ_STRIDE x FATTN_KQ_STRIDE square to be skipped or // multiple sequences of possibly different lengths. - if (mask && K->ne[1] % FATTN_KQ_STRIDE == 0 && (Q->ne[1] >= 1024 || Q->ne[3] > 1)) { + if (!use_sparse && mask && K->ne[1] % FATTN_KQ_STRIDE == 0 && (Q->ne[1] >= 1024 || Q->ne[3] > 1)) { const int64_t s31 = mask->nb[1] / sizeof(half2); const int64_t s33 = mask->nb[3] / sizeof(half2); @@ -1114,16 +1132,26 @@ void launch_fattn( GGML_ASSERT(max_blocks_per_sm > 0); int parallel_blocks = max_blocks_per_sm; - const int ntiles_KV = (K->ne[1] + nbatch_fa - 1) / nbatch_fa; // Max. number of parallel blocks limited by KV cache length. + const int64_t n_kv = use_sparse ? n_kv_max : K->ne[1]; + const int ntiles_KV = (n_kv + nbatch_fa - 1) / nbatch_fa; // Max. number of parallel blocks limited by KV cache length. dim3 blocks_num; if (stream_k) { - // For short contexts it can be faster to have the SMs work on whole tiles because this lets us skip the fixup. - const int max_blocks = max_blocks_per_sm*nsm; - const int tiles_nwaves = (ntiles_dst + max_blocks - 1) / max_blocks; - const int tiles_efficiency_percent = 100 * ntiles_dst / (max_blocks*tiles_nwaves); + auto should_use_stream_k = [](const int cc, const int ntiles_dst, const int max_blocks, const int DKQ) { + const int tiles_nwaves = (ntiles_dst + max_blocks - 1) / max_blocks; + const int tiles_efficiency_percent = 100 * ntiles_dst / (max_blocks*tiles_nwaves); + + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_ADA_LOVELACE) { + return true; + } + if (amd_wmma_available(cc) && DKQ == 64) { + return true; // TODO better configuration + } + return tiles_efficiency_percent < 75; + }; - const bool use_stream_k = cc >= GGML_CUDA_CC_ADA_LOVELACE || amd_wmma_available(cc) || tiles_efficiency_percent < 75; + const int max_blocks = max_blocks_per_sm*nsm; + const bool use_stream_k = should_use_stream_k(cc, ntiles_dst, max_blocks, Q->ne[0]); blocks_num.x = ntiles_dst; blocks_num.y = 1; @@ -1207,8 +1235,8 @@ void launch_fattn( GGML_ASSERT(block_dim.x % warp_size == 0); - ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(blocks_num, block_dim, nbytes_shared, main_stream); - ggml_cuda_kernel_launch(fattn_kernel, launch_params, + ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(blocks_num, block_dim, nbytes_shared, main_stream); + ggml_cuda_kernel_launch(fattn_kernel, launch_params, (const char *) Q->data, K_data, V_data, @@ -1218,7 +1246,7 @@ void launch_fattn( !stream_k && parallel_blocks > 1 ? dst_tmp.ptr : (float *) KQV->data, dst_tmp_meta.ptr, scale, max_bias, m0, m1, n_head_log2, logit_softcap, Q->ne[0], ne01, Q->ne[2], Q->ne[3], Q->nb[1], Q->nb[2], Q->nb[3], - K->ne[0], K->ne[1], K->ne[2], K->ne[3], nb11, nb12, nb13, + K->ne[0], n_kv, K->ne[2], K->ne[3], nb11, nb12, nb13, nb21, nb22, nb23, mask ? mask->ne[1] : 0, mask ? mask->ne[2] : 0, mask ? mask->ne[3] : 0, mask ? mask->nb[1] : 0, mask ? mask->nb[2] : 0, mask ? mask->nb[3] : 0 diff --git a/ggml/src/ggml-cuda/fattn-mma-f16.cuh b/ggml/src/ggml-cuda/fattn-mma-f16.cuh index 7f4cfd55..449a77c5 100644 --- a/ggml/src/ggml-cuda/fattn-mma-f16.cuh +++ b/ggml/src/ggml-cuda/fattn-mma-f16.cuh @@ -66,17 +66,17 @@ static constexpr __host__ __device__ fattn_mma_config ggml_cuda_fattn_mma_get_co GGML_CUDA_FATTN_MMA_CONFIG_CASE(192, 128, 32, 128, 2, 32, 96, 64, 64, 2, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(192, 128, 64, 128, 2, 32, 96, 64, 64, 2, true); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 8, 64, 4, 64, 128, 128, 128, 2, true); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 16, 64, 4, 32, 128, 128, 128, 2, true); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 8, 128, 2, 64, 128, 128, 128, 2, true); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 16, 256, 1, 64, 128, 128, 128, 2, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 32, 128, 2, 32, 128, 128, 128, 2, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 64, 128, 2, 32, 128, 128, 128, 2, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(320, 256, 32, 128, 2, 32, 128, 128, 128, 1, false); GGML_CUDA_FATTN_MMA_CONFIG_CASE(320, 256, 64, 256, 1, 32, 128, 128, 128, 1, false); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 8, 64, 4, 32, 256, 256, 128, 1, false); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 16, 64, 4, 32, 256, 256, 128, 1, false); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 32, 128, 2, 32, 128, 128, 128, 1, false); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 8, 128, 2, 64, 128, 128, 128, 1, false); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 16, 256, 1, 64, 128, 128, 128, 1, false); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 32, 256, 1, 32, 128, 128, 128, 1, false); GGML_CUDA_FATTN_MMA_CONFIG_CASE(512, 512, 64, 256, 1, 32, 128, 128, 128, 1, false); GGML_CUDA_FATTN_MMA_CONFIG_CASE(576, 512, 8, 64, 4, 32, 288, 256, 128, 1, false); @@ -157,8 +157,8 @@ static constexpr __host__ __device__ fattn_mma_config ggml_cuda_fattn_mma_get_co GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 8, 64, 2, 32, 128, 128, 128, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 16, 64, 2, 32, 128, 128, 128, 1, true); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 32, 128, 2, 64, 128, 128, 64, 1, true); - GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 64, 128, 2, 64, 128, 128, 64, 1, true); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 32, 256, 2, 64, 128, 128, 64, 1, true); + GGML_CUDA_FATTN_MMA_CONFIG_CASE(256, 256, 64, 256, 2, 64, 128, 128, 64, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(320, 256, 32, 128, 2, 32, 160, 128, 128, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE(320, 256, 64, 128, 2, 32, 160, 128, 128, 1, true); @@ -180,7 +180,7 @@ static constexpr __host__ __device__ fattn_mma_config ggml_cuda_fattn_mma_get_co GGML_CUDA_FATTN_MMA_CONFIG_CASE( 64, 64, 8, 128, 1, 64, 32, 32, 32, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE( 64, 64, 16, 256, 2, 64, 32, 32, 32, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE( 64, 64, 32, 256, 2, 64, 32, 32, 32, 1, true); - GGML_CUDA_FATTN_MMA_CONFIG_CASE( 64, 64, 64, 256, 4, 64, 32, 32, 32, 1, true); + GGML_CUDA_FATTN_MMA_CONFIG_CASE( 64, 64, 64, 256, 3, 64, 32, 32, 32, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE( 80, 80, 8, 256, 2, 64, 40, 40, 40, 1, true); GGML_CUDA_FATTN_MMA_CONFIG_CASE( 80, 80, 16, 256, 2, 64, 40, 40, 40, 1, true); @@ -326,6 +326,32 @@ static constexpr __device__ bool ggml_cuda_fattn_mma_get_Q_in_reg(const int DKQ, return ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols).Q_in_reg; } +// Swizzling needs a tile stride that is a multiple of 32 half2 columns. +static constexpr __host__ __device__ bool ggml_cuda_fattn_mma_bank_aligned(const int nbatch_2) { + return nbatch_2 >= 32 && nbatch_2 % 32 == 0; +} + +// Swizzling needs ldmatrix, on other hardware the tiles keep the row padding. +static __host__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols1, const int ncols2, const int cc) { + const fattn_mma_config cfg = ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols1*ncols2, cc); + return turing_mma_available(cc) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_K2) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_V2); +} + +static constexpr __device__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols1, const int ncols2) { +#if defined(TURING_MMA_AVAILABLE) + const fattn_mma_config cfg = ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols1*ncols2); + return ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_K2) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_V2); +#else + GGML_UNUSED_VARS(DKQ, DV, ncols1, ncols2); + return false; +#endif // defined(TURING_MMA_AVAILABLE) +} + +// Row padding is only needed if the tile is not swizzled. +static constexpr __host__ __device__ int ggml_cuda_fattn_mma_get_stride_tile(const int nbatch_2, const bool swizzled) { + return swizzled ? nbatch_2 : nbatch_2 + 4; +} + static constexpr __device__ int get_cols_per_thread() { #if defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) return 1; // AMD has a single column per thread. @@ -349,20 +375,24 @@ static __host__ int ggml_cuda_fattn_mma_get_nstages(const int DKQ, const int DV, return cp_async_available(cc) && ncols2 >= 2 ? ggml_cuda_fattn_mma_get_nstages_target(DKQ, DV, ncols1*ncols2, cc) : 0; } -static constexpr __device__ int ggml_cuda_fattn_mma_get_nstages(const int DKQ, const int DV, const int ncols1, const int ncols2) { +static constexpr __device__ int ggml_cuda_fattn_mma_get_nstages( + const int DKQ, const int DV, const int ncols1, const int ncols2, const bool use_sparse) { #ifdef CP_ASYNC_AVAILABLE - return ncols2 >= 2 ? ggml_cuda_fattn_mma_get_nstages_target(DKQ, DV, ncols1*ncols2) : 0; + const int nstages_target = ncols2 >= 2 ? ggml_cuda_fattn_mma_get_nstages_target(DKQ, DV, ncols1*ncols2) : 0; + // sparse gather is not implemented for multi-stage loading + return use_sparse && nstages_target > 1 ? 1 : nstages_target; #else - GGML_UNUSED_VARS(DKQ, DV, ncols1, ncols2); + GGML_UNUSED_VARS(DKQ, DV, ncols1, ncols2, use_sparse); return 0; #endif // CP_ASYNC_AVAILABLE } // ------------------------------------------------------------------------------------------------------------------ -template +template static __device__ __forceinline__ void flash_attn_ext_f16_load_tile( - const half2 * const __restrict__ KV, half2 * const __restrict__ tile_KV, const int D2, const int stride_KV, const int i_sup) { + const half2 * const __restrict__ KV, half2 * const __restrict__ tile_KV, const int D2, const int stride_KV, + const int k_VKQ_0, const int i_sup, const int32_t * const __restrict__ indices) { constexpr int warp_size = ggml_cuda_get_physical_warp_size(); // K/V data is loaded with decreasing granularity for D for better memory bandwidth. // The minimum granularity is 16 bytes. @@ -370,7 +400,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile( const int chunks_per_row = D2 / h2_per_chunk; if constexpr (use_cp_async) { static_assert(warp_size == 32, "bad warp_size"); - static_assert(!oob_check, "OOB check not compatible with cp_async"); + static_assert(!oob_check || use_sparse, "OOB check not compatible with cp_async"); constexpr int preload = 64; const unsigned int tile_KV_32 = ggml_cuda_cvta_generic_to_shared(tile_KV); @@ -393,11 +423,20 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile( break; } + int64_t i_KV; + if constexpr (use_sparse) { + // padded slots gather row 0, the -inf mask removes their contribution + const int32_t index = i < i_sup ? indices[k_VKQ_0 + i] : 0; + i_KV = index >= 0 ? index : 0; + } else { + i_KV = k_VKQ_0 + i; + } + #pragma unroll for (int k0 = k0_start; k0 < k0_stop; k0 += stride_k) { const int k = k0 + (stride_k == warp_size ? threadIdx.x : threadIdx.x % stride_k); - cp_async_cg_16(tile_KV_32 + i*(stride_tile*sizeof(half2)) + k*16, KV + i*stride_KV + k*h2_per_chunk); + cp_async_cg_16(tile_KV_32 + swizzle_bytes(i, k*h2_per_chunk, stride_tile), KV + i_KV*stride_KV + k*h2_per_chunk); } } }; @@ -432,8 +471,14 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile( for (int k0 = k0_start; k0 < k0_stop; k0 += stride_k) { const int k = k0 + (stride_k == warp_size ? threadIdx.x : threadIdx.x % stride_k); - ggml_cuda_memcpy_1<16>(tile_KV + i*stride_tile + k*4, - !oob_check || i < i_sup ? KV + i*stride_KV + k*h2_per_chunk : zero); + const half2 * src; + if constexpr (use_sparse) { + const int32_t index = i < i_sup ? indices[k_VKQ_0 + i] : -1; + src = index >= 0 ? KV + int64_t(index)*stride_KV + k*h2_per_chunk : zero; + } else { + src = !oob_check || i < i_sup ? KV + int64_t(k_VKQ_0 + i)*stride_KV + k*h2_per_chunk : zero; + } + ggml_cuda_memcpy_1<16>((char *) tile_KV + swizzle_bytes(i, k*h2_per_chunk, stride_tile), src); } } }; @@ -447,14 +492,16 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile( } } -template +template static __device__ __forceinline__ void flash_attn_ext_f16_load_mask( const half * const __restrict__ mask_h, half * const __restrict__ tile_mask, - const int stride_mask, const int i_sup, const int j0, const uint3 ne01) { + const int stride_mask, const int k_VKQ_0, const int i_sup, const int j0, const uint3 ne01, + const int32_t * const __restrict__ indices) { constexpr int warp_size = ggml_cuda_get_physical_warp_size(); if constexpr (use_cp_async) { static_assert(nbatch_fa <= 8*warp_size && nbatch_fa % 8 == 0, "bad nbatch_fa"); static_assert(!oob_check, "OOB check incompatible with cp_async"); + static_assert(!use_sparse, "sparse gather incompatible with cp_async"); constexpr int preload = nbatch_fa >= 32 ? nbatch_fa * sizeof(half) : 64; constexpr int cols_per_warp = 8*warp_size/nbatch_fa; constexpr int stride_j = nwarps * cols_per_warp; @@ -472,9 +519,9 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_mask( const int i = 8 * (threadIdx.x % (nbatch_fa/8)); - cp_async_cg_16(tile_mask_32 + j_sram*(nbatch_fa*sizeof(half) + 16) + i*sizeof(half), mask_h + int64_t(j_vram)*stride_mask + i); + cp_async_cg_16(tile_mask_32 + j_sram*(nbatch_fa*sizeof(half) + 16) + i*sizeof(half), mask_h + int64_t(j_vram)*stride_mask + k_VKQ_0 + i); } - } else if constexpr (oob_check) { + } else if constexpr (oob_check || use_sparse) { #pragma unroll for (int j1 = 0; j1 < ncols1; j1 += nwarps) { const int j_sram = j1 + threadIdx.y; @@ -488,7 +535,12 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_mask( for (int i0 = 0; i0 < nbatch_fa; i0 += warp_size) { const int i = i0 + threadIdx.x; - tile_mask[j_sram*(nbatch_fa + 8) + i] = i < i_sup ? mask_h[int64_t(j_vram)*stride_mask + i] : half(0.0f); + if constexpr (use_sparse) { + const int32_t index = i < i_sup ? indices[k_VKQ_0 + i] : -1; + tile_mask[j_sram*(nbatch_fa + 8) + i] = index >= 0 ? mask_h[int64_t(j_vram)*stride_mask + index] : half(-INFINITY); + } else { + tile_mask[j_sram*(nbatch_fa + 8) + i] = i < i_sup ? mask_h[int64_t(j_vram)*stride_mask + k_VKQ_0 + i] : half(0.0f); + } } } } else if constexpr (nbatch_fa < 2*warp_size) { @@ -505,7 +557,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_mask( const int i = threadIdx.x % (warp_size/cols_per_warp); - ggml_cuda_memcpy_1(tile_mask + j_sram*(nbatch_fa + 8) + 2*i, mask_h + int64_t(j_vram)*stride_mask + 2*i); + ggml_cuda_memcpy_1(tile_mask + j_sram*(nbatch_fa + 8) + 2*i, mask_h + int64_t(j_vram)*stride_mask + k_VKQ_0 + 2*i); } } else { #pragma unroll @@ -521,20 +573,21 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_mask( for (int i0 = 0; i0 < nbatch_fa; i0 += 2*warp_size) { const int i = i0 + 2*threadIdx.x; - ggml_cuda_memcpy_1(tile_mask + j_sram*(nbatch_fa + 8) + i, mask_h + int64_t(j_vram)*stride_mask + i); + ggml_cuda_memcpy_1(tile_mask + j_sram*(nbatch_fa + 8) + i, mask_h + int64_t(j_vram)*stride_mask + k_VKQ_0 + i); } } } } template static __device__ __forceinline__ void flash_attn_ext_f16_iter( const float2 * const __restrict__ Q_f2, const half2 * const __restrict__ K_h2, const half2 * const __restrict__ V_h2, const half * const __restrict__ mask_h, + const int32_t * const __restrict__ indices, float2 * const __restrict__ dstk, float2 * const __restrict__ dstk_fixup, const float scale, @@ -566,11 +619,11 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( constexpr int nbatch_K2 = ggml_cuda_fattn_mma_get_nbatch_K2(DKQ, DV, ncols); constexpr int nbatch_V2 = ggml_cuda_fattn_mma_get_nbatch_V2(DKQ, DV, ncols); constexpr bool Q_in_reg = ggml_cuda_fattn_mma_get_Q_in_reg (DKQ, DV, ncols); - constexpr int nstages = ggml_cuda_fattn_mma_get_nstages (DKQ, DV, ncols1, ncols2); + constexpr int nstages = ggml_cuda_fattn_mma_get_nstages (DKQ, DV, ncols1, ncols2, use_sparse); - constexpr int stride_tile_K = nbatch_K2 + 4; - - constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : nbatch_V2 + 4; + constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2); + constexpr int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swz); + constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swz); const int k_VKQ_0 = kb0 * nbatch_fa; #if defined(TURING_MMA_AVAILABLE) @@ -588,13 +641,14 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( constexpr bool use_cp_async = true; cp_async_wait_all(); __syncthreads(); - flash_attn_ext_f16_load_tile - (V_h2 + int64_t(k_VKQ_0)*stride_V, tile_V, nbatch_V2, stride_V, k_VKQ_sup); + flash_attn_ext_f16_load_tile + (V_h2, tile_V, nbatch_V2, stride_V, k_VKQ_0, k_VKQ_sup, nullptr); } else { - constexpr bool use_cp_async = nstages == 1; + // the sparse mask values are gathered per element, always load them synchronously + constexpr bool use_cp_async = nstages == 1 && !use_sparse; if (ncols2 > 1 || mask_h) { - flash_attn_ext_f16_load_mask - (mask_h + k_VKQ_0, tile_mask, stride_mask, k_VKQ_sup, jt*ncols1, ne01); + flash_attn_ext_f16_load_mask + (mask_h, tile_mask, stride_mask, k_VKQ_0, k_VKQ_sup, jt*ncols1, ne01, indices); } } @@ -607,8 +661,8 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( if constexpr (nstages <= 1) { const int k0_diff = k0_stop - k0_start; constexpr bool use_cp_async = nstages == 1; - flash_attn_ext_f16_load_tile - (K_h2 + int64_t(k_VKQ_0)*stride_K + k0_start, tile_K, k0_diff, stride_K, k_VKQ_sup); + flash_attn_ext_f16_load_tile + (K_h2 + k0_start, tile_K, k0_diff, stride_K, k_VKQ_0, k_VKQ_sup, indices); if (use_cp_async) { cp_async_wait_all(); } @@ -623,7 +677,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( #pragma unroll for (int k_KQ_0 = k0_start; k_KQ_0 < k0_stop; k_KQ_0 += T_A_KQ::J) { T_A_KQ K_A; - load_ldmatrix(K_A, tile_K + i_KQ_0*stride_tile_K + (k_KQ_0 - k0_start), stride_tile_K); + load_ldmatrix(K_A, tile_K, i_KQ_0, k_KQ_0 - k0_start, stride_tile_K); if constexpr (cols_per_warp == 8) { mma(KQ_C[i_KQ_00/(np*T_A_KQ::I)], K_A, Q_B[k_KQ_0/T_A_KQ::J]); } else { @@ -649,7 +703,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( const int i_KQ_0 = i_KQ_00 + (threadIdx.y % np)*T_A_KQ::I; T_A_KQ K_A; - load_ldmatrix(K_A, tile_K + i_KQ_0*stride_tile_K + (k_KQ_0 - k0_start), stride_tile_K); + load_ldmatrix(K_A, tile_K, i_KQ_0, k_KQ_0 - k0_start, stride_tile_K); if constexpr (cols_per_warp == 8) { mma(KQ_C[i_KQ_00/(np*T_A_KQ::I)], K_A, Q_B[0]); @@ -933,6 +987,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( } if constexpr (nstages > 1) { + static_assert(!use_sparse, "sparse gather not implemented for multi-stage loading"); static_assert(!V_is_K_view, "K data reuse not implemented multi-stage loading"); // Preload K tile for next iteration: constexpr bool use_cp_async = true; @@ -940,11 +995,11 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( __syncthreads(); if (!last_iter) { if (ncols2 > 1 || mask_h) { - flash_attn_ext_f16_load_mask - (mask_h + k_VKQ_0 + nbatch_fa, tile_mask, stride_mask, k_VKQ_sup, jt*ncols1, ne01); + flash_attn_ext_f16_load_mask + (mask_h, tile_mask, stride_mask, k_VKQ_0 + nbatch_fa, k_VKQ_sup, jt*ncols1, ne01, nullptr); } - flash_attn_ext_f16_load_tile - (K_h2 + int64_t(k_VKQ_0 + nbatch_fa)*stride_K, tile_K, nbatch_K2, stride_K, k_VKQ_sup); + flash_attn_ext_f16_load_tile + (K_h2, tile_K, nbatch_K2, stride_K, k_VKQ_0 + nbatch_fa, k_VKQ_sup, nullptr); } } @@ -959,8 +1014,8 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( const int i0_diff = i0_stop - i0_start; if (!V_is_K_view || i0_stop > 2*nbatch_K2) { constexpr bool use_cp_async = nstages == 1; - flash_attn_ext_f16_load_tile - (V_h2 + int64_t(k_VKQ_0)*stride_V + i0_start/2, tile_V, i0_diff/2, stride_V, k_VKQ_sup); + flash_attn_ext_f16_load_tile + (V_h2 + i0_start/2, tile_V, i0_diff/2, stride_V, k_VKQ_0, k_VKQ_sup, indices); if (use_cp_async) { cp_async_wait_all(); } @@ -978,7 +1033,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( const int k0 = k00 + (threadIdx.y % np)*T_A_VKQ::J; T_A_VKQ A; // Transposed in SRAM but not in registers, gets transposed on load. - load_ldmatrix_trans(A, tile_V_i + 2*k0*stride_tile_V + (i_VKQ_0 - i0_start)/2, stride_tile_V); + load_ldmatrix_trans(A, tile_V, 2*k0, (int)(tile_V_i - tile_V) + (i_VKQ_0 - i0_start)/2, stride_tile_V); if constexpr (T_B_KQ::I == 8) { mma(VKQ_C[i_VKQ_0/T_A_VKQ::I], A, B[k00/(np*T_A_VKQ::J)]); } else { @@ -1004,6 +1059,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( const int k0 = k00 + (threadIdx.y % np)*T_A_VKQ::I; T_A_VKQ A; // Transposed in both SRAM and registers, load normally. + static_assert(!swz, "Volta has no ldmatrix"); load_ldmatrix(A, tile_V_i + k0*stride_tile_V + (i_VKQ_0 - i0_start)/2, stride_tile_V); mma(VKQ_C[i_VKQ_0/i0_stride], B[k00/(np*T_A_VKQ::I)], A); } @@ -1015,7 +1071,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( } } #else - GGML_UNUSED_VARS(Q_f2, K_h2, V_h2, mask_h, dstk, dstk_fixup, + GGML_UNUSED_VARS(Q_f2, K_h2, V_h2, mask_h, indices, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, stride_K, stride_V, stride_mask, tile_Q, tile_K, tile_V, tile_mask, @@ -1025,7 +1081,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter( } #if defined(TURING_MMA_AVAILABLE) -template struct mma_tile_sizes { +template struct mma_tile_sizes { using T_A_KQ = tile<16, 8, half2>; // row-major using T_B_KQ = tile<16, 8, half2>; // column-major using T_C_KQ = tile<16, 16, float>; // column-major @@ -1033,7 +1089,33 @@ template struct mma_tile_sizes { using T_B_VKQ = tile<16, 8, half2>; // column-major using T_C_VKQ = tile<16, 8, half2>; // column-major }; -template struct mma_tile_sizes { +// If there are only 8 columns, use thinner B tiles to avoid wasting compute: +template struct mma_tile_sizes { + using T_A_KQ = tile<16, 8, half2>; // row-major + using T_B_KQ = tile< 8, 8, half2>; // column-major + using T_C_KQ = tile<16, 8, float>; // row-major + using T_A_VKQ = tile<16, 8, half2>; // row-major + using T_B_VKQ = tile< 8, 8, half2>; // column-major + using T_C_VKQ = tile<16, 4, half2>; // row-major +}; +// For very large head sizes, use thinner B tiles to reduce register pressure: +template<> struct mma_tile_sizes<256, 16> { + using T_A_KQ = tile<16, 8, half2>; // row-major + using T_B_KQ = tile< 8, 8, half2>; // column-major + using T_C_KQ = tile<16, 8, float>; // row-major + using T_A_VKQ = tile<16, 8, half2>; // row-major + using T_B_VKQ = tile< 8, 8, half2>; // column-major + using T_C_VKQ = tile<16, 4, half2>; // row-major +}; +template<> struct mma_tile_sizes<512, 16> { + using T_A_KQ = tile<16, 8, half2>; // row-major + using T_B_KQ = tile< 8, 8, half2>; // column-major + using T_C_KQ = tile<16, 8, float>; // row-major + using T_A_VKQ = tile<16, 8, half2>; // row-major + using T_B_VKQ = tile< 8, 8, half2>; // column-major + using T_C_VKQ = tile<16, 4, half2>; // row-major +}; +template<> struct mma_tile_sizes<512, 32> { using T_A_KQ = tile<16, 8, half2>; // row-major using T_B_KQ = tile< 8, 8, half2>; // column-major using T_C_KQ = tile<16, 8, float>; // row-major @@ -1043,7 +1125,7 @@ template struct mma_tile_sizes { }; #elif defined(AMD_WMMA_AVAILABLE) #ifdef RDNA3 -template struct mma_tile_sizes { +template struct mma_tile_sizes { using T_A_KQ = tile<16, 8, half2, DATA_LAYOUT_I_MAJOR_MIRRORED>; // row-major using T_B_KQ = tile<16, 8, half2, DATA_LAYOUT_I_MAJOR_MIRRORED>; // column-major using T_C_KQ = tile<16, 16, float, DATA_LAYOUT_I_MAJOR>; // column-major @@ -1068,7 +1150,7 @@ template struct mma_tile_sizes<112, ncols> { using T_C_VKQ = tile<16, 16, float, DATA_LAYOUT_I_MAJOR>; // column-major }; #else -template struct mma_tile_sizes { +template struct mma_tile_sizes { using T_A_KQ = tile<16, 8, half2, DATA_LAYOUT_I_MAJOR>; // row-major using T_B_KQ = tile<16, 8, half2, DATA_LAYOUT_I_MAJOR>; // column-major using T_C_KQ = tile<16, 16, float, DATA_LAYOUT_I_MAJOR>; // column-major @@ -1094,16 +1176,16 @@ template struct mma_tile_sizes<112, ncols> { }; #endif // RDNA3 #elif defined(AMD_MFMA_AVAILABLE) -template struct mma_tile_sizes { +template struct mma_tile_sizes { using T_A_KQ = tile<16, 8, half2>; // row-major using T_B_KQ = tile<16, 8, half2>; // column-major using T_C_KQ = tile<16, 16, float>; // column-major using T_A_VKQ = tile<16, 8, half2>; // row-major using T_B_VKQ = tile<16, 8, half2>; // column-major - using T_C_VKQ = tile<16, 8, half2>; // column-major + using T_C_VKQ = tile<16, 16, float>; // column-major }; #else // Volta -template struct mma_tile_sizes { +template struct mma_tile_sizes { using T_A_KQ = tile< 8, 4, half2, DATA_LAYOUT_I_MAJOR_MIRRORED>; // row-major using T_B_KQ = tile<32, 4, half2, DATA_LAYOUT_I_MAJOR>; // column-major using T_C_KQ = tile<32, 8, float, DATA_LAYOUT_I_MAJOR>; // column-major @@ -1113,12 +1195,13 @@ template struct mma_tile_sizes { }; #endif // defined(TURING_MMA_AVAILABLE) -template +template static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( const float2 * const __restrict__ Q_f2, const half2 * const __restrict__ K_h2, const half2 * const __restrict__ V_h2, const half * const __restrict__ mask_h, + const int32_t * const __restrict__ indices, const float * const __restrict__ sinks_f, float2 * const __restrict__ dstk, float2 * const __restrict__ dstk_fixup, @@ -1143,12 +1226,12 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( constexpr int warp_size = ggml_cuda_get_physical_warp_size(); constexpr int ncols = ncols1 * ncols2; - using T_A_KQ = typename mma_tile_sizes::T_A_KQ; - using T_B_KQ = typename mma_tile_sizes::T_B_KQ; - using T_C_KQ = typename mma_tile_sizes::T_C_KQ; - using T_A_VKQ = typename mma_tile_sizes::T_A_VKQ; - using T_B_VKQ = typename mma_tile_sizes::T_B_VKQ; - using T_C_VKQ = typename mma_tile_sizes::T_C_VKQ; + using T_A_KQ = typename mma_tile_sizes::T_A_KQ; + using T_B_KQ = typename mma_tile_sizes::T_B_KQ; + using T_C_KQ = typename mma_tile_sizes::T_C_KQ; + using T_A_VKQ = typename mma_tile_sizes::T_A_VKQ; + using T_B_VKQ = typename mma_tile_sizes::T_B_VKQ; + using T_C_VKQ = typename mma_tile_sizes::T_C_VKQ; constexpr int cols_per_warp = T_B_KQ::I; constexpr int cols_per_thread = get_cols_per_thread(); @@ -1158,7 +1241,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( constexpr int nbatch_V2 = ggml_cuda_fattn_mma_get_nbatch_V2 (DKQ, DV, ncols); constexpr int nbatch_combine = ggml_cuda_fattn_mma_get_nbatch_combine(DKQ, DV, ncols); constexpr bool Q_in_reg = ggml_cuda_fattn_mma_get_Q_in_reg (DKQ, DV, ncols); - constexpr int nstages = ggml_cuda_fattn_mma_get_nstages (DKQ, DV, ncols1, ncols2); + constexpr int nstages = ggml_cuda_fattn_mma_get_nstages (DKQ, DV, ncols1, ncols2, use_sparse); if (cols_per_warp > ncols) { NO_DEVICE_CODE; @@ -1168,9 +1251,9 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( static_assert(nwarps * (cols_per_warp/ncols2) % ncols1 == 0, "bad nwarps"); constexpr int stride_tile_Q = DKQ/2 + 4; - constexpr int stride_tile_K = nbatch_K2 + 4; - - constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : nbatch_V2 + 4; + constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2); + constexpr int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swz); + constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swz); constexpr int stride_tile_KV_max = stride_tile_K > stride_tile_V ? stride_tile_K : stride_tile_V; extern __shared__ half2 tile_Q[]; @@ -1183,7 +1266,9 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( T_C_VKQ VKQ_C[cols_per_warp == 8 ? DV/T_C_VKQ::I : DV/(2*T_C_VKQ::J)]; #elif defined(AMD_WMMA_AVAILABLE) && defined(RDNA3) T_C_VKQ VKQ_C[DV % 32 != 0 ? DV/T_C_VKQ::J : DV/(2*T_C_VKQ::J)]; -#elif defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) +#elif defined(AMD_MFMA_AVAILABLE) + T_C_VKQ VKQ_C[ DV/T_C_VKQ::J]; +#elif defined(AMD_WMMA_AVAILABLE) T_C_VKQ VKQ_C[ DV/(2*T_C_VKQ::J)]; #else // Volta T_C_VKQ VKQ_C[ DV/(2*T_C_VKQ::J)]; @@ -1257,37 +1342,38 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( // Preload mask and K data for first iteration when using cp_async with multiple stages: if constexpr (nstages > 1) { + static_assert(!use_sparse, "sparse gather not implemented for multi-stage loading"); static_assert(nbatch_K2 == DKQ/2, "batching not implemented for multi-stage pipeline"); constexpr bool use_cp_async = true; constexpr bool oob_check = false; constexpr int k_VKQ_sup = nbatch_fa; if (ncols2 > 1 || mask_h) { - flash_attn_ext_f16_load_mask - (mask_h + kb0*nbatch_fa, tile_mask, stride_mask, k_VKQ_sup, jt*ncols1, ne01); + flash_attn_ext_f16_load_mask + (mask_h, tile_mask, stride_mask, kb0*nbatch_fa, k_VKQ_sup, jt*ncols1, ne01, nullptr); } - flash_attn_ext_f16_load_tile - (K_h2 + int64_t(kb0)*nbatch_fa*stride_K, tile_K, nbatch_K2, stride_K, k_VKQ_sup); + flash_attn_ext_f16_load_tile + (K_h2, tile_K, nbatch_K2, stride_K, kb0*nbatch_fa, k_VKQ_sup, nullptr); } // kb0_start is always < kb0_stop so the last iter can be executed unconditionally. - if constexpr (ncols2 == 1) { + if constexpr (ncols2 == 1 || use_sparse) { constexpr bool oob_check = true; for (; kb0 < kb0_stop-1; ++kb0) { constexpr bool last_iter = false; constexpr int k_VKQ_sup = nbatch_fa; flash_attn_ext_f16_iter - - (Q_f2, K_h2, V_h2, mask_h, dstk, dstk_fixup, scale, slope, logit_softcap, + (Q_f2, K_h2, V_h2, mask_h, indices, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, stride_K, stride_V, stride_mask, tile_Q, tile_K, tile_V, tile_mask, Q_B, VKQ_C, KQ_max, KQ_rowsum, jt, kb0, k_VKQ_sup); } constexpr bool last_iter = true; const int k_VKQ_sup = ne11 - kb0*nbatch_fa; flash_attn_ext_f16_iter - - (Q_f2, K_h2, V_h2, mask_h, dstk, dstk_fixup, scale, slope, logit_softcap, + (Q_f2, K_h2, V_h2, mask_h, indices, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, stride_K, stride_V, stride_mask, tile_Q, tile_K, tile_V, tile_mask, Q_B, VKQ_C, KQ_max, KQ_rowsum, jt, kb0, k_VKQ_sup); } else { @@ -1296,18 +1382,18 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( constexpr bool last_iter = false; constexpr int k_VKQ_sup = nbatch_fa; flash_attn_ext_f16_iter - - (Q_f2, K_h2, V_h2, mask_h, dstk, dstk_fixup, scale, slope, logit_softcap, + (Q_f2, K_h2, V_h2, mask_h, indices, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, stride_K, stride_V, stride_mask, tile_Q, tile_K, tile_V, tile_mask, Q_B, VKQ_C, KQ_max, KQ_rowsum, jt, kb0, k_VKQ_sup); } constexpr bool last_iter = true; constexpr int k_VKQ_sup = nbatch_fa; flash_attn_ext_f16_iter - - (Q_f2, K_h2, V_h2, mask_h, dstk, dstk_fixup, scale, slope, logit_softcap, + (Q_f2, K_h2, V_h2, mask_h, indices, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, stride_K, stride_V, stride_mask, tile_Q, tile_K, tile_V, tile_mask, Q_B, VKQ_C, KQ_max, KQ_rowsum, jt, kb0, k_VKQ_sup); } @@ -1435,6 +1521,10 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( const int jc_cwm = threadIdx.y*(2*T_C_VKQ::J) + 2*T_C_VKQ::get_j(-1) + jc_cwmo; // jc combine write meta const float2 KQ_cmr = make_float2(KQ_max[jc_cwmo], KQ_rowsum[jc_cwmo]); // KQ combine max rowsum + if constexpr (swz) { + __syncthreads(); + } + if (((!needs_fixup && !is_fixup) || np > 1) && threadIdx.x < 2*T_C_VKQ::J) { // Use the 16 bytes of padding in each row to store the meta data: KQ max, KQ rowsum, KQ max scale. ((float2 *) tile_Q)[jc_cwm*(tile_stride/2) + nbatch_combine/2] = KQ_cmr; @@ -1471,6 +1561,10 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( const bool thread_should_write = T_C_KQ::J == 8 || T_C_KQ::get_j(threadIdx.x & 2) < 8; #endif // defined(TURING_MMA_AVAILABLE) + if constexpr (swz) { + __syncthreads(); + } + if (((!needs_fixup && !is_fixup) || np > 1) && thread_should_write) { ((float2 *) tile_Q)[jc_cwm*(tile_stride/2) + nbatch_combine/2] = KQ_cmr; } @@ -1490,77 +1584,77 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( } } - if (np > 1 && threadIdx.y % np == 0) { - // Combine the meta data for parallel warps via shared memory. - // Warps with threadIdx.y % np != 0 must NOT return early. - // All threads must return simultaneously to avoid race conditions with work on the next tile. - + if (np > 1) { constexpr int nmeta = np*cols_per_warp >= warp_size ? np*cols_per_warp/warp_size : 1; + float KQ_cmn; + float KQ_cms[nmeta]; + float KQ_crs; + const int jc_meta = threadIdx.y*cols_per_warp + (np*cols_per_warp < warp_size ? threadIdx.x % (np*cols_per_warp) : threadIdx.x); float2 * const meta_ptr = ((float2 *) tile_Q) + jc_meta*(tile_stride/2) + nbatch_combine/2; - float2 meta[nmeta]; + + if (threadIdx.y % np == 0) { + // Combine the meta data for parallel warps via shared memory. + float2 meta[nmeta]; #pragma unroll - for (int imeta = 0; imeta < nmeta; ++imeta) { - meta[imeta] = meta_ptr[imeta * warp_size * tile_stride/2]; - } + for (int imeta = 0; imeta < nmeta; ++imeta) { + meta[imeta] = meta_ptr[imeta * warp_size * tile_stride/2]; + } - float KQ_cmn = meta[0].x; // KQ combine max new, max between all parallel warps. + KQ_cmn = meta[0].x; // KQ combine max new, max between all parallel warps. #pragma unroll - for (int imeta = 1; imeta < nmeta; ++imeta) { - KQ_cmn = fmaxf(KQ_cmn, meta[imeta].x); - } + for (int imeta = 1; imeta < nmeta; ++imeta) { + KQ_cmn = fmaxf(KQ_cmn, meta[imeta].x); + } #pragma unroll - for (int offset = np*cols_per_warp/2; offset >= cols_per_warp; offset >>= 1) { - if (offset < warp_size) { - KQ_cmn = fmaxf(KQ_cmn, __shfl_xor_sync(0xFFFFFFFF, KQ_cmn, offset, warp_size)); + for (int offset = np*cols_per_warp/2; offset >= cols_per_warp; offset >>= 1) { + if (offset < warp_size) { + KQ_cmn = fmaxf(KQ_cmn, __shfl_xor_sync(0xFFFFFFFF, KQ_cmn, offset, warp_size)); + } } - } - float KQ_cms[nmeta]; // KQ combine max scale per warp. #pragma unroll - for (int imeta = 0; imeta < nmeta; ++imeta) { - KQ_cms[imeta] = expf(meta[imeta].x - KQ_cmn); - } + for (int imeta = 0; imeta < nmeta; ++imeta) { + KQ_cms[imeta] = expf(meta[imeta].x - KQ_cmn); + } - float KQ_crs = KQ_cms[0]*meta[0].y; // KQ combine rowsum, scaled sum of all parallel warps. + KQ_crs = KQ_cms[0]*meta[0].y; // KQ combine rowsum, scaled sum of all parallel warps. #pragma unroll - for (int imeta = 1; imeta < nmeta; ++imeta) { - KQ_crs += KQ_cms[imeta]*meta[imeta].y; - } + for (int imeta = 1; imeta < nmeta; ++imeta) { + KQ_crs += KQ_cms[imeta]*meta[imeta].y; + } #pragma unroll - for (int offset = np*cols_per_warp/2; offset >= cols_per_warp; offset >>= 1) { - if (offset < warp_size) { - KQ_crs += __shfl_xor_sync(0xFFFFFFFF, KQ_crs, offset, warp_size); + for (int offset = np*cols_per_warp/2; offset >= cols_per_warp; offset >>= 1) { + if (offset < warp_size) { + KQ_crs += __shfl_xor_sync(0xFFFFFFFF, KQ_crs, offset, warp_size); + } } } __syncthreads(); - // Write back combined meta data: + if (threadIdx.y % np == 0) { + // Write back combined meta data: #pragma unroll - for (int imeta = 0; imeta < nmeta; ++imeta) { - if (np*cols_per_warp >= warp_size || threadIdx.x < np*cols_per_warp) { - // Combined KQ max scale + rowsum. - meta_ptr[imeta * warp_size * tile_stride/2] = make_float2(KQ_cms[imeta], KQ_crs); + for (int imeta = 0; imeta < nmeta; ++imeta) { + if (np*cols_per_warp >= warp_size || threadIdx.x < np*cols_per_warp) { + // Combined KQ max scale + rowsum. + meta_ptr[imeta * warp_size * tile_stride/2] = make_float2(KQ_cms[imeta], KQ_crs); + } } - } - // Combined KQ max + rowsum. - static_assert(cols_per_warp <= warp_size); - if (needs_fixup && (cols_per_warp == warp_size || threadIdx.x < cols_per_warp)) { - float2 * dstk_fixup_meta = dstk_fixup + blockIdx.x*ncols; - dstk_fixup_meta[(threadIdx.y/np)*cols_per_warp + threadIdx.x] = make_float2(KQ_cmn, KQ_crs); - } - if (is_fixup && (cols_per_warp == warp_size || threadIdx.x < cols_per_warp)) { - float2 * dstk_fixup_meta = dstk_fixup + (gridDim.x + blockIdx.x)*ncols; - dstk_fixup_meta[(threadIdx.y/np)*cols_per_warp + threadIdx.x] = make_float2(KQ_cmn, KQ_crs); + // Combined KQ max + rowsum. + static_assert(cols_per_warp <= warp_size); + if (needs_fixup && (cols_per_warp == warp_size || threadIdx.x < cols_per_warp)) { + float2 * dstk_fixup_meta = dstk_fixup + blockIdx.x*ncols; + dstk_fixup_meta[(threadIdx.y/np)*cols_per_warp + threadIdx.x] = make_float2(KQ_cmn, KQ_crs); + } + if (is_fixup && (cols_per_warp == warp_size || threadIdx.x < cols_per_warp)) { + float2 * dstk_fixup_meta = dstk_fixup + (gridDim.x + blockIdx.x)*ncols; + dstk_fixup_meta[(threadIdx.y/np)*cols_per_warp + threadIdx.x] = make_float2(KQ_cmn, KQ_crs); + } } - } else if (np > 1) { - // Warps with threadIdx.y % np == 0 execute a __syncthreads() in the if branch. - // Therefore, all other warps also need to execute a __syncthreads(). - // Otherwise the points at which warps synchronize with each other would become misaligned. - __syncthreads(); } #pragma unroll @@ -1692,7 +1786,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( } } #else - GGML_UNUSED_VARS(Q_f2, K_h2, V_h2, mask_h, sinks_f, dstk, dstk_fixup, + GGML_UNUSED_VARS(Q_f2, K_h2, V_h2, mask_h, indices, sinks_f, dstk, dstk_fixup, scale, slope, logit_softcap, ne01, ne02, gqa_ratio, stride_Q1, stride_Q2, stride_K, stride_V, stride_mask, jt, kb0_start, kb0_stop); @@ -1700,7 +1794,15 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile( #endif // defined(VOLTA_MMA_AVAILABLE) || defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE) } -template +static constexpr __host__ __device__ bool ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse( + const int DKQ, const int DV, const int ncols1, const int ncols2) { + return (DKQ == 512 && DV == 512 && ncols1 == 1 && ncols2 == 8) || + (DKQ == 576 && DV == 512 && ncols1 == 1 && ncols2 == 16) || + (DKQ == 256 && DV == 256 && ncols1 == 1 && ncols2 == 8) || + (DKQ == 256 && DV == 256 && ncols1 == 8 && ncols2 == 8); +} + +template __launch_bounds__(ggml_cuda_fattn_mma_get_nthreads(DKQ, DV, ncols1*ncols2), ggml_cuda_fattn_mma_get_occupancy(DKQ, DV, ncols1*ncols2)) static __global__ void flash_attn_ext_f16( const char * Q_ptr, @@ -1726,14 +1828,16 @@ static __global__ void flash_attn_ext_f16( const int32_t nb31, const int32_t nb32, const int64_t nb33) { ggml_cuda_pdl_sync(); // TODO optimize placement #if defined(FLASH_ATTN_AVAILABLE) && (defined(VOLTA_MMA_AVAILABLE) || defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE)) - const char * GGML_CUDA_RESTRICT Q = Q_ptr; - const char * GGML_CUDA_RESTRICT K = K_ptr; - const char * GGML_CUDA_RESTRICT V = V_ptr; - const char * GGML_CUDA_RESTRICT mask = mask_ptr; - const char * GGML_CUDA_RESTRICT sinks = sinks_ptr; - const int * GGML_CUDA_RESTRICT KV_max = KV_max_ptr; - float * GGML_CUDA_RESTRICT dst = dst_ptr; - float2 * GGML_CUDA_RESTRICT dst_meta = dst_meta_ptr; + const char * GGML_CUDA_RESTRICT Q = Q_ptr; + const char * GGML_CUDA_RESTRICT K = K_ptr; + const char * GGML_CUDA_RESTRICT V = V_ptr; + const char * GGML_CUDA_RESTRICT mask = mask_ptr; + const char * GGML_CUDA_RESTRICT sinks = sinks_ptr; + // sparse: one index list per (sequence, query tile), the live count of each list follows the lists + const int * GGML_CUDA_RESTRICT sparse_indices = use_sparse ? KV_max_ptr : nullptr; + const int * GGML_CUDA_RESTRICT KV_max = KV_max_ptr; + float * GGML_CUDA_RESTRICT dst = dst_ptr; + float2 * GGML_CUDA_RESTRICT dst_meta = dst_meta_ptr; // Skip unused kernel variants for faster compilation: if (use_logit_softcap && !(DKQ == 128 || DKQ == 256 || DKQ == 512)) { @@ -1744,6 +1848,11 @@ static __global__ void flash_attn_ext_f16( NO_DEVICE_CODE; return; } + + if (!ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse(DKQ, DV, ncols1, ncols2) && use_sparse) { + NO_DEVICE_CODE; + return; + } #ifdef VOLTA_MMA_AVAILABLE if (ncols1*ncols2 < 32) { NO_DEVICE_CODE; @@ -1759,7 +1868,7 @@ static __global__ void flash_attn_ext_f16( #endif // __CUDA_ARCH__ == GGML_CUDA_CC_TURING #if defined(AMD_WMMA_AVAILABLE) - if (ncols1*ncols2 < 16 || ncols2 == 1 || DKQ > 128) { + if (ncols1*ncols2 < 16 || ncols2 == 1 || DKQ > 256) { NO_DEVICE_CODE; return; } @@ -1791,6 +1900,10 @@ static __global__ void flash_attn_ext_f16( const int iter_j = (ne01.z + (ncols1 - 1)) / ncols1; const int iter_z_gqa = (gqa_ratio + (ncols2 - 1)) / ncols2; + if (use_sparse) { + KV_max = KV_max_ptr + int64_t(iter_j)*ne33*ne11; + } + // kbc == k block continuous, current index in continuous ijk space. int kbc = int64_t(blockIdx.x + 0)*(iter_k*iter_j*iter_z_gqa*ne12*ne03) / gridDim.x; const int kbc_stop = int64_t(blockIdx.x + 1)*(iter_k*iter_j*iter_z_gqa*ne12*ne03) / gridDim.x; @@ -1820,22 +1933,25 @@ static __global__ void flash_attn_ext_f16( const half2 * V_h2 = V_is_K_view ? K_h2 : (const half2 *) (V + nb23*sequence + nb22*z_KV); const float * sinks_f = sinks ? (const float *) sinks + zt_Q : nullptr; + const int32_t * indices = use_sparse ? sparse_indices + (int64_t(sequence % ne33)*iter_j + jt)*ne11 : nullptr; const float slope = ncols2 == 1 ? get_alibi_slope(max_bias, zt_Q, n_head_log2, m0, m1) : 1.0f; - if (KV_max) { + if (use_sparse) { + kb0_stop = min(kb0_stop, (KV_max[(sequence % ne33)*iter_j + jt] + nbatch_fa - 1) / nbatch_fa); + } else if (KV_max) { kb0_stop = min(kb0_stop, KV_max[sequence*iter_j + jt] / nbatch_fa); } constexpr bool is_fixup = false; // All but (potentially) the last iterations write their data to dst rather than the fixup buffer. if (kb0_start == 0) { constexpr bool needs_fixup = false; // CUDA block is working on an entire tile. - flash_attn_ext_f16_process_tile - (Q_f2, K_h2, V_h2, mask_h, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, + flash_attn_ext_f16_process_tile + (Q_f2, K_h2, V_h2, mask_h, indices, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, ne01, ne02, gqa_ratio, ne11, stride_Q1, stride_Q2, stride_K, stride_V, stride_mask, jt, zt_gqa, kb0_start, kb0_stop); } else { constexpr bool needs_fixup = true; // CUDA block is missing the beginning of a tile. - flash_attn_ext_f16_process_tile - (Q_f2, K_h2, V_h2, mask_h, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, + flash_attn_ext_f16_process_tile + (Q_f2, K_h2, V_h2, mask_h, indices, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, ne01, ne02, gqa_ratio, ne11, stride_Q1, stride_Q2, stride_K, stride_V, stride_mask, jt, zt_gqa, kb0_start, kb0_stop); } @@ -1866,17 +1982,20 @@ static __global__ void flash_attn_ext_f16( const half2 * V_h2 = V_is_K_view ? K_h2 : (const half2 *) (V + nb23*sequence + nb22*z_KV); const float * sinks_f = sinks ? (const float *) sinks + zt_Q : nullptr; + const int32_t * indices = use_sparse ? sparse_indices + (int64_t(sequence % ne33)*iter_j + jt)*ne11 : nullptr; const float slope = ncols2 == 1 ? get_alibi_slope(max_bias, zt_Q, n_head_log2, m0, m1) : 1.0f; - if (KV_max) { + if (use_sparse) { + kb0_stop = min(kb0_stop, (KV_max[(sequence % ne33)*iter_j + jt] + nbatch_fa - 1) / nbatch_fa); + } else if (KV_max) { kb0_stop = min(kb0_stop, KV_max[sequence*iter_j + jt] / nbatch_fa); } constexpr bool is_fixup = true; // Last index writes its data to fixup buffer to avoid data races with other blocks. constexpr bool needs_fixup = false; - flash_attn_ext_f16_process_tile - (Q_f2, K_h2, V_h2, mask_h, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, + flash_attn_ext_f16_process_tile + (Q_f2, K_h2, V_h2, mask_h, indices, sinks_f, dstk, dst_meta, scale, slope, logit_softcap, ne01, ne02, gqa_ratio, ne11, stride_Q1, stride_Q2, stride_K, stride_V, stride_mask, jt, zt_gqa, kb0_start, kb0_stop); #else GGML_UNUSED_VARS(Q_ptr, K_ptr, V_ptr, mask_ptr, sinks_ptr, KV_max_ptr, dst_ptr, dst_meta_ptr, scale, @@ -1892,6 +2011,8 @@ static __global__ void flash_attn_ext_f16( #endif // defined(FLASH_ATTN_AVAILABLE) && (defined(VOLTA_MMA_AVAILABLE) || defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE)) } +bool ggml_cuda_flash_attn_ext_mma_f16_shall_use_sparse(const int cc, const ggml_tensor * dst, const int ncols1, const int ncols2); + template void ggml_cuda_flash_attn_ext_mma_f16_case(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const ggml_tensor * KQV = dst; @@ -1914,8 +2035,12 @@ void ggml_cuda_flash_attn_ext_mma_f16_case(ggml_backend_cuda_context & ctx, ggml constexpr bool V_is_K_view = DKQ == 576; // Guaranteed by the kernel selection logic in fattn.cu - const size_t nbytes_shared_KV_1stage = nbatch_fa * std::max(nbatch_K2 + 4, nbatch_V2 + 4) * sizeof(half2); - const size_t nbytes_shared_KV_2stage = nbatch_fa * (nbatch_K2 + 4 + nbatch_V2 + 4) * sizeof(half2); + // KV tile strides must match flash_attn_ext_f16_iter / _process_tile. + const bool swizzled = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2, cc); + const int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swizzled); + const int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swizzled); + const size_t nbytes_shared_KV_1stage = nbatch_fa * std::max(stride_tile_K, stride_tile_V) * sizeof(half2); + const size_t nbytes_shared_KV_2stage = nbatch_fa * (stride_tile_K + stride_tile_V) * sizeof(half2); const size_t nbytes_shared_Q = ncols * (DKQ/2 + 4) * sizeof(half2); const size_t nbytes_shared_mask = ncols1 * (nbatch_fa/2 + 4) * sizeof(half2); const size_t nbytes_shared_combine = nwarps*cols_per_warp * (nbatch_combine + 4) * sizeof(half2); @@ -1935,20 +2060,49 @@ void ggml_cuda_flash_attn_ext_mma_f16_case(ggml_backend_cuda_context & ctx, ggml using fattn_kernel_ptr_t = fattn_kernel_t; #endif // defined(GGML_USE_HIP) fattn_kernel_t fattn_kernel; + bool use_sparse = false; if (logit_softcap == 0.0f) { constexpr bool use_logit_softcap = false; - fattn_kernel = flash_attn_ext_f16; +#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) + if constexpr (ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse(DKQ, DV, ncols1, ncols2)) { + if (ggml_cuda_flash_attn_ext_mma_f16_shall_use_sparse(cc, dst, ncols1, ncols2)) { + constexpr bool use_sparse_kernel = true; + fattn_kernel = flash_attn_ext_f16; + use_sparse = true; + + static bool shared_memory_limit_raised[GGML_CUDA_MAX_DEVICES] = {false}; + if (!shared_memory_limit_raised[id]) { + CUDA_CHECK(cudaFuncSetAttribute(reinterpret_cast(fattn_kernel), cudaFuncAttributeMaxDynamicSharedMemorySize, nbytes_shared_total)); + shared_memory_limit_raised[id] = true; + } + } else { + constexpr bool use_sparse_kernel = false; + fattn_kernel = flash_attn_ext_f16; + + static bool shared_memory_limit_raised[GGML_CUDA_MAX_DEVICES] = {false}; + if (!shared_memory_limit_raised[id]) { + CUDA_CHECK(cudaFuncSetAttribute(reinterpret_cast(fattn_kernel), cudaFuncAttributeMaxDynamicSharedMemorySize, nbytes_shared_total)); + shared_memory_limit_raised[id] = true; + } + } + } else +#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) + { + constexpr bool use_sparse_kernel = false; + fattn_kernel = flash_attn_ext_f16; #if !defined(GGML_USE_MUSA) - static bool shared_memory_limit_raised[GGML_CUDA_MAX_DEVICES] = {false}; - if (!shared_memory_limit_raised[id]) { - CUDA_CHECK(cudaFuncSetAttribute(reinterpret_cast(fattn_kernel), cudaFuncAttributeMaxDynamicSharedMemorySize, nbytes_shared_total)); - shared_memory_limit_raised[id] = true; - } + static bool shared_memory_limit_raised[GGML_CUDA_MAX_DEVICES] = {false}; + if (!shared_memory_limit_raised[id]) { + CUDA_CHECK(cudaFuncSetAttribute(reinterpret_cast(fattn_kernel), cudaFuncAttributeMaxDynamicSharedMemorySize, nbytes_shared_total)); + shared_memory_limit_raised[id] = true; + } #endif // !defined(GGML_USE_MUSA) + } } else { constexpr bool use_logit_softcap = true; - fattn_kernel = flash_attn_ext_f16; + constexpr bool use_sparse_kernel = false; + fattn_kernel = flash_attn_ext_f16; #if !defined(GGML_USE_MUSA) static bool shared_memory_limit_raised[GGML_CUDA_MAX_DEVICES] = {false}; @@ -1960,7 +2114,7 @@ void ggml_cuda_flash_attn_ext_mma_f16_case(ggml_backend_cuda_context & ctx, ggml } launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared_total, nbatch_fa, true, true, true, warp_size_host); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared_total, nbatch_fa, true, true, true, use_sparse, warp_size_host); } diff --git a/ggml/src/ggml-cuda/fattn-tile.cuh b/ggml/src/ggml-cuda/fattn-tile.cuh index d1164b85..8981ab80 100644 --- a/ggml/src/ggml-cuda/fattn-tile.cuh +++ b/ggml/src/ggml-cuda/fattn-tile.cuh @@ -1163,7 +1163,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } } @@ -1179,7 +1179,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } } @@ -1191,7 +1191,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } } @@ -1203,7 +1203,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } } @@ -1215,7 +1215,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } } @@ -1226,7 +1226,7 @@ static void launch_fattn_tile_switch_ncols1(ggml_backend_cuda_context & ctx, ggm const int nbatch_fa = ggml_cuda_fattn_tile_get_nbatch_fa(DKQ, DV, cols_per_block, cc); fattn_kernel_t fattn_kernel = flash_attn_tile; launch_fattn - (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, warp_size); + (ctx, dst, fattn_kernel, nwarps, nbytes_shared, nbatch_fa, true, true, false, false, warp_size); return; } diff --git a/ggml/src/ggml-cuda/fattn-vec.cuh b/ggml/src/ggml-cuda/fattn-vec.cuh index 69dd9368..57a28556 100644 --- a/ggml/src/ggml-cuda/fattn-vec.cuh +++ b/ggml/src/ggml-cuda/fattn-vec.cuh @@ -317,9 +317,7 @@ static __global__ void flash_attn_ext_vec( #endif // V_DOT2_F32_F16_AVAILABLE } -#ifndef GGML_USE_HIP - __syncwarp(); -#endif // GGML_USE_HIP + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < WARP_SIZE; k0 += V_cols_per_iter) { @@ -540,7 +538,7 @@ void ggml_cuda_flash_attn_ext_vec_case_impl(ggml_backend_cuda_context & ctx, ggm const bool need_f16_K = type_K == GGML_TYPE_F16; const bool need_f16_V = type_V == GGML_TYPE_F16; constexpr size_t nbytes_shared = 0; - launch_fattn(ctx, dst, fattn_kernel, nwarps, nbytes_shared, D, need_f16_K, need_f16_V, false); + launch_fattn(ctx, dst, fattn_kernel, nwarps, nbytes_shared, D, need_f16_K, need_f16_V, false, false); } template diff --git a/ggml/src/ggml-cuda/fattn.cu b/ggml/src/ggml-cuda/fattn.cu index ab7a3b29..d1fcf58c 100644 --- a/ggml/src/ggml-cuda/fattn.cu +++ b/ggml/src/ggml-cuda/fattn.cu @@ -5,11 +5,168 @@ #include "fattn-vec.cuh" #include "fattn.cuh" +#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) +// one list per group of ncols1 queries: a column is selected if any query of the group can see it +template +__launch_bounds__(256, 1) +static __global__ void flash_attn_mask_to_sparse_indices( + const half * mask_ptr, int32_t * indices_ptr, int32_t * counts_ptr, const int ne30, const int n_queries, + const int n_kv_max, const int64_t s31, const int64_t s33) { + ggml_cuda_pdl_sync(); + + constexpr int values_per_lane = 8; + const int tid = threadIdx.x; + const int warp = tid / WARP_SIZE; + const int lane = tid % WARP_SIZE; + const int sequence = blockIdx.y; + const int group = blockIdx.x; + + const int q0 = group*ncols1; + const int q1 = min(q0 + ncols1, n_queries); + + const half * mask = mask_ptr + sequence*s33 + q0*s31; + int32_t * indices = indices_ptr + (int64_t(sequence)*gridDim.x + group)*n_kv_max; + + __shared__ int warp_offsets[256/WARP_SIZE]; + __shared__ int row_count; + __shared__ int chunk_count; + + if (tid == 0) { + row_count = 0; + } + __syncthreads(); + + for (int i0 = 0; i0 < ne30; i0 += blockDim.x*values_per_lane) { + uint32_t selected_warp[values_per_lane]; + int warp_count = 0; +#pragma unroll + for (int item = 0; item < values_per_lane; ++item) { + const int i = i0 + (warp*values_per_lane + item)*WARP_SIZE + lane; + bool selected = false; + if (i < ne30) { +#pragma unroll + for (int q = 0; q < ncols1; ++q) { + selected |= (!oob || q < q1 - q0) && isfinite(__half2float(mask[q*s31 + i])); + } + } + selected_warp[item] = __ballot_sync(0xFFFFFFFF, selected); + warp_count += __popc(selected_warp[item]); + } + + if (lane == 0) { + warp_offsets[warp] = warp_count; + } + __syncthreads(); + + if (tid == 0) { + int offset = 0; +#pragma unroll + for (int iw = 0; iw < 256/WARP_SIZE; ++iw) { + const int count = warp_offsets[iw]; + warp_offsets[iw] = offset; + offset += count; + } + chunk_count = offset; + } + __syncthreads(); + + const uint32_t lane_mask = lane == 0 ? 0 : (1u << lane) - 1; + int warp_item_offset = 0; +#pragma unroll + for (int item = 0; item < values_per_lane; ++item) { + const int i = i0 + (warp*values_per_lane + item)*WARP_SIZE + lane; + const int dst = row_count + warp_offsets[warp] + warp_item_offset + __popc(selected_warp[item] & lane_mask); + if ((selected_warp[item] & (uint32_t(1) << lane)) && dst < n_kv_max) { + indices[dst] = i; + } + warp_item_offset += __popc(selected_warp[item]); + } + __syncthreads(); + + if (tid == 0) { + row_count += chunk_count; + } + __syncthreads(); + } + + const int count = min(row_count, n_kv_max); + for (int i = count + tid; i < n_kv_max; i += blockDim.x) { + indices[i] = -1; + } + if (tid == 0) { + counts_ptr[int64_t(sequence)*gridDim.x + group] = count; + } + __syncthreads(); + + // the dependent grid reads indices, signal once the row is complete + ggml_cuda_pdl_lc(); +} +#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) + +void ggml_cuda_flash_attn_ext_compact_mask( + const ggml_tensor * mask, int32_t * indices, int32_t * counts, int32_t n_queries, int32_t ncols1, int32_t n_kv_max, cudaStream_t stream) { +#if defined(GGML_USE_HIP) || defined(GGML_USE_MUSA) + GGML_UNUSED_VARS(mask, indices, counts, n_queries, ncols1, n_kv_max, stream); + GGML_ABORT("sparse flash attention is only supported on NVIDIA CUDA"); +#else + const int64_t s31 = mask->nb[1] / sizeof(half); + const int64_t s33 = mask->nb[3] / sizeof(half); + const dim3 blocks_num((n_queries + ncols1 - 1)/ncols1, mask->ne[3], 1); + const dim3 block_dim(256, 1, 1); + const ggml_cuda_kernel_launch_params launch_params(blocks_num, block_dim, 0, stream); + // the last group of queries is partial only if ncols1 does not divide n_queries + GGML_ASSERT(ncols1 == 1 || ncols1 == 8); + const auto kernel = ncols1 == 1 ? flash_attn_mask_to_sparse_indices<1, false> : + n_queries % 8 != 0 ? flash_attn_mask_to_sparse_indices<8, true> : + flash_attn_mask_to_sparse_indices<8, false>; + ggml_cuda_kernel_launch(kernel, launch_params, + (const half *) mask->data, indices, counts, int(mask->ne[0]), n_queries, n_kv_max, s31, s33); + CUDA_CHECK(cudaGetLastError()); +#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) +} + +bool ggml_cuda_flash_attn_ext_mma_f16_shall_use_sparse(const int cc, const ggml_tensor * dst, const int ncols1, const int ncols2) { +#if defined(GGML_USE_HIP) || defined(GGML_USE_MUSA) + GGML_UNUSED_VARS(cc, dst, ncols1, ncols2); + return false; +#else + const ggml_tensor * Q = dst->src[0]; + const ggml_tensor * K = dst->src[1]; + const ggml_tensor * mask = dst->src[3]; + + float max_bias = 0.0f; + float logit_softcap = 0.0f; + memcpy(&max_bias, (const float *) dst->op_params + 1, sizeof(float)); + memcpy(&logit_softcap, (const float *) dst->op_params + 2, sizeof(float)); + + const int32_t n_kv_max = ggml_get_op_params_i32(dst, 4); + + // the dense kernel handles up to 64/ncols2 queries per K/V pass, the single-query gather has to beat that + const int64_t n_gather = (ncols1 == 1 ? std::min(Q->ne[1], 64/ncols2) : ncols1) * (int64_t) n_kv_max; + + return GGML_CUDA_CC_IS_NVIDIA(cc) && turing_mma_available(cc) && + mask != nullptr && n_kv_max > 0 && max_bias == 0.0f && logit_softcap == 0.0f && + mask->ne[0] == K->ne[1] && mask->ne[1] >= Q->ne[1] && mask->ne[2] == 1 && + K->ne[1] >= std::max(4096, 2*n_gather); +#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) +} + template static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; const ggml_tensor * Q = dst->src[0]; +#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) + if constexpr (ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse(DKQ, DV, 1, ncols2)) { + // a sparse variant at the full tile width gathers the union of its queries once, prefer it for large batches + constexpr bool has_wide_sparse = ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse(DKQ, DV, 64/ncols2, ncols2); + if (!(has_wide_sparse && Q->ne[1] > 32/ncols2) && ggml_cuda_flash_attn_ext_mma_f16_shall_use_sparse(cc, dst, 1, ncols2)) { + ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); + return; + } + } +#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) + if constexpr (ncols2 <= 8) { if (turing_mma_available(cc) && Q->ne[1] <= 8/ncols2) { ggml_cuda_flash_attn_ext_mma_f16_case(ctx, dst); @@ -88,6 +245,24 @@ static void ggml_cuda_flash_attn_ext_mma_f16_switch_ncols2(ggml_backend_cuda_con } } + // On RDNA it is preferable to minimize wasted compute vs. duplicate I/O for the mask. + if (amd_wmma_available(cc)) { + if (use_gqa_opt && gqa_ratio % 8 == 0) { + ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ctx, dst); + return; + } + + if (use_gqa_opt && gqa_ratio % 4 == 0) { + ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ctx, dst); + return; + } + + if (use_gqa_opt && gqa_ratio % 2 == 0) { + ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ctx, dst); + return; + } + } + if (use_gqa_opt && gqa_ratio > 4) { ggml_cuda_flash_attn_ext_mma_f16_switch_ncols1(ctx, dst); return; @@ -241,90 +416,101 @@ static void ggml_cuda_flash_attn_ext_mma_f16(ggml_backend_cuda_context & ctx, gg } } -#define FATTN_VEC_CASE(D, type_K, type_V) \ - { \ - const bool type_K_okay = K->type == (type_K) || (K->type == GGML_TYPE_F32 && (type_K) == GGML_TYPE_F16); \ - const bool type_V_okay = V->type == (type_V) || (V->type == GGML_TYPE_F32 && (type_V) == GGML_TYPE_F16); \ - if (Q->ne[0] == (D) && type_K_okay && type_V_okay) { \ - ggml_cuda_flash_attn_ext_vec_case(ctx, dst); \ - return; \ - } \ - } \ - -#define FATTN_VEC_CASES_ALL_D(type_K, type_V) \ - FATTN_VEC_CASE( 64, type_K, type_V) \ - FATTN_VEC_CASE(128, type_K, type_V) \ - FATTN_VEC_CASE(256, type_K, type_V) \ +#define FATTN_VEC_CASE(D, type_K_case, type_V_case) \ + if constexpr (GGML_CUDA_FA_##type_K_case##_##type_V_case) { \ + const bool type_K_okay = type_K == GGML_TYPE_##type_K_case || (type_K == GGML_TYPE_F32 && GGML_TYPE_##type_K_case == GGML_TYPE_F16); \ + const bool type_V_okay = type_V == GGML_TYPE_##type_V_case || (type_V == GGML_TYPE_F32 && GGML_TYPE_##type_V_case == GGML_TYPE_F16); \ + if (head_size == (D) && type_K_okay && type_V_okay) { \ + return ggml_cuda_flash_attn_ext_vec_case; \ + } \ + } \ + +#define FATTN_VEC_CASES_ALL_D(type_K_case, type_V_case) \ + FATTN_VEC_CASE( 64, type_K_case, type_V_case) \ + FATTN_VEC_CASE(128, type_K_case, type_V_case) \ + FATTN_VEC_CASE(256, type_K_case, type_V_case) \ + +typedef void (* fattn_vec_case_t)(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + +// Vector kernel for the given head size and K/V types, nullptr if its template instance was not compiled: +static fattn_vec_case_t ggml_cuda_get_fattn_vec_case(const int64_t head_size, const ggml_type type_K, const ggml_type type_V) { + FATTN_VEC_CASES_ALL_D(F16, F16) + FATTN_VEC_CASES_ALL_D(Q4_0, F16) + FATTN_VEC_CASES_ALL_D(Q4_1, F16) + FATTN_VEC_CASES_ALL_D(Q5_0, F16) + FATTN_VEC_CASES_ALL_D(Q5_1, F16) + FATTN_VEC_CASES_ALL_D(Q8_0, F16) + FATTN_VEC_CASES_ALL_D(BF16, F16) + + FATTN_VEC_CASES_ALL_D(F16, Q4_0) + FATTN_VEC_CASES_ALL_D(Q4_0, Q4_0) + FATTN_VEC_CASES_ALL_D(Q4_1, Q4_0) + FATTN_VEC_CASES_ALL_D(Q5_0, Q4_0) + FATTN_VEC_CASES_ALL_D(Q5_1, Q4_0) + FATTN_VEC_CASES_ALL_D(Q8_0, Q4_0) + FATTN_VEC_CASES_ALL_D(BF16, Q4_0) + + FATTN_VEC_CASES_ALL_D(F16, Q4_1) + FATTN_VEC_CASES_ALL_D(Q4_0, Q4_1) + FATTN_VEC_CASES_ALL_D(Q4_1, Q4_1) + FATTN_VEC_CASES_ALL_D(Q5_0, Q4_1) + FATTN_VEC_CASES_ALL_D(Q5_1, Q4_1) + FATTN_VEC_CASES_ALL_D(Q8_0, Q4_1) + FATTN_VEC_CASES_ALL_D(BF16, Q4_1) + + FATTN_VEC_CASES_ALL_D(F16, Q5_0) + FATTN_VEC_CASES_ALL_D(Q4_0, Q5_0) + FATTN_VEC_CASES_ALL_D(Q4_1, Q5_0) + FATTN_VEC_CASES_ALL_D(Q5_0, Q5_0) + FATTN_VEC_CASES_ALL_D(Q5_1, Q5_0) + FATTN_VEC_CASES_ALL_D(Q8_0, Q5_0) + FATTN_VEC_CASES_ALL_D(BF16, Q5_0) + + FATTN_VEC_CASES_ALL_D(F16, Q5_1) + FATTN_VEC_CASES_ALL_D(Q4_0, Q5_1) + FATTN_VEC_CASES_ALL_D(Q4_1, Q5_1) + FATTN_VEC_CASES_ALL_D(Q5_0, Q5_1) + FATTN_VEC_CASES_ALL_D(Q5_1, Q5_1) + FATTN_VEC_CASES_ALL_D(Q8_0, Q5_1) + FATTN_VEC_CASES_ALL_D(BF16, Q5_1) + + FATTN_VEC_CASES_ALL_D(F16, Q8_0) + FATTN_VEC_CASES_ALL_D(Q4_0, Q8_0) + FATTN_VEC_CASES_ALL_D(Q4_1, Q8_0) + FATTN_VEC_CASES_ALL_D(Q5_0, Q8_0) + FATTN_VEC_CASES_ALL_D(Q5_1, Q8_0) + FATTN_VEC_CASES_ALL_D(Q8_0, Q8_0) + FATTN_VEC_CASES_ALL_D(BF16, Q8_0) + + FATTN_VEC_CASES_ALL_D(F16, BF16) + FATTN_VEC_CASES_ALL_D(Q4_0, BF16) + FATTN_VEC_CASES_ALL_D(Q4_1, BF16) + FATTN_VEC_CASES_ALL_D(Q5_0, BF16) + FATTN_VEC_CASES_ALL_D(Q5_1, BF16) + FATTN_VEC_CASES_ALL_D(Q8_0, BF16) + FATTN_VEC_CASES_ALL_D(BF16, BF16) + + return nullptr; +} static void ggml_cuda_flash_attn_ext_vec(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_tensor * Q = dst->src[0]; - ggml_tensor * K = dst->src[1]; - ggml_tensor * V = dst->src[2]; - -#ifdef GGML_CUDA_FA_ALL_QUANTS - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_F16) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_Q4_0) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q4_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_Q4_1) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q5_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_Q5_0) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q5_1) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_Q5_1) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_Q8_0) - - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_1, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_0, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q5_1, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_BF16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_BF16) -#else - FATTN_VEC_CASES_ALL_D(GGML_TYPE_F16, GGML_TYPE_F16) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q4_0, GGML_TYPE_Q4_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_Q8_0, GGML_TYPE_Q8_0) - FATTN_VEC_CASES_ALL_D(GGML_TYPE_BF16, GGML_TYPE_BF16) -#endif // GGML_CUDA_FA_ALL_QUANTS + const ggml_tensor * Q = dst->src[0]; + const ggml_tensor * K = dst->src[1]; + const ggml_tensor * V = dst->src[2]; - GGML_ABORT("fatal error"); + fattn_vec_case_t vec_case = ggml_cuda_get_fattn_vec_case(Q->ne[0], K->type, V->type); + if (vec_case == nullptr) { + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("%s: no FlashAttention vector kernel compiled for K/V types %s-%s, converting K and V to f16 instead (slow). " + "Add \"%s-%s\" to GGML_CUDA_FA_QUANTS to compile it.\n", + __func__, ggml_type_name(K->type), ggml_type_name(V->type), ggml_type_name(K->type), ggml_type_name(V->type)); + warned = true; + } + vec_case = ggml_cuda_get_fattn_vec_case(Q->ne[0], GGML_TYPE_F16, GGML_TYPE_F16); + } + GGML_ASSERT(vec_case != nullptr); + vec_case(ctx, dst); } // Best FlashAttention kernel for a specific GPU: @@ -335,20 +521,17 @@ enum best_fattn_kernel { BEST_FATTN_KERNEL_MMA_F16 = 400, }; -static bool ggml_cuda_fattn_kv_type_supported(ggml_type type) { +// K/V types for which there is a vector kernel template instance, other kernels convert these to f16: +static bool ggml_cuda_fattn_kv_type_supported(const ggml_type type) { switch (type) { case GGML_TYPE_F32: case GGML_TYPE_F16: - return true; + case GGML_TYPE_BF16: + case GGML_TYPE_Q4_0: case GGML_TYPE_Q4_1: case GGML_TYPE_Q5_0: case GGML_TYPE_Q5_1: -#ifndef GGML_CUDA_FA_ALL_QUANTS - return false; -#endif // GGML_CUDA_FA_ALL_QUANTS - case GGML_TYPE_Q4_0: case GGML_TYPE_Q8_0: - case GGML_TYPE_BF16: return true; default: return false; @@ -439,12 +622,6 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const return BEST_FATTN_KERNEL_NONE; } -#ifndef GGML_CUDA_FA_ALL_QUANTS - if (K->type != V->type) { - return BEST_FATTN_KERNEL_NONE; - } -#endif // GGML_CUDA_FA_ALL_QUANTS - if (!ggml_cuda_fattn_kv_type_supported(K->type) || !ggml_cuda_fattn_kv_type_supported(V->type)) { return BEST_FATTN_KERNEL_NONE; } @@ -461,7 +638,12 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const if (turing_mma_available(cc) && Q->ne[0] != 40 && Q->ne[0] != 72) { if (can_use_vector_kernel) { if (!ggml_is_quantized(K->type) && !ggml_is_quantized(V->type)) { - if (cc >= GGML_CUDA_CC_ADA_LOVELACE && Q->ne[1] == 1 && Q->ne[3] == 1 && !(gqa_ratio > 4 && K->ne[1] >= 8192)) { + // the sparse gather exists only in the MMA kernel: (DKQ, DV, 1, 8) with GQA > 4 + const bool sparse_decode = gqa_opt_applies && gqa_ratio > 4 && + ggml_cuda_flash_attn_ext_mma_f16_may_use_sparse(K->ne[0], V->ne[0], 1, 8) && + ggml_cuda_flash_attn_ext_mma_f16_shall_use_sparse(cc, dst, 1, 8); + if (!sparse_decode && cc >= GGML_CUDA_CC_ADA_LOVELACE && Q->ne[1] == 1 && Q->ne[3] == 1 && + !(gqa_ratio > 4 && (Q->ne[0] >= 256 || K->ne[1] >= 8192))) { return BEST_FATTN_KERNEL_VEC; } } else { @@ -511,8 +693,9 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const } } - // AMD WMMA is always faster than the tile kernel if the full tile width of 16 can be utilized. - if ((amd_wmma_available(cc) && gqa_opt_applies && Q->ne[0] <= 128) && Q->ne[0] != 40 && Q->ne[0] != 72 && Q->ne[1] * gqa_ratio_eff > 8) { + // AMD WMMA is faster than the tile kernel if the wide tiles with high arithmetic intensity can be utilized. + if ((amd_wmma_available(cc) && gqa_opt_applies && Q->ne[0] <= 256) && Q->ne[0] != 40 && Q->ne[0] != 72 && + Q->ne[1] * gqa_ratio_eff > (Q->ne[0] <= 128 ? 8 : 16)) { return BEST_FATTN_KERNEL_MMA_F16; } @@ -536,6 +719,7 @@ static best_fattn_kernel ggml_cuda_get_best_fattn_kernel(const int device, const size_t ggml_cuda_flash_attn_ext_get_alloc_size(int device, const ggml_tensor * dst) { GGML_ASSERT(dst->op == GGML_OP_FLASH_ATTN_EXT); + const ggml_tensor * Q = dst->src[0]; const ggml_tensor * K = dst->src[1]; const ggml_tensor * V = dst->src[2]; @@ -553,10 +737,11 @@ size_t ggml_cuda_flash_attn_ext_get_alloc_size(int device, const ggml_tensor * d need_f16_K = true; need_f16_V = true; break; - case BEST_FATTN_KERNEL_VEC: - need_f16_K = K->type == GGML_TYPE_F32; - need_f16_V = V->type == GGML_TYPE_F32; - break; + case BEST_FATTN_KERNEL_VEC: { + const bool f16_fallback = ggml_cuda_get_fattn_vec_case(Q->ne[0], K->type, V->type) == nullptr; + need_f16_K = K->type == GGML_TYPE_F32 || f16_fallback; + need_f16_V = V->type == GGML_TYPE_F32 || f16_fallback; + } break; case BEST_FATTN_KERNEL_NONE: break; } diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu index f2e381ee..27b83503 100644 --- a/ggml/src/ggml-cuda/ggml-cuda.cu +++ b/ggml/src/ggml-cuda/ggml-cuda.cu @@ -32,12 +32,14 @@ #include "ggml-cuda/mmq.cuh" #include "ggml-cuda/mmvf.cuh" #include "ggml-cuda/mmvq.cuh" +#include "ggml-cuda/moe-weighted-reduction.cuh" #include "ggml-cuda/norm.cuh" #include "ggml-cuda/opt-step-adamw.cuh" #include "ggml-cuda/opt-step-sgd.cuh" #include "ggml-cuda/out-prod.cuh" #include "ggml-cuda/pad.cuh" #include "ggml-cuda/pool2d.cuh" +#include "ggml-cuda/pool1d.cuh" #include "ggml-cuda/quantize.cuh" #include "ggml-cuda/rope.cuh" #include "ggml-cuda/roll.cuh" @@ -210,6 +212,7 @@ static int ggml_cuda_parse_id(char devName[]) { } archNum += archMajor * 0x100; archNum += archMinor; + return archNum; } #endif // defined(GGML_USE_HIP) @@ -301,11 +304,7 @@ static ggml_cuda_device_info ggml_cuda_init() { info.default_tensor_split[id] = total_vram; total_vram += device_vram; -#if defined(GGML_USE_HIP) - info.devices[id].integrated = prop.integrated; -#else info.devices[id].integrated = false; // Temporarily disabled due to issues with corrupted output (e.g. #15034) -#endif info.devices[id].nsm = prop.multiProcessorCount; info.devices[id].smpb = prop.sharedMemPerBlock; info.devices[id].warp_size = prop.warpSize; @@ -711,9 +710,12 @@ ggml_backend_cuda_context::~ggml_backend_cuda_context() { if (streams[i][j] != nullptr) { CUDA_CHECK(cudaStreamDestroy(streams[i][j])); } - } - if (cublas_handles[i] != nullptr) { - CUBLAS_CHECK(cublasDestroy(cublas_handles[i])); + if (cublas_handles[i][j] != nullptr) { + CUBLAS_CHECK(cublasDestroy(cublas_handles[i][j])); + } + if (cublas_workspaces[i][j] != nullptr) { + CUDA_CHECK(cudaFree(cublas_workspaces[i][j])); + } } } } @@ -911,6 +913,7 @@ static size_t ggml_backend_cuda_buffer_type_get_alloc_size(ggml_backend_buffer_t : ggml_nbytes(tensor); int64_t ne0 = tensor->ne[0]; + // [TAG_ALLOC_SIZE_EXPAND] if (ggml_is_quantized(tensor->type)) { if (ne0 % MATRIX_ROW_PADDING != 0) { GGML_ASSERT(tensor->nb[0] == ggml_element_size(tensor)); @@ -1416,7 +1419,7 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const const int64_t ne_dst = ggml_nelements(dst); cudaStream_t main_stream = ctx.stream(); - CUBLAS_CHECK(cublasSetStream(ctx.cublas_handle(), main_stream)); + cublasHandle_t cublas_h = ctx.cublas_handle(); const size_t src0_ts = ggml_type_size(src0->type); GGML_ASSERT(nb00 == src0_ts); @@ -1539,14 +1542,14 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const // probably because the internal kernel selection logic is suboptimal. if (compute_type == GGML_TYPE_F32 && ne12 == 1 && ne13 == 1) { CUBLAS_CHECK( - cublasSgemm(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + cublasSgemm(cublas_h, CUBLAS_OP_T, CUBLAS_OP_N, ne01, ne11, ne10, (const float *) alpha, (const float *) src0_ptr, s01, (const float *) src1_ptr, s11, (const float *) beta, (float *) dst_ptr, ne0)); } else if (ne12 == 1 && ne13 == 1) { CUBLAS_CHECK( - cublasGemmEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + cublasGemmEx(cublas_h, CUBLAS_OP_T, CUBLAS_OP_N, ne01, ne11, ne10, alpha, src0_ptr, cu_data_type_a, s01, src1_ptr, cu_data_type_b, s11, @@ -1561,7 +1564,7 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const // there is no broadcast and src0, src1 are contiguous across dims 2, 3 // use cublasGemmStridedBatchedEx CUBLAS_CHECK( - cublasGemmStridedBatchedEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + cublasGemmStridedBatchedEx(cublas_h, CUBLAS_OP_T, CUBLAS_OP_N, ne01, ne11, ne10, alpha, src0_ptr, cu_data_type_a, s01, sma, // strideA src1_ptr, cu_data_type_b, s11, smb, // strideB @@ -1599,7 +1602,7 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const CUDA_CHECK(cudaGetLastError()); CUBLAS_CHECK( - cublasGemmBatchedEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, + cublasGemmBatchedEx(cublas_h, CUBLAS_OP_T, CUBLAS_OP_N, ne01, ne11, ne10, alpha, (const void **) (ptrs_src.get() + 0*ne23), cu_data_type_a, s01, (const void **) (ptrs_src.get() + 1*ne23), cu_data_type_b, s11, @@ -1617,11 +1620,19 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const } static void ggml_cuda_mul_mat_cublas(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const int cc = ggml_cuda_info().devices[ctx.device].cc; ggml_type compute_type = src0->type; if (ggml_is_quantized(compute_type)) { - compute_type = fast_fp16_hardware_available(ggml_cuda_info().devices[ctx.device].cc) ? GGML_TYPE_F16 : GGML_TYPE_F32; - } else if (compute_type == GGML_TYPE_F16 && !fast_fp16_hardware_available(ggml_cuda_info().devices[ctx.device].cc)) { + compute_type = fast_fp16_hardware_available(cc) ? GGML_TYPE_F16 : GGML_TYPE_F32; + } else if (compute_type == GGML_TYPE_F16 && !fast_fp16_hardware_available(cc)) { compute_type = GGML_TYPE_F32; + } else if (compute_type == GGML_TYPE_BF16 && !fast_bf16_hardware_available(cc)) { + if (GGML_CUDA_CC_IS_AMD(cc) && src1->ne[1] > 32) { + compute_type = GGML_TYPE_F32; + } + if (GGML_CUDA_CC_IS_NVIDIA(cc) && src1->ne[1] > (cc >= GGML_CUDA_CC_VOLTA ? 8 : 128)) { + compute_type = GGML_TYPE_F32; + } } if (dst->op_params[0] == GGML_PREC_F32) { compute_type = GGML_TYPE_F32; @@ -1740,7 +1751,7 @@ static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up, return false; } - static constexpr std::array valid_glu_ops = { GGML_GLU_OP_SWIGLU, GGML_GLU_OP_GEGLU, GGML_GLU_OP_SWIGLU_OAI }; + static constexpr std::array valid_glu_ops = { GGML_GLU_OP_SWIGLU, GGML_GLU_OP_GEGLU, GGML_GLU_OP_SWIGLU_OAI, GGML_GLU_OP_SWIGLU_CLAMP }; if (std::find(valid_glu_ops.begin(), valid_glu_ops.end(), ggml_get_glu_op(glu)) == valid_glu_ops.end()) { return false; @@ -1802,7 +1813,7 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) { return false; } - if (tensor->op == GGML_OP_MUL_MAT_ID && dst->ne[2] != 1) { + if (tensor->op == GGML_OP_MUL_MAT_ID && dst->ne[2] > get_mmvq_mmid_max_batch(src0->type, cc)) { return false; } @@ -2199,6 +2210,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg case GGML_GLU_OP_GEGLU_QUICK: ggml_cuda_op_geglu_quick(ctx, dst); break; + case GGML_GLU_OP_SWIGLU_CLAMP: + ggml_cuda_op_swiglu_clamp(ctx, dst); + break; default: return false; } @@ -2323,6 +2337,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg case GGML_OP_POOL_2D: ggml_cuda_op_pool2d(ctx, dst); break; + case GGML_OP_POOL_1D: + ggml_cuda_op_pool1d(ctx, dst); + break; case GGML_OP_SUM: ggml_cuda_op_sum(ctx, dst); break; @@ -2723,6 +2740,12 @@ static bool ggml_cuda_should_fuse_rms_norm_mul_rope(const ggml_tensor * rms_norm return false; } + // ggml_rope_set_offset is not yet supported in the fused kernel + const int n_offs = ((const int32_t *) rope->op_params)[15]; + if (n_offs != 0) { + return false; + } + return true; } @@ -2966,9 +2989,10 @@ static bool ggml_cuda_check_fusion_memory_ranges(const ggml_cgraph * cgraph, }; bool is_ok = true; - // exception for topk-moe, as each row is read entirely before writing - if (ggml_nrows(cgraph->nodes[node_idx]) == 1 && is_topk_moe) { - return true; + // one block reads all logits before it writes, so logits may alias the out nodes + const ggml_tensor * logits_may_alias = nullptr; + if (is_topk_moe && ggml_nrows(cgraph->nodes[node_idx]) <= TOPK_MOE_ROWS_PER_BLOCK) { + logits_may_alias = cgraph->nodes[node_idx]->src[0]; } for (int i = 0; i < out_count; ++i) { @@ -2982,7 +3006,7 @@ static bool ggml_cuda_check_fusion_memory_ranges(const ggml_cgraph * cgraph, for (int src_idx = 0; src_idx < GGML_MAX_SRC; ++src_idx) { const ggml_tensor * src = cgraph->nodes[j]->src[src_idx]; - if (!src || src->op == GGML_OP_NONE) { + if (!src || src->op == GGML_OP_NONE || src == logits_may_alias) { continue; } @@ -3008,6 +3032,150 @@ static bool ggml_cuda_check_fusion_memory_ranges(const ggml_cgraph * cgraph, return is_ok; } +// The long form spans 2*k + 1 nodes. ggml_can_fuse_subgraph() accepts at most +// 31 nodes, so k <= 15; larger values use the per-operation path. +static constexpr int MOE_WEIGHTED_REDUCTION_MAX_EXPERTS = 15; + +struct ggml_cuda_moe_weighted_reduction_match { + const ggml_tensor * experts = nullptr; + const ggml_tensor * expert_scale = nullptr; + const ggml_tensor * weights = nullptr; + ggml_tensor * dst = nullptr; + int node_count = 0; +}; + +static bool ggml_cuda_match_moe_weighted_reduction( + const ggml_cgraph * cgraph, + int node_idx, + ggml_cuda_moe_weighted_reduction_match & match) { + const ggml_tensor * first = cgraph->nodes[node_idx]; + if (first->op != GGML_OP_MUL || first->type != GGML_TYPE_F32 || !ggml_is_contiguous(first)) { + return false; + } + + auto split_mul = [](const ggml_tensor * mul, const ggml_tensor *& full, const ggml_tensor *& broadcast) { + auto is_weights = [mul](const ggml_tensor * tensor) { + return tensor && tensor->type == GGML_TYPE_F32 && ggml_is_contiguous(tensor) && tensor->ne[0] == 1 && + tensor->ne[1] == mul->ne[1] && tensor->ne[2] == mul->ne[2] && tensor->ne[3] == mul->ne[3]; + }; + auto is_experts = [mul](const ggml_tensor * tensor) { + return tensor && tensor->type == GGML_TYPE_F32 && ggml_is_contiguous(tensor) && + ggml_are_same_shape(tensor, mul); + }; + + if (is_experts(mul->src[0]) && is_weights(mul->src[1])) { + full = mul->src[0]; + broadcast = mul->src[1]; + return true; + } + if (is_experts(mul->src[1]) && is_weights(mul->src[0])) { + full = mul->src[1]; + broadcast = mul->src[0]; + return true; + } + return false; + }; + + const ggml_tensor * weighted = first; + const ggml_tensor * experts = nullptr; + const ggml_tensor * expert_scale = nullptr; + const ggml_tensor * weights = nullptr; + int mul_count = 1; + + // Match both structural forms: + // (experts * expert_scale) * router_weight + // experts * router_weight + // The matcher does not depend on the model or quantization type. + if (node_idx + 1 < cgraph->n_nodes) { + const ggml_tensor * second = cgraph->nodes[node_idx + 1]; + const ggml_tensor * scaled = nullptr; + const ggml_tensor * route = nullptr; + const ggml_tensor * raw = nullptr; + const ggml_tensor * scale = nullptr; + if (second->op == GGML_OP_MUL && second->type == GGML_TYPE_F32 && ggml_is_contiguous(second) && + split_mul(second, scaled, route) && scaled == first && split_mul(first, raw, scale)) { + weighted = second; + experts = raw; + expert_scale = scale; + weights = route; + mul_count = 2; + } + } + + if (experts == nullptr && !split_mul(first, experts, weights)) { + return false; + } + + const int n_expert_used = (int) weighted->ne[1]; + const int64_t n_tokens = weighted->ne[2] * weighted->ne[3]; + if (n_expert_used < 2 || n_expert_used > MOE_WEIGHTED_REDUCTION_MAX_EXPERTS || n_tokens <= 0) { + return false; + } + + const int node_count = 2 * n_expert_used + mul_count - 1; + if (node_idx + node_count > cgraph->n_nodes) { + return false; + } + + std::vector ops(node_count, GGML_OP_VIEW); + ops[0] = GGML_OP_MUL; + if (mul_count == 2) { + ops[1] = GGML_OP_MUL; + } + std::vector views; + views.reserve(n_expert_used); + const ggml_tensor * previous = nullptr; + int n_adds = 0; + for (int offset = mul_count; offset < node_count; ++offset) { + const ggml_tensor * candidate = cgraph->nodes[node_idx + offset]; + ops[offset] = candidate->op; + + if (candidate->op == GGML_OP_VIEW) { + const int expert = (int) views.size(); + if (expert >= n_expert_used || candidate->src[0] != weighted || candidate->view_src != weighted || + candidate->type != GGML_TYPE_F32 || candidate->ne[0] != weighted->ne[0] || + candidate->ne[1] != n_tokens || candidate->ne[2] != 1 || candidate->ne[3] != 1 || + candidate->nb[0] != weighted->nb[0] || candidate->nb[1] != weighted->nb[2] || + candidate->view_offs != (size_t) expert * weighted->nb[1]) { + return false; + } + views.push_back(candidate); + continue; + } + + if (candidate->op != GGML_OP_ADD || views.size() < 2 || n_adds + 1 >= (int) views.size()) { + return false; + } + const ggml_tensor * lhs = n_adds == 0 ? views[0] : previous; + const ggml_tensor * rhs = views[n_adds + 1]; + if (candidate->src[0] != lhs || candidate->src[1] != rhs || candidate->type != GGML_TYPE_F32) { + return false; + } + previous = candidate; + ++n_adds; + } + + if ((int) views.size() != n_expert_used || n_adds != n_expert_used - 1 || previous == nullptr) { + return false; + } + if (!ggml_is_contiguous(previous) || previous->ne[0] != weighted->ne[0] || + previous->ne[1] != n_tokens || previous->ne[2] != 1 || previous->ne[3] != 1) { + return false; + } + + const int output_idx = node_idx + node_count - 1; + if (!ggml_can_fuse_subgraph(cgraph, node_idx, node_count, ops.data(), &output_idx, 1)) { + return false; + } + + match.experts = experts; + match.expert_scale = expert_scale; + match.weights = weights; + match.dst = cgraph->nodes[output_idx]; + match.node_count = node_count; + return true; +} + static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph, int node_idx, @@ -3270,6 +3438,18 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph ggml_tensor * node = cgraph->nodes[i]; + if (node->op == GGML_OP_MUL) { + ggml_cuda_moe_weighted_reduction_match match; + if (ggml_cuda_match_moe_weighted_reduction(cgraph, i, match)) { + const int output_idx = i + match.node_count - 1; + if (ggml_cuda_check_fusion_memory_ranges(cgraph, i, match.node_count, &output_idx, 1)) { + ggml_cuda_op_moe_weighted_reduction( + *cuda_ctx, match.experts, match.expert_scale, match.weights, match.dst); + return match.node_count - 1; + } + } + } + // gated_delta_net -> cpy: scatter recurrent-state snapshots into the cache if (node->op == GGML_OP_GATED_DELTA_NET) { ggml_cuda_gated_delta_net_fused_cache fused_state_cpy; @@ -3290,6 +3470,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph ggml_cuda_topk_moe_args args; const bool can_fuse = ggml_cuda_topk_moe_fusion(cgraph, i, args); std::vector ops; + ops.reserve(13); // max ops; avoids gcc -Wstringop-overflow false positive if (can_fuse) { const ggml_tensor * logits = node->src[0]; @@ -3582,6 +3763,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph fusion_data.x_scale = up_scale; fusion_data.gate_scale = gate_scale; fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); if (ggml_cuda_should_fuse_mul_mat_vec_q(up_n)) { ggml_cuda_mul_mat_vec_q(*cuda_ctx, src0, src1, ids, cgraph->nodes[glu_idx], &fusion_data); @@ -3675,6 +3857,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph fusion_data.x_scale = up_scale; fusion_data.gate_scale = gate_scale; fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); if (ggml_cuda_should_fuse_mul_mat_vec_q(up_n)) { ggml_cuda_mul_mat_vec_q(*cuda_ctx, src0, src1, ids, cgraph->nodes[glu_idx], &fusion_data); @@ -3731,6 +3914,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph fusion_data.x_bias = up_bias_tensor; fusion_data.gate_bias = gate_bias_tensor; fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); ggml_cuda_mul_mat_vec_f(*cuda_ctx, src0, src1, ids, glu, &fusion_data); fused_mul_mat_vec = true; @@ -3744,6 +3928,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph fusion_data.x_bias = up_bias_tensor; fusion_data.gate_bias = gate_bias_tensor; fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); ggml_cuda_mul_mat_vec_q(*cuda_ctx, src0, src1, ids, glu, &fusion_data); fused_mul_mat_vec = true; @@ -3768,8 +3953,9 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph if (ggml_cuda_should_fuse_mul_mat_vec_f(up)) { ggml_cuda_mm_fusion_args_host fusion_data{}; - fusion_data.gate = gate->src[0]; - fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.gate = gate->src[0]; + fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); ggml_cuda_mul_mat_vec_f(*cuda_ctx, src0, src1, ids, glu, &fusion_data); fused_mul_mat_vec = true; @@ -3779,8 +3965,9 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph if (ggml_cuda_should_fuse_mul_mat_vec_q(up)) { ggml_cuda_mm_fusion_args_host fusion_data{}; - fusion_data.gate = gate->src[0]; - fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.gate = gate->src[0]; + fusion_data.glu_op = ggml_get_glu_op(glu); + fusion_data.glu_limit = ggml_get_op_params_f32(glu, 3); ggml_cuda_mul_mat_vec_q(*cuda_ctx, src0, src1, ids, glu, &fusion_data); fused_mul_mat_vec = true; @@ -4315,9 +4502,98 @@ static void ggml_backend_cuda_event_wait(ggml_backend_t backend, ggml_backend_ev } } -static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph) { +static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph, ggml_backend_graph_optimize_params * params) { ggml_backend_cuda_context * cuda_ctx = (ggml_backend_cuda_context *) backend->context; + static const bool disable_fusion = getenv("GGML_CUDA_DISABLE_FUSION") != nullptr && std::atoi(getenv("GGML_CUDA_DISABLE_FUSION")); + + auto add_alloc_deps = [&](size_t start, size_t last_node) { + + for (size_t i = start; i < last_node; ++i) { + params->add_alloc_dep(params->user_data, cgraph->nodes[i], cgraph->nodes[last_node]); + + for (int j = 0; j < GGML_MAX_SRC; ++j) { + if (cgraph->nodes[i]->src[j]) { + params->add_alloc_dep(params->user_data, cgraph->nodes[i]->src[j], cgraph->nodes[last_node]); + } + } + } + }; + + if (!disable_fusion) { + // add alloc deps for performance positive fusions. This may increase the overall compute buffer size. + // TODO: consolidate fusion paths in graph_optimize and graph_compute + for (int i = 0; i < cgraph->n_nodes; ++i) { + ggml_cuda_moe_weighted_reduction_match match; + if (ggml_cuda_match_moe_weighted_reduction(cgraph, i, match)) { + params->add_alloc_dep(params->user_data, const_cast(match.experts), match.dst); + params->add_alloc_dep(params->user_data, const_cast(match.weights), match.dst); + if (match.expert_scale != nullptr) { + params->add_alloc_dep( + params->user_data, const_cast(match.expert_scale), match.dst); + } + i += match.node_count - 1; + } + + if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX || + cgraph->nodes[i]->op == GGML_OP_ARGSORT) { + ggml_cuda_topk_moe_args args; + const bool can_fuse = ggml_cuda_topk_moe_fusion(cgraph, i, args); + std::vector ops; + ops.reserve(13); // max ops; avoids gcc -Wstringop-overflow false positive + + const ggml_tensor * node = cgraph->nodes[i]; + + if (can_fuse) { + const ggml_tensor * logits = node->src[0]; + ggml_tensor * weights = nullptr; + ggml_tensor * ids = nullptr; + + if (!args.delayed_softmax) { + int out_nodes[2]; // nodes which can't be elided + + if (args.sigmoid) { + ops.insert(ops.end(), { GGML_OP_UNARY }); + } else if (args.sqrt_softplus) { + ops.insert(ops.end(), { GGML_OP_UNARY, GGML_OP_SQRT }); + } else { + ops.insert(ops.end(), { GGML_OP_SOFT_MAX }); + } + const int i_probs = i + (int) ops.size() - 1; // last node of the gating activation + + if (args.prob_bias) { + ops.insert(ops.end(), { GGML_OP_RESHAPE, GGML_OP_ADD, GGML_OP_ARGSORT, GGML_OP_VIEW, + GGML_OP_GET_ROWS }); + out_nodes[0] = i_probs + 4; + } else { + ops.insert(ops.end(), { GGML_OP_RESHAPE, GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS }); + out_nodes[0] = i_probs + 3; + } + ids = cgraph->nodes[out_nodes[0]]; + + if (args.norm) { + ops.insert(ops.end(), + { GGML_OP_RESHAPE, GGML_OP_SUM_ROWS, GGML_OP_CLAMP, GGML_OP_DIV, GGML_OP_RESHAPE }); + } + if (args.scale) { + ops.insert(ops.end(), { GGML_OP_SCALE }); + } + + weights = cgraph->nodes[i + ops.size() - 1]; + out_nodes[1] = i + ops.size() - 1; + + if (ggml_can_fuse_subgraph(cgraph, i, ops.size(), ops.data(), out_nodes, 2) && + ggml_cuda_should_use_topk_moe(node, logits, weights, ids)) { + + add_alloc_deps(i, i + ops.size()); + i += ops.size() - 1; + } + } + } + } + } + } + #ifdef USE_CUDA_GRAPH const void * graph_key = ggml_cuda_graph_get_key(cgraph); const bool use_cuda_graph = ggml_cuda_graph_set_enabled(cuda_ctx, graph_key); @@ -4339,10 +4615,12 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph ggml_cuda_stream_context & stream_context = cuda_ctx->stream_context(); stream_context.reset(); - if (!use_cuda_graph || ggml_backend_cuda_get_device_count() != 1) { + if (!use_cuda_graph) { return; } + ggml_cuda_set_device(cuda_ctx->device); + // number of out-degrees for a particular node std::unordered_map fan_out; // reverse mapping of node to index in the cgraph @@ -4598,8 +4876,8 @@ static std::string ggml_cuda_device_description(int device) { const ggml_cuda_device_info & info = ggml_cuda_info(); std::string description = prop.name; if (info.device_count > info.physical_device_count) { - description += " (physical device " + std::to_string(info.devices[device].physical_device) + - ", virtual device " + std::to_string(info.devices[device].virtual_index) + ")"; + description += " (dev p" + std::to_string(info.devices[device].physical_device) + + "/v" + std::to_string(info.devices[device].virtual_index) + ")"; } return description; } @@ -4904,6 +5182,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return ggml_is_contiguous_1(op->src[0]); default: return false; @@ -4920,6 +5199,9 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g if (b->type == GGML_TYPE_F16 && a->type != GGML_TYPE_F16) { return false; } + if (op->op == GGML_OP_MUL_MAT_ID && ggml_get_op_params_i32(op, 3) == GGML_PREC_F32) { + return false; + } #ifdef GGML_USE_MUSA const int cc = ggml_cuda_info().devices[dev_ctx->device].cc; if (b->ne[2]*b->ne[3] > 1 && !ggml_is_transposed(a) && !ggml_is_transposed(b)) { @@ -5087,10 +5369,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g return false; } break; case GGML_OP_DUP: - { - ggml_type src0_type = op->src[0]->type; - return src0_type != GGML_TYPE_I32 && src0_type != GGML_TYPE_I16; - } break; + return true; case GGML_OP_ARGMAX: case GGML_OP_COUNT_EQUAL: { @@ -5236,6 +5515,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g case GGML_OP_CONV_2D_DW: return op->src[0]->type == GGML_TYPE_F32; case GGML_OP_CONV_TRANSPOSE_2D: + case GGML_OP_POOL_1D: case GGML_OP_POOL_2D: return true; case GGML_OP_ACC: @@ -5245,6 +5525,11 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g case GGML_OP_SUM: return ggml_is_contiguous_rows(op->src[0]); case GGML_OP_TOP_K: +#if defined(GGML_USE_HIP) || defined(GGML_CUDA_USE_CUB) + return true; +#else + return op->src[0]->ne[0] <= 1024; +#endif // defined(GGML_USE_HIP) || defined(GGML_CUDA_USE_CUB) case GGML_OP_ARGSORT: #ifndef GGML_CUDA_USE_CUB return op->src[0]->ne[0] <= 1024; @@ -5252,7 +5537,9 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g return true; #endif case GGML_OP_SUM_ROWS: + return op->src[0]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32 && ggml_is_contiguous_rows(op->src[0]); case GGML_OP_MEAN: + return op->src[0]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32 && ggml_is_contiguous_rows(op->src[0]); case GGML_OP_GROUP_NORM: return ggml_is_contiguous(op->src[0]); case GGML_OP_PAD: @@ -5281,7 +5568,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g op->type == GGML_TYPE_F32; case GGML_OP_DSV4_HC_POST: return op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && - op->src[2]->type == GGML_TYPE_F32 && op->src[3]->type == GGML_TYPE_F32 && + op->src[2]->type == GGML_TYPE_F32 && (op->src[3] == nullptr || op->src[3]->type == GGML_TYPE_F32) && op->type == GGML_TYPE_F32; case GGML_OP_FLASH_ATTN_EXT: return ggml_cuda_flash_attn_ext_supported(dev_ctx->device, op); @@ -5431,8 +5718,8 @@ static ggml_backend_feature * ggml_backend_cuda_get_features(ggml_backend_reg_t features.push_back({ "USE_GRAPHS", "1" }); #endif - #ifdef GGML_CUDA_FA_ALL_QUANTS - features.push_back({ "FA_ALL_QUANTS", "1" }); + #ifdef GGML_CUDA_FA_QUANTS + features.push_back({ "FA_QUANTS", GGML_CUDA_FA_QUANTS }); #endif { diff --git a/ggml/src/ggml-cuda/im2col.cu b/ggml/src/ggml-cuda/im2col.cu index 28c79ab4..d377f285 100644 --- a/ggml/src/ggml-cuda/im2col.cu +++ b/ggml/src/ggml-cuda/im2col.cu @@ -7,40 +7,41 @@ template static __global__ void im2col_kernel( const float * x, T * dst, int64_t IC, int64_t IW, int64_t IH, int64_t OH, int64_t OW, int64_t KW, int64_t KH, - int64_t IC_IH_IW, int64_t IH_IW, int64_t N_OH, int64_t KH_KW, int64_t IC_KH_KW, + int64_t N, int64_t IC_IH_IW, int64_t IH_IW, int64_t N_OH, int64_t KH_KW, int64_t IC_KH_KW, int s0, int s1, int p0, int p1, int d0, int d1) { - const int64_t i = threadIdx.x + blockIdx.x * blockDim.x; - if (i >= IC_KH_KW) { - return; - } - - const int64_t iic = i / (KH_KW); - const int64_t rem = i - iic * KH_KW; - const int64_t ikh = rem / KW; - const int64_t ikw = rem - ikh * KW; - - for (int64_t iow = blockIdx.y; iow < OW; iow += MAX_GRIDDIM_Y) { - for (int64_t iz = blockIdx.z; iz < N_OH; iz += MAX_GRIDDIM_Z) { - const int64_t in = iz / OH; - const int64_t ioh = iz - in * OH; - - const int64_t iiw = iow * s0 + ikw * d0 - p0; - const int64_t iih = ioh * s1 + ikh * d1 - p1; - - const int64_t offset_dst = - ((in * OH + ioh) * OW + iow) * IC_KH_KW + iic * KH_KW + ikh * KW + ikw; - - if (iih < 0 || iih >= IH || iiw < 0 || iiw >= IW) { - dst[offset_dst] = 0.0f; - } else { - const int64_t offset_src = iic * IC_IH_IW + in * IH_IW; - dst[offset_dst] = x[offset_src + iih * IW + iiw]; + const int tid = threadIdx.x; + + const int64_t total_channels = IC * KH * KW; + const int threads_per_pos = blockDim.x; + const int64_t start_ch = tid; + const int64_t stride_ch = threads_per_pos; + + for (int64_t iow = blockIdx.x; iow < OW; iow += MAX_GRIDDIM_Y) { + for (int64_t iz = blockIdx.y; iz < N_OH; iz += MAX_GRIDDIM_Z) { + const int64_t in = iz / OH; + const int64_t ioh = iz - in * OH; + + for (int64_t iic_khw = start_ch; iic_khw < total_channels; iic_khw += stride_ch) { + const int64_t iic = iic_khw / KH_KW; + const int64_t rem = iic_khw - iic * KH_KW; + const int64_t ikh = rem / KW; + const int64_t ikw = rem - ikh * KW; + + const int64_t iiw = iow * s0 + ikw * d0 - p0; + const int64_t iih = ioh * s1 + ikh * d1 - p1; + + const int64_t offset_dst = + ((in * OH + ioh) * OW + iow) * IC_KH_KW + iic * KH_KW + ikh * KW + ikw; + + if (iih < 0 || iih >= IH || iiw < 0 || iiw >= IW) { + dst[offset_dst] = 0.0f; + } else { + const int64_t offset_src = iic * IC_IH_IW + in * IH_IW; + dst[offset_dst] = x[offset_src + iih * IW + iiw]; + } } } } - - GGML_UNUSED(IC); - GGML_UNUSED(KH); } // im2col: [N, IC, IH, IW] => [N, OH, OW, IC*KH*KW] @@ -50,13 +51,15 @@ static void im2col_cuda(const float * x, T* dst, int64_t N, int64_t IC_IH_IW, int64_t IH_IW, int s0,int s1,int p0,int p1,int d0,int d1, cudaStream_t stream) { const int64_t IC_KH_KW = IC * KH * KW; - const int64_t num_blocks = (IC_KH_KW + CUDA_IM2COL_BLOCK_SIZE - 1) / CUDA_IM2COL_BLOCK_SIZE; const int64_t N_OH = N * OH; const int64_t KH_KW = KW*KH; - dim3 block_nums(num_blocks, MIN(OW, MAX_GRIDDIM_Y), MIN(N_OH, MAX_GRIDDIM_Z)); - im2col_kernel<<>>(x, dst, IC, IW, IH, OH, OW, KW, KH, - IC_IH_IW, IH_IW, N_OH, KH_KW, IC_KH_KW, - s0, s1, p0, p1, d0, d1); + const int threads_per_block = MIN((int)IC_KH_KW, CUDA_IM2COL_BLOCK_SIZE); + dim3 block_nums(MIN(OW, MAX_GRIDDIM_Y), MIN(N_OH, MAX_GRIDDIM_Z)); + + im2col_kernel<<>>( + x, dst, IC, IW, IH, OH, OW, KW, KH, + N, IC_IH_IW, IH_IW, N_OH, KH_KW, IC_KH_KW, + s0, s1, p0, p1, d0, d1); } static void im2col_cuda_f16(const float * x, half * dst, diff --git a/ggml/src/ggml-cuda/mean.cu b/ggml/src/ggml-cuda/mean.cu index a8f6046e..64ad7e1d 100644 --- a/ggml/src/ggml-cuda/mean.cu +++ b/ggml/src/ggml-cuda/mean.cu @@ -18,7 +18,7 @@ void ggml_cuda_op_mean(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { GGML_ASSERT(src0->type == GGML_TYPE_F32); GGML_ASSERT(dst->type == GGML_TYPE_F32); - GGML_ASSERT(ggml_is_contiguous(src0)); + GGML_ASSERT(ggml_is_contiguous_rows(src0)); const int64_t ncols = src0->ne[0]; const int64_t nrows = ggml_nrows(src0); @@ -65,13 +65,20 @@ void ggml_cuda_op_mean(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { // Heuristic for block size selection to optimize occupancy. // See discussion in: https://github.com/ggml-org/llama.cpp/pull/15132 + dim3 block_dims; if ((nrows / nsm) < 2) { - const dim3 block_dims(512, 1, 1); - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); - ggml_cuda_kernel_launch(reduce_rows_f32, launch_params, src0_d, dst_d, ncols); + block_dims = dim3(512, 1, 1); } else { - const dim3 block_dims(ncols < 1024 ? 32 : 128, 1, 1); - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); + block_dims = dim3(ncols < 1024 ? 32 : 128, 1, 1); + } + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); + + if (ggml_is_contiguous(src0)) { ggml_cuda_kernel_launch(reduce_rows_f32, launch_params, src0_d, dst_d, ncols); + return; } + + const char * src0_d_bytes = (const char *) src0->data; + ggml_cuda_kernel_launch(reduce_rows_f32_strided, launch_params, src0_d_bytes, dst_d, ncols, + src0->ne[1], src0->ne[2], src0->nb[1], src0->nb[2], src0->nb[3]); } diff --git a/ggml/src/ggml-cuda/mma.cuh b/ggml/src/ggml-cuda/mma.cuh index 8d7c69dc..3583ba5e 100644 --- a/ggml/src/ggml-cuda/mma.cuh +++ b/ggml/src/ggml-cuda/mma.cuh @@ -782,6 +782,20 @@ namespace ggml_cuda_mma { } } + // Byte offset of tile element (i, j). If swz, XOR swizzle it to avoid bank conflicts without row padding. + template + static __device__ __forceinline__ int swizzle_bytes(const int i, const int j, const int stride) { + static_assert(!swz || sizeof(T) == 4, "swizzled tiles need 32 bit elements"); + const int off = (i*stride + j) * (int) sizeof(T); + return swz ? off ^ ((i & 7) << 4) : off; + } + + template + static __device__ __forceinline__ const T * swizzle( + const T * __restrict__ tile_base, const int i, const int j, const int stride) { + return (const T *) ((const char *) tile_base + swizzle_bytes(i, j, stride)); + } + template static __device__ __forceinline__ void load_ldmatrix( tile<8, 8, T> & t, const T * __restrict__ xs0, const int stride) { @@ -858,6 +872,29 @@ namespace ggml_cuda_mma { #endif // TURING_MMA_AVAILABLE } + // Load from tile element (i0, j0), swz tells if the tile is stored swizzled. + template + static __device__ __forceinline__ void load_ldmatrix( + tile & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) { + if constexpr (!swz) { + load_ldmatrix(t, tile_base + i0*stride + j0, stride); + return; + } +#if defined(TURING_MMA_AVAILABLE) + static_assert(I == 16, "bad tile width"); + static_assert(J == 8, "bad tile height"); + const int i = i0 + threadIdx.x % t.I; + const int j = j0 + (threadIdx.x / t.I) * (t.J / 2); + int * xi = (int *) t.x; + asm volatile("ldmatrix.sync.aligned.m8n8.x4.b16 {%0, %1, %2, %3}, [%4];" + : "=r"(xi[0]), "=r"(xi[1]), "=r"(xi[2]), "=r"(xi[3]) + : "l"(swizzle(tile_base, i, j, stride))); +#else + GGML_UNUSED_VARS(t, tile_base, i0, j0, stride); + NO_DEVICE_CODE; +#endif // defined(TURING_MMA_AVAILABLE) + } + static __device__ __forceinline__ void load_ldmatrix( tile<8, 4, half2, DATA_LAYOUT_I_MAJOR_MIRRORED> & t, const half2 * __restrict__ xs0, const int stride) { ggml_cuda_memcpy_1<4*sizeof(half2)>(t.x, xs0 + t.get_i(0)*stride); @@ -917,6 +954,29 @@ namespace ggml_cuda_mma { #endif // TURING_MMA_AVAILABLE } + // Load from tile element (i0, j0), swz tells if the tile is stored swizzled. + template + static __device__ __forceinline__ void load_ldmatrix_trans( + tile & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) { + if constexpr (!swz) { + load_ldmatrix_trans(t, tile_base + i0*stride + j0, stride); + return; + } +#if defined(TURING_MMA_AVAILABLE) + static_assert(I == 16, "bad tile width"); + static_assert(dl == DATA_LAYOUT_I_MAJOR, "bad data layout"); + const int i = i0 + threadIdx.x % t.I; + const int j = j0 + (threadIdx.x / t.I) * (t.J / 2); + int * xi = (int *) t.x; + asm volatile("ldmatrix.sync.aligned.m8n8.x4.trans.b16 {%0, %1, %2, %3}, [%4];" + : "=r"(xi[0]), "=r"(xi[2]), "=r"(xi[1]), "=r"(xi[3]) + : "l"(swizzle(tile_base, i, j, stride))); +#else + GGML_UNUSED_VARS(t, tile_base, i0, j0, stride); + NO_DEVICE_CODE; +#endif // defined(TURING_MMA_AVAILABLE) + } + static __device__ __forceinline__ void mma( tile<16, 8, int> & D, const tile<16, 4, int> & A, const tile<8, 4, int> & B) { #ifdef TURING_MMA_AVAILABLE diff --git a/ggml/src/ggml-cuda/mmf.cuh b/ggml/src/ggml-cuda/mmf.cuh index d55cc1ec..879a8652 100644 --- a/ggml/src/ggml-cuda/mmf.cuh +++ b/ggml/src/ggml-cuda/mmf.cuh @@ -143,6 +143,7 @@ static __global__ void mul_mat_f( if (threadIdx.x == 0) { slot_map[j] = -1; } + ggml_cuda_syncwarp(); if (col_base + j >= ncols_dst_total) { continue; @@ -171,10 +172,12 @@ static __global__ void mul_mat_f( tile_A A[ntA][warp_size / tile_A::J]; #pragma unroll for (int itA = 0; itA < ntA; ++itA) { + ggml_cuda_syncwarp(); #pragma unroll for (int i = 0; i < tile_A::I; ++i) { tile_xy[i*tile_k_padded + threadIdx.x] = x[(itA*tile_A::I + i)*stride_row + col]; } + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < warp_size; k0 += tile_A::J) { load_ldmatrix(A[itA][k0/tile_A::J], tile_xy + k0, tile_k_padded); @@ -183,6 +186,7 @@ static __global__ void mul_mat_f( #pragma unroll for (int itB = 0; itB < ntB; ++itB) { + ggml_cuda_syncwarp(); if constexpr (std::is_same_v) { #pragma unroll for (int j0 = 0; j0 < tile_B::I; ++j0) { @@ -212,6 +216,7 @@ static __global__ void mul_mat_f( } else { static_assert(std::is_same_v, "unsupported type"); } + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < warp_size; k0 += tile_B::J) { tile_B B; @@ -229,6 +234,8 @@ static __global__ void mul_mat_f( if (nwarps > 1) { __syncthreads(); + } else { + ggml_cuda_syncwarp(); } #pragma unroll for (int itB = 0; itB < ntB; ++itB) { @@ -245,6 +252,8 @@ static __global__ void mul_mat_f( if (nwarps > 1) { __syncthreads(); + } else { + ggml_cuda_syncwarp(); } #pragma unroll @@ -382,10 +391,12 @@ static __global__ void mul_mat_f_ids( tile_A A[ntA][warp_size / tile_A::J]; #pragma unroll for (int itA = 0; itA < ntA; ++itA) { + ggml_cuda_syncwarp(); #pragma unroll for (int i = 0; i < tile_A::I; ++i) { tile_xy[i*tile_k_padded + threadIdx.x] = x[(itA*tile_A::I + i)*stride_row + col]; } + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < warp_size; k0 += tile_A::J) { load_ldmatrix(A[itA][k0/tile_A::J], tile_xy + k0, tile_k_padded); @@ -419,6 +430,7 @@ static __global__ void mul_mat_f_ids( int next_buf = 1; #pragma unroll for (int itB = 0; itB < ntB; ++itB) { + ggml_cuda_syncwarp(); #pragma unroll for (int j0 = 0; j0 < tile_B::I; ++j0) { tile_xy[j0*tile_k_padded + threadIdx.x] = vals_buf[curr_buf][j0]; @@ -428,6 +440,7 @@ static __global__ void mul_mat_f_ids( gather_tile(itB + 1, vals_buf[next_buf]); } + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < warp_size; k0 += tile_B::J) { tile_B B; @@ -472,6 +485,7 @@ static __global__ void mul_mat_f_ids( int next_buf = 1; #pragma unroll for (int itB = 0; itB < ntB; ++itB) { + ggml_cuda_syncwarp(); #pragma unroll for (int j0 = 0; j0 < tile_B::I; ++j0) { const float2 tmp = vals_buf[curr_buf][j0]; @@ -482,6 +496,7 @@ static __global__ void mul_mat_f_ids( gather_tile(itB + 1, vals_buf[next_buf]); } + ggml_cuda_syncwarp(); #pragma unroll for (int k0 = 0; k0 < warp_size; k0 += tile_B::J) { tile_B B; @@ -507,6 +522,8 @@ static __global__ void mul_mat_f_ids( if (nwarps > 1) { __syncthreads(); + } else { + ggml_cuda_syncwarp(); } #pragma unroll for (int itB = 0; itB < ntB; ++itB) { @@ -523,6 +540,8 @@ static __global__ void mul_mat_f_ids( if (nwarps > 1) { __syncthreads(); + } else { + ggml_cuda_syncwarp(); } #pragma unroll diff --git a/ggml/src/ggml-cuda/mmid.cu b/ggml/src/ggml-cuda/mmid.cu index f80442fb..0b222e63 100644 --- a/ggml/src/ggml-cuda/mmid.cu +++ b/ggml/src/ggml-cuda/mmid.cu @@ -19,6 +19,11 @@ struct mm_ids_helper_store { }; static_assert(sizeof(mm_ids_helper_store) == 4, "unexpected size for mm_ids_helper_store"); +// the generic path passes 0, which needs no padding since it never groups lanes by token +template struct mm_ids_pow2 { static constexpr int value = 2*mm_ids_pow2<(n + 1)/2>::value; }; +template <> struct mm_ids_pow2<1> { static constexpr int value = 1; }; +template <> struct mm_ids_pow2<0> { static constexpr int value = 1; }; + // Helper function for mul_mat_id, converts ids to a more convenient format. // ids_src1 describes how to permute the flattened column indices of src1 in order to get a compact src1 tensor sorted by expert. // ids_dst describes the same mapping but for the dst tensor. @@ -32,6 +37,9 @@ static __global__ void mm_ids_helper( const int n_expert_used = n_expert_used_template == 0 ? n_expert_used_var : n_expert_used_template; const int expert = blockIdx.x; + // token slots per warp lane group, padded to a power of 2 so a warp divides evenly + constexpr int neu_padded = mm_ids_pow2::value; + extern __shared__ char data_mm_ids_helper[]; mm_ids_helper_store * store = (mm_ids_helper_store *) data_mm_ids_helper; @@ -60,8 +68,8 @@ static __global__ void mm_ids_helper( } } else { // Implementation optimized for specific numbers of experts used: - static_assert(n_expert_used == 6 || warp_size % n_expert_used == 0, "bad n_expert_used"); - const int neu_padded = n_expert_used == 6 ? 8 : n_expert_used; // Padded to next higher power of 2. + // a warp holds a whole number of token slots, so the slot count is padded to a power of 2 + static_assert(neu_padded <= warp_size && warp_size % neu_padded == 0, "bad n_expert_used"); for (int it0 = 0; it0 < n_tokens; it0 += warp_size/neu_padded) { const int it = it0 + threadIdx.x / neu_padded; @@ -93,6 +101,7 @@ static __global__ void mm_ids_helper( } } nex_prev = warp_reduce_sum(nex_prev); + ggml_cuda_syncwarp(); for (int itc = threadIdx.x; itc < it_compact; itc += warp_size) { const mm_ids_helper_store store_it = store[itc]; @@ -156,6 +165,9 @@ void ggml_cuda_launch_mm_ids_helper( case 8: launch_mm_ids_helper< 8>(ids, ids_src1, ids_dst, expert_bounds, n_experts, n_tokens, n_expert_used, nchannels_y, si1, sis1, write_inverse, stream); break; + case 10: + launch_mm_ids_helper<10>(ids, ids_src1, ids_dst, expert_bounds, n_experts, n_tokens, n_expert_used, nchannels_y, si1, sis1, write_inverse, stream); + break; case 16: launch_mm_ids_helper<16>(ids, ids_src1, ids_dst, expert_bounds, n_experts, n_tokens, n_expert_used, nchannels_y, si1, sis1, write_inverse, stream); break; diff --git a/ggml/src/ggml-cuda/mmq-config-gcn.cuh b/ggml/src/ggml-cuda/mmq-config-gcn.cuh new file mode 100644 index 00000000..24af2ef2 --- /dev/null +++ b/ggml/src/ggml-cuda/mmq-config-gcn.cuh @@ -0,0 +1,281 @@ +static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_gcn(ggml_type type, int J, bool fallback) { + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 64, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 512, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_Q2_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 512, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 3, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 2, 128, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 512, 2, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 512, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 128, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 128, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + + return ggml_cuda_mmq_config(GGML_TYPE_COUNT, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, 256, false, true); +} diff --git a/ggml/src/ggml-cuda/mmq-config-pascal.cuh b/ggml/src/ggml-cuda/mmq-config-pascal-dp4a.cuh similarity index 99% rename from ggml/src/ggml-cuda/mmq-config-pascal.cuh rename to ggml/src/ggml-cuda/mmq-config-pascal-dp4a.cuh index e7d4a9a3..83eb7c14 100644 --- a/ggml/src/ggml-cuda/mmq-config-pascal.cuh +++ b/ggml/src/ggml-cuda/mmq-config-pascal-dp4a.cuh @@ -1,4 +1,4 @@ -static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_pascal(ggml_type type, int J, bool fallback) { +static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_pascal_dp4a(ggml_type type, int J, bool fallback) { CASE(GGML_TYPE_Q1_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q1_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q1_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); diff --git a/ggml/src/ggml-cuda/mmq-config-pascal-older.cuh b/ggml/src/ggml-cuda/mmq-config-pascal-older.cuh new file mode 100644 index 00000000..2a8dc9e1 --- /dev/null +++ b/ggml/src/ggml-cuda/mmq-config-pascal-older.cuh @@ -0,0 +1,273 @@ +static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_pascal_older(ggml_type type, int J, bool fallback) { + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 256, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + +// --------------------------------------------------------------------------------------------- + + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 8, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 24, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 40, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 256, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + + return ggml_cuda_mmq_config(GGML_TYPE_COUNT, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, 256, false, true); +} diff --git a/ggml/src/ggml-cuda/mmq-config-rdna3.cuh b/ggml/src/ggml-cuda/mmq-config-rdna3.cuh index 676f27fe..3a3ef7bd 100644 --- a/ggml/src/ggml-cuda/mmq-config-rdna3.cuh +++ b/ggml/src/ggml-cuda/mmq-config-rdna3.cuh @@ -1,289 +1,273 @@ static constexpr __host__ __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config_rdna3(ggml_type type, int J, bool fallback) { CASE(GGML_TYPE_Q1_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q1_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q1_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q1_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q1_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q1_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q1_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q1_0, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q1_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q1_0, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q1_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q1_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q1_0, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q1_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q2_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q2_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q2_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q2_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q2_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_0, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_0, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_0, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q2_0, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q2_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q4_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q4_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 128, 4, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_0, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q4_1, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_1, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_1, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_1, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_1, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_1, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_1, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q5_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 4, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_0, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q5_1, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_1, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_1, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_1, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_1, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_1, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 4, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_1, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q8_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q8_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q8_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q8_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q8_0, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q8_0, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q8_0, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); // --------------------------------------------------------------------------------------------- CASE(GGML_TYPE_Q2_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q2_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q2_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q2_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q2_K, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 1, 128, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q2_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q2_K, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q2_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q2_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q3_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q3_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q3_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q3_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q3_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q3_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q3_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q3_K, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q4_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q4_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q4_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q4_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q4_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q4_K, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q5_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q5_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q5_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q5_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q5_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 4, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 4, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q5_K, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q6_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q6_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_Q6_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_Q6_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_Q6_K, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 128, 4, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_Q6_K, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_Q6_K, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_Q6_K, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q6_K, MMQ_ITER_K, false, false); // --------------------------------------------------------------------------------------------- CASE(GGML_TYPE_IQ1_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ1_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ1_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ1_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 128, 4, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 128, 4, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 128, 1, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ1_S, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XXS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 128, 4, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 128, 4, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XXS, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_XS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 128, 4, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 128, 4, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_XS, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ2_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 256, 1, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 128, 4, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ2_S, 128, 4, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q3_K, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_XXS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 128, 4, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 128, 4, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_XXS, 128, 1, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ3_S, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 128, 4, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 128, 4, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ3_S, 128, 2, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 128, 4, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_XS, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_XS, 128, 2, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, true); CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_IQ4_NL, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 128, 1, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_IQ4_NL, 128, 2, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, MMQ_ITER_K, false, false); // --------------------------------------------------------------------------------------------- CASE(GGML_TYPE_MXFP4, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_MXFP4, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_MXFP4, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); CASE(GGML_TYPE_MXFP4, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_MXFP4, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_MXFP4, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 1, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 1, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 1, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 2, 64, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_MXFP4, 128, 2, 64, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_1, MMQ_ITER_K, false, false); CASE(GGML_TYPE_NVFP4, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); CASE(GGML_TYPE_NVFP4, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); - CASE(GGML_TYPE_NVFP4, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); + CASE(GGML_TYPE_NVFP4, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); CASE(GGML_TYPE_NVFP4, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, true); CASE(GGML_TYPE_NVFP4, 128, 2, 64, 16, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); CASE(GGML_TYPE_NVFP4, 128, 2, 64, 32, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_NVFP4, 256, 2, 128, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_NVFP4, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_NVFP4, 256, 2, 128, 80, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 128, 2, 64, 48, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); + CASE(GGML_TYPE_NVFP4, 128, 2, 64, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); CASE(GGML_TYPE_NVFP4, 256, 2, 128, 96, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); - CASE(GGML_TYPE_NVFP4, 256, 2, 128, 112, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); CASE(GGML_TYPE_NVFP4, 256, 2, 128, 128, GGML_CUDA_MMQ_SRAM_LAYOUT_NVFP4, MMQ_ITER_K, false, false); return ggml_cuda_mmq_config(GGML_TYPE_COUNT, 256, 2, 128, 64, GGML_CUDA_MMQ_SRAM_LAYOUT_Q8_0, 256, false, true); diff --git a/ggml/src/ggml-cuda/mmq-load-tiles.cuh b/ggml/src/ggml-cuda/mmq-load-tiles.cuh index 8ed704c2..7f00bad9 100644 --- a/ggml/src/ggml-cuda/mmq-load-tiles.cuh +++ b/ggml/src/ggml-cuda/mmq-load-tiles.cuh @@ -138,12 +138,20 @@ template static __device__ __forceinline_ for (int j = 0; j < 4; ++j) { const int q = qxi[j]; +#if defined(GGML_USE_HIP) + const uint32_t qx_indices = (q & 0x03) | ((q & 0x0C) << 6) | ((q & 0x30) << 12) | ((q & 0xC0) << 18); + const uint32_t qy_bits = q >> 8; + const uint32_t qy_indices = (qy_bits & 0x03) | ((qy_bits & 0x0C) << 6) | ((qy_bits & 0x30) << 12) | ((qy_bits & 0xC0) << 18); + const int qx = __builtin_amdgcn_perm(0x020100FF, 0x020100FF, qx_indices); + const int qy = __builtin_amdgcn_perm(0x020100FF, 0x020100FF, qy_indices); +#else // unpack even and odd crumbs into byte values const int qe = __byte_perm(0x020100FF, 0x020100FF, q >> 0); const int qo = __byte_perm(0x020100FF, 0x020100FF, q >> 2); // unshuffle values const int qx = __byte_perm(qe, qo, 0x5140); const int qy = __byte_perm(qe, qo, 0x7362); +#endif // defined(GGML_USE_HIP) #if defined(AMD_MFMA_AVAILABLE) || defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) x_qs[i*sram_stride + dst_offset + j*2+0] = qx; diff --git a/ggml/src/ggml-cuda/mmq-vec-dot.cuh b/ggml/src/ggml-cuda/mmq-vec-dot.cuh index d5734338..4d1c398f 100644 --- a/ggml/src/ggml-cuda/mmq-vec-dot.cuh +++ b/ggml/src/ggml-cuda/mmq-vec-dot.cuh @@ -148,7 +148,6 @@ static __device__ __forceinline__ void ggml_cuda_mmq_vec_dot_q8_0_q8_1_mma( typedef tile<16, 8, int, input_layout> tile_B; typedef tile<16, 16, int, DATA_LAYOUT_J_MAJOR> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -204,7 +203,6 @@ static __device__ __forceinline__ void ggml_cuda_mmq_vec_dot_q8_0_q8_1_mma( typedef tile< 8, 8, int> tile_B; typedef tile<16, 8, int> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -320,7 +318,6 @@ template static __device__ __forceinline_ typedef tile<16, 8, int, input_layout> tile_B; typedef tile<16, 16, int, DATA_LAYOUT_J_MAJOR> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -371,7 +368,6 @@ template static __device__ __forceinline_ typedef tile< 8, 8, int> tile_B; typedef tile<16, 8, int> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -486,7 +482,6 @@ template static __device__ __forceinline_ typedef tile<16, 4, int, input_layout> tile_B; typedef tile<16, 16, int, DATA_LAYOUT_J_MAJOR> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -537,7 +532,6 @@ template static __device__ __forceinline_ typedef tile< 8, 4, int> tile_B; typedef tile<16, 8, int> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -686,7 +680,6 @@ template static __device__ __forceinline_ typedef tile<16, 4, int, input_layout> tile_B; typedef tile<16, 16, int, DATA_LAYOUT_J_MAJOR> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -756,7 +749,6 @@ template static __device__ __forceinline_ typedef tile< 8, 4, int> tile_B; typedef tile<16, 8, int> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -1023,7 +1015,6 @@ template static __device__ __forceinline_ typedef tile<16, 4, int, input_layout> tile_B; typedef tile<16, 16, int, DATA_LAYOUT_J_MAJOR> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -1075,7 +1066,6 @@ template static __device__ __forceinline_ typedef tile< 8, 4, int> tile_B; typedef tile<16, 8, int> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -1190,7 +1180,6 @@ template static __device__ __forceinline_ typedef tile<8, 8, int> tile_B; typedef tile<16, 8, float> tile_C; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int sram_stride = ggml_cuda_mmq_get_sram_stride(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp / tile_C::I; diff --git a/ggml/src/ggml-cuda/mmq.cu b/ggml/src/ggml-cuda/mmq.cu index 707437ea..b13b34ee 100644 --- a/ggml/src/ggml-cuda/mmq.cu +++ b/ggml/src/ggml-cuda/mmq.cu @@ -171,7 +171,7 @@ void ggml_cuda_mul_mat_q( ne00, ne01, ne1, s01, ne11, s1, ne02, ne12, s02, s12, s2, ne03, ne13, s03, s13, s3, - ne1}; + ne1, ne1}; ggml_cuda_mul_mat_q_switch_type(ctx, args, stream); return; } @@ -244,6 +244,13 @@ void ggml_cuda_mul_mat_q( ne11 * ne10_padded * sizeof(block_q8_1) / (QK8_1 * sizeof(int)); const int64_t s13 = ne12*s12; + // Each expert only sees ne12*n_expert_used/ne02 tokens on average. + // On RDNA3 and RDNA4 it is faster to pick the tile size against this value instead of ne12. + int64_t ncols_opt = ne12; + if (GGML_CUDA_CC_IS_RDNA3(cc) || GGML_CUDA_CC_IS_RDNA4(cc)) { + ncols_opt = (ne12*n_expert_used + ne02 - 1) / ne02; + } + // Note that ne02 is used instead of ne12 because the number of y channels determines the z dimension of the CUDA grid. const mmq_args args = { src0_d, src0->type, (const int *) src1_q8_1.get(), ids_dst.get(), expert_bounds.get(), dst_d, @@ -251,7 +258,7 @@ void ggml_cuda_mul_mat_q( ne00, ne01, ne_get_rows, s01, ne_get_rows, s1, ne02, ne02, s02, s12, s2, ne03, ne13, s03, s13, s3, - ne12}; + ne12, ncols_opt}; ggml_cuda_mul_mat_q_switch_type(ctx, args, stream); } @@ -314,7 +321,9 @@ bool ggml_cuda_should_use_mmq(enum ggml_type type, int cc, int64_t ne11, int64_t } if (ggml_cuda_highest_compiled_arch(cc) < GGML_CUDA_CC_DP4A) { - return false; + // for MoE, mmq is faster even without native dp4a + // TODO: check if cards older than pascal might benefit from this as well + return cc >= GGML_CUDA_CC_PASCAL && n_experts > 0; } #ifdef GGML_CUDA_FORCE_MMQ @@ -373,10 +382,10 @@ bool ggml_cuda_should_use_mmq(enum ggml_type type, int cc, int64_t ne11, int64_t return true; } - // gfx900 (Vega 10) lacks native dp4a, loses to dequant + hipBLAS + // gfx900 (Vega 10), gfx909, and gfx90c lack native dp4a, losing to dequant + hipBLAS // for dense matrices; keep MMQ only for MoE, where the // hipBLAS path is much slower. - if (cc == GGML_CUDA_CC_VEGA) { + if (cc == GGML_CUDA_CC_VEGA || GGML_CUDA_CC_IS_GCN_APU(cc)) { return n_experts > 0; } diff --git a/ggml/src/ggml-cuda/mmq.cuh b/ggml/src/ggml-cuda/mmq.cuh index 2eb15fdf..6923f351 100644 --- a/ggml/src/ggml-cuda/mmq.cuh +++ b/ggml/src/ggml-cuda/mmq.cuh @@ -213,10 +213,12 @@ struct ggml_cuda_mmq_config { return ggml_cuda_mmq_config((type_), (nthreads_), (occupancy_), (I_), (J_), (sram_layout_), (K_vram_), (stream_k_), (fallback_)); \ } \ -#include "mmq-config-pascal.cuh" +#include "mmq-config-pascal-older.cuh" +#include "mmq-config-pascal-dp4a.cuh" #include "mmq-config-ampere.cuh" #include "mmq-config-blackwell.cuh" +#include "mmq-config-gcn.cuh" #include "mmq-config-cdna.cuh" #include "mmq-config-rdna2.cuh" #include "mmq-config-rdna3.cuh" @@ -227,6 +229,9 @@ struct ggml_cuda_mmq_config { static __host__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config(const ggml_type type, const int J, const bool fallback, const int cc) { if (GGML_CUDA_CC_IS_AMD(cc)) { + if (GGML_CUDA_CC_IS_GCN(cc)) { + return ggml_cuda_mmq_get_config_gcn(type, J, fallback); + } if (GGML_CUDA_CC_IS_CDNA(cc)) { return ggml_cuda_mmq_get_config_cdna(type, J, fallback); } @@ -247,12 +252,17 @@ static __host__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config(const ggml_type ty if (ggml_cuda_highest_compiled_arch(cc) >= GGML_CUDA_CC_VOLTA) { return ggml_cuda_mmq_get_config_ampere(type, J, fallback); } - return ggml_cuda_mmq_get_config_pascal(type, J, fallback); + if (ggml_cuda_highest_compiled_arch(cc) >= GGML_CUDA_CC_DP4A) { + return ggml_cuda_mmq_get_config_pascal_dp4a(type, J, fallback); + } + return ggml_cuda_mmq_get_config_pascal_older(type, J, fallback); } static constexpr __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config(ggml_type type, int J, bool fallback) { #ifdef GGML_USE_HIP -#ifdef CDNA +#ifdef GCN + return ggml_cuda_mmq_get_config_gcn(type, J, fallback); +#elif defined(CDNA) return ggml_cuda_mmq_get_config_cdna(type, J, fallback); #elif defined(RDNA4) return ggml_cuda_mmq_get_config_rdna4(type, J, fallback); @@ -268,8 +278,10 @@ static constexpr __device__ ggml_cuda_mmq_config ggml_cuda_mmq_get_config(ggml_t return ggml_cuda_mmq_get_config_blackwell(type, J, fallback); #elif __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA return ggml_cuda_mmq_get_config_ampere(type, J, fallback); +#elif __CUDA_ARCH__ >= GGML_CUDA_CC_DP4A + return ggml_cuda_mmq_get_config_pascal_dp4a(type, J, fallback); #else - return ggml_cuda_mmq_get_config_pascal(type, J, fallback); + return ggml_cuda_mmq_get_config_pascal_older(type, J, fallback); #endif // BLACKWELL_MMA_AVAILABLE #endif // GGML_USE_HIP GGML_UNUSED_VARS(type, J, fallback); @@ -475,9 +487,6 @@ static __device__ __forceinline__ void ggml_cuda_mmq_write_back_mma( typedef tile<16, 8, int> tile_C; #endif // defined(AMD_MFMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) - constexpr int warp_size = ggml_cuda_get_physical_warp_size(); - constexpr int nwarps = ggml_cuda_mmq_get_nthreads(type, J, fallback) / warp_size; - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); constexpr int rows_per_warp = ggml_cuda_mmq_get_rows_per_warp(type, J, fallback); constexpr int ntx = rows_per_warp/tile_C::I; // Number of x minitiles per warp. @@ -534,8 +543,6 @@ struct ggml_cuda_mmq_util_funcs { template static constexpr __device__ ggml_cuda_mmq_util_funcs ggml_cuda_mmq_get_util_funcs() { - constexpr int I = ggml_cuda_mmq_get_I(type, J, fallback); - if (!ggml_cuda_mmq_get_config(type, J, fallback).use_mma_data_layout()) { switch (type) { case GGML_TYPE_Q1_0: @@ -1375,6 +1382,7 @@ struct mmq_args { int64_t nchannels_x; int64_t nchannels_y; int64_t stride_channel_x; int64_t stride_channel_y; int64_t stride_channel_dst; int64_t nsamples_x; int64_t nsamples_y; int64_t stride_sample_x; int64_t stride_sample_y; int64_t stride_sample_dst; int64_t ncols_max; + int64_t ncols_opt; // value to optimize the tile size against, launch grid still uses ncols_max }; static size_t mmq_get_nbytes_shared(const ggml_cuda_mmq_config & config, const int cc) { @@ -1485,7 +1493,7 @@ void mul_mat_q_switch_J(ggml_backend_cuda_context & ctx, const mmq_args & args, continue; } - const int ntiles_x = (args.ncols_max + config.J - 1) / config.J; + const int ntiles_x = (args.ncols_opt + config.J - 1) / config.J; if (ntiles_x < ntiles_J_best) { J_best = J; diff --git a/ggml/src/ggml-cuda/mmvf.cu b/ggml/src/ggml-cuda/mmvf.cu index d7dbc8b9..bd5c5d42 100644 --- a/ggml/src/ggml-cuda/mmvf.cu +++ b/ggml/src/ggml-cuda/mmvf.cu @@ -56,6 +56,7 @@ static __global__ void mul_mat_vec_f( bool use_bias = false; bool use_gate_bias = false; ggml_glu_op glu_op = ggml_glu_op::GGML_GLU_OP_SWIGLU; + float glu_limit = 0.0f; const T * gate_x = nullptr; const float * x_bias = nullptr; const float * gate_bias = nullptr; @@ -65,6 +66,7 @@ static __global__ void mul_mat_vec_f( use_bias = fusion.x_bias != nullptr; use_gate_bias = fusion.gate_bias != nullptr; glu_op = fusion.glu_op; + glu_limit = fusion.glu_limit; if (use_gate) { gate_x = static_cast(fusion.gate); @@ -365,6 +367,9 @@ static __global__ void mul_mat_vec_f( value = ggml_cuda_op_swiglu_oai_single(gate_value, value); break; } + case GGML_GLU_OP_SWIGLU_CLAMP: + value = ggml_cuda_op_swiglu_clamp_single(gate_value, value, glu_limit); + break; default: break; } @@ -374,7 +379,7 @@ static __global__ void mul_mat_vec_f( dst[tid*stride_col_dst + row] = value; if constexpr (!has_fusion) { - GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, glu_op, gate_x, x_bias, gate_bias, sumf_gate); + GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, glu_op, glu_limit, gate_x, x_bias, gate_bias, sumf_gate); } } @@ -675,6 +680,7 @@ void ggml_cuda_mul_mat_vec_f(ggml_backend_cuda_context & ctx, const ggml_tensor fusion_local.gate_bias = fusion->gate_bias->data; } fusion_local.glu_op = fusion->glu_op; + fusion_local.glu_limit = fusion->glu_limit; } const int64_t s01 = src0->nb[1] / ts_src0; diff --git a/ggml/src/ggml-cuda/mmvq.cu b/ggml/src/ggml-cuda/mmvq.cu index c9992380..a8515536 100644 --- a/ggml/src/ggml-cuda/mmvq.cu +++ b/ggml/src/ggml-cuda/mmvq.cu @@ -6,6 +6,35 @@ #include #include +// only enabled on DGX Spark, where it is a gain on every type below. On the higher-bandwidth parts the kernel +// has little exposed latency left to hide and the extra requests cost more than they save. +// For perf data, see https://github.com/ggml-org/llama.cpp/pull/26705#issuecomment-5569335031 +#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ == GGML_CUDA_CC_DGX_SPARK +// returns true only for those quants that benefit from prefetch and false otherwise +static constexpr __host__ __device__ bool mmvq_should_prefetch(ggml_type type) { + switch (type) { + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q8_0: + case GGML_TYPE_MXFP4: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ4_NL: + case GGML_TYPE_IQ4_XS: + return true; + default: + return false; + } +} + +static __device__ __forceinline__ void mmvq_prefetch_l2(const void * p) { + asm volatile("prefetch.global.L2 [%0];" :: "l"(p)); +} +#endif + typedef float (*vec_dot_q_cuda_t)(const void * __restrict__ vbq, const block_q8_1 * __restrict__ bq8_1, const int & kbx, const int & iqs); static constexpr __device__ vec_dot_q_cuda_t get_vec_dot_q_cuda(ggml_type type) { @@ -290,6 +319,68 @@ bool ggml_cuda_should_use_mmvq(enum ggml_type type, int cc, int64_t ne11) { if (!ggml_is_quantized(type)) { return false; } + // k-quants cost more to decode and mvq redoes that per column, so MMQ wins sooner. + // Only list quant-types MMQ supports, others would fall back to cuBLAS. + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_ADA_LOVELACE) { + switch (type) { // tuned on RTX 4090 + case GGML_TYPE_Q2_K: + return ne11 <= 4; + case GGML_TYPE_Q3_K: + return ne11 <= 6; + default: + return ne11 <= MMVQ_MAX_BATCH_SIZE; + } + } + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_BLACKWELL) { + switch (type) { // tuned on RTX 5090 + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + return ne11 <= 5; + case GGML_TYPE_Q5_K: + return ne11 <= 6; + case GGML_TYPE_Q6_K: + return ne11 <= 7; + default: + return ne11 <= MMVQ_MAX_BATCH_SIZE; + } + } + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_DGX_SPARK) { + switch (type) { // tuned on DGX Spark GB10 + case GGML_TYPE_Q2_K: + return ne11 <= 6; + default: + return ne11 <= MMVQ_MAX_BATCH_SIZE; + } + } + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_ORIN) { + switch (type) { // tuned for Jetson Orin + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + return ne11 <= 1; + default: + return ne11 <= MMVQ_MAX_BATCH_SIZE; + } + } + if (GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_VOLTA) { + switch (type) { + case GGML_TYPE_Q2_K: + return ne11 <= 4; + case GGML_TYPE_Q3_K: + return ne11 <= 6; + case GGML_TYPE_Q4_K: + return ne11 <= 5; + case GGML_TYPE_Q5_K: + return ne11 <= 6; + case GGML_TYPE_Q6_K: + return ne11 <= 7; + default: + return ne11 <= MMVQ_MAX_BATCH_SIZE; + } + } if (GGML_CUDA_CC_IS_CDNA(cc)) { if (GGML_CUDA_CC_IS_CDNA1(cc)) { switch (type) { @@ -559,6 +650,7 @@ static __global__ void mul_mat_vec_q( const float * x_scale = nullptr; const float * gate_scale = nullptr; ggml_glu_op active_glu; + float glu_limit = 0.0f; if constexpr (has_fusion) { use_gate = fusion.gate != nullptr; @@ -568,6 +660,7 @@ static __global__ void mul_mat_vec_q( x_bias = (const float *) fusion.x_bias; gate_bias = (const float *) fusion.gate_bias; active_glu = fusion.glu_op; + glu_limit = fusion.glu_limit; if constexpr (type == GGML_TYPE_NVFP4) { use_scale = fusion.x_scale != nullptr; use_gate_scale = fusion.gate_scale != nullptr && use_gate; @@ -625,6 +718,26 @@ static __global__ void mul_mat_vec_q( // x block quant index when casting the quants to int const int kqs = vdr * (tid % (qi/vdr)); +#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ == GGML_CUDA_CC_DGX_SPARK + // start the next iterations' weight loads early + if constexpr (mmvq_should_prefetch(type)) { + constexpr int pf_dist = 2; // loop iterations, not blocks + const int kbx_pf = kbx + pf_dist*blocks_per_iter; + if (kbx_pf < blocks_per_row_x) { +#pragma unroll + for (int i = 0; i < rows_per_cuda_block; ++i) { + const size_t off = (size_t)(kbx_offset + i*stride_row_x + kbx_pf) * ggml_cuda_type_traits::bs; + mmvq_prefetch_l2((const char *) vx + off); + if constexpr (has_fusion) { + if (use_gate) { + mmvq_prefetch_l2((const char *) vgate + off); + } + } + } + } + } +#endif + #pragma unroll for (int j = 0; j < ncols_dst; ++j) { #pragma unroll @@ -709,6 +822,9 @@ static __global__ void mul_mat_vec_q( case GGML_GLU_OP_SWIGLU_OAI: result = ggml_cuda_op_swiglu_oai_single(gate_value, result); break; + case GGML_GLU_OP_SWIGLU_CLAMP: + result = ggml_cuda_op_swiglu_clamp_single(gate_value, result, glu_limit); + break; default: result = result * gate_value; break; @@ -721,7 +837,7 @@ static __global__ void mul_mat_vec_q( } if constexpr (!has_fusion) { - GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_scale, use_gate_scale, active_glu, gate_bias, x_bias, x_scale, gate_scale, tmp_gate); + GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_scale, use_gate_scale, active_glu, glu_limit, gate_bias, x_bias, x_scale, gate_scale, tmp_gate); } if constexpr (type != GGML_TYPE_NVFP4) { GGML_UNUSED_VARS(use_scale, use_gate_scale, x_scale, gate_scale, x_scales, gate_scales); @@ -732,10 +848,10 @@ static __global__ void mul_mat_vec_q( // Grid: (ceil(nrows_x / c_rows_per_block), nchannels_dst) // Block: (warp_size, ncols_dst) - each warp handles one token independently. // No shared memory reduction needed since each warp works alone. -template +template __launch_bounds__(get_mmvq_mmid_max_batch_for_device()*ggml_cuda_get_physical_warp_size(), 1) static __global__ void mul_mat_vec_q_moe( - const void * vx_ptr, const void * vy_ptr, const int32_t * ids_ptr, + const void * vx_ptr, const void * vy_ptr, const int32_t * ids_ptr, const ggml_cuda_mm_fusion_args_device fusion, float * dst_ptr, const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t nrows_x, const uint32_t stride_row_x, const uint32_t stride_col_y, const uint32_t stride_col_dst, @@ -753,6 +869,29 @@ static __global__ void mul_mat_vec_q_moe( constexpr vec_dot_q_cuda_t vec_dot_q_cuda = get_vec_dot_q_cuda(type); + // fuse gate, bias, scales, and glu_op into the up projection + bool use_gate = false; + const void * vgate = nullptr; + const float * x_bias = nullptr; + const float * gate_bias = nullptr; + const float * x_scale = nullptr; + const float * gate_scale = nullptr; + ggml_glu_op active_glu = GGML_GLU_OP_SWIGLU; + float glu_limit = 0.0f; + + if constexpr (has_fusion) { + use_gate = fusion.gate != nullptr; + vgate = fusion.gate; + x_bias = (const float *) fusion.x_bias; + gate_bias = (const float *) fusion.gate_bias; + active_glu = fusion.glu_op; + glu_limit = fusion.glu_limit; + if constexpr (type == GGML_TYPE_NVFP4) { + x_scale = (const float *) fusion.x_scale; + gate_scale = (const float *) fusion.gate_scale; + } + } + const uint32_t token_idx = threadIdx.y; const int row0 = c_rows_per_block*blockIdx.x; const int blocks_per_row_x = ncols_x / qk; @@ -773,6 +912,7 @@ static __global__ void mul_mat_vec_q_moe( // partial sum for each thread float tmp[c_rows_per_block] = {0.0f}; + float tmp_gate[c_rows_per_block] = {0.0f}; for (int kbx = threadIdx.x / (qi/vdr); kbx < blocks_per_row_x; kbx += blocks_per_iter) { const int kby = kbx * (qk/QK8_1); @@ -781,6 +921,11 @@ static __global__ void mul_mat_vec_q_moe( #pragma unroll for (int i = 0; i < c_rows_per_block; ++i) { tmp[i] += vec_dot_q_cuda(vx, &y[kby], kbx_offset + i*stride_row_x + kbx, kqs); + if constexpr (has_fusion) { + if (use_gate) { + tmp_gate[i] += vec_dot_q_cuda(vgate, &y[kby], kbx_offset + i*stride_row_x + kbx, kqs); + } + } } } @@ -790,11 +935,63 @@ static __global__ void mul_mat_vec_q_moe( #pragma unroll for (int i = 0; i < c_rows_per_block; ++i) { tmp[i] = warp_reduce_sum(tmp[i]); + if constexpr (has_fusion) { + if (use_gate) { + tmp_gate[i] = warp_reduce_sum(tmp_gate[i]); + } + } } // Write results if (threadIdx.x < c_rows_per_block && (c_rows_per_block == 1 || uint32_t(row0 + threadIdx.x) < nrows_x)) { - dst[channel_dst*stride_channel_dst + token_idx*stride_col_dst + row0 + threadIdx.x] = tmp[threadIdx.x]; + float result = tmp[threadIdx.x]; + if constexpr (has_fusion) { + const uint32_t bias_idx = channel_x*stride_channel_dst + row0 + threadIdx.x; + + if constexpr (type == GGML_TYPE_NVFP4) { + if (x_scale) { + result *= x_scale[channel_x]; + } + } + if (x_bias) { + result += x_bias[bias_idx]; + } + if (use_gate) { + float gate_value = tmp_gate[threadIdx.x]; + if constexpr (type == GGML_TYPE_NVFP4) { + if (gate_scale) { + gate_value *= gate_scale[channel_x]; + } + } + if (gate_bias) { + gate_value += gate_bias[bias_idx]; + } + switch (active_glu) { + case GGML_GLU_OP_SWIGLU: + result *= ggml_cuda_op_silu_single(gate_value); + break; + case GGML_GLU_OP_GEGLU: + result *= ggml_cuda_op_gelu_single(gate_value); + break; + case GGML_GLU_OP_SWIGLU_OAI: + result = ggml_cuda_op_swiglu_oai_single(gate_value, result); + break; + case GGML_GLU_OP_SWIGLU_CLAMP: + result = ggml_cuda_op_swiglu_clamp_single(gate_value, result, glu_limit); + break; + default: + result = result * gate_value; + break; + } + } + } + dst[channel_dst*stride_channel_dst + token_idx*stride_col_dst + row0 + threadIdx.x] = result; + } + + if constexpr (!has_fusion) { + GGML_UNUSED_VARS(use_gate, tmp_gate, vgate, x_bias, gate_bias, active_glu, glu_limit, x_scale, gate_scale); + } else if constexpr (type != GGML_TYPE_NVFP4) { + GGML_UNUSED_VARS(x_scale, gate_scale); } } @@ -844,7 +1041,7 @@ static void mul_mat_vec_q_switch_fusion( template static void mul_mat_vec_q_moe_launch( - const void * vx, const void * vy, const int32_t * ids, float * dst, + const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst, const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t nrows_x, const uint32_t stride_row_x, const uint32_t stride_col_y, const uint32_t stride_col_dst, const uint32_t stride_channel_x, const uint32_t stride_channel_y, const uint32_t stride_channel_dst, @@ -857,11 +1054,22 @@ static void mul_mat_vec_q_moe_launch( const dim3 block_dims(warp_size, ncols_dst); const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); - ggml_cuda_kernel_launch(mul_mat_vec_q_moe, launch_params, - vx, vy, ids, dst, ncols_x, nchannels_y, nrows_x, - stride_row_x, stride_col_y, stride_col_dst, - stride_channel_x, stride_channel_y, stride_channel_dst, - ncols_dst, ids_stride); + const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr || + fusion.x_scale != nullptr || fusion.gate_scale != nullptr; + + if (has_fusion) { + ggml_cuda_kernel_launch(mul_mat_vec_q_moe, launch_params, + vx, vy, ids, fusion, dst, ncols_x, nchannels_y, nrows_x, + stride_row_x, stride_col_y, stride_col_dst, + stride_channel_x, stride_channel_y, stride_channel_dst, + ncols_dst, ids_stride); + } else { + ggml_cuda_kernel_launch(mul_mat_vec_q_moe, launch_params, + vx, vy, ids, fusion, dst, ncols_x, nchannels_y, nrows_x, + stride_row_x, stride_col_y, stride_col_dst, + stride_channel_x, stride_channel_y, stride_channel_dst, + ncols_dst, ids_stride); + } } template @@ -957,7 +1165,7 @@ static void mul_mat_vec_q_switch_ncols_dst( if (has_ids && ncols_dst > 1) { // Multi-token MUL_MAT_ID path - dedicated MoE kernel mul_mat_vec_q_moe_launch( - vx, vy, ids, dst, ncols_x, nchannels_y_fd, nrows_x, + vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, nrows_x, stride_row_x, stride_col_y, stride_col_dst, stride_channel_x, stride_channel_y, stride_channel_dst, ncols_dst, ids_stride, warp_size, nchannels_dst, stream); @@ -1239,7 +1447,8 @@ void ggml_cuda_mul_mat_vec_q( ggml_cuda_mm_fusion_args_device fusion_local{}; if (fusion) { - GGML_ASSERT( !ids || dst->ne[2] == 1); + const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; + GGML_ASSERT( !ids || dst->ne[2] <= get_mmvq_mmid_max_batch(src0->type, cc)); GGML_ASSERT( ids || dst->ne[1] == 1); // Scale fusion is only allowed for NVFP4 currently as the cost of checking this at run-time in the prologue is // non-negligible for some models such as gpt-oss-20b @@ -1274,6 +1483,7 @@ void ggml_cuda_mul_mat_vec_q( fusion_local.gate_scale = fusion->gate_scale->data; } fusion_local.glu_op = fusion->glu_op; + fusion_local.glu_limit = fusion->glu_limit; } // If src0 is a temporary compute buffer, clear any potential padding. diff --git a/ggml/src/ggml-cuda/moe-weighted-reduction.cu b/ggml/src/ggml-cuda/moe-weighted-reduction.cu new file mode 100644 index 00000000..11ec5849 --- /dev/null +++ b/ggml/src/ggml-cuda/moe-weighted-reduction.cu @@ -0,0 +1,65 @@ +#include "moe-weighted-reduction.cuh" + +static __global__ void moe_weighted_reduction_f32(const float * __restrict__ experts, + const float * __restrict__ expert_scale, + const float * __restrict__ weights, + float * __restrict__ dst, + const int64_t n_embd, + const int n_expert_used) { + const int64_t token = blockIdx.x; + const int64_t col = (int64_t) blockIdx.y * blockDim.x + threadIdx.x; + if (col >= n_embd) { + return; + } + + const uint64_t first_row = (uint64_t) token * n_expert_used; + const float first_scale = expert_scale != nullptr ? expert_scale[first_row] : 1.0f; + float sum = (experts[first_row * n_embd + col] * first_scale) * weights[first_row]; + + for (int expert = 1; expert < n_expert_used; ++expert) { + const uint64_t row = first_row + expert; + const float scale = expert_scale != nullptr ? expert_scale[row] : 1.0f; + sum += (experts[row * n_embd + col] * scale) * weights[row]; + } + dst[token * n_embd + col] = sum; +} + +static void launch_moe_weighted_reduction(const float * experts, + const float * expert_scale, + const float * weights, + float * dst, + int64_t n_embd, + int64_t n_tokens, + int n_expert_used, + cudaStream_t stream) { + constexpr int threads = 256; + const dim3 blocks(n_tokens, (n_embd + threads - 1) / threads, 1); + moe_weighted_reduction_f32 + <<>>(experts, expert_scale, weights, dst, n_embd, n_expert_used); +} + +void ggml_cuda_op_moe_weighted_reduction(ggml_backend_cuda_context & ctx, + const ggml_tensor * experts, + const ggml_tensor * expert_scale, + const ggml_tensor * weights, + ggml_tensor * dst) { + GGML_ASSERT(experts->type == GGML_TYPE_F32); + GGML_ASSERT(weights->type == GGML_TYPE_F32); + GGML_ASSERT(expert_scale == nullptr || expert_scale->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(ggml_is_contiguous(experts)); + GGML_ASSERT(ggml_is_contiguous(weights)); + GGML_ASSERT(expert_scale == nullptr || ggml_is_contiguous(expert_scale)); + GGML_ASSERT(ggml_is_contiguous(dst)); + + const int64_t n_embd = experts->ne[0]; + const int64_t n_expert_used = experts->ne[1]; + const int64_t n_tokens = experts->ne[2] * experts->ne[3]; + cudaStream_t stream = ctx.stream(); + + launch_moe_weighted_reduction((const float *) experts->data, + expert_scale ? (const float *) expert_scale->data : nullptr, + (const float *) weights->data, + (float *) dst->data, n_embd, n_tokens, (int) n_expert_used, stream); + CUDA_CHECK(cudaGetLastError()); +} diff --git a/ggml/src/ggml-cuda/moe-weighted-reduction.cuh b/ggml/src/ggml-cuda/moe-weighted-reduction.cuh new file mode 100644 index 00000000..b72f947a --- /dev/null +++ b/ggml/src/ggml-cuda/moe-weighted-reduction.cuh @@ -0,0 +1,7 @@ +#include "common.cuh" + +void ggml_cuda_op_moe_weighted_reduction(ggml_backend_cuda_context & ctx, + const ggml_tensor * experts, + const ggml_tensor * expert_scale, + const ggml_tensor * weights, + ggml_tensor * dst); diff --git a/ggml/src/ggml-cuda/out-prod.cu b/ggml/src/ggml-cuda/out-prod.cu index 46b9f3a6..c46e0455 100644 --- a/ggml/src/ggml-cuda/out-prod.cu +++ b/ggml/src/ggml-cuda/out-prod.cu @@ -54,8 +54,6 @@ void ggml_cuda_out_prod(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const float alpha = 1.0f; const float beta = 0.0f; - CUBLAS_CHECK(cublasSetStream(handle, stream)); - const int64_t lda = nb01 / sizeof(float); const int64_t ldc = nb1 / sizeof(float); diff --git a/ggml/src/ggml-cuda/pool1d.cu b/ggml/src/ggml-cuda/pool1d.cu new file mode 100644 index 00000000..ac6fb0cb --- /dev/null +++ b/ggml/src/ggml-cuda/pool1d.cu @@ -0,0 +1,85 @@ +#include "pool1d.cuh" + +static __global__ void pool1d_nchw_kernel( + const int iw, const int ow, + const int kw, const int sw, const int pw, + const int parallel_elements, + const float * src, float * dst, const enum ggml_op_pool op) { + const int idx = threadIdx.x + blockIdx.x * blockDim.x; + if (idx >= parallel_elements) { + return; + } + + const int nc = idx / ow; + const int cur_ow = idx % ow; + + const float * i_ptr = src + nc * iw; + float * o_ptr = dst + nc * ow; + + const int start = cur_ow * sw - pw; + const int b = max(0, start); + const int e = min(iw, start + kw); + + float res; + switch (op) { + case GGML_OP_POOL_AVG: res = 0.0f; break; + case GGML_OP_POOL_MAX: res = -FLT_MAX; break; + default: return; + } + + int count = 0; + for (int i = b; i < e; i++) { +#if __CUDA_ARCH__ >= 350 + float cur = __ldg(i_ptr + i); +#else + float cur = i_ptr[i]; +#endif + switch (op) { + case GGML_OP_POOL_AVG: res += cur; break; + case GGML_OP_POOL_MAX: res = max(res, cur); break; + default: break; + } + count++; + } + + if (op == GGML_OP_POOL_AVG) { + res = (count > 0) ? (res / count) : 0.0f; + } + + o_ptr[cur_ow] = res; +} + +static void pool1d_nchw_kernel_f32_f32_cuda( + const int iw, const int ow, + const int kw, const int sw, const int pw, + const int parallel_elements, + const float * src, float * dst, const enum ggml_op_pool op, + cudaStream_t stream) { + const int num_blocks = (parallel_elements + CUDA_POOL1D_BLOCK_SIZE - 1) / CUDA_POOL1D_BLOCK_SIZE; + dim3 block_nums(num_blocks); + pool1d_nchw_kernel<<>>(iw, ow, kw, sw, pw, parallel_elements, src, dst, op); +} + +void ggml_cuda_op_pool1d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const float * src0_d = (const float *)src0->data; + float * dst_d = (float *)dst->data; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT( dst->type == GGML_TYPE_F32); + + const int32_t * opts = (const int32_t *)dst->op_params; + enum ggml_op_pool op = static_cast(opts[0]); + const int k0 = opts[1]; + const int s0 = opts[2]; + const int p0 = opts[3]; + + const int64_t IW = src0->ne[0]; + const int64_t OW = dst->ne[0]; + const int64_t nr = ggml_nrows(src0); + + const int parallel_elements = (int)(nr * OW); + + pool1d_nchw_kernel_f32_f32_cuda(IW, OW, k0, s0, p0, parallel_elements, src0_d, dst_d, op, stream); +} diff --git a/ggml/src/ggml-cuda/pool1d.cuh b/ggml/src/ggml-cuda/pool1d.cuh new file mode 100644 index 00000000..c79461dd --- /dev/null +++ b/ggml/src/ggml-cuda/pool1d.cuh @@ -0,0 +1,5 @@ +#include "common.cuh" + +#define CUDA_POOL1D_BLOCK_SIZE 256 + +void ggml_cuda_op_pool1d(ggml_backend_cuda_context & ctx, ggml_tensor * dst); diff --git a/ggml/src/ggml-cuda/reduce_rows.cuh b/ggml/src/ggml-cuda/reduce_rows.cuh index 968c47aa..111fd838 100644 --- a/ggml/src/ggml-cuda/reduce_rows.cuh +++ b/ggml/src/ggml-cuda/reduce_rows.cuh @@ -1,11 +1,6 @@ #include "common.cuh" -// Row reduction kernel template - compute sum (norm=false) or mean (norm=true) -template -static __global__ void reduce_rows_f32(const float * x_ptr, float * dst_ptr, const int ncols) { - const float * GGML_CUDA_RESTRICT x = x_ptr; - float * GGML_CUDA_RESTRICT dst = dst_ptr; - const int row = blockIdx.x; +static __device__ __forceinline__ float reduce_row_f32(const float * x, const int ncols) { const int col = threadIdx.x; float sum = 0.0f; @@ -17,7 +12,7 @@ static __global__ void reduce_rows_f32(const float * x_ptr, float * dst_ptr, con for (int i = col; i < ncols;) { for (int j = 0; j < num_unroll; ++j) { if (i < ncols) { - temp[j] = x[row * ncols + i]; + temp[j] = x[i]; } else { temp[j] = 0; } @@ -35,6 +30,40 @@ static __global__ void reduce_rows_f32(const float * x_ptr, float * dst_ptr, con __shared__ float shared_vals[32]; sum = block_reduce(sum, shared_vals); + return sum; +} + +// Row reduction kernel template - compute sum (norm=false) or mean (norm=true) +template +static __global__ void reduce_rows_f32(const float * x_ptr, float * dst_ptr, const int ncols) { + float * GGML_CUDA_RESTRICT dst = dst_ptr; + const int64_t row = blockIdx.x; + const int col = threadIdx.x; + + const float * GGML_CUDA_RESTRICT x = x_ptr + row*ncols; + const float sum = reduce_row_f32(x, ncols); + + if (col != 0) { + return; + } + + dst[row] = norm ? sum / ncols : sum; +} + +template +static __global__ void reduce_rows_f32_strided(const char * x_ptr, float * dst_ptr, const int ncols, + const int64_t ne1, const int64_t ne2, const int64_t nb1, const int64_t nb2, const int64_t nb3) { + float * GGML_CUDA_RESTRICT dst = dst_ptr; + const int64_t row = blockIdx.x; + const int col = threadIdx.x; + + const int64_t i1 = row % ne1; + const int64_t i2 = (row / ne1) % ne2; + const int64_t i3 = row / (ne1 * ne2); + + const float * GGML_CUDA_RESTRICT x = (const float *) (x_ptr + i1*nb1 + i2*nb2 + i3*nb3); + const float sum = reduce_row_f32(x, ncols); + if (col != 0) { return; } diff --git a/ggml/src/ggml-cuda/rope.cu b/ggml/src/ggml-cuda/rope.cu index 504c6b81..e546fb65 100644 --- a/ggml/src/ggml-cuda/rope.cu +++ b/ggml/src/ggml-cuda/rope.cu @@ -53,6 +53,7 @@ static __global__ void rope_norm(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int32_t * pos, const float freq_scale, const float ext_factor, @@ -61,7 +62,8 @@ static __global__ void rope_norm(const T * x, const float theta_scale, const float * freq_factors, const int64_t * row_indices, - const int set_rows_stride) { + const int set_rows_stride, + const bool inplace) { const int i0 = 2*(blockDim.y*blockIdx.y + threadIdx.y); if (i0 >= ne00) { @@ -92,19 +94,24 @@ static __global__ void rope_norm(const T * x, ggml_cuda_memcpy_1<4>(dst + idst, &v); } }; - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { + if (inplace) { + return; + } store_coaelsced(x[ix + 0], x[ix + 1]); return; } - const float theta_base = pos[i2]*powf(theta_scale, i0/2.0f); + const int iw = i0 - n_offs; // relative idx - const float freq_factor = has_ff ? freq_factors[i0/2] : 1.0f; + const float theta_base = pos[i2]*powf(theta_scale, iw/2.0f); + + const float freq_factor = has_ff ? freq_factors[iw/2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor, cos_theta, sin_theta); + rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); const float x0 = x[ix + 0]; const float x1 = x[ix + 1]; @@ -125,6 +132,7 @@ static __global__ void rope_neox(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int32_t * pos, const float freq_scale, const float ext_factor, @@ -133,7 +141,8 @@ static __global__ void rope_neox(const T * x, const float theta_scale, const float * freq_factors, const int64_t * row_indices, - const int set_rows_stride) { + const int set_rows_stride, + const bool inplace) { ggml_cuda_pdl_lc(); const int i0 = 2*(blockDim.y*blockIdx.y + threadIdx.y); @@ -158,27 +167,33 @@ static __global__ void rope_neox(const T * x, idst += row_indices[i2] * set_rows_stride; } - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { + if (inplace) { + return; + } dst[idst + i0 / 2 + 0] = ggml_cuda_cast(x[ix + i0 / 2 + 0]); dst[idst + i0 / 2 + 1] = ggml_cuda_cast(x[ix + i0 / 2 + 1]); return; } - const float theta_base = pos[i2]*powf(theta_scale, i0/2.0f); + const int iw = i0 - n_offs; // relative idx - const float freq_factor = has_ff ? freq_factors[i0/2] : 1.0f; + const float theta_base = pos[i2]*powf(theta_scale, iw/2.0f); + + const float freq_factor = has_ff ? freq_factors[iw/2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor, cos_theta, sin_theta); + rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); - const float x0 = x[ix + 0]; - const float x1 = x[ix + n_dims/2]; + // idst/ix point at channel i0/2; the first channel of the rotated pair is n_offs + iw/2 = i0/2 + n_offs/2 + const float x0 = x[ix + n_offs/2 + 0]; + const float x1 = x[ix + n_offs/2 + n_dims/2]; - dst[idst + 0] = ggml_cuda_cast(x0 * cos_theta - x1 * sin_theta); - dst[idst + n_dims / 2] = ggml_cuda_cast(x0 * sin_theta + x1 * cos_theta); + dst[idst + n_offs/2 + 0] = ggml_cuda_cast(x0 * cos_theta - x1 * sin_theta); + dst[idst + n_offs/2 + n_dims / 2] = ggml_cuda_cast(x0 * sin_theta + x1 * cos_theta); } template @@ -194,6 +209,7 @@ static __global__ void rope_multi(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int32_t * pos, const float freq_scale, const float ext_factor, @@ -202,7 +218,8 @@ static __global__ void rope_multi(const T * x, const float theta_scale, const float * freq_factors, const mrope_sections sections, - const bool is_imrope) { + const bool is_imrope, + const bool inplace) { const int i0 = 2 * (blockDim.y * blockIdx.y + threadIdx.y); if (i0 >= ne00) { @@ -219,52 +236,58 @@ static __global__ void rope_multi(const T * x, const int ix = i0 / 2 + i1 * s01 + i2 * s02 + i3 * s03; ggml_cuda_pdl_sync(); - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { + if (inplace) { + return; + } dst[idst + i0/2 + 0] = x[ix + i0/2 + 0]; dst[idst + i0/2 + 1] = x[ix + i0/2 + 1]; return; } + const int iw = i0 - n_offs; // relative idx + const int sect_dims = sections.v[0] + sections.v[1] + sections.v[2] + sections.v[3]; const int sec_w = sections.v[1] + sections.v[0]; - const int sector = (i0 / 2) % sect_dims; + const int sector = (iw / 2) % sect_dims; float theta_base = 0.0; if (is_imrope) { if (sector % 3 == 1 && sector < 3 * sections.v[1]) { // h - theta_base = pos[i2 + ne02 * 1] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 1] * powf(theta_scale, iw / 2.0f); } else if (sector % 3 == 2 && sector < 3 * sections.v[2]) { // w - theta_base = pos[i2 + ne02 * 2] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 2] * powf(theta_scale, iw / 2.0f); } else if (sector % 3 == 0 && sector < 3 * sections.v[0]) { // t - theta_base = pos[i2] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2] * powf(theta_scale, iw / 2.0f); } else { - theta_base = pos[i2 + ne02 * 3] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 3] * powf(theta_scale, iw / 2.0f); } } else { if (sector < sections.v[0]) { - theta_base = pos[i2] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2] * powf(theta_scale, iw / 2.0f); } else if (sector >= sections.v[0] && sector < sec_w) { - theta_base = pos[i2 + ne02 * 1] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 1] * powf(theta_scale, iw / 2.0f); } else if (sector >= sec_w && sector < sec_w + sections.v[2]) { - theta_base = pos[i2 + ne02 * 2] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 2] * powf(theta_scale, iw / 2.0f); } else if (sector >= sec_w + sections.v[2]) { - theta_base = pos[i2 + ne02 * 3] * powf(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 3] * powf(theta_scale, iw / 2.0f); } } - const float freq_factor = has_ff ? freq_factors[i0/2] : 1.0f; + const float freq_factor = has_ff ? freq_factors[iw/2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor, cos_theta, sin_theta); + rope_yarn(theta_base/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); - const float x0 = x[ix + 0]; - const float x1 = x[ix + n_dims/2]; + // idst/ix point at channel i0/2; the first channel of the rotated pair is n_offs + iw/2 = i0/2 + n_offs/2 + const float x0 = x[ix + n_offs/2 + 0]; + const float x1 = x[ix + n_offs/2 + n_dims/2]; - dst[idst + 0] = x0*cos_theta - x1*sin_theta; - dst[idst + n_dims/2] = x0*sin_theta + x1*cos_theta; + dst[idst + n_offs/2 + 0] = x0*cos_theta - x1*sin_theta; + dst[idst + n_offs/2 + n_dims/2] = x0*sin_theta + x1*cos_theta; } template @@ -344,6 +367,7 @@ static void rope_norm_cuda(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int nr, const int32_t * pos, const float freq_scale, @@ -354,6 +378,7 @@ static void rope_norm_cuda(const T * x, const float * freq_factors, const int64_t * row_indices, const int set_rows_stride, + const bool inplace, cudaStream_t stream) { GGML_ASSERT(ne00 % 2 == 0); const dim3 block_dims(1, CUDA_ROPE_BLOCK_SIZE, 1); @@ -364,12 +389,12 @@ static void rope_norm_cuda(const T * x, if (freq_factors == nullptr) { rope_norm<<>>( - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride, inplace); } else { rope_norm<<>>( - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride, inplace); } } @@ -386,6 +411,7 @@ static void rope_neox_cuda(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int nr, const int32_t * pos, const float freq_scale, @@ -396,6 +422,7 @@ static void rope_neox_cuda(const T * x, const float * freq_factors, const int64_t * row_indices, const int set_rows_stride, + const bool inplace, cudaStream_t stream) { GGML_ASSERT(ne00 % 2 == 0); const dim3 block_dims(1, CUDA_ROPE_BLOCK_SIZE, 1); @@ -407,12 +434,12 @@ static void rope_neox_cuda(const T * x, if (freq_factors == nullptr) { ggml_cuda_kernel_launch(rope_neox, launch_params, - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride, inplace); } else { ggml_cuda_kernel_launch(rope_neox, launch_params, - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride, inplace); } } @@ -429,6 +456,7 @@ static void rope_multi_cuda(const T * x, const int s2, const int s3, const int n_dims, + const int n_offs, const int nr, const int32_t * pos, const float freq_scale, @@ -439,6 +467,7 @@ static void rope_multi_cuda(const T * x, const float * freq_factors, const mrope_sections sections, const bool is_imrope, + const bool inplace, cudaStream_t stream) { GGML_ASSERT(ne00 % 2 == 0); const dim3 block_dims(1, CUDA_ROPE_BLOCK_SIZE, 1); @@ -450,13 +479,13 @@ static void rope_multi_cuda(const T * x, if (freq_factors == nullptr) { const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); ggml_cuda_kernel_launch(rope_multi, launch_params, - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope, inplace); } else { const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); ggml_cuda_kernel_launch(rope_multi, launch_params, - x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, pos, freq_scale, ext_factor, - attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope); + x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, pos, freq_scale, ext_factor, + attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope, inplace); } } @@ -552,8 +581,12 @@ void ggml_cuda_op_rope_impl(ggml_backend_cuda_context & ctx, const int mode = ((int32_t *) dst->op_params)[2]; //const int n_ctx = ((int32_t *) dst->op_params)[3]; const int n_ctx_orig = ((int32_t *) dst->op_params)[4]; + const int n_offs = ((int32_t *) dst->op_params)[15]; mrope_sections sections; + // when dst aliases src0, the channels outside the rotated window already hold the correct data + const bool inplace = dst_d == src0->data; + // RoPE alteration for extended context float freq_base; float freq_scale; @@ -581,6 +614,7 @@ void ggml_cuda_op_rope_impl(ggml_backend_cuda_context & ctx, if (is_vision) { GGML_ASSERT(n_dims == ne00/2); + GGML_ASSERT(n_offs == 0); // offset not supported for vision, as the rotated pairs span the whole row } const int32_t * pos = (const int32_t *) src1_d; @@ -597,31 +631,31 @@ void ggml_cuda_op_rope_impl(ggml_backend_cuda_context & ctx, if (is_neox) { if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F32) { rope_neox_cuda((const float *) src0_d, (float *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F16) { rope_neox_cuda((const float *) src0_d, (half *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else if (src0->type == GGML_TYPE_F16 && dst_type == GGML_TYPE_F16) { rope_neox_cuda((const half *) src0_d, (half *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else { GGML_ABORT("fatal error"); } } else if (is_mrope && !is_vision) { if (src0->type == GGML_TYPE_F32) { rope_multi_cuda((const float *) src0_d, (float *) dst_d, ne00, ne01, ne02, s01, s02, s03, s1, - s2, s3, n_dims, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, - corr_dims, freq_factors, sections, is_imrope, stream); + s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, + corr_dims, freq_factors, sections, is_imrope, inplace, stream); } else if (src0->type == GGML_TYPE_F16) { rope_multi_cuda((const half *) src0_d, (half *) dst_d, ne00, ne01, ne02, s01, s02, s03, s1, - s2, s3, n_dims, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, - corr_dims, freq_factors, sections, is_imrope, stream); + s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, + corr_dims, freq_factors, sections, is_imrope, inplace, stream); } else { GGML_ABORT("fatal error"); } @@ -640,19 +674,19 @@ void ggml_cuda_op_rope_impl(ggml_backend_cuda_context & ctx, } else { if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F32) { rope_norm_cuda((const float *) src0_d, (float *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F16) { rope_norm_cuda((const float *) src0_d, (half *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else if (src0->type == GGML_TYPE_F16 && dst_type == GGML_TYPE_F16) { rope_norm_cuda((const half *) src0_d, (half *) dst_d, ne00, ne01, ne02, s01, s02, - s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, - set_rows_stride, stream); + set_rows_stride, inplace, stream); } else { GGML_ABORT("fatal error"); } diff --git a/ggml/src/ggml-cuda/solve_tri.cu b/ggml/src/ggml-cuda/solve_tri.cu index 07ca33f5..d9678342 100644 --- a/ggml/src/ggml-cuda/solve_tri.cu +++ b/ggml/src/ggml-cuda/solve_tri.cu @@ -65,15 +65,13 @@ static void solve_tri_f32_cublas(ggml_backend_cuda_context & ctx, get_batch_pointers<<<(total_batches + 255) / 256, 256, 0, stream>>>(A, X, A_ptrs_dev, X_ptrs_dev, ne02, total_batches, s02, s03, s2, s3); - CUBLAS_CHECK(cublasSetStream(ctx.cublas_handle(id), stream)); - // Yes, this is necessary, without this we get RMSE errors - CUBLAS_CHECK(cublasSetMathMode(ctx.cublas_handle(id), CUBLAS_DEFAULT_MATH)); - CUBLAS_CHECK(cublasStrsmBatched(ctx.cublas_handle(id), CUBLAS_SIDE_RIGHT, CUBLAS_FILL_MODE_UPPER, CUBLAS_OP_N, + CUBLAS_CHECK(cublasSetMathMode(ctx.cublas_handle(), CUBLAS_DEFAULT_MATH)); + CUBLAS_CHECK(cublasStrsmBatched(ctx.cublas_handle(), CUBLAS_SIDE_RIGHT, CUBLAS_FILL_MODE_UPPER, CUBLAS_OP_N, CUBLAS_DIAG_NON_UNIT, k, n, &alpha, A_ptrs_dev, n, X_ptrs_dev, k, total_batches)); // revert to standard mode from common.cuh - CUBLAS_CHECK(cublasSetMathMode(ctx.cublas_handle(id), CUBLAS_TF32_TENSOR_OP_MATH)); + CUBLAS_CHECK(cublasSetMathMode(ctx.cublas_handle(), CUBLAS_TF32_TENSOR_OP_MATH)); GGML_UNUSED_VARS(s12, s13); } diff --git a/ggml/src/ggml-cuda/ssm-scan.cu b/ggml/src/ggml-cuda/ssm-scan.cu index ef342f01..40cb38de 100644 --- a/ggml/src/ggml-cuda/ssm-scan.cu +++ b/ggml/src/ggml-cuda/ssm-scan.cu @@ -632,7 +632,6 @@ static void ssm_scan_ssd_f32_cuda( // Step 3: chunked SSD loop // Per chunk: pre_matmul (incl. M) + 4 cuBLAS (CB, Y, S@C, state update) + scale_state cublasHandle_t handle = ctx.cublas_handle(); - CUBLAS_CHECK(cublasSetStream(handle, stream)); const float alpha_one = 1.0f; const float beta_zero = 0.0f; const float beta_one = 1.0f; diff --git a/ggml/src/ggml-cuda/sumrows.cu b/ggml/src/ggml-cuda/sumrows.cu index 0003658c..aa8342b5 100644 --- a/ggml/src/ggml-cuda/sumrows.cu +++ b/ggml/src/ggml-cuda/sumrows.cu @@ -24,24 +24,30 @@ void ggml_cuda_op_sum_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { GGML_ASSERT(src0->type == GGML_TYPE_F32); GGML_ASSERT( dst->type == GGML_TYPE_F32); - GGML_ASSERT(ggml_is_contiguous(src0)); + GGML_ASSERT(ggml_is_contiguous_rows(src0)); const int64_t ncols = src0->ne[0]; const int64_t nrows = ggml_nrows(src0); + if (ggml_is_contiguous(src0)) { + sum_rows_f32_cuda(src0_d, dst_d, ncols, nrows, stream); + return; + } + const dim3 block_nums(nrows, 1, 1); const int id = ggml_cuda_get_device(); const int nsm = ggml_cuda_info().devices[id].nsm; + dim3 block_dims; if ((nrows / nsm) < 2) { // Increase num threads to 512 for small nrows to better hide the latency - const dim3 block_dims(512, 1, 1); - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); - ggml_cuda_kernel_launch(reduce_rows_f32, launch_params, src0_d, dst_d, ncols); + block_dims = dim3(512, 1, 1); } else { // Enough active SMs to hide latency, use smaller blocks to allow better scheduling - const dim3 block_dims(ncols < 1024 ? 32 : 128, 1, 1); - const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); - ggml_cuda_kernel_launch(reduce_rows_f32, launch_params, src0_d, dst_d, ncols); + block_dims = dim3(ncols < 1024 ? 32 : 128, 1, 1); } + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream); + const char * src0_d_bytes = (const char *) src0->data; + ggml_cuda_kernel_launch(reduce_rows_f32_strided, launch_params, src0_d_bytes, dst_d, ncols, + src0->ne[1], src0->ne[2], src0->nb[1], src0->nb[2], src0->nb[3]); } diff --git a/ggml/src/ggml-cuda/top-k.cu b/ggml/src/ggml-cuda/top-k.cu index 9681cd29..c7a0c831 100644 --- a/ggml/src/ggml-cuda/top-k.cu +++ b/ggml/src/ggml-cuda/top-k.cu @@ -48,6 +48,168 @@ static int next_power_of_2(int x) { #endif // CUB_TOP_K_AVAILABLE +#if !defined(GGML_CUDA_USE_CUB) && defined(GGML_USE_HIP) + +static __device__ __forceinline__ uint32_t top_k_float_to_ordered(float value) { + const uint32_t bits = __float_as_uint(value); + const uint32_t mask = (uint32_t) (-(int32_t) (bits >> 31)) | 0x80000000U; + return bits ^ mask; +} + +struct top_k_radix_state { + uint32_t prefix; + uint32_t prefix_mask; + int rank; + int greater_count; + int equal_count; +}; + +static __global__ void top_k_radix_init(top_k_radix_state * states, int nrows, int k) { + const int row = blockIdx.x * blockDim.x + threadIdx.x; + if (row < nrows) { + states[row] = {0, 0, k, 0, 0}; + } +} + +template +static __global__ void top_k_radix_histogram( + const float * __restrict__ src, + const top_k_radix_state * __restrict__ states, + int * __restrict__ block_histograms, + int ncols, + int blocks_per_row, + int shift) { + constexpr int NBINS = 1 << RADIX_BITS; + + const int row = blockIdx.x / blocks_per_row; + const int row_block = blockIdx.x % blocks_per_row; + const int tid = threadIdx.x; + const float * row_src = src + (size_t) row * ncols; + __shared__ int histogram[NBINS]; + + histogram[tid] = 0; + __syncthreads(); + + const top_k_radix_state state = states[row]; + for (int col = row_block * BLOCK_SIZE + tid; + col < ncols; + col += blocks_per_row * BLOCK_SIZE) { + const uint32_t key = top_k_float_to_ordered(row_src[col]); + if ((key & state.prefix_mask) == state.prefix) { + atomicAdd(&histogram[(key >> shift) & (NBINS - 1)], 1); + } + } + __syncthreads(); + + const size_t histogram_offset = + ((size_t) row * blocks_per_row + row_block) * NBINS; + block_histograms[histogram_offset + tid] = histogram[tid]; +} + +template +static __global__ void top_k_radix_select( + const int * __restrict__ block_histograms, + top_k_radix_state * __restrict__ states, + int blocks_per_row, + int shift) { + constexpr int NBINS = 1 << RADIX_BITS; + + const int row = blockIdx.x; + const int tid = threadIdx.x; + __shared__ int histogram[NBINS]; + + int count = 0; + for (int row_block = 0; row_block < blocks_per_row; ++row_block) { + const size_t offset = ((size_t) row * blocks_per_row + row_block) * NBINS; + count += block_histograms[offset + tid]; + } + histogram[tid] = count; + __syncthreads(); + + if (tid == 0) { + top_k_radix_state state = states[row]; + int bin = NBINS - 1; + while (bin > 0 && histogram[bin] < state.rank) { + state.rank -= histogram[bin--]; + } + state.prefix |= (uint32_t) bin << shift; + state.prefix_mask |= (uint32_t) (NBINS - 1) << shift; + states[row] = state; + } +} + +static __global__ void top_k_radix_reset_counters(top_k_radix_state * states, int nrows) { + const int row = blockIdx.x * blockDim.x + threadIdx.x; + if (row < nrows) { + states[row].greater_count = 0; + states[row].equal_count = 0; + } +} + +template +static __global__ void top_k_radix_gather( + const float * __restrict__ src, + int * __restrict__ dst, + top_k_radix_state * __restrict__ states, + int ncols, + int k, + int blocks_per_row) { + const int row = blockIdx.x / blocks_per_row; + const int row_block = blockIdx.x % blocks_per_row; + const int tid = threadIdx.x; + const float * row_src = src + (size_t) row * ncols; + int * row_dst = dst + (size_t) row * k; + top_k_radix_state * state = &states[row]; + + for (int col = row_block * BLOCK_SIZE + tid; + col < ncols; + col += blocks_per_row * BLOCK_SIZE) { + const uint32_t key = top_k_float_to_ordered(row_src[col]); + if (key > state->prefix) { + const int pos = atomicAdd(&state->greater_count, 1); + row_dst[pos] = col; + } else if (key == state->prefix) { + const int pos = atomicAdd(&state->equal_count, 1); + if (pos < state->rank) { + row_dst[k - state->rank + pos] = col; + } + } + } +} + +static void top_k_radix_cuda( + ggml_cuda_pool & pool, + const float * src, int * dst, int ncols, int nrows, int k, cudaStream_t stream) { + constexpr int BLOCK_SIZE = 256; + constexpr int RADIX_BITS = 8; + constexpr int NBINS = 1 << RADIX_BITS; + const int blocks_per_row = std::min((ncols + 1023) / 1024, 64); + + ggml_cuda_pool_alloc states_alloc(pool, nrows); + ggml_cuda_pool_alloc histograms_alloc(pool, (size_t) nrows * blocks_per_row * NBINS); + top_k_radix_state * states = states_alloc.get(); + int * histograms = histograms_alloc.get(); + + top_k_radix_init<<<(nrows + BLOCK_SIZE - 1) / BLOCK_SIZE, BLOCK_SIZE, 0, stream>>>(states, nrows, k); + + const dim3 row_grid(blocks_per_row * nrows); + for (int shift = 32 - RADIX_BITS; shift >= 0; shift -= RADIX_BITS) { + top_k_radix_histogram + <<>>( + src, states, histograms, ncols, blocks_per_row, shift); + top_k_radix_select + <<>>(histograms, states, blocks_per_row, shift); + } + + top_k_radix_reset_counters + <<<(nrows + BLOCK_SIZE - 1) / BLOCK_SIZE, BLOCK_SIZE, 0, stream>>>(states, nrows); + top_k_radix_gather + <<>>( + src, dst, states, ncols, k, blocks_per_row); +} + +#endif // !defined(GGML_CUDA_USE_CUB) && defined(GGML_USE_HIP) + void ggml_cuda_op_top_k(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const ggml_tensor * src0 = dst->src[0]; const float * src0_d = (const float *) src0->data; @@ -96,10 +258,18 @@ void ggml_cuda_op_top_k(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { dst_d += k * iter_nrows; } #else // GGML_CUDA_USE_CUB - ggml_cuda_pool_alloc temp_dst_alloc(pool, ncols * nrows); - int * tmp_dst = temp_dst_alloc.get(); - argsort_f32_i32_cuda_bitonic(src0_d, tmp_dst, ncols, nrows, GGML_SORT_ORDER_DESC, stream); - CUDA_CHECK(cudaMemcpy2DAsync(dst_d, k * sizeof(int), tmp_dst, ncols * sizeof(int), k * sizeof(int), nrows, - cudaMemcpyDeviceToDevice, stream)); +#if defined(GGML_USE_HIP) + if (ncols > 1024) { + top_k_radix_cuda(pool, src0_d, dst_d, ncols, nrows, k, stream); + } else { +#endif // defined(GGML_USE_HIP) + ggml_cuda_pool_alloc temp_dst_alloc(pool, ncols * nrows); + int * tmp_dst = temp_dst_alloc.get(); + argsort_f32_i32_cuda_bitonic(src0_d, tmp_dst, ncols, nrows, GGML_SORT_ORDER_DESC, stream); + CUDA_CHECK(cudaMemcpy2DAsync(dst_d, k * sizeof(int), tmp_dst, ncols * sizeof(int), k * sizeof(int), nrows, + cudaMemcpyDeviceToDevice, stream)); +#if defined(GGML_USE_HIP) + } +#endif // defined(GGML_USE_HIP) #endif } diff --git a/ggml/src/ggml-cuda/topk-moe.cu b/ggml/src/ggml-cuda/topk-moe.cu index c8cec70b..dadcd601 100644 --- a/ggml/src/ggml-cuda/topk-moe.cu +++ b/ggml/src/ggml-cuda/topk-moe.cu @@ -88,15 +88,16 @@ __device__ void sqrt_softplus_warp_inplace(float (&vals)[experts_per_thread], co It is intended as fusion of softmax->top-k->get_rows pipeline for MoE models */ template -__launch_bounds__(4 * WARP_SIZE, 1) __global__ void topk_moe_cuda(const float * logits, - float * weights, - int32_t * ids, - float * bias, - const int n_rows, - const int n_expert_used, - const float clamp_val, - const float scale_val, - const topk_moe_config config) { +__launch_bounds__(TOPK_MOE_ROWS_PER_BLOCK * WARP_SIZE, 1) +__global__ void topk_moe_cuda(const float * logits, + float * weights, + int32_t * ids, + float * bias, + const int n_rows, + const int n_expert_used, + const float clamp_val, + const float scale_val, + const topk_moe_config config) { const int row = blockIdx.x * blockDim.y + threadIdx.y; if (row >= n_rows) { return; @@ -123,6 +124,9 @@ __launch_bounds__(4 * WARP_SIZE, 1) __global__ void topk_moe_cuda(const float * wt[i / WARP_SIZE] = (n_experts % WARP_SIZE == 0 || expert < n_experts) ? logits[expert] : -INFINITY; } + // Weights and IDs can alias logits, so wait until every row in the block reads its logits. + __syncthreads(); + if (!config.delayed_softmax) { if (config.use_sigmoid) { sigmoid_warp_inplace(wt, n_experts, threadIdx.x); @@ -282,7 +286,7 @@ static void launch_topk_moe_cuda(ggml_backend_cuda_context & ctx, const topk_moe_config config) { GGML_ASSERT(!(config.with_norm && config.delayed_softmax) && "delayed softmax is not supported with weight normalization"); - const int rows_per_block = 4; + const int rows_per_block = TOPK_MOE_ROWS_PER_BLOCK; dim3 grid_dims((n_rows + rows_per_block - 1) / rows_per_block, 1, 1); dim3 block_dims(WARP_SIZE, rows_per_block, 1); cudaStream_t stream = ctx.stream(); diff --git a/ggml/src/ggml-cuda/topk-moe.cuh b/ggml/src/ggml-cuda/topk-moe.cuh index 091ef02a..061b37e2 100644 --- a/ggml/src/ggml-cuda/topk-moe.cuh +++ b/ggml/src/ggml-cuda/topk-moe.cuh @@ -3,6 +3,9 @@ #include +// Rows that one CUDA block handles. +#define TOPK_MOE_ROWS_PER_BLOCK 8 + struct ggml_cuda_topk_moe_args { bool sigmoid{}; bool sqrt_softplus{}; diff --git a/ggml/src/ggml-cuda/unary.cu b/ggml/src/ggml-cuda/unary.cu index 4cb805fa..d3e59487 100644 --- a/ggml/src/ggml-cuda/unary.cu +++ b/ggml/src/ggml-cuda/unary.cu @@ -427,6 +427,81 @@ void ggml_cuda_op_swiglu_oai(ggml_backend_cuda_context & ctx, ggml_tensor * dst) swiglu_oai_cuda(src0_p, src1_p, (float *)dst_d, ggml_nelements(dst), nc, src0_o / sizeof(float), src1_o / sizeof(float), alpha, limit, stream); } +// swiglu_clamp + +template +static __global__ void swiglu_clamp_kernel(const T * gate, const T * up, T * dst, const int64_t k, const int64_t n, const int64_t o0, const int64_t o1, float limit) { + const int64_t i = int64_t(blockDim.x)*blockIdx.x + threadIdx.x; + + if (i >= k) { + return; + } + + const int64_t j0 = (i / n) * o0 + (i % n); + const int64_t j1 = o0 == o1 ? j0 : (i / n) * o1 + (i % n); + + dst[i] = (T) ggml_cuda_op_swiglu_clamp_single((float) gate[j0], (float) up[j1], limit); +} + +template +static void swiglu_clamp_cuda(const T * gate, const T * up, T * dst, const int64_t k, const int64_t n, const int64_t o0, const int64_t o1, const float limit, cudaStream_t stream) { + const int64_t num_blocks = (k + CUDA_GLU_BLOCK_SIZE - 1) / CUDA_GLU_BLOCK_SIZE; + swiglu_clamp_kernel<<>>(gate, up, dst, k, n, o0, o1, limit); +} + +void ggml_cuda_op_swiglu_clamp(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + void * src0_d = src0->data; + void * src1_d = src1 ? src1->data : src0->data; + const int64_t src0_o = src0->nb[1]; + const int64_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; + void * dst_d = dst->data; + const int64_t nc = src1 ? src0->ne[0] : src0->ne[0] / 2; + cudaStream_t stream = ctx.stream(); + + GGML_ASSERT(ggml_is_contiguous_1(src0)); + GGML_ASSERT(src0->nb[0] == ggml_element_size(src0)); + GGML_ASSERT(ggml_is_contiguous(dst)); + + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT(src0->type == dst->type); + GGML_ASSERT(dst->ne[0] == nc); + GGML_ASSERT(ggml_nrows(dst) == ggml_nrows(src0)); + + if (src1) { + GGML_ASSERT(ggml_is_contiguous_1(src1)); + GGML_ASSERT(src1->nb[0] == ggml_element_size(src1)); + GGML_ASSERT(src1->ne[0] == nc); + GGML_ASSERT(src0->type == src1->type); + } + + const int32_t swapped = ggml_get_op_params_i32(dst, 1); + const float limit = ggml_get_op_params_f32(dst, 3); + + if (src0->type == GGML_TYPE_F16) { + half * src0_p = (half *) src0_d; + half * src1_p = (half *) src1_d; + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + swiglu_clamp_cuda(src0_p, src1_p, (half *) dst_d, ggml_nelements(dst), nc, src0_o / sizeof(half), src1_o / sizeof(half), limit, stream); + } else { + float * src0_p = (float *) src0_d; + float * src1_p = (float *) src1_d; + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + swiglu_clamp_cuda(src0_p, src1_p, (float *) dst_d, ggml_nelements(dst), nc, src0_o / sizeof(float), src1_o / sizeof(float), limit, stream); + } +} + /* CUDA kernel + launcher for xIELU */ template diff --git a/ggml/src/ggml-cuda/unary.cuh b/ggml/src/ggml-cuda/unary.cuh index 81ed873e..04f3af64 100644 --- a/ggml/src/ggml-cuda/unary.cuh +++ b/ggml/src/ggml-cuda/unary.cuh @@ -83,6 +83,8 @@ void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); void ggml_cuda_op_swiglu_oai(ggml_backend_cuda_context & ctx, ggml_tensor * dst); +void ggml_cuda_op_swiglu_clamp(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst); void ggml_cuda_op_geglu_quick(ggml_backend_cuda_context & ctx, ggml_tensor * dst); @@ -112,3 +114,10 @@ __device__ __forceinline__ float ggml_cuda_op_swiglu_oai_single(float x, float g out_glu = out_glu * (1.0f + g); return out_glu; } + +__device__ __forceinline__ float ggml_cuda_op_swiglu_clamp_single(float gate, float up, float limit) { + gate = fminf(gate, limit); + up = fmaxf(fminf(up, limit), -limit); + + return ggml_cuda_op_silu_single(gate) * up; +} diff --git a/ggml/src/ggml-cuda/vecdotq.cuh b/ggml/src/ggml-cuda/vecdotq.cuh index 0f039c73..f2a6f200 100644 --- a/ggml/src/ggml-cuda/vecdotq.cuh +++ b/ggml/src/ggml-cuda/vecdotq.cuh @@ -747,12 +747,20 @@ static __device__ __forceinline__ float vec_dot_q2_0_q8_1( const int u = get_int_b4(bq8_1_chunk->qs, j*2+0); const int v = get_int_b4(bq8_1_chunk->qs, j*2+1); +#if defined(GGML_USE_HIP) + const uint32_t qx_indices = (q & 0x03) | ((q & 0x0C) << 6) | ((q & 0x30) << 12) | ((q & 0xC0) << 18); + const uint32_t qy_bits = q >> 8; + const uint32_t qy_indices = (qy_bits & 0x03) | ((qy_bits & 0x0C) << 6) | ((qy_bits & 0x30) << 12) | ((qy_bits & 0xC0) << 18); + const int qx = __builtin_amdgcn_perm(0x020100FF, 0x020100FF, qx_indices); + const int qy = __builtin_amdgcn_perm(0x020100FF, 0x020100FF, qy_indices); +#else // unpack even and odd crumbs into byte values const int qe = __byte_perm(0x020100FF, 0x020100FF, q >> 0); const int qo = __byte_perm(0x020100FF, 0x020100FF, q >> 2); // unshuffle values const int qx = __byte_perm(qe, qo, 0x5140); const int qy = __byte_perm(qe, qo, 0x7362); +#endif // defined(GGML_USE_HIP) sumi = ggml_cuda_dp4a(u, qx, sumi); sumi = ggml_cuda_dp4a(v, qy, sumi); @@ -928,16 +936,20 @@ static __device__ __forceinline__ float vec_dot_q4_K_q8_1( v[0] = q4[0]; v[1] = q4[4]; + // branchless so nvcc can hoist this out of the ncols_dst loop const uint16_t * scales = (const uint16_t *)bq4_K->scales; + const int j = bq8_offset/2; + const int jm = j & 1; + + const uint32_t s0 = scales[jm + 0]; + const uint32_t s2 = scales[jm + 2]; + const uint32_t s4 = scales[jm + 4]; + + const uint32_t hi = (uint32_t) -(int32_t) (j >= 2); + uint16_t aux[2]; - const int j = bq8_offset/2; - if (j < 2) { - aux[0] = scales[j+0] & 0x3f3f; - aux[1] = scales[j+2] & 0x3f3f; - } else { - aux[0] = ((scales[j+2] >> 0) & 0x0f0f) | ((scales[j-2] & 0xc0c0) >> 2); - aux[1] = ((scales[j+2] >> 4) & 0x0f0f) | ((scales[j-0] & 0xc0c0) >> 2); - } + aux[0] = (uint16_t) (((s0 & 0x3f3f) & ~hi) | ((((s4 >> 0) & 0x0f0f) | ((s0 & 0xc0c0) >> 2)) & hi)); + aux[1] = (uint16_t) (((s2 & 0x3f3f) & ~hi) | ((((s4 >> 4) & 0x0f0f) | ((s2 & 0xc0c0) >> 2)) & hi)); const uint8_t * sc = (const uint8_t *)aux; const uint8_t * m = sc + 2; @@ -973,16 +985,21 @@ static __device__ __forceinline__ float vec_dot_q5_K_q8_1( vh[0] = qh[0] >> bq8_offset; vh[1] = qh[4] >> bq8_offset; + // same as q4_K const uint16_t * scales = (const uint16_t *)bq5_K->scales; + const int j = bq8_offset/2; + const int jm = j & 1; + + const uint32_t s0 = scales[jm + 0]; + const uint32_t s2 = scales[jm + 2]; + const uint32_t s4 = scales[jm + 4]; + + const uint32_t hi = (uint32_t) -(int32_t) (j >= 2); + uint16_t aux[2]; - const int j = bq8_offset/2; - if (j < 2) { - aux[0] = scales[j+0] & 0x3f3f; - aux[1] = scales[j+2] & 0x3f3f; - } else { - aux[0] = ((scales[j+2] >> 0) & 0x0f0f) | ((scales[j-2] & 0xc0c0) >> 2); - aux[1] = ((scales[j+2] >> 4) & 0x0f0f) | ((scales[j-0] & 0xc0c0) >> 2); - } + aux[0] = (uint16_t) (((s0 & 0x3f3f) & ~hi) | ((((s4 >> 0) & 0x0f0f) | ((s0 & 0xc0c0) >> 2)) & hi)); + aux[1] = (uint16_t) (((s2 & 0x3f3f) & ~hi) | ((((s4 >> 4) & 0x0f0f) | ((s2 & 0xc0c0) >> 2)) & hi)); + const uint8_t * sc = (const uint8_t *)aux; const uint8_t * m = sc + 2; diff --git a/ggml/src/ggml-cuda/vendors/hip.h b/ggml/src/ggml-cuda/vendors/hip.h index 9aa558f3..0a2f2829 100644 --- a/ggml/src/ggml-cuda/vendors/hip.h +++ b/ggml/src/ggml-cuda/vendors/hip.h @@ -73,6 +73,10 @@ #define cudaGetDeviceProperties hipGetDeviceProperties #define cudaGetErrorString hipGetErrorString #define cudaGetLastError hipGetLastError +#define cudaHostAlloc hipHostMalloc +#define cudaHostAllocPortable hipHostMallocPortable +#define cudaHostAllocMapped hipHostMallocMapped +#define cudaHostGetDevicePointer hipHostGetDevicePointer #define cudaHostRegister hipHostRegister #define cudaHostRegisterPortable hipHostRegisterPortable #define cudaHostRegisterReadOnly hipHostRegisterReadOnly @@ -176,9 +180,9 @@ #define __CUDA_ARCH__ 1300 -#if defined(__gfx900__) || defined(__gfx906__) +#if defined(__gfx900__) || defined(__gfx906__) || defined(__gfx909__) || defined(__gfx90c__) #define GCN5 -#endif // defined(__gfx900__) || defined(__gfx906__) +#endif // defined(__gfx900__) || defined(__gfx906__) || defined(__gfx909__) || defined(__gfx90c__) #if defined(__gfx803__) #define GCN4 @@ -273,7 +277,15 @@ static __device__ __forceinline__ int __vsubss4(const int a, const int b) { } static __device__ __forceinline__ int __vsub4(const int a, const int b) { - return __vsubss4(a, b); + // do some small modifications to a and b to make the subtraction not underflow + const unsigned int a_large = a | 0x80808080; + const unsigned int b_small = b & 0x7f7f7f7f; + const unsigned int result_low_7bits = a_large - b_small; + + // if two ops share the same high bit, we should flip the high bit of the result + const unsigned int should_flip_high_1bit = (a ^ ~b) & 0x80808080; + + return result_low_7bits ^ should_flip_high_1bit; } static __device__ __forceinline__ unsigned int __vcmpeq4(unsigned int a, unsigned int b) { @@ -289,13 +301,13 @@ static __device__ __forceinline__ unsigned int __vcmpeq4(unsigned int a, unsigne } static __device__ __forceinline__ unsigned int __vcmpne4(unsigned int a, unsigned int b) { - const uint8x4_t& va = reinterpret_cast(a); - const uint8x4_t& vb = reinterpret_cast(b); - unsigned int c; - uint8x4_t& vc = reinterpret_cast(c); -#pragma unroll - for (int i = 0; i < 4; ++i) { - vc[i] = va[i] == vb[i] ? 0x00 : 0xff; - } - return c; + const unsigned int x = a ^ b; + + // any non-equal bit in a byte will set the high bit of that byte here + // the addition will not overflow in the byte as op1 and op2 are both less than 0x80 + const unsigned int ne_low_7bits = ((x & 0x7f7f7f7f) + 0x7f7f7f7f) & 0x80808080; + const unsigned int ne_high_1bit = x & 0x80808080; + const unsigned int ne_any_bit = ne_low_7bits | ne_high_1bit; + + return (ne_any_bit >> 7) * 0xff; } diff --git a/ggml/src/ggml-et/et-kernels/src/glu_f32.c b/ggml/src/ggml-et/et-kernels/src/glu_f32.c index 95fe5721..d376d6f5 100644 --- a/ggml/src/ggml-et/et-kernels/src/glu_f32.c +++ b/ggml/src/ggml-et/et-kernels/src/glu_f32.c @@ -17,7 +17,7 @@ struct ggml_et_glu_params { int32_t glu_op_type; // GLU operation type (REGLU=0, GEGLU=1, SWIGLU=2, etc.) int32_t swapped; // Whether gate and value are swapped float alpha; // SWIGLU_OAI: sigmoid scaling factor - float limit; // SWIGLU_OAI: clamp limit + float limit; // GLU clamp limit }; // SiLU activation function: silu(x) = x * sigmoid(x) = x / (1 + exp(-x)) @@ -332,6 +332,57 @@ static inline void block_swiglu_oai(float * dst_block, } } +static inline void block_swiglu_clamp(float * dst_block, + const float * gate_block, + const float * up_block, + int elements, + float limit) { + int32_t vec_end = (elements / 8) * 8; + + unsigned long temp_mask; + __asm__ volatile("mova.x.m %0" : "=r"(temp_mask)); + __asm__ volatile("mov.m.x m0, x0, 0xFF"); + + float one_const = 1.0f; + float limit_pos = limit; + float limit_neg = -limit; + float neg_log2e = -1.4426950408889634f; + + for (int32_t i = 0; i < vec_end; i += 8) { + __asm__ volatile( + "flw.ps f10, %[gate_vec]\n" + "flw.ps f11, %[up_vec]\n" + "fbc.ps f21, %[one_ptr]\n" + "fbc.ps f23, %[lim_pos]\n" + "fbc.ps f24, %[lim_neg]\n" + "fbc.ps f25, %[k_ptr]\n" + "fmin.ps f12, f10, f23\n" + "fmax.ps f13, f11, f24\n" + "fmin.ps f13, f13, f23\n" + "fmul.ps f14, f12, f25\n" + "fexp.ps f15, f14\n" + "fadd.ps f15, f15, f21\n" + "frcp.ps f16, f15\n" + "fmul.ps f17, f12, f16\n" + "fmul.ps f18, f17, f13\n" + "fsw.ps f18, %[dst_out]\n" + : [dst_out] "=m"(*(float (*)[8]) & dst_block[i]) + : [gate_vec] "m"(*(const float (*)[8]) & gate_block[i]), [up_vec] "m"(*(const float (*)[8]) & up_block[i]), + [one_ptr] "m"(one_const), [lim_pos] "m"(limit_pos), [lim_neg] "m"(limit_neg), [k_ptr] "m"(neg_log2e) + : "f10", "f11", "f12", "f13", "f14", "f15", "f16", "f17", "f18", "f21", "f23", "f24", "f25"); + } + + __asm__ volatile("mova.m.x %0" :: "r"(temp_mask)); + + for (int32_t i = vec_end; i < elements; i++) { + float gate = gate_block[i] > limit ? limit : gate_block[i]; + float up = up_block[i]; + up = up > limit ? limit : up; + up = up < -limit ? -limit : up; + dst_block[i] = silu_f32(gate) * up; + } +} + // Scalar erf approximation (Abramowitz & Stegun 7.1.26, max error ~1.5e-7) static inline float erf_approx(float x) { const float a1 = 0.254829592f; @@ -386,6 +437,7 @@ int entry_point(struct ggml_et_glu_params * params, void * env) { switch (params->glu_op_type) { case GGML_GLU_OP_SWIGLU: case GGML_GLU_OP_SWIGLU_OAI: + case GGML_GLU_OP_SWIGLU_CLAMP: case GGML_GLU_OP_GEGLU: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: @@ -531,6 +583,9 @@ int entry_point(struct ggml_et_glu_params * params, void * env) { case GGML_GLU_OP_SWIGLU_OAI: block_swiglu_oai(dst_ptr, x_ptr, g_ptr, (int) elements_to_process, params->alpha, params->limit); break; + case GGML_GLU_OP_SWIGLU_CLAMP: + block_swiglu_clamp(dst_ptr, x_ptr, g_ptr, (int) elements_to_process, params->limit); + break; default: return -1; } diff --git a/ggml/src/ggml-et/ggml-et-cpu-compare.cpp b/ggml/src/ggml-et/ggml-et-cpu-compare.cpp index b37f6d26..5771679b 100644 --- a/ggml/src/ggml-et/ggml-et-cpu-compare.cpp +++ b/ggml/src/ggml-et/ggml-et-cpu-compare.cpp @@ -261,7 +261,12 @@ bool ggml_et_cpu_compare_compute_and_check(ggml_et_cpu_compare_ctx * ct GGML_LOG_ERROR("ET: GLU CPU comparison requires split tensor mode\n"); return false; } - ctx->cpu_dst = ggml_glu_split(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, glu_op); + if (glu_op == GGML_GLU_OP_SWIGLU_CLAMP) { + const float limit = ggml_get_op_params_f32(node, 3); + ctx->cpu_dst = ggml_swiglu_clamp(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, limit); + } else { + ctx->cpu_dst = ggml_glu_split(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, glu_op); + } } break; case GGML_OP_SOFT_MAX: diff --git a/ggml/src/ggml-et/ggml-et-ops.cpp b/ggml/src/ggml-et/ggml-et-ops.cpp index 7871d524..87651386 100644 --- a/ggml/src/ggml-et/ggml-et-ops.cpp +++ b/ggml/src/ggml-et/ggml-et-ops.cpp @@ -636,6 +636,7 @@ bool ggml_et_op_glu(ggml_backend_et_device_context * dev_ctx, const ggml_tensor case GGML_GLU_OP_GEGLU: case GGML_GLU_OP_SWIGLU: case GGML_GLU_OP_SWIGLU_OAI: + case GGML_GLU_OP_SWIGLU_CLAMP: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: break; @@ -661,6 +662,8 @@ bool ggml_et_op_glu(ggml_backend_et_device_context * dev_ctx, const ggml_tensor params.limit = 0.0f; if (glu_op_type == GGML_GLU_OP_SWIGLU_OAI) { params.alpha = ggml_get_op_params_f32(node, 2); + } + if (glu_op_type == GGML_GLU_OP_SWIGLU_OAI || glu_op_type == GGML_GLU_OP_SWIGLU_CLAMP) { params.limit = ggml_get_op_params_f32(node, 3); } // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel) diff --git a/ggml/src/ggml-et/ggml-et.cpp b/ggml/src/ggml-et/ggml-et.cpp index e8482f73..61c31d6f 100644 --- a/ggml/src/ggml-et/ggml-et.cpp +++ b/ggml/src/ggml-et/ggml-et.cpp @@ -1061,9 +1061,11 @@ static bool ggml_backend_et_device_supports_op(ggml_backend_dev_t dev, const ggm const bool zero_view_offset = op->src[0]->view_src == nullptr || op->src[0]->view_offs == 0; const bool has_sections = ggml_get_op_params_i32(op, 11) > 0 || ggml_get_op_params_i32(op, 12) > 0 || ggml_get_op_params_i32(op, 13) > 0; + // FIXME: support ggml_rope_set_offset + const bool zero_rot_offset = ggml_get_op_params_i32(op, 15) == 0; supported = - zero_view_offset && ndims <= 512 && + zero_view_offset && zero_rot_offset && ndims <= 512 && (is_normal || (is_neox && ndims % 16 == 0) || (is_imrope && ndims % 16 == 0 && has_sections)); } else { supported = false; @@ -1208,7 +1210,8 @@ static bool ggml_backend_et_device_supports_op(ggml_backend_dev_t dev, const ggm // Check GLU variant - support SWIGLU, SWIGLU_OAI, GEGLU, GEGLU_ERF, GEGLU_QUICK, REGLU ggml_glu_op glu_type = ggml_get_glu_op(op); const bool supported_variant = glu_type == GGML_GLU_OP_SWIGLU || glu_type == GGML_GLU_OP_SWIGLU_OAI || - glu_type == GGML_GLU_OP_GEGLU || glu_type == GGML_GLU_OP_GEGLU_ERF || + glu_type == GGML_GLU_OP_SWIGLU_CLAMP || glu_type == GGML_GLU_OP_GEGLU || + glu_type == GGML_GLU_OP_GEGLU_ERF || glu_type == GGML_GLU_OP_GEGLU_QUICK || glu_type == GGML_GLU_OP_REGLU; if (op->src[1]) { diff --git a/ggml/src/ggml-hexagon/ggml-hexagon.cpp b/ggml/src/ggml-hexagon/ggml-hexagon.cpp index f80c60a5..58806e37 100644 --- a/ggml/src/ggml-hexagon/ggml-hexagon.cpp +++ b/ggml/src/ggml-hexagon/ggml-hexagon.cpp @@ -6,6 +6,7 @@ #include #include +#include #include #include #include @@ -18,7 +19,10 @@ #include #include #include +#include #include +#include +#include #ifdef _WIN32 # define WIN32_LEAN_AND_MEAN @@ -50,8 +54,15 @@ #include "htp-opnode.h" #include "htp-ops.h" #include "htp/matmul-ops.h" +#include "htp/binary-ops.h" #include "htp/flash-attn-ops.h" #include "htp/unary-ops.h" +#include "htp/get-rows-ops.h" +#include "htp/set-rows-ops.h" +#include "htp/softmax-ops.h" +#include "htp/rope-ops.h" +#include "htp/ssm-conv.h" +#include "htp/gated-delta-net-ops.h" #include "htp_iface.h" #include "htp-drv.h" @@ -59,6 +70,22 @@ using intvec = std::vector; using uintvec = std::vector; using u32vec = std::vector; +#define GGML_HEXAGON_MAX_SESSIONS 16 + +#define GGML_HEXAGON_FENCE_SLOT_SIZE 128 + +struct ggml_hexagon_device_config { + int physical_idx = 0; + int virtual_idx = 0; + int domain_id = 0; + std::string domain_name; + std::string name; + + std::vector mdev_group; +}; + +static ggml_hexagon_device_config opt_device_configs[GGML_HEXAGON_MAX_SESSIONS]; + static int opt_arch = 0; // autodetect static size_t opt_ndev = 1; static size_t opt_nhvx = 0; // use all @@ -68,24 +95,40 @@ static size_t opt_mbuf = 1ul * 1024 * 1024 * 1024; // max buffer size static int opt_etm = 0; static int opt_verbose = 0; static int opt_profile = 0; // profiling mode (0-disabled, 1-basic, 2-pmu) -static int opt_hostbuf = 1; // hostbuf ON by default +static bool opt_hostbuf = false; +static bool opt_dma64 = false; -static int opt_mm_select = 3; // 3 = HMX -> Tiled -> Flat -> CPU, 2 = Tiled -> Flat -> CPU, 1 = Flat -> CPU +static int opt_mm_select = 2; // 2 = HMX -> HVX -> CPU, 1 = HVX -> CPU, 0 = CPU (unsupported) static int opt_fa_select = 2; // 2 = HMX -> HVX -> CPU, 1 = HVX -> CPU, 0 = CPU (unsupported) +static int opt_gdn_select = 2; // 2 = HMX -> HVX, 1 = HVX, 0 = CPU (unsupported) +static int opt_ar_select = 2; // 2 = fused ALLREDUCE+ADD (DMA, default), 1 = unfused ALLREDUCE (DMA), 0 = fallback to CPY+FENCE // Default PMU events, if profiling with PMU (mode=2) is enabled // See https://docs.qualcomm.com/doc/80-N2040-60/topic/pmu-events.html // https://docs.qualcomm.com/doc/80-N2040-61/topic/hvx-pmu-events.html static u32vec opt_pmu_evt { 0x3, 0x111, 0x100, 0x105, 0x240, 0x256, 0x7D, 0x8C }; -// Enable all stages by default -static int opt_opstage = HTP_OPSTAGE_QUEUE | HTP_OPSTAGE_COMPUTE; -static int opt_opbatch = 1024; // max number of ops in a batch -static int opt_opqueue = 16; // max number of pending batches +static int opt_opbatch = 1280; // max number of ops in a batch +static int opt_opqueue = 32; // max number of pending batches static int opt_optrace = 0; // trace buffer size per thread (0 means default) static int opt_oppoll = 0; // polling for batch completions static int opt_opfusion = 1; // enable/disable op fusion +enum ggml_hexagon_fusion_flags { + GGML_HEXAGON_FUSE_ALLREDUCE_ADD = (1 << 1), // 2 + GGML_HEXAGON_FUSE_RMS_NORM_MUL = (1 << 2), // 4 + GGML_HEXAGON_FUSE_MUL_MAT_ADD = (1 << 3), // 8 + GGML_HEXAGON_FUSE_MUL_MAT_NX = (1 << 4), // 16 + GGML_HEXAGON_FUSE_MUL_MAT_ID_NX = (1 << 5), // 32 + GGML_HEXAGON_FUSE_GDN_CPY = (1 << 6), // 64 +}; + +static inline bool ggml_hexagon_is_fusion_enabled(int flag) { + if (opt_opfusion <= 0) return false; + if (opt_opfusion == 1) return true; // 1 enables all + return (opt_opfusion & flag) != 0; +} + static std::regex* opt_opfilter = NULL; // regex of ops to not claim #define HEX_VERBOSE(...) \ @@ -121,7 +164,7 @@ static void ggml_hexagon_dump_op_exec(const std::string &sess_name, const htp_op static void ggml_hexagon_dump_op_supp(const std::string &sess_name, const struct ggml_tensor * op, bool supp) { if (!opt_verbose) return; - htp_opformat fmt(htp_opformat(htp_opnode{const_cast(op), {}, HTP_OP_INVALID})); + htp_opformat fmt(htp_opformat(htp_opnode(HTP_OP_INVALID, const_cast(op)))); GGML_LOG_DEBUG("ggml-hex: %s supports-op %s|%s|%s|%s|%s|%s|%s\n", sess_name.c_str(), ggml_op_desc(op), fmt.names, fmt.dims, fmt.types, fmt.strides, fmt.buffs, supp ? "yes" : "no"); } @@ -140,10 +183,18 @@ static const char * htp_event_name(uint16_t id) { case HTP_TRACE_EVT_HVX_FA_Q_PREP: return "HVX_Q_PREP"; case HTP_TRACE_EVT_HVX_FA_K_PREP: return "HVX_K_PREP"; case HTP_TRACE_EVT_HVX_FA_V_PREP: return "HVX_V_PREP"; + case HTP_TRACE_EVT_HVX_GDN_PREP: return "HVX_GDN_PREP"; + case HTP_TRACE_EVT_HVX_GDN_SOLVE: return "HVX_GDN_SOLVE"; + case HTP_TRACE_EVT_HVX_GDN_V_PREP: return "HVX_GDN_V_PREP"; + case HTP_TRACE_EVT_HVX_GDN_D_PREP: return "HVX_GDN_D_PREP"; + case HTP_TRACE_EVT_HVX_GDN_OUT: return "HVX_GDN_OUT"; + case HTP_TRACE_EVT_HVX_GDN_STATE: return "HVX_GDN_STATE"; + case HTP_TRACE_EVT_HVX_GDN_REM: return "HVX_GDN_REM"; case HTP_TRACE_EVT_HMX_COMP: return "HMX_COMP"; case HTP_TRACE_EVT_L2FLUSH: return "L2FLUSH"; case HTP_TRACE_EVT_INIT: return "INIT"; case HTP_TRACE_EVT_BUFF: return "BUFF"; + case HTP_TRACE_EVT_FENCE: return "FENCE"; default: return "UNKNOWN"; } } @@ -205,12 +256,30 @@ static void ggml_hexagon_dump_trace_events(const std::string & sess_name, const } } -// ** +enum ggml_hexagon_tensor_flags { + GGML_HEXAGON_TENSOR_REPACK = (1 << 0), + GGML_HEXAGON_TENSOR_WEIGHT = (1 << 1), + GGML_HEXAGON_TENSOR_FENCE = (1 << 2), + GGML_HEXAGON_TENSOR_FUSEABLE = (1 << 3), +}; static inline bool ggml_hexagon_is_repack_type(enum ggml_type type) { return type == GGML_TYPE_Q4_0 || type == GGML_TYPE_Q4_1 || type == GGML_TYPE_Q8_0 || type == GGML_TYPE_IQ4_NL || - type == GGML_TYPE_MXFP4; + type == GGML_TYPE_MXFP4 || type == GGML_TYPE_Q6_K || + type == GGML_TYPE_Q4_K; +} + +// Size of one repacked row in the DSP tiled layout. The Q6_K and Q4_K tiles store uncompressed scales/mins, +// so they are larger than the ggml blocks. For the other repack types the tile has the same size as the ggml blocks. +static inline size_t ggml_hexagon_tiled_row_size(enum ggml_type type, int64_t ne0) { + if (type == GGML_TYPE_Q6_K) { + return (size_t) (ne0 / 32) * (HTP_MM_WEIGHT_TILE_SIZE_Q6_K / 32); + } + if (type == GGML_TYPE_Q4_K) { + return (size_t) (ne0 / 32) * (HTP_MM_WEIGHT_TILE_SIZE_Q4_1 / 32); + } + return ggml_row_size(type, ne0); } static inline bool ggml_hexagon_is_hmx_weight_type(enum ggml_type type) { @@ -227,6 +296,15 @@ static void ggml_hexagon_precompute_matmul_params( struct htp_mm_kernel_params * kparams ); +static void ggml_hexagon_precompute_fused_matmul_add_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * src2, + const struct ggml_tensor * dst, + struct htp_mm_kernel_params * kparams +); + static void ggml_hexagon_precompute_unary_params( const struct ggml_hexagon_session * sess, uint32_t op, @@ -236,25 +314,145 @@ static void ggml_hexagon_precompute_unary_params( struct htp_unary_kernel_params * kparams ); -static void ggml_hexagon_precompute_fused_qkv_params( +static bool ggml_hexagon_precompute_binary_params( + const struct ggml_hexagon_session * sess, + uint32_t op, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_binary_kernel_params * kparams +); + +static void ggml_hexagon_precompute_get_rows_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_get_rows_kernel_params * kparams +); + +static void ggml_hexagon_precompute_set_rows_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_set_rows_kernel_params * kparams +); + +static void ggml_hexagon_precompute_softmax_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * op, + struct htp_softmax_kernel_params * kparams +); + +static void ggml_hexagon_precompute_rope_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * op, + struct htp_rope_kernel_params * kparams +); + +static void ggml_hexagon_precompute_ssm_conv_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_ssm_conv_kernel_params * kparams +); + +static void ggml_hexagon_precompute_gated_delta_net_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * op, + struct htp_gdn_kernel_params * kparams +); + +static void ggml_hexagon_precompute_fused_mmnx_params( const struct ggml_hexagon_session * sess, const struct ggml_tensor * src0, const struct ggml_tensor * src1, + int32_t n_weights, struct htp_mm_kernel_params * kparams ); -static void ggml_hexagon_precompute_fused_ffn_params( +static void ggml_hexagon_precompute_fused_mmidnx_params( const struct ggml_hexagon_session * sess, const struct ggml_tensor * src0, const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + int32_t n_weights, struct htp_mm_kernel_params * kparams ); +static bool ggml_hexagon_precompute_allreduce_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * dst, + uint32_t rank, + uint32_t n_ranks, + bool has_add, + bool is_row_bcast, + struct htp_allreduce_kernel_params * kparams +); + +static bool mm_is_hmx_eligible(const ggml_tensor * t); +static htp_op_code op_remap_to_htp(const ggml_tensor * t); +static bool is_supported_mul_mat_nx_kernel(const ggml_tensor * src0, const struct htp_mm_kernel_params * kparams); +static bool is_supported_mul_mat_id_nx_kernel(const ggml_tensor * src0, const struct htp_mm_kernel_params * kparams); +static bool is_mergeable_mul_mat(const ggml_tensor * t); +static bool is_mergeable_mul_mat_pair(const ggml_tensor * n1, const ggml_tensor * n2); +static bool is_mergeable_mul_mat_id(const ggml_tensor * t); +static bool is_mergeable_mul_mat_id_pair(const ggml_tensor * n1, const ggml_tensor * n2); + // ** backend sessions +struct ggml_hexagon_tensor_extra { + std::vector shadow_buf; + size_t shadow_size { 0 }; + uint32_t flags { 0 }; +}; + +static inline bool ggml_hexagon_tensor_is_fuseable(const struct ggml_tensor * t) { + if (!t->extra) return false; + auto extra = (const struct ggml_hexagon_tensor_extra *) t->extra; + return (extra->flags & GGML_HEXAGON_TENSOR_FUSEABLE) != 0; +} + +static inline bool ggml_hexagon_tensors_overlap(const struct ggml_tensor * a, const struct ggml_tensor * b) { + const uintptr_t a0 = (uintptr_t) a->data; + const uintptr_t b0 = (uintptr_t) b->data; + const uintptr_t a1 = a0 + ggml_nbytes(a); + const uintptr_t b1 = b0 + ggml_nbytes(b); + + return a0 < b1 && b0 < a1; +} + +struct htp_opnode; + struct ggml_hexagon_opbatch; struct ggml_hexagon_opqueue; -struct htp_opnode; +struct ggml_hexagon_shared_buffer; +struct ggml_hexagon_fence_buffer; +struct ggml_hexagon_session; +struct ggml_backend_hexagon_device_context; + +struct ggml_hexagon_mdev_group { + uint32_t idx = 0; + uint32_t count = 1; + std::vector> sessions; +}; + +struct ggml_backend_hexagon_comm_context { + std::vector backends; + size_t n_backends = 0; + volatile uint32_t * fence_slots[GGML_HEXAGON_MAX_SESSIONS] = {}; + ggml_tensor fence_tensors[GGML_HEXAGON_MAX_SESSIONS] = {}; +}; + +struct ggml_hexagon_event { + ggml_hexagon_session * sess = nullptr; + ggml_hexagon_session * fence_sess = nullptr; + volatile uint32_t * fence_slot = nullptr; + ggml_tensor fence_tensor = {}; + uint32_t seq = 0; +}; struct ggml_hexagon_session { std::string name; @@ -263,78 +461,178 @@ struct ggml_hexagon_session { uint32_t session_id; uint32_t domain_id; uint64_t queue_id; - int dev_id; + int phys_idx; + int virt_idx; bool valid_session; bool valid_handle; bool valid_queue; bool valid_iface; - std::atomic op_pending; ggml_hexagon_opbatch* op_batch; ggml_hexagon_opqueue* op_queue; - ggml_backend_buffer_type buffer_type = {}; - ggml_backend_buffer_type repack_buffer_type = {}; + std::unordered_map> cloned_buffers; + std::unordered_set virt_peers; + std::unordered_set phys_peers; - uint32_t n_threads = 0; - uint32_t n_hvx = 0; - uint32_t n_hmx = 0; - uint64_t vtcm_size = 0; - size_t max_vmem = 0; - size_t max_bufsize = 0; + uint32_t n_threads = 0; + uint32_t n_hvx = 0; + uint32_t n_hmx = 0; + uint64_t vtcm_size = 0; + size_t max_vmem = 0; + uint32_t fence_seq = 0; - struct { - uint64_t uid = 0; - std::vector htp_nodes; - } cached_graph; + std::atomic batch_req_seq{0}; + std::atomic batch_rsp_seq{0}; + std::atomic last_error{HTP_STATUS_OK}; - ggml_hexagon_session(int dev_id, ggml_backend_dev_t dev) noexcept(false); + uint64_t cached_uid = 0; + std::vector cached_nodes; + + mutable std::unordered_set needs_repack; + + ggml_hexagon_mdev_group mdev; + ggml_backend_dev_t dev = nullptr; + ggml_backend_hexagon_device_context * dev_ctx = nullptr; + ggml_hexagon_fence_buffer * fence_buf = nullptr; + + ggml_hexagon_session(const ggml_hexagon_device_config & config, ggml_backend_dev_t dev = nullptr, uint32_t mdev_idx = 0, uint32_t mdev_count = 0) noexcept(false); ~ggml_hexagon_session() noexcept(true); const char* c_name() const { return name.c_str(); } - void allocate(int dev_id) noexcept(false); + void allocate(const ggml_hexagon_device_config & config) noexcept(false); void release() noexcept(true); - void enqueue_op(const htp_opnode & node); - void flush(bool all = true); + uint8_t * alloc_fence(uint32_t n_slots = 1); + void free_fence(void * ptr, uint32_t n_slots = 1); - void flush_pending(bool all = false); - void flush_batch(); + uint8_t * mdev_fence_slot = nullptr; + std::unordered_map cpy_fence_slots; + + void enqueue_mdev_group(); + void enqueue_op(const htp_opnode & node); + void enqueue_cpy(const ggml_tensor * src, ggml_tensor * dst, const ggml_tensor * sync_tensor = nullptr, uint32_t fence_seq = 0); + void enqueue_fence(const ggml_tensor * sync_tensor, uint32_t fence_seq = 0, bool wait = true); + void enqueue_allreduce(const ggml_tensor * dst, const std::vector & src_tensors, + const std::vector & sync_tensors, uint32_t rank, uint32_t n_ranks, + uint32_t fence_seq_entry = 0, uint32_t fence_seq_exit = 0); + + void start_batch(); + void flush_sync(bool all = true); + void flush_async(); + void flush_batch(size_t min_ops = 1); + void flush_peers(); + void flush_pending(bool all = true); + + ggml_hexagon_shared_buffer * mmap_tensor(const ggml_tensor * t); + bool clone_buffer(const ggml_hexagon_shared_buffer*); + void release_buffer(const ggml_hexagon_shared_buffer*); + void unclone_buffer(const ggml_hexagon_shared_buffer*); + + void add_peer(ggml_hexagon_session * peer) { + if (this->phys_idx == peer->phys_idx) { + virt_peers.insert(peer); + } else { + phys_peers.insert(peer); + } + } }; // ** backend buffers +struct ggml_backend_hexagon_device_context { + int dev_id; + ggml_hexagon_device_config config; + ggml_backend_dev_t dev = nullptr; + + ggml_backend_buffer_type buffer_type = {}; + ggml_backend_buffer_type host_buffer_type = {}; + ggml_backend_buffer_type fence_buffer_type = {}; + + std::unique_ptr sess; + + ggml_backend_hexagon_device_context(int dev_id, const ggml_hexagon_device_config & config, ggml_backend_dev_t dev); + ~ggml_backend_hexagon_device_context(); + + const char * c_name() const { return config.name.c_str(); } + + ggml_hexagon_session * session() { + if (!sess) { + sess = std::make_unique(config, dev); + } + return sess.get(); + } +}; + struct ggml_backend_hexagon_buffer_type_context { - ggml_backend_hexagon_buffer_type_context(const std::string & name, ggml_hexagon_session * sess) { - this->sess = sess; - this->name = name; + ggml_backend_hexagon_buffer_type_context(const std::string & name, ggml_backend_hexagon_device_context * dev_ctx) { + this->dev_ctx = dev_ctx; + this->name = name; + } + + ggml_backend_hexagon_device_context * dev_ctx; + std::string name; +}; + +struct ggml_hexagon_rpcmem_block { + uint8_t * base = nullptr; + int fd = -1; + size_t size = 0; + + std::unordered_set mapped_clones; + + ggml_hexagon_rpcmem_block(size_t size) { + base = (uint8_t *) rpcmem_alloc2(RPCMEM_HEAP_ID_SYSTEM, RPCMEM_DEFAULT_FLAGS, size); + if (!base) { + throw std::runtime_error("ggml-hex: rpcmem_alloc failed"); + } + fd = rpcmem_to_fd(base); + if (fd < 0) { + rpcmem_free(base); + throw std::runtime_error("ggml-hex: rpcmem_to_fd failed"); + } + this->size = size; } - ggml_hexagon_session * sess; - std::string name; + ~ggml_hexagon_rpcmem_block() { + if (base) { + rpcmem_free(base); + } + } }; struct ggml_hexagon_shared_buffer { - ggml_hexagon_session * sess; - uint8_t * base; - size_t size; - int fd; - bool mapped; - bool pinned; + ggml_hexagon_session * sess; + std::shared_ptr mem; + std::vector tensor_extra; + bool mapped; + bool pinned; + bool extended; + + const char * c_name() const { return sess->c_name(); } + uint8_t * base() const { return mem ? mem->base : nullptr; } + size_t size() const { return mem ? mem->size : 0; } + int fd() const { return mem ? mem->fd : -1; } + + void mmap(bool extended = false) { + if (!this->mem) return; + if (this->mapped) return; - void mmap() { - fastrpc_map_flags flags = this->pinned ? FASTRPC_MAP_FD : FASTRPC_MAP_FD_DELAYED; + GGML_ASSERT(!this->pinned || !extended); - int err = fastrpc_mmap(sess->domain_id, this->fd, (void *) this->base, 0, this->size, flags); + this->extended = extended; + fastrpc_map_flags flags = this->pinned ? FASTRPC_MAP_FD : (extended ? FASTRPC_MAP_FD_DELAYED_EXTENDED : FASTRPC_MAP_FD_DELAYED); + + int err = fastrpc_mmap(sess->domain_id, fd(), (void *) base(), 0, size(), flags); if (err != 0) { GGML_LOG_ERROR("ggml-hex: %s buffer mapping failed : domain_id %d size %zu fd %d error 0x%08x\n", sess->c_name(), - sess->domain_id, this->size, this->fd, (unsigned) err); + sess->domain_id, size(), fd(), (unsigned) err); throw std::runtime_error("ggml-hex: fastrpc_mmap failed (see log for details)"); } - HEX_VERBOSE("ggml-hex: %s mapped buffer: base %p size %zu fd %d pinned %u\n", - sess->c_name(), (void *) this->base, this->size, this->fd, pinned); + HEX_VERBOSE("ggml-hex: %s mapped buffer: base %p size %zu fd %d pinned %u extended %u\n", + sess->c_name(), (void *) base(), size(), fd(), pinned, extended); this->mapped = true; } @@ -342,92 +640,156 @@ struct ggml_hexagon_shared_buffer { void unmap() { if (!this->mapped) return; - if (!this->pinned) { + if (!this->pinned && mem) { // HTP might still hold a reference, tell it drop it - htp_iface_munmap(sess->handle, this->fd); + htp_iface_munmap(sess->handle, fd()); } - fastrpc_munmap(sess->domain_id, this->fd, (void *) this->base, this->size); + if (mem) { + fastrpc_munmap(sess->domain_id, fd(), (void *) base(), size()); + } HEX_VERBOSE("ggml-hex: %s unmapped buffer: base %p size %zu fd %d\n", sess->c_name(), - (void *) this->base, size, this->fd); + (void *) base(), size(), fd()); this->mapped = false; - this->fd = -1; } void alloc(size_t size) { - if (this->base) return; - - this->base = (uint8_t *) rpcmem_alloc2(RPCMEM_HEAP_ID_SYSTEM, RPCMEM_DEFAULT_FLAGS, size); - if (!this->base) { - GGML_LOG_ERROR("ggml-hex: %s failed to allocate buffer : size %zu\n", sess->c_name(), size); - throw std::runtime_error("ggml-hex: rpcmem_alloc failed (see log for details)"); - } + if (this->mem) return; - this->fd = rpcmem_to_fd(this->base); - if (this->fd < 0) { - GGML_LOG_ERROR("ggml-hex: %s failed to get FD for buffer %p\n", sess->c_name(), (void *) this->base); - throw std::runtime_error("ggml-hex: rpcmem_to_fd failed (see log for details)"); - } - this->size = size; + this->mem = std::make_shared(size); HEX_VERBOSE("ggml-hex: %s allocated buffer: base %p size %zu fd %d pinned %d\n", sess->c_name(), - (void *) this->base, this->size, this->fd, (int) pinned); - mmap(); + (void *) base(), this->size(), fd(), (int) pinned); + if (this->pinned) { + mmap(); + } } void free() { - if (!this->base) return; - unmap(); - rpcmem_free(this->base); + // The memory is freed when the shared_ptr refcount drops to 0. + HEX_VERBOSE("ggml-hex: %s release ref on buffer: base %p size %zu fd %d\n", sess->c_name(), + (void *) base(), size(), fd()); + this->mem = nullptr; + } - HEX_VERBOSE("ggml-hex: %s freed buffer: base %p size %zu fd %d\n", sess->c_name(), - (void *) this->base, size, this->fd); + ggml_hexagon_shared_buffer(ggml_hexagon_session * sess, size_t size, bool pinned = false) { + this->sess = sess; + this->mapped = false; + this->pinned = pinned; + this->extended = false; + + // Size adjustment inside the buffer class: 4K aligned data size + 4K guard page + size_t guard_offset = (size + 4095) & ~4095; + size_t total_size = guard_offset + 4096; + if (!pinned && opt_dma64) { + constexpr size_t extended_align = 2 * 1024 * 1024; + total_size = (total_size + extended_align - 1) & ~(extended_align - 1); + } - this->base = NULL; + alloc(total_size); } - ggml_hexagon_shared_buffer(ggml_hexagon_session * sess, size_t size, bool pinned = false) { + // Clone constructor for cross-session mapping + ggml_hexagon_shared_buffer(ggml_hexagon_session * sess, const ggml_hexagon_shared_buffer & other) { this->sess = sess; - this->size = 0; - this->base = nullptr; - this->fd = -1; + this->mem = other.mem; this->mapped = false; - this->pinned = pinned; - - alloc(size); + this->pinned = other.pinned; + this->extended = other.extended; } ~ggml_hexagon_shared_buffer() { free(); + for (auto * extra : tensor_extra) { + delete extra; + } + } +}; + +struct ggml_hexagon_fence_buffer : public ggml_hexagon_shared_buffer { + uint32_t slot_count = 0; + uint32_t slot_head = 0; + std::vector free_slots; + ggml_backend_buffer backend_buffer{}; + + ggml_hexagon_fence_buffer(ggml_hexagon_session * sess, ggml_backend_buffer_type_t buft, size_t size) + : ggml_hexagon_shared_buffer(sess, size, false /* pinned */), + slot_count(size / GGML_HEXAGON_FENCE_SLOT_SIZE), + slot_head(0) { + backend_buffer.buft = buft; + backend_buffer.context = static_cast(this); + backend_buffer.size = size; + mmap(false); + } + + uint8_t * alloc_slot(uint32_t n_slots = 1) { + uint8_t * ptr = nullptr; + if (n_slots == 1 && !free_slots.empty()) { + uint32_t slot = free_slots.back(); + free_slots.pop_back(); + ptr = base() + (size_t) slot * GGML_HEXAGON_FENCE_SLOT_SIZE; + } else if (slot_head + n_slots <= slot_count) { + uint32_t slot = slot_head; + slot_head += n_slots; + ptr = base() + (size_t) slot * GGML_HEXAGON_FENCE_SLOT_SIZE; + } + if (ptr) { + memset(ptr, 0, (size_t) n_slots * GGML_HEXAGON_FENCE_SLOT_SIZE); + } + return ptr; + } + + void free_slot(void * ptr, uint32_t n_slots = 1) { + if (!ptr) return; + uint32_t slot = ((uint8_t *) ptr - base()) / GGML_HEXAGON_FENCE_SLOT_SIZE; + for (uint32_t i = 0; i < n_slots; i++) { + free_slots.push_back(slot + i); + } } }; -static ggml_hexagon_session * ggml_backend_hexagon_buffer_get_sess(ggml_backend_buffer_t buffer) { - return static_cast(buffer->buft->context)->sess; +inline uint8_t * ggml_hexagon_session::alloc_fence(uint32_t n_slots) { + uint8_t * ptr = fence_buf->alloc_slot(n_slots); + GGML_ASSERT(ptr); + return ptr; +} + +inline void ggml_hexagon_session::free_fence(void * ptr, uint32_t n_slots) { + if (fence_buf) { + fence_buf->free_slot(ptr, n_slots); + } } static void ggml_backend_hexagon_buffer_free_buffer(ggml_backend_buffer_t buffer) { auto sbuf = static_cast(buffer->context); + sbuf->sess->unclone_buffer(sbuf); delete sbuf; } static void * ggml_backend_hexagon_buffer_get_base(ggml_backend_buffer_t buffer) { auto sbuf = static_cast(buffer->context); - return sbuf->base; + return sbuf->base(); } static enum ggml_status ggml_backend_hexagon_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor) { auto sbuf = static_cast(buffer->context); auto sess = sbuf->sess; - HEX_VERBOSE("ggml-hex: %s init-tensor %s : base %p data %p nbytes %zu usage %d\n", sess->c_name(), - tensor->name, (void *) sbuf->base, tensor->data, ggml_nbytes(tensor), (int) buffer->usage); + HEX_VERBOSE("ggml-hex: %s init-tensor %s : base %p data %p nbytes %zu\n", sess->c_name(), + tensor->name, (void *) sbuf->base(), tensor->data, ggml_nbytes(tensor)); + + auto extra = new ggml_hexagon_tensor_extra(); + sbuf->tensor_extra.push_back(extra); - if (tensor->view_src != NULL && tensor->view_offs == 0) { - return GGML_STATUS_SUCCESS; // nothing to do for the view + tensor->extra = extra; + if (ggml_hexagon_is_repack_type(tensor->type)) { + if (sess->needs_repack.count(tensor)) { + extra->flags |= GGML_HEXAGON_TENSOR_REPACK; + sess->needs_repack.erase(tensor); + } } return GGML_STATUS_SUCCESS; @@ -499,7 +861,7 @@ static void pack_mxfp4_quants(block_mxfp4 * x, const uint8_t * qs, unsigned int } // repack q4_0 data into q4_0_tiled tensor -static void repack_q4_0_tiled(ggml_tensor * t, const void * data, size_t size) { +static void repack_q4_0_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { const block_q4_0 * src_matrix = (const block_q4_0 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -513,46 +875,49 @@ static void repack_q4_0_tiled(ggml_tensor * t, const void * data, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_0; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - const block_q4_0 * src_expert = src_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + const block_q4_0 * src_slice = src_matrix + (slice_idx - start_slice) * (ne1 * (ne0 / 32)); + uint8_t * matrix_dst = (uint8_t *) t->data + slice_idx * matrix_size; - uint8_t tile_quants[32][32]; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - unpack_q4_0_quants(tile_quants[row], &src_expert[r * (ne0 / 32) + kt], 0); - } else { - memset(tile_quants[row], 8, 32); - } - } + for (int ct = 0; ct < n_col_tiles; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; - } + uint8_t tile_quants[32][32]; + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r < ne1 && kt < ne0 / 32) { + unpack_q4_0_quants(tile_quants[row], &src_slice[r * (ne0 / 32) + kt], 0); + } else { + memset(tile_quants[row], 8, 32); } + } - ggml_half * scale_dst = (ggml_half *)(tile_dst + 512); + for (int cp = 0; cp < 16; cp++) { for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_expert[r * (ne0 / 32) + kt].d : 0; + tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; } } + + ggml_half * scale_dst = (ggml_half *)(tile_dst + 512); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_slice[r * (ne0 / 32) + kt].d : 0; + } } } } - - GGML_UNUSED(size); } // repack q4_0_tiled tensor into q4_0 data -static void repack_tiled_q4_0(void * data, const ggml_tensor * t, size_t size) { +static void repack_tiled_q4_0(void * data, const ggml_tensor * t, size_t offset, size_t size) { block_q4_0 * dst_matrix = (block_q4_0 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -566,48 +931,65 @@ static void repack_tiled_q4_0(void * data, const ggml_tensor * t, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_0; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - block_q4_0 * dst_expert = dst_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + size_t row_size_bytes = ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + size_t cur_start_byte = (std::max)(offset, (size_t) slice_idx * slice_size); + size_t cur_end_byte = (std::min)(offset + size, (size_t) (slice_idx + 1) * slice_size); + size_t slice_offset_start = cur_start_byte - (size_t) slice_idx * slice_size; + size_t slice_offset_end = cur_end_byte - (size_t) slice_idx * slice_size; - uint8_t tile_quants[32][32]; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - uint8_t val = tile_src[cp * 32 + row]; - tile_quants[row][2 * cp + 0] = val & 0x0F; - tile_quants[row][2 * cp + 1] = val >> 4; - } - } + int64_t start_row = slice_offset_start / row_size_bytes; + int64_t end_row = (slice_offset_end + row_size_bytes - 1) / row_size_bytes; + end_row = (std::min)(end_row, ne1); + + int start_ct = start_row / 32; + int end_ct = (end_row + 31) / 32; + end_ct = (std::min)(end_ct, n_col_tiles); + + block_q4_0 * dst_slice = dst_matrix + (cur_start_byte - offset) / sizeof(block_q4_0); + const uint8_t * matrix_src = (const uint8_t *) t->data + slice_idx * matrix_size; + + for (int ct = start_ct; ct < end_ct; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + uint8_t tile_quants[32][32]; + for (int cp = 0; cp < 16; cp++) { for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - pack_q4_0_quants(&dst_expert[r * (ne0 / 32) + kt], tile_quants[row], 0); - } + uint8_t val = tile_src[cp * 32 + row]; + tile_quants[row][2 * cp + 0] = val & 0x0F; + tile_quants[row][2 * cp + 1] = val >> 4; } + } - const ggml_half * scale_src = (const ggml_half *)(tile_src + 512); - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - dst_expert[r * (ne0 / 32) + kt].d = scale_src[row]; - } + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + pack_q4_0_quants(&dst_slice[(r - start_row) * (ne0 / 32) + kt], tile_quants[row], 0); + } + } + + const ggml_half * scale_src = (const ggml_half *)(tile_src + 512); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + dst_slice[(r - start_row) * (ne0 / 32) + kt].d = scale_src[row]; } } } } } - - GGML_UNUSED(size); } // repack q4_1 data into q4_1_tiled tensor -static void repack_q4_1_tiled(ggml_tensor * t, const void * data, size_t size) { +static void repack_q4_1_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { const block_q4_1 * src_matrix = (const block_q4_1 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -621,52 +1003,55 @@ static void repack_q4_1_tiled(ggml_tensor * t, const void * data, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_1; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - const block_q4_1 * src_expert = src_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + const block_q4_1 * src_slice = src_matrix + (slice_idx - start_slice) * (ne1 * (ne0 / 32)); + uint8_t * matrix_dst = (uint8_t *) t->data + slice_idx * matrix_size; - uint8_t tile_quants[32][32]; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - unpack_q4_1_quants(tile_quants[row], &src_expert[r * (ne0 / 32) + kt], 0); - } else { - memset(tile_quants[row], 0, 32); - } - } + for (int ct = 0; ct < n_col_tiles; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; - } + uint8_t tile_quants[32][32]; + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r < ne1 && kt < ne0 / 32) { + unpack_q4_1_quants(tile_quants[row], &src_slice[r * (ne0 / 32) + kt], 0); + } else { + memset(tile_quants[row], 0, 32); } + } - ggml_half * scale_dst = (ggml_half *)(tile_dst + 512); + for (int cp = 0; cp < 16; cp++) { for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - scale_dst[2 * row + 0] = src_expert[r * (ne0 / 32) + kt].d; - scale_dst[2 * row + 1] = src_expert[r * (ne0 / 32) + kt].m; - } else { - scale_dst[2 * row + 0] = 0; - scale_dst[2 * row + 1] = 0; - } + tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; + } + } + + ggml_half * scale_dst = (ggml_half *)(tile_dst + 512); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r < ne1 && kt < ne0 / 32) { + scale_dst[2 * row + 0] = src_slice[r * (ne0 / 32) + kt].d; + scale_dst[2 * row + 1] = src_slice[r * (ne0 / 32) + kt].m; + } else { + scale_dst[2 * row + 0] = 0; + scale_dst[2 * row + 1] = 0; } } } } } - - GGML_UNUSED(size); } // repack q4_1_tiled tensor into q4_1 data -static void repack_tiled_q4_1(void * data, const ggml_tensor * t, size_t size) { +static void repack_tiled_q4_1(void * data, const ggml_tensor * t, size_t offset, size_t size) { block_q4_1 * dst_matrix = (block_q4_1 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -680,49 +1065,66 @@ static void repack_tiled_q4_1(void * data, const ggml_tensor * t, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_1; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - block_q4_1 * dst_expert = dst_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + size_t row_size_bytes = ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + size_t cur_start_byte = (std::max)(offset, (size_t) slice_idx * slice_size); + size_t cur_end_byte = (std::min)(offset + size, (size_t) (slice_idx + 1) * slice_size); + size_t slice_offset_start = cur_start_byte - (size_t) slice_idx * slice_size; + size_t slice_offset_end = cur_end_byte - (size_t) slice_idx * slice_size; - uint8_t tile_quants[32][32]; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - uint8_t val = tile_src[cp * 32 + row]; - tile_quants[row][2 * cp + 0] = val & 0x0F; - tile_quants[row][2 * cp + 1] = val >> 4; - } - } + int64_t start_row = slice_offset_start / row_size_bytes; + int64_t end_row = (slice_offset_end + row_size_bytes - 1) / row_size_bytes; + end_row = (std::min)(end_row, ne1); + int start_ct = start_row / 32; + int end_ct = (end_row + 31) / 32; + end_ct = (std::min)(end_ct, n_col_tiles); + + block_q4_1 * dst_slice = dst_matrix + (cur_start_byte - offset) / sizeof(block_q4_1); + const uint8_t * matrix_src = (const uint8_t *) t->data + slice_idx * matrix_size; + + for (int ct = start_ct; ct < end_ct; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + + uint8_t tile_quants[32][32]; + for (int cp = 0; cp < 16; cp++) { for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - pack_q4_1_quants(&dst_expert[r * (ne0 / 32) + kt], tile_quants[row], 0); - } + uint8_t val = tile_src[cp * 32 + row]; + tile_quants[row][2 * cp + 0] = val & 0x0F; + tile_quants[row][2 * cp + 1] = val >> 4; } + } - const ggml_half * scale_src = (const ggml_half *)(tile_src + 512); - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - dst_expert[r * (ne0 / 32) + kt].d = scale_src[2 * row]; - dst_expert[r * (ne0 / 32) + kt].m = scale_src[2 * row + 1]; - } + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + pack_q4_1_quants(&dst_slice[(r - start_row) * (ne0 / 32) + kt], tile_quants[row], 0); + } + } + + const ggml_half * scale_src = (const ggml_half *)(tile_src + 512); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + dst_slice[(r - start_row) * (ne0 / 32) + kt].d = scale_src[2 * row]; + dst_slice[(r - start_row) * (ne0 / 32) + kt].m = scale_src[2 * row + 1]; } } } } } - - GGML_UNUSED(size); } // repack q8_0 data into q8_0_tiled tensor -static void repack_q8_0_tiled(ggml_tensor * t, const void * data, size_t size) { +static void repack_q8_0_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { const block_q8_0 * src_matrix = (const block_q8_0 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -736,41 +1138,44 @@ static void repack_q8_0_tiled(ggml_tensor * t, const void * data, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q8_0; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - const block_q8_0 * src_expert = src_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + const block_q8_0 * src_slice = src_matrix + (slice_idx - start_slice) * (ne1 * (ne0 / 32)); + uint8_t * matrix_dst = (uint8_t *) t->data + slice_idx * matrix_size; - for (int cp = 0; cp < 16; cp++) { - int col0 = cp * 2; - int col1 = col0 + 1; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - const block_q8_0 * b = (r < ne1 && kt < ne0 / 32) ? &src_expert[r * (ne0 / 32) + kt] : NULL; - tile_dst[cp * 64 + 2 * row + 0] = b ? b->qs[col0] : 0; - tile_dst[cp * 64 + 2 * row + 1] = b ? b->qs[col1] : 0; - } - } + for (int ct = 0; ct < n_col_tiles; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; - ggml_half * scale_dst = (ggml_half *)(tile_dst + 1024); + for (int cp = 0; cp < 16; cp++) { + int col0 = cp * 2; + int col1 = col0 + 1; for (int row = 0; row < 32; row++) { int64_t r = ct * 32 + row; - scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_expert[r * (ne0 / 32) + kt].d : 0; + const block_q8_0 * b = (r < ne1 && kt < ne0 / 32) ? &src_slice[r * (ne0 / 32) + kt] : NULL; + tile_dst[cp * 64 + 2 * row + 0] = b ? b->qs[col0] : 0; + tile_dst[cp * 64 + 2 * row + 1] = b ? b->qs[col1] : 0; } } + + ggml_half * scale_dst = (ggml_half *)(tile_dst + 1024); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_slice[r * (ne0 / 32) + kt].d : 0; + } } } } - - GGML_UNUSED(size); } // repack q8_0_tiled tensor into q8_0 data -static void repack_tiled_q8_0(void * data, const ggml_tensor * t, size_t size) { +static void repack_tiled_q8_0(void * data, const ggml_tensor * t, size_t offset, size_t size) { block_q8_0 * dst_matrix = (block_q8_0 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -784,45 +1189,62 @@ static void repack_tiled_q8_0(void * data, const ggml_tensor * t, size_t size) { const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q8_0; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - block_q8_0 * dst_expert = dst_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + size_t row_size_bytes = ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + size_t cur_start_byte = (std::max)(offset, (size_t) slice_idx * slice_size); + size_t cur_end_byte = (std::min)(offset + size, (size_t) (slice_idx + 1) * slice_size); + size_t slice_offset_start = cur_start_byte - (size_t) slice_idx * slice_size; + size_t slice_offset_end = cur_end_byte - (size_t) slice_idx * slice_size; - for (int cp = 0; cp < 16; cp++) { - int col0 = cp * 2; - int col1 = col0 + 1; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - block_q8_0 & b = dst_expert[r * (ne0 / 32) + kt]; - b.qs[col0] = tile_src[cp * 64 + 2 * row + 0]; - b.qs[col1] = tile_src[cp * 64 + 2 * row + 1]; - } - } - } + int64_t start_row = slice_offset_start / row_size_bytes; + int64_t end_row = (slice_offset_end + row_size_bytes - 1) / row_size_bytes; + end_row = (std::min)(end_row, ne1); + + int start_ct = start_row / 32; + int end_ct = (end_row + 31) / 32; + end_ct = (std::min)(end_ct, n_col_tiles); - const ggml_half * scale_src = (const ggml_half *)(tile_src + 1024); + block_q8_0 * dst_slice = dst_matrix + (cur_start_byte - offset) / sizeof(block_q8_0); + const uint8_t * matrix_src = (const uint8_t *) t->data + slice_idx * matrix_size; + + for (int ct = start_ct; ct < end_ct; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + + for (int cp = 0; cp < 16; cp++) { + int col0 = cp * 2; + int col1 = col0 + 1; for (int row = 0; row < 32; row++) { int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - dst_expert[r * (ne0 / 32) + kt].d = scale_src[row]; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + block_q8_0 & b = dst_slice[(r - start_row) * (ne0 / 32) + kt]; + b.qs[col0] = tile_src[cp * 64 + 2 * row + 0]; + b.qs[col1] = tile_src[cp * 64 + 2 * row + 1]; } } } + + const ggml_half * scale_src = (const ggml_half *)(tile_src + 1024); + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + dst_slice[(r - start_row) * (ne0 / 32) + kt].d = scale_src[row]; + } + } } } } - - GGML_UNUSED(size); } // repack mxfp4 data into mxfp4_tiled tensor -static void repack_mxfp4_tiled(ggml_tensor * t, const void * data, size_t size) { +static void repack_mxfp4_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { const block_mxfp4 * src_matrix = (const block_mxfp4 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -836,46 +1258,49 @@ static void repack_mxfp4_tiled(ggml_tensor * t, const void * data, size_t size) const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_MXFP4; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; - for (int i3 = 0; i3 < ne3; i3++) { - for (int i2 = 0; i2 < ne2; i2++) { - const block_mxfp4 * src_expert = src_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } - for (int ct = 0; ct < n_col_tiles; ct++) { - for (int kt = 0; kt < n_k_tiles; kt++) { - uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + const block_mxfp4 * src_slice = src_matrix + (slice_idx - start_slice) * (ne1 * (ne0 / 32)); + uint8_t * matrix_dst = (uint8_t *) t->data + slice_idx * matrix_size; - uint8_t tile_quants[32][32]; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - unpack_mxfp4_quants(tile_quants[row], &src_expert[r * (ne0 / 32) + kt], 0); - } else { - memset(tile_quants[row], 0, 32); - } - } + for (int ct = 0; ct < n_col_tiles; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + uint8_t * tile_dst = matrix_dst + (ct * n_k_tiles + kt) * tile_size; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; - } + uint8_t tile_quants[32][32]; + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r < ne1 && kt < ne0 / 32) { + unpack_mxfp4_quants(tile_quants[row], &src_slice[r * (ne0 / 32) + kt], 0); + } else { + memset(tile_quants[row], 0, 32); } + } - uint8_t * scale_dst = tile_dst + 512; + for (int cp = 0; cp < 16; cp++) { for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_expert[r * (ne0 / 32) + kt].e : 0; + tile_dst[cp * 32 + row] = (tile_quants[row][2 * cp + 1] << 4) | tile_quants[row][2 * cp]; } } + + uint8_t * scale_dst = tile_dst + 512; + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + scale_dst[row] = (r < ne1 && kt < ne0 / 32) ? src_slice[r * (ne0 / 32) + kt].e : 0; + } } } } - - GGML_UNUSED(size); } // repack mxfp4_tiled tensor into mxfp4 data -static void repack_tiled_mxfp4(void * data, const ggml_tensor * t, size_t size) { +static void repack_tiled_mxfp4(void * data, const ggml_tensor * t, size_t offset, size_t size) { block_mxfp4 * dst_matrix = (block_mxfp4 *) data; int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; @@ -889,36 +1314,240 @@ static void repack_tiled_mxfp4(void * data, const ggml_tensor * t, size_t size) const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_MXFP4; const size_t matrix_size = n_col_tiles * n_k_tiles * tile_size; + size_t slice_size = ne1 * ggml_row_size(t->type, ne0); + size_t row_size_bytes = ggml_row_size(t->type, ne0); + int64_t start_slice = offset / slice_size; + int64_t end_slice = (offset + size + slice_size - 1) / slice_size; + if (end_slice > ne2 * ne3) { + end_slice = ne2 * ne3; + } + + for (int64_t slice_idx = start_slice; slice_idx < end_slice; slice_idx++) { + size_t cur_start_byte = (std::max)(offset, (size_t) slice_idx * slice_size); + size_t cur_end_byte = (std::min)(offset + size, (size_t) (slice_idx + 1) * slice_size); + size_t slice_offset_start = cur_start_byte - (size_t) slice_idx * slice_size; + size_t slice_offset_end = cur_end_byte - (size_t) slice_idx * slice_size; + + int64_t start_row = slice_offset_start / row_size_bytes; + int64_t end_row = (slice_offset_end + row_size_bytes - 1) / row_size_bytes; + end_row = (std::min)(end_row, ne1); + + int start_ct = start_row / 32; + int end_ct = (end_row + 31) / 32; + end_ct = (std::min)(end_ct, n_col_tiles); + + block_mxfp4 * dst_slice = dst_matrix + (cur_start_byte - offset) / sizeof(block_mxfp4); + const uint8_t * matrix_src = (const uint8_t *) t->data + slice_idx * matrix_size; + + for (int ct = start_ct; ct < end_ct; ct++) { + for (int kt = 0; kt < n_k_tiles; kt++) { + const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + + uint8_t tile_quants[32][32]; + for (int cp = 0; cp < 16; cp++) { + for (int row = 0; row < 32; row++) { + uint8_t val = tile_src[cp * 32 + row]; + tile_quants[row][2 * cp + 0] = val & 0x0F; + tile_quants[row][2 * cp + 1] = val >> 4; + } + } + + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + pack_mxfp4_quants(&dst_slice[(r - start_row) * (ne0 / 32) + kt], tile_quants[row], 0); + } + } + + const uint8_t * scale_src = tile_src + 512; + for (int row = 0; row < 32; row++) { + int64_t r = ct * 32 + row; + if (r >= start_row && r < end_row && kt < ne0 / 32) { + dst_slice[(r - start_row) * (ne0 / 32) + kt].e = scale_src[row]; + } + } + } + } + } +} + +// unsigned 6-bit value (0..63) of element e of a Q6_K block, same bit layout as dequantize_row_q6_K +static inline uint8_t q6_K_get_quant(const block_q6_K * b, int e) { + const int c = e / 128; + const int w = e % 128; + const int g = w / 32; + const int l = w % 32; + const uint8_t * ql = b->ql + c * 64; + const uint8_t * qh = b->qh + c * 32; + uint8_t lo, hi; + switch (g) { + case 0: lo = ql[l] & 0xF; hi = (qh[l] >> 0) & 3; break; + case 1: lo = ql[l + 32] & 0xF; hi = (qh[l] >> 2) & 3; break; + case 2: lo = ql[l] >> 4; hi = (qh[l] >> 4) & 3; break; + default: lo = ql[l + 32] >> 4; hi = (qh[l] >> 6) & 3; break; + } + return (uint8_t) (lo | (hi << 4)); +} + +// tile layout: see HTP_MM_WEIGHT_TILE_SIZE_Q6_K in htp/matmul-ops.h +static void repack_q6_K_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { + GGML_ASSERT(offset == 0); + + const block_q6_K * src_matrix = (const block_q6_K *) data; + int64_t ne0 = t->ne[0]; + int64_t ne1 = t->ne[1]; + int64_t ne2 = t->ne[2]; + int64_t ne3 = t->ne[3]; + int64_t ne0_padded = hex_round_up(ne0, 32); + int64_t ne1_padded = hex_round_up(ne1, 32); + + GGML_ASSERT(ne0 % QK_K == 0); + + const int n_col_tiles = ne1_padded / 32; + const int n_k_tiles = ne0_padded / 32; + const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q6_K; + const size_t matrix_size = (size_t) n_col_tiles * n_k_tiles * tile_size; + + const int64_t sb_per_row = ne0 / QK_K; + for (int i3 = 0; i3 < ne3; i3++) { for (int i2 = 0; i2 < ne2; i2++) { - block_mxfp4 * dst_expert = dst_matrix + (i3 * ne2 + i2) * (ne1 * (ne0 / 32)); - const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + const block_q6_K * src_slice = src_matrix + (i3 * ne2 + i2) * (ne1 * sb_per_row); + uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + + memset(matrix_dst, 0, matrix_size); // padding rows and the OR-ed nibbles below need zeroed tiles + + for (int64_t r = 0; r < ne1; r++) { + const int ct = (int) (r / 32); + const int row = (int) (r % 32); + const block_q6_K * src_row = src_slice + r * sb_per_row; - for (int ct = 0; ct < n_col_tiles; ct++) { for (int kt = 0; kt < n_k_tiles; kt++) { - const uint8_t * tile_src = matrix_src + (ct * n_k_tiles + kt) * tile_size; + const int kt_local = kt % 8; // k-tile within the super-block + const block_q6_K * b = &src_row[kt / 8]; + const float d = GGML_FP16_TO_FP32(b->d); + + uint8_t * tile = matrix_dst + ((size_t) ct * n_k_tiles + kt) * tile_size; + uint8_t * lo_pl = tile; + uint8_t * hi_pl = tile + 512; + ggml_half * sc_pl = (ggml_half *) (tile + 768); + + for (int lk = 0; lk < 32; lk++) { + const uint8_t q6 = q6_K_get_quant(b, kt_local * 32 + lk); + const int g = lk >> 2; + const int pos = row * 4 + (lk & 3); + lo_pl[(g >> 1) * 128 + pos] |= (uint8_t) ((q6 & 0xF) << ((g & 1) * 4)); + hi_pl[(g >> 2) * 128 + pos] |= (uint8_t) ((q6 >> 4) << ((g & 3) * 2)); + } + for (int sub = 0; sub < 2; sub++) { + sc_pl[sub * 32 + row] = GGML_FP32_TO_FP16(d * (float) b->scales[kt_local * 2 + sub]); + } + } + } + } + } - uint8_t tile_quants[32][32]; - for (int cp = 0; cp < 16; cp++) { - for (int row = 0; row < 32; row++) { - uint8_t val = tile_src[cp * 32 + row]; - tile_quants[row][2 * cp + 0] = val & 0x0F; - tile_quants[row][2 * cp + 1] = val >> 4; + GGML_UNUSED(size); +} + +// Reverse of repack_q6_K_tiled. Unpacks quants losslessly and normalizes sub-block scales. Read-back only. +static void repack_tiled_q6_K(void * data, const ggml_tensor * t, size_t offset, size_t size) { + GGML_ASSERT(offset == 0); + + block_q6_K * dst_matrix = (block_q6_K *) data; + int64_t ne0 = t->ne[0]; + int64_t ne1 = t->ne[1]; + int64_t ne2 = t->ne[2]; + int64_t ne3 = t->ne[3]; + int64_t ne0_padded = hex_round_up(ne0, 32); + int64_t ne1_padded = hex_round_up(ne1, 32); + + GGML_ASSERT(ne0 % QK_K == 0); + + const int n_col_tiles = ne1_padded / 32; + const int n_k_tiles = ne0_padded / 32; + const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q6_K; + const size_t matrix_size = (size_t) n_col_tiles * n_k_tiles * tile_size; + + const int64_t sb_per_row = ne0 / QK_K; + + for (int i3 = 0; i3 < ne3; i3++) { + for (int i2 = 0; i2 < ne2; i2++) { + block_q6_K * dst_slice = dst_matrix + (i3 * ne2 + i2) * (ne1 * sb_per_row); + const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + + for (int64_t r = 0; r < ne1; r++) { + const int ct = (int) (r / 32); + const int row = (int) (r % 32); + block_q6_K * dst_row = dst_slice + r * sb_per_row; + + for (int64_t sb = 0; sb < sb_per_row; sb++) { + block_q6_K * b = &dst_row[sb]; + memset(b, 0, sizeof(block_q6_K)); + + float sub_scales[16]; + for (int kt_local = 0; kt_local < 8; kt_local++) { + const int kt = sb * 8 + kt_local; + const uint8_t * tile = matrix_src + ((size_t) ct * n_k_tiles + kt) * tile_size; + const uint8_t * lo_pl = tile; + const uint8_t * hi_pl = tile + 512; + const ggml_half * sc_pl = (const ggml_half *) (tile + 768); + + const int c = kt_local / 4; + const int g = kt_local % 4; + uint8_t * ql = b->ql + c * 64; + uint8_t * qh = b->qh + c * 32; + + for (int lk = 0; lk < 32; lk++) { + const int g_tile = lk >> 2; + const int pos = row * 4 + (lk & 3); + const uint8_t lo = (lo_pl[(g_tile >> 1) * 128 + pos] >> ((g_tile & 1) * 4)) & 0xF; + const uint8_t hi = (hi_pl[(g_tile >> 2) * 128 + pos] >> ((g_tile & 3) * 2)) & 3; + + switch (g) { + case 0: + ql[lk] |= lo; + qh[lk] |= (hi << 0); + break; + case 1: + ql[lk + 32] |= lo; + qh[lk] |= (hi << 2); + break; + case 2: + ql[lk] |= (lo << 4); + qh[lk] |= (hi << 4); + break; + default: + ql[lk + 32] |= (lo << 4); + qh[lk] |= (hi << 6); + break; + } + } + + for (int sub = 0; sub < 2; sub++) { + sub_scales[kt_local * 2 + sub] = GGML_FP16_TO_FP32(sc_pl[sub * 32 + row]); } } - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - pack_mxfp4_quants(&dst_expert[r * (ne0 / 32) + kt], tile_quants[row], 0); + float max_abs_scale = 0.0f; + for (int s = 0; s < 16; s++) { + float abs_scale = fabsf(sub_scales[s]); + if (abs_scale > max_abs_scale) { + max_abs_scale = abs_scale; } } - const uint8_t * scale_src = tile_src + 512; - for (int row = 0; row < 32; row++) { - int64_t r = ct * 32 + row; - if (r < ne1 && kt < ne0 / 32) { - dst_expert[r * (ne0 / 32) + kt].e = scale_src[row]; + if (max_abs_scale == 0.0f) { + b->d = GGML_FP32_TO_FP16(0.0f); + memset(b->scales, 0, sizeof(b->scales)); + } else { + float d_flt = max_abs_scale / 127.0f; + b->d = GGML_FP32_TO_FP16(d_flt); + float d_actual = GGML_FP16_TO_FP32(b->d); + float inv_d = (d_actual != 0.0f) ? (1.0f / d_actual) : 0.0f; + for (int s = 0; s < 16; s++) { + int sc = (int) roundf(sub_scales[s] * inv_d); + b->scales[s] = (int8_t) (std::max)(-128, (std::min)(127, sc)); } } } @@ -929,93 +1558,326 @@ static void repack_tiled_mxfp4(void * data, const ggml_tensor * t, size_t size) GGML_UNUSED(size); } -static void ggml_backend_hexagon_buffer_set_tensor(ggml_backend_buffer_t buffer, - ggml_tensor * tensor, - const void * data, - size_t offset, - size_t size) { - auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; - auto sess = sbuf->sess; +static inline void get_scale_min_k4(int j, const uint8_t * q, uint8_t * d, uint8_t * m) { + if (j < 4) { + *d = q[j] & 63; + *m = q[j + 4] & 63; + } else { + *d = (q[j + 4] & 0xF) | ((q[j - 4] >> 6) << 4); + *m = (q[j + 4] >> 4) | ((q[j - 0] >> 6) << 4); + } +} + +// tile layout: see HTP_MM_WEIGHT_TILE_SIZE_Q4_1 in htp/matmul-ops.h +static void repack_q4_K_tiled(ggml_tensor * t, const void * data, size_t offset, size_t size) { + GGML_ASSERT(offset == 0); + + const block_q4_K * src_matrix = (const block_q4_K *) data; + int64_t ne0 = t->ne[0]; + int64_t ne1 = t->ne[1]; + int64_t ne2 = t->ne[2]; + int64_t ne3 = t->ne[3]; + int64_t ne0_padded = hex_round_up(ne0, 32); + int64_t ne1_padded = hex_round_up(ne1, 32); + + GGML_ASSERT(ne0 % QK_K == 0); + + const int n_col_tiles = ne1_padded / 32; + const int n_k_tiles = ne0_padded / 32; + const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_1; + const size_t matrix_size = (size_t) n_col_tiles * n_k_tiles * tile_size; + + const int64_t sb_per_row = ne0 / QK_K; + + for (int i3 = 0; i3 < ne3; i3++) { + for (int i2 = 0; i2 < ne2; i2++) { + const block_q4_K * src_slice = src_matrix + (i3 * ne2 + i2) * (ne1 * sb_per_row); + uint8_t * matrix_dst = (uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + + memset(matrix_dst, 0, matrix_size); + + for (int64_t r = 0; r < ne1; r++) { + const int ct = (int) (r / 32); + const int row = (int) (r % 32); + const block_q4_K * src_row = src_slice + r * sb_per_row; + + for (int kt = 0; kt < n_k_tiles; kt++) { + const int kt_local = kt % 8; + const block_q4_K * b = &src_row[kt / 8]; + const float d = GGML_FP16_TO_FP32(b->d); + const float dmin = GGML_FP16_TO_FP32(b->dmin); + + uint8_t * tile_dst = matrix_dst + ((size_t) ct * n_k_tiles + kt) * tile_size; + + uint8_t sc, m; + get_scale_min_k4(kt_local, b->scales, &sc, &m); + + const float D = d * (float) sc; + const float M = -dmin * (float) m; + + const uint8_t * qs_sub = b->qs + (kt_local / 2) * 32; + const int shift = (kt_local & 1) ? 4 : 0; + + for (int cp = 0; cp < 16; cp++) { + const uint8_t q0 = (qs_sub[2 * cp + 0] >> shift) & 0x0F; + const uint8_t q1 = (qs_sub[2 * cp + 1] >> shift) & 0x0F; + tile_dst[cp * 32 + row] = (uint8_t) ((q1 << 4) | q0); + } + + ggml_half * scale_dst = (ggml_half *) (tile_dst + 512); + scale_dst[2 * row + 0] = GGML_FP32_TO_FP16(D); + scale_dst[2 * row + 1] = GGML_FP32_TO_FP16(M); + } + } + } + } - HEX_VERBOSE("ggml-hex: %s set-tensor %s : data %p offset %zu size %zu\n", sess->c_name(), tensor->name, data, offset, size); + GGML_UNUSED(size); +} + +// Reverse of repack_q4_K_tiled. Unpacks quants and normalizes scales/mins. Read-back only. +static void repack_tiled_q4_K(void * data, const ggml_tensor * t, size_t offset, size_t size) { + GGML_ASSERT(offset == 0); + + block_q4_K * dst_matrix = (block_q4_K *) data; + int64_t ne0 = t->ne[0]; + int64_t ne1 = t->ne[1]; + int64_t ne2 = t->ne[2]; + int64_t ne3 = t->ne[3]; + int64_t ne0_padded = hex_round_up(ne0, 32); + int64_t ne1_padded = hex_round_up(ne1, 32); + + GGML_ASSERT(ne0 % QK_K == 0); + + const int n_col_tiles = ne1_padded / 32; + const int n_k_tiles = ne0_padded / 32; + const size_t tile_size = HTP_MM_WEIGHT_TILE_SIZE_Q4_1; + const size_t matrix_size = (size_t) n_col_tiles * n_k_tiles * tile_size; + + const int64_t sb_per_row = ne0 / QK_K; + + for (int i3 = 0; i3 < ne3; i3++) { + for (int i2 = 0; i2 < ne2; i2++) { + block_q4_K * dst_slice = dst_matrix + (i3 * ne2 + i2) * (ne1 * sb_per_row); + const uint8_t * matrix_src = (const uint8_t *) t->data + (i3 * ne2 + i2) * matrix_size; + + for (int64_t r = 0; r < ne1; r++) { + const int ct = (int) (r / 32); + const int row = (int) (r % 32); + block_q4_K * dst_row = dst_slice + r * sb_per_row; + + for (int64_t sb = 0; sb < sb_per_row; sb++) { + block_q4_K * b = &dst_row[sb]; + memset(b, 0, sizeof(block_q4_K)); + + float sub_scales[8]; + float sub_mins[8]; + + for (int kt_local = 0; kt_local < 8; kt_local++) { + const int kt = sb * 8 + kt_local; + const uint8_t * tile_src = matrix_src + ((size_t) ct * n_k_tiles + kt) * tile_size; + const ggml_half * scale_src = (const ggml_half *) (tile_src + 512); + + uint8_t * qs_sub = b->qs + (kt_local / 2) * 32; + const int shift = (kt_local & 1) ? 4 : 0; + + for (int cp = 0; cp < 16; cp++) { + const uint8_t val = tile_src[cp * 32 + row]; + const uint8_t q0 = val & 0x0F; + const uint8_t q1 = val >> 4; + qs_sub[2 * cp + 0] |= (uint8_t) (q0 << shift); + qs_sub[2 * cp + 1] |= (uint8_t) (q1 << shift); + } + + const float D = GGML_FP16_TO_FP32(scale_src[2 * row + 0]); + const float M = GGML_FP16_TO_FP32(scale_src[2 * row + 1]); + sub_scales[kt_local] = (D > 0.0f) ? D : 0.0f; + sub_mins[kt_local] = (-M > 0.0f) ? -M : 0.0f; + } + + float max_scale = 0.0f; + float max_min = 0.0f; + for (int j = 0; j < 8; j++) { + if (sub_scales[j] > max_scale) max_scale = sub_scales[j]; + if (sub_mins[j] > max_min) max_min = sub_mins[j]; + } + + float inv_scale = 0.0f; + if (max_scale > 0.0f) { + b->d = GGML_FP32_TO_FP16(max_scale / 63.0f); + const float d_actual = GGML_FP16_TO_FP32(b->d); + inv_scale = (d_actual > 0.0f) ? (1.0f / d_actual) : 0.0f; + } else { + b->d = GGML_FP32_TO_FP16(0.0f); + } + + float inv_min = 0.0f; + if (max_min > 0.0f) { + b->dmin = GGML_FP32_TO_FP16(max_min / 63.0f); + const float dmin_actual = GGML_FP16_TO_FP32(b->dmin); + inv_min = (dmin_actual > 0.0f) ? (1.0f / dmin_actual) : 0.0f; + } else { + b->dmin = GGML_FP32_TO_FP16(0.0f); + } + for (int j = 0; j < 8; j++) { + uint8_t ls = (uint8_t) roundf(inv_scale * sub_scales[j]); + uint8_t lm = (uint8_t) roundf(inv_min * sub_mins[j]); + ls = (std::min)((uint8_t) 63, ls); + lm = (std::min)((uint8_t) 63, lm); + if (j < 4) { + b->scales[j] = ls; + b->scales[j + 4] = lm; + } else { + b->scales[j + 4] = (ls & 0xF) | ((lm & 0xF) << 4); + b->scales[j - 4] |= ((ls >> 4) << 6); + b->scales[j - 0] |= ((lm >> 4) << 6); + } + } + } + } + } + } + + GGML_UNUSED(size); +} + +static void repack_tensor_tiled(ggml_tensor * tensor, const void * data, size_t size) { switch (tensor->type) { case GGML_TYPE_Q4_0: - GGML_ASSERT(offset == 0); - GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_q4_0_tiled(tensor, data, size); + repack_q4_0_tiled(tensor, data, 0, size); break; case GGML_TYPE_Q4_1: - GGML_ASSERT(offset == 0); - GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_q4_1_tiled(tensor, data, size); + repack_q4_1_tiled(tensor, data, 0, size); + break; + + case GGML_TYPE_Q4_K: + repack_q4_K_tiled(tensor, data, 0, size); break; case GGML_TYPE_Q8_0: - GGML_ASSERT(offset == 0); - GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_q8_0_tiled(tensor, data, size); + repack_q8_0_tiled(tensor, data, 0, size); break; case GGML_TYPE_IQ4_NL: - GGML_ASSERT(offset == 0); - GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - // IQ4_NL has identical block layout to Q4_0 (ggml_half d + uint8_t qs[16]) - repack_q4_0_tiled(tensor, data, size); + repack_q4_0_tiled(tensor, data, 0, size); break; case GGML_TYPE_MXFP4: - GGML_ASSERT(offset == 0); - GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_mxfp4_tiled(tensor, data, size); + repack_mxfp4_tiled(tensor, data, 0, size); + break; + + case GGML_TYPE_Q6_K: + repack_q6_K_tiled(tensor, data, 0, size); break; default: - memcpy((char *) tensor->data + offset, data, size); break; } } +static void ggml_backend_hexagon_buffer_set_tensor(ggml_backend_buffer_t buffer, + ggml_tensor * tensor, + const void * data, + size_t offset, + size_t size) { + auto extra = (ggml_hexagon_tensor_extra *) tensor->extra; + auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; + auto sess = sbuf->sess; + + if (ggml_backend_buffer_get_usage(buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) { + extra->flags |= GGML_HEXAGON_TENSOR_WEIGHT; + if (ggml_hexagon_is_repack_type(tensor->type)) { + extra->flags |= GGML_HEXAGON_TENSOR_REPACK; + } + } + + HEX_VERBOSE("ggml-hex: %s set-tensor %s : data %p offset %zu size %zu usage %d flags 0x%x\n", + sess->c_name(), tensor->name, data, offset, size, (int) buffer->usage, extra->flags); + + if ((extra->flags & GGML_HEXAGON_TENSOR_REPACK) == 0) { + memcpy((char *) tensor->data + offset, data, size); + return; + } + + if (offset == 0 && size == ggml_nbytes(tensor) && extra->shadow_buf.empty()) { + repack_tensor_tiled(tensor, data, size); + return; + } + + if (extra->shadow_buf.size() < ggml_nbytes(tensor)) { + extra->shadow_buf.resize(ggml_nbytes(tensor)); + } + memcpy(extra->shadow_buf.data() + offset, data, size); + extra->shadow_size += size; + + if (extra->shadow_size >= ggml_nbytes(tensor)) { + repack_tensor_tiled(tensor, extra->shadow_buf.data(), extra->shadow_buf.size()); + extra->shadow_buf.clear(); + extra->shadow_buf.shrink_to_fit(); + extra->shadow_size = 0; + } +} + static void ggml_backend_hexagon_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { - auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; - auto sess = sbuf->sess; + auto extra = (ggml_hexagon_tensor_extra *) tensor->extra; + auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; + auto sess = sbuf->sess; + + HEX_VERBOSE("ggml-hex: %s get-tensor %s : data %p offset %zu size %zu usage %d flags 0x%x\n", + sess->c_name(), tensor->name, data, offset, size, (int) buffer->usage, extra->flags); - HEX_VERBOSE("ggml-hex: %s get-tensor %s : data %p offset %zu size %zu\n", sess->c_name(), tensor->name, data, offset, size); + if ((extra->flags & GGML_HEXAGON_TENSOR_REPACK) == 0) { + memcpy(data, (const char *) tensor->data + offset, size); + return; + } switch (tensor->type) { case GGML_TYPE_Q4_0: GGML_ASSERT(offset == 0); GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_tiled_q4_0(data, tensor, size); + repack_tiled_q4_0(data, tensor, offset, size); break; case GGML_TYPE_Q4_1: GGML_ASSERT(offset == 0); GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_tiled_q4_1(data, tensor, size); + repack_tiled_q4_1(data, tensor, offset, size); + break; + + case GGML_TYPE_Q4_K: + GGML_ASSERT(offset == 0); + GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); + repack_tiled_q4_K(data, tensor, offset, size); break; case GGML_TYPE_Q8_0: GGML_ASSERT(offset == 0); GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_tiled_q8_0(data, tensor, size); + repack_tiled_q8_0(data, tensor, offset, size); break; case GGML_TYPE_IQ4_NL: GGML_ASSERT(offset == 0); GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_tiled_q4_0(data, tensor, size); + repack_tiled_q4_0(data, tensor, offset, size); break; case GGML_TYPE_MXFP4: GGML_ASSERT(offset == 0); GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); - repack_tiled_mxfp4(data, tensor, size); + repack_tiled_mxfp4(data, tensor, offset, size); + break; + + case GGML_TYPE_Q6_K: + GGML_ASSERT(offset == 0); + GGML_ASSERT(offset + size <= ggml_nbytes(tensor)); + repack_tiled_q6_K(data, tensor, offset, size); break; default: @@ -1035,71 +1897,233 @@ static bool ggml_backend_hexagon_buffer_cpy_tensor(ggml_backend_buffer_t bu GGML_UNUSED(dst); } -static void ggml_backend_hexagon_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) { - auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; - auto sess = sbuf->sess; - HEX_VERBOSE("ggml-hex: %s clear-buff base %p size %zu\n", sess->c_name(), (void *) sbuf->base, sbuf->size); - memset(sbuf->base, value, sbuf->size); -} +static void ggml_backend_hexagon_buffer_set_tensor_2d(ggml_backend_buffer_t buffer, + ggml_tensor * tensor, + const void * data, + size_t offset, + size_t size, + size_t n_copies, + size_t stride_tensor, + size_t stride_data) { + auto extra = (ggml_hexagon_tensor_extra *) tensor->extra; + auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; + auto sess = sbuf->sess; + + if (ggml_backend_buffer_get_usage(buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) { + extra->flags |= GGML_HEXAGON_TENSOR_WEIGHT; + if (ggml_hexagon_is_repack_type(tensor->type)) { + extra->flags |= GGML_HEXAGON_TENSOR_REPACK; + } + } -static ggml_backend_buffer_i ggml_backend_hexagon_buffer_interface = { - /* .free_buffer = */ ggml_backend_hexagon_buffer_free_buffer, - /* .get_base = */ ggml_backend_hexagon_buffer_get_base, - /* .init_tensor = */ ggml_backend_hexagon_buffer_init_tensor, - /* .memset_tensor = */ NULL, - /* .set_tensor = */ ggml_backend_hexagon_buffer_set_tensor, - /* .get_tensor = */ ggml_backend_hexagon_buffer_get_tensor, - /* .set_tensor_2d = */ NULL, - /* .get_tensor_2d = */ NULL, - /* .cpy_tensor = */ ggml_backend_hexagon_buffer_cpy_tensor, - /* .clear = */ ggml_backend_hexagon_buffer_clear, - /* .reset = */ NULL, -}; + HEX_VERBOSE("ggml-hex: %s set-tensor-2d %s : data %p offset %zu size %zu n_copies %zu stride_tensor %zu stride_data %zu usage %d flags 0x%x\n", + sess->c_name(), tensor->name, data, offset, size, n_copies, stride_tensor, stride_data, (int) buffer->usage, extra->flags); -// ** backend buffer type + if ((extra->flags & GGML_HEXAGON_TENSOR_REPACK) == 0) { + for (size_t i = 0; i < n_copies; i++) { + memcpy((uint8_t *) tensor->data + offset + i * stride_tensor, (const uint8_t *) data + i * stride_data, size); + } + return; + } -static const char * ggml_backend_hexagon_buffer_type_name(ggml_backend_buffer_type_t buffer_type) { - return static_cast(buffer_type->context)->name.c_str(); -} + if (extra->shadow_buf.size() < ggml_nbytes(tensor)) { + extra->shadow_buf.resize(ggml_nbytes(tensor)); + } + for (size_t i = 0; i < n_copies; i++) { + memcpy(extra->shadow_buf.data() + offset + i * stride_tensor, (const uint8_t *) data + i * stride_data, size); + } + extra->shadow_size += n_copies * size; -static ggml_backend_buffer_t ggml_backend_hexagon_buffer_type_alloc_buffer( - ggml_backend_buffer_type_t buffer_type, size_t size) { - auto sess = static_cast(buffer_type->context)->sess; - try { - size += 4 * 1024; // guard page - ggml_hexagon_shared_buffer * sbuf = new ggml_hexagon_shared_buffer(sess, size); - return ggml_backend_buffer_init(buffer_type, ggml_backend_hexagon_buffer_interface, sbuf, size); - } catch (const std::exception & exc) { - GGML_LOG_ERROR("ggml-hex: %s failed to allocate buffer context (host): %s\n", sess->c_name(), exc.what()); - return nullptr; + if (extra->shadow_size >= ggml_nbytes(tensor)) { + repack_tensor_tiled(tensor, extra->shadow_buf.data(), extra->shadow_buf.size()); + extra->shadow_buf.clear(); + extra->shadow_buf.shrink_to_fit(); + extra->shadow_size = 0; } } -static ggml_backend_buffer_t ggml_backend_hexagon_repack_buffer_type_alloc_buffer( - ggml_backend_buffer_type_t buffer_type, size_t size) { - auto sess = static_cast(buffer_type->context)->sess; - try { - size += 4 * 1024; // guard page - ggml_hexagon_shared_buffer * sbuf = new ggml_hexagon_shared_buffer(sess, size); - return ggml_backend_buffer_init(buffer_type, ggml_backend_hexagon_buffer_interface, sbuf, size); - } catch (const std::exception & exc) { - GGML_LOG_ERROR("ggml-hex: %s failed to allocate buffer context (repack): %s\n", sess->c_name(), exc.what()); - return nullptr; +static void ggml_backend_hexagon_buffer_get_tensor_2d(ggml_backend_buffer_t buffer, + const ggml_tensor * tensor, + void * data, + size_t offset, + size_t size, + size_t n_copies, + size_t stride_tensor, + size_t stride_data) { + auto extra = (ggml_hexagon_tensor_extra *) tensor->extra; + auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; + auto sess = sbuf->sess; + + HEX_VERBOSE("ggml-hex: %s get-tensor-2d %s : data %p offset %zu size %zu n_copies %zu stride_tensor %zu stride_data %zu usage %d\n", + sess->c_name(), tensor->name, data, offset, size, n_copies, stride_tensor, stride_data, (int) buffer->usage); + + if ((extra->flags & GGML_HEXAGON_TENSOR_REPACK) == 0) { + for (size_t i = 0; i < n_copies; i++) { + memcpy((uint8_t *)data + i * stride_data, (const uint8_t *)tensor->data + offset + i * stride_tensor, size); + } + return; } -} -static size_t ggml_backend_hexagon_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) { + size_t temp_size = n_copies > 0 ? (n_copies - 1) * stride_tensor + size : 0; + size_t slice_size = tensor->ne[1] * ggml_row_size(tensor->type, tensor->ne[0]); + size_t slice_offset = offset % slice_size; + size_t row_size_bytes = ggml_row_size(tensor->type, tensor->ne[0]); + + GGML_ASSERT((slice_offset % row_size_bytes) == 0 && "offset must be aligned to row boundary"); + GGML_ASSERT((temp_size % row_size_bytes) == 0 && "temp_size must be a multiple of row size"); + GGML_ASSERT((slice_offset / row_size_bytes) % 32 == 0 && "offset must be aligned to tile size (32 rows)"); + GGML_ASSERT((offset + temp_size) <= ggml_nbytes(tensor)); + + std::vector temp_buf(temp_size); + + switch (tensor->type) { + case GGML_TYPE_Q4_0: + repack_tiled_q4_0(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_Q4_1: + repack_tiled_q4_1(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_Q4_K: + repack_tiled_q4_K(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_Q8_0: + repack_tiled_q8_0(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_IQ4_NL: + repack_tiled_q4_0(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_MXFP4: + repack_tiled_mxfp4(temp_buf.data(), tensor, offset, temp_size); + break; + + case GGML_TYPE_Q6_K: + repack_tiled_q6_K(temp_buf.data(), tensor, offset, temp_size); + break; + + default: + memcpy(temp_buf.data(), (const uint8_t *) tensor->data + offset, temp_size); + break; + } + + for (size_t i = 0; i < n_copies; i++) { + memcpy((uint8_t *) data + i * stride_data, temp_buf.data() + i * stride_tensor, size); + } +} + +static void ggml_backend_hexagon_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) { + auto sbuf = (ggml_hexagon_shared_buffer *) buffer->context; + auto sess = sbuf->sess; + HEX_VERBOSE("ggml-hex: %s clear-buff base %p size %zu\n", sess->c_name(), (void *) sbuf->base(), sbuf->size()); + memset(sbuf->base(), value, sbuf->size()); +} + +static ggml_backend_buffer_i ggml_backend_hexagon_buffer_interface = { + /* .free_buffer = */ ggml_backend_hexagon_buffer_free_buffer, + /* .get_base = */ ggml_backend_hexagon_buffer_get_base, + /* .init_tensor = */ ggml_backend_hexagon_buffer_init_tensor, + /* .memset_tensor = */ NULL, + /* .set_tensor = */ ggml_backend_hexagon_buffer_set_tensor, + /* .get_tensor = */ ggml_backend_hexagon_buffer_get_tensor, + /* .set_tensor_2d = */ ggml_backend_hexagon_buffer_set_tensor_2d, + /* .get_tensor_2d = */ ggml_backend_hexagon_buffer_get_tensor_2d, + /* .cpy_tensor = */ ggml_backend_hexagon_buffer_cpy_tensor, + /* .clear = */ ggml_backend_hexagon_buffer_clear, + /* .reset = */ NULL, +}; + +// ** backend buffer type + +static void ggml_backend_hexagon_host_buffer_set_tensor(ggml_backend_buffer_t buffer, + ggml_tensor * tensor, + const void * data, + size_t offset, + size_t size) { + memcpy((char *) tensor->data + offset, data, size); + GGML_UNUSED(buffer); +} + +static void ggml_backend_hexagon_host_buffer_get_tensor(ggml_backend_buffer_t buffer, + const ggml_tensor * tensor, + void * data, + size_t offset, + size_t size) { + memcpy(data, (const char *) tensor->data + offset, size); + GGML_UNUSED(buffer); +} + +static ggml_backend_buffer_i ggml_backend_hexagon_host_buffer_interface = { + /* .free_buffer = */ ggml_backend_hexagon_buffer_free_buffer, + /* .get_base = */ ggml_backend_hexagon_buffer_get_base, + /* .init_tensor = */ ggml_backend_hexagon_buffer_init_tensor, + /* .memset_tensor = */ NULL, + /* .set_tensor = */ ggml_backend_hexagon_host_buffer_set_tensor, + /* .get_tensor = */ ggml_backend_hexagon_host_buffer_get_tensor, + /* .set_tensor_2d = */ NULL, + /* .get_tensor_2d = */ NULL, + /* .cpy_tensor = */ ggml_backend_hexagon_buffer_cpy_tensor, + /* .clear = */ ggml_backend_hexagon_buffer_clear, + /* .reset = */ NULL, +}; + +// ** backend buffer type + +static const char * ggml_backend_hexagon_buffer_type_name(ggml_backend_buffer_type_t buffer_type) { + return static_cast(buffer_type->context)->name.c_str(); +} + +static ggml_backend_buffer_t ggml_backend_hexagon_buffer_type_alloc_buffer( + ggml_backend_buffer_type_t buffer_type, size_t size) { + auto dev_ctx = static_cast(buffer_type->context)->dev_ctx; + auto sess = dev_ctx->session(); + if (sess && sess->max_vmem && size > sess->max_vmem) { + GGML_LOG_ERROR("ggml-hex: %s buffer size %zu exceeds max_vmem %zu\n", + dev_ctx->c_name(), size, sess->max_vmem); + return nullptr; + } + try { + ggml_hexagon_shared_buffer * sbuf = new ggml_hexagon_shared_buffer(sess, size, false); + return ggml_backend_buffer_init(buffer_type, ggml_backend_hexagon_buffer_interface, sbuf, size); + } catch (const std::exception & exc) { + GGML_LOG_ERROR("ggml-hex: %s failed to allocate device buffer context: %s\n", dev_ctx->c_name(), exc.what()); + return nullptr; + } +} + +static ggml_backend_buffer_t ggml_backend_hexagon_host_buffer_type_alloc_buffer( + ggml_backend_buffer_type_t buffer_type, size_t size) { + auto dev_ctx = static_cast(buffer_type->context)->dev_ctx; + auto sess = dev_ctx->session(); + if (sess && sess->max_vmem && size > sess->max_vmem) { + GGML_LOG_ERROR("ggml-hex: %s host buffer size %zu exceeds max_vmem %zu\n", + dev_ctx->c_name(), size, sess->max_vmem); + return nullptr; + } + try { + ggml_hexagon_shared_buffer * sbuf = new ggml_hexagon_shared_buffer(sess, size, false); + return ggml_backend_buffer_init(buffer_type, ggml_backend_hexagon_host_buffer_interface, sbuf, size); + } catch (const std::exception & exc) { + GGML_LOG_ERROR("ggml-hex: %s failed to allocate host buffer context: %s\n", dev_ctx->c_name(), exc.what()); + return nullptr; + } +} + +static size_t ggml_backend_hexagon_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) { return 128; // HVX alignment GGML_UNUSED(buft); } static size_t ggml_backend_hexagon_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const struct ggml_tensor * t) { - if (t->type == GGML_TYPE_Q4_0 || t->type == GGML_TYPE_Q4_1 || t->type == GGML_TYPE_Q8_0 || t->type == GGML_TYPE_IQ4_NL || t->type == GGML_TYPE_MXFP4) { + if (ggml_hexagon_is_repack_type(t->type)) { int64_t ne0 = hex_round_up(t->ne[0], 32); int64_t ne1 = hex_round_up(t->ne[1], 32); int64_t ne2 = t->ne[2]; int64_t ne3 = t->ne[3]; - return ggml_row_size(t->type, ne0) * ne1 * ne2 * ne3; + return ggml_hexagon_tiled_row_size(t->type, ne0) * ne1 * ne2 * ne3; } return ggml_nbytes(t); @@ -1107,19 +2131,17 @@ static size_t ggml_backend_hexagon_buffer_type_get_alloc_size(ggml_backend_buffe } static size_t ggml_backend_hexagon_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) { - auto * context = static_cast(buft->context); - return context->sess->max_bufsize; + return opt_mbuf; + GGML_UNUSED(buft); } static bool ggml_backend_hexagon_buffer_type_is_host(ggml_backend_buffer_type_t buft) { - return opt_hostbuf; - + return false; GGML_UNUSED(buft); } -static bool ggml_backend_hexagon_repack_buffer_type_is_host(ggml_backend_buffer_type_t buft) { - return false; - +static bool ggml_backend_hexagon_host_buffer_type_is_host(ggml_backend_buffer_type_t buft) { + return true; GGML_UNUSED(buft); } @@ -1132,24 +2154,38 @@ static ggml_backend_buffer_type_i ggml_backend_hexagon_buffer_type_interface = { /* .is_host = */ ggml_backend_hexagon_buffer_type_is_host, }; -static ggml_backend_buffer_type_i ggml_backend_hexagon_repack_buffer_type_interface = { +static ggml_backend_buffer_type_i ggml_backend_hexagon_host_buffer_type_interface = { /* .get_name = */ ggml_backend_hexagon_buffer_type_name, - /* .alloc_buffer = */ ggml_backend_hexagon_repack_buffer_type_alloc_buffer, + /* .alloc_buffer = */ ggml_backend_hexagon_host_buffer_type_alloc_buffer, /* .get_alignment = */ ggml_backend_hexagon_buffer_type_get_alignment, /* .get_max_size = */ ggml_backend_hexagon_buffer_type_get_max_size, /* .get_alloc_size = */ ggml_backend_hexagon_buffer_type_get_alloc_size, - /* .is_host = */ ggml_backend_hexagon_repack_buffer_type_is_host, + /* .is_host = */ ggml_backend_hexagon_host_buffer_type_is_host, }; -static bool ggml_backend_buffer_is_hexagon(const struct ggml_backend_buffer * b) { - return b->buft->iface.get_alignment == ggml_backend_hexagon_buffer_type_get_alignment; +ggml_backend_hexagon_device_context::ggml_backend_hexagon_device_context(int dev_id, const ggml_hexagon_device_config & config, ggml_backend_dev_t dev) + : dev_id(dev_id), config(config), dev(dev) { + buffer_type.device = dev; + buffer_type.iface = ggml_backend_hexagon_buffer_type_interface; + buffer_type.context = new ggml_backend_hexagon_buffer_type_context(config.name, this); + + host_buffer_type.device = dev; + host_buffer_type.iface = ggml_backend_hexagon_host_buffer_type_interface; + host_buffer_type.context = new ggml_backend_hexagon_buffer_type_context(config.name + "-HOST", this); + + fence_buffer_type.device = dev; + fence_buffer_type.iface = ggml_backend_hexagon_buffer_type_interface; + fence_buffer_type.context = new ggml_backend_hexagon_buffer_type_context(config.name + "-FENCE", this); } -static inline bool ggml_backend_buffer_is_hexagon_repack(const struct ggml_backend_buffer * b) { - if (!opt_hostbuf) { - return ggml_backend_buffer_is_hexagon(b); - } - return b->buft->iface.alloc_buffer == ggml_backend_hexagon_repack_buffer_type_alloc_buffer; +ggml_backend_hexagon_device_context::~ggml_backend_hexagon_device_context() { + delete static_cast(buffer_type.context); + delete static_cast(host_buffer_type.context); + delete static_cast(fence_buffer_type.context); +} + +static bool ggml_backend_buffer_is_hexagon(const struct ggml_backend_buffer * b) { + return b->buft->iface.get_alignment == ggml_backend_hexagon_buffer_type_get_alignment; } struct ggml_hexagon_opbatch { @@ -1165,12 +2201,10 @@ struct ggml_hexagon_opbatch { std::unordered_map t_map; // tensor ptr to index std::unordered_multimap d_map; // tensor data to index - - unsigned int n_bufs; // num buffers in the batch unsigned int n_tens; // num tensors ... unsigned int n_ops; // num ops ... - size_t b_vmem; // sum of all buffer sizes + size_t b_vmem; // sum of non-extended buffer sizes unsigned int n_bufs_max; unsigned int n_tens_max; @@ -1186,6 +2220,7 @@ struct ggml_hexagon_opbatch { b_map.clear(); t_map.clear(); d_map.clear(); + ops.resize(n_ops_max); } ggml_hexagon_opbatch(ggml_hexagon_session *sess, size_t batch_size, size_t max_vmem) { @@ -1218,39 +2253,42 @@ struct ggml_hexagon_opbatch { // add buffer and return its index int add_buffer(ggml_hexagon_shared_buffer * sbuf) { // Lookup by fd - auto it = b_map.find(sbuf->fd); + auto it = b_map.find(sbuf->fd()); if (it != b_map.end()) { return it->second; } // Add new buffer to the batch - int bi = n_bufs++; GGML_ASSERT(n_bufs < HTP_OP_MAX_BUFS); + int bi = n_bufs++; - b_map.insert({sbuf->fd, bi}); + b_map.insert({sbuf->fd(), bi}); htp_buf_desc &b = h_bufs[bi]; - b.base = (uint64_t) sbuf->base; - b.fd = sbuf->fd; - b.size = sbuf->size; + b.base = (uint64_t) sbuf->base(); + b.fd = sbuf->fd(); + b.size = sbuf->size(); + b.flags = sbuf->extended ? HTP_BUF_EXTENDED : 0; - b_vmem += b.size; + if (!sbuf->extended) { + b_vmem += b.size; + } - HEX_VERBOSE("ggml-hex: %s add-buffer #%u : fd %d base %p size %zu : vmem %zu\n", sess->c_name(), bi, b.fd, (void*) sbuf->base, (size_t) b.size, b_vmem); + HEX_VERBOSE("ggml-hex: %s add-buffer #%u : fd %d base %p size %zu : vmem %zu\n", sess->c_name(), bi, b.fd, (void*) sbuf->base(), (size_t) b.size, b_vmem); return bi; } - - bool same_shape(const htp_tensor * h, const ggml_tensor * t) const { + auto extra = (ggml_hexagon_tensor_extra *) t->extra; + int64_t ne0 = t->ne[0]; int64_t ne1 = t->ne[1]; - const bool is_repack = ggml_backend_buffer_is_hexagon_repack(t->buffer) && ggml_hexagon_is_repack_type(t->type); + const bool is_repack = (extra->flags & GGML_HEXAGON_TENSOR_REPACK) != 0; if (is_repack) { ne0 = hex_round_up(ne0, 32); ne1 = hex_round_up(ne1, 32); } - int64_t nb1 = is_repack ? ggml_row_size(t->type, ne0) : t->nb[1]; - int64_t nb2 = is_repack ? nb1 * ne1 : t->nb[2]; + int64_t nb1 = is_repack ? (int64_t) ggml_hexagon_tiled_row_size(t->type, ne0) : t->nb[1]; + int64_t nb2 = is_repack ? nb1 * ne1 : t->nb[2]; int64_t nb3 = is_repack ? nb2 * t->ne[2] : t->nb[3]; return (h->type == t->type) && @@ -1260,7 +2298,8 @@ struct ggml_hexagon_opbatch { // add tensor and return its index int add_tensor(const ggml_tensor * t) { - auto sbuf = static_cast(t->buffer->context); + auto extra = (ggml_hexagon_tensor_extra *) t->extra; + auto sbuf = static_cast(t->buffer->context); // First lookup by tensor data auto range = d_map.equal_range(t->data); @@ -1280,131 +2319,839 @@ struct ggml_hexagon_opbatch { t_map.insert({t, ti}); d_map.insert({t->data, ti}); - uint64_t t_offset = (uint8_t *) t->data - sbuf->base; - size_t t_size = ggml_nbytes(t); + uint64_t t_offset = (uint8_t *) t->data - sbuf->base(); + size_t t_size = ggml_nbytes(t); + + htp_tensor &h = h_tens[ti]; + h.bi = add_buffer(sbuf); + h.ti = ti; + h.data = t_offset; + h.type = t->type; + + const bool is_repack = (extra->flags & GGML_HEXAGON_TENSOR_REPACK) != 0; + if (is_repack) { + h.ne[0] = hex_round_up(t->ne[0], 32); + h.ne[1] = hex_round_up(t->ne[1], 32); + h.ne[2] = t->ne[2]; + h.ne[3] = t->ne[3]; + + h.nb[0] = t->nb[0]; + h.nb[1] = ggml_hexagon_tiled_row_size(t->type, h.ne[0]); + h.nb[2] = h.nb[1] * h.ne[1]; + h.nb[3] = h.nb[2] * h.ne[2]; + h.size = h.nb[3] * h.ne[3]; + t_size = h.size; + } else { + h.size = t_size; + h.ne[0] = t->ne[0]; h.ne[1] = t->ne[1]; h.ne[2] = t->ne[2]; h.ne[3] = t->ne[3]; + h.nb[0] = t->nb[0]; h.nb[1] = t->nb[1]; h.nb[2] = t->nb[2]; h.nb[3] = t->nb[3]; + } + + h.flags = 0; + if ((extra->flags & GGML_HEXAGON_TENSOR_WEIGHT) != 0) { + h.flags |= HTP_TENSOR_WEIGHT; + } + if ((extra->flags & GGML_HEXAGON_TENSOR_REPACK) != 0) { + h.flags |= HTP_TENSOR_REPACK; + } + if ((extra->flags & GGML_HEXAGON_TENSOR_FENCE) != 0) { + h.flags |= HTP_TENSOR_FENCE; + } + + HEX_VERBOSE("ggml-hex: %s add-tensor #%u %s : bi %d data %p offset %zu size %zu flags 0x%x : %zu:%zu:%zu:%zu\n", sess->c_name(), + ti, t->name, h.bi, (void*) t->data, (size_t) t_offset, t_size, h.flags, + (size_t) h.ne[0], (size_t) h.ne[1], (size_t) h.ne[2], (size_t) h.ne[3]); + + return ti; + } + + bool fit_op(const htp_opnode & node) const { + if (n_ops >= n_ops_max) return false; + + // check how much extras we will need + size_t extra_bufs = 0; + size_t extra_vmem = 0; + size_t extra_tens = 0; + + int seen_bufs[HTP_OP_MAX_BUFS]; + int n_seen_bufs = 0; + + auto fit_tensor = [&](const ggml_tensor *t) { + if (!t) return; + if (!t_map.count(t)) { + extra_tens++; + + auto sbuf = static_cast(t->buffer->context); + int fd = sbuf->fd(); + if (!b_map.count(fd)) { + for (int i = 0; i < n_seen_bufs; i++) { + if (seen_bufs[i] == fd) return; + } + if (n_seen_bufs < HTP_OP_MAX_BUFS) { + seen_bufs[n_seen_bufs++] = fd; + } + if (!sbuf->extended) { + extra_vmem += sbuf->size(); + } + extra_bufs += 1; + } + } + }; + + for (const auto * src : node.get_inputs()) { + fit_tensor(src); + } + for (const auto * output : node.get_outputs()) { + fit_tensor(output); + } + + if ((extra_bufs + n_bufs) > n_bufs_max) return false; + if ((extra_tens + n_tens) > n_tens_max) return false; + if ((extra_vmem + b_vmem) > b_vmem_max) return false; + + return true; + } + + // assumes that fit_op() was called first and returned true + void add_op(const htp_opnode & node) { + // Add new op + + unsigned int n = n_ops++; + GGML_ASSERT(n_ops <= n_ops_max); + + ops[n] = node; + + htp_op_desc &o = h_ops[n]; + memcpy(o.params, node.node->op_params, sizeof(node.node->op_params)); + memcpy(o.kernel_params, node.kernel_params, sizeof(o.kernel_params)); + o.opcode = node.opcode; + o.flags = 0; + + ggml_hexagon_dump_op_exec(sess->c_name(), ops[n], o.flags); + + auto inputs = node.get_inputs(); + for (unsigned int i=0; i < HTP_OP_MAX_INPUTS; i++) { + o.src[i] = (i < inputs.size() && inputs[i]) ? add_tensor(inputs[i]) : 0xffff; + } + + auto outputs = node.get_outputs(); + for (unsigned int i=0; i < HTP_OP_MAX_OUTPUTS; i++) { + o.dst[i] = (i < outputs.size() && outputs[i]) ? add_tensor(outputs[i]) : 0xffff; + } + } + + void sort_buffers() { + if (n_bufs <= 1) return; + + std::vector order(n_bufs); + for (unsigned int i = 0; i < n_bufs; i++) { order[i] = (int) i; } + + std::stable_sort(order.begin(), order.end(), [&](int a, int b) { + return h_bufs[a].size > h_bufs[b].size; + }); + + bool already_sorted = true; + for (unsigned int i = 0; i < n_bufs; i++) { + if (order[i] != (int) i) { + already_sorted = false; + break; + } + } + if (already_sorted) return; + + std::vector remap(n_bufs); + std::vector sorted_bufs(n_bufs); + for (unsigned int new_bi = 0; new_bi < n_bufs; new_bi++) { + int old_bi = order[new_bi]; + remap[old_bi] = (uint16_t) new_bi; + sorted_bufs[new_bi] = h_bufs[old_bi]; + } + + for (unsigned int i = 0; i < n_bufs; i++) { + h_bufs[i] = sorted_bufs[i]; + } + + for (unsigned int i = 0; i < n_tens; i++) { + h_tens[i].bi = remap[h_tens[i].bi]; + } + } + + void update_mdev_group(uint32_t mdev_idx) { + if (n_ops > 0 && h_ops[0].opcode == HTP_OP_MDEV_GROUP) { + h_ops[0].params[0] = (int32_t) mdev_idx; + } + } + + bool try_fuse_common(std::initializer_list tensors) const { + size_t extra_bufs = 0, extra_vmem = 0, extra_tens = 0; + + int seen_bufs[HTP_OP_MAX_BUFS]; + int n_seen_bufs = 0; + + for (const auto * t : tensors) { + if (!t || t_map.count(t)) { + continue; + } + extra_tens++; + auto sbuf = static_cast(t->buffer->context); + int fd = sbuf->fd(); + if (!b_map.count(fd)) { + bool found = false; + for (int i = 0; i < n_seen_bufs; i++) { + if (seen_bufs[i] == fd) { + found = true; + break; + } + } + if (!found) { + if (n_seen_bufs < HTP_OP_MAX_BUFS) { + seen_bufs[n_seen_bufs++] = fd; + } + if (!sbuf->extended) { + extra_vmem += sbuf->size(); + } + extra_bufs += 1; + } + } + } + + if ((extra_bufs + n_bufs) > n_bufs_max || (extra_tens + n_tens) > n_tens_max || (extra_vmem + b_vmem) > b_vmem_max) { + return false; + } + + return true; + } + + bool try_fuse_common(const ggml_tensor * t1, const ggml_tensor * t2) const { + return try_fuse_common({t1, t2}); + } + + bool try_fuse_allreduce_add(const htp_opnode & node) { + if (n_ops == 0 || opt_ar_select != 2) return false; + if (node.opcode != HTP_OP_ADD) return false; + + htp_opnode & last_node = ops[n_ops - 1]; + if (last_node.opcode != HTP_OP_ALLREDUCE) return false; + + auto * ar_kparams = (struct htp_allreduce_kernel_params *) last_node.kernel_params; + const uint32_t rank = (uint32_t) ar_kparams->rank; + const uint32_t n_ranks = (uint32_t) ar_kparams->n_ranks; + const ggml_tensor * ar_local = last_node.inputs[rank]; + const ggml_tensor * add_src0 = node.src0(); + const ggml_tensor * add_src1 = node.src1(); + const ggml_tensor * add_dst = node.dst(); + + if (!ggml_hexagon_tensor_is_fuseable(ar_local)) return false; + + const ggml_tensor * res_tensor; + if (add_src0 == ar_local || add_src0->data == ar_local->data) { + res_tensor = add_src1; + } else if (add_src1 == ar_local || add_src1->data == ar_local->data) { + res_tensor = add_src0; + } else { + return false; + } + + if (ar_local->type != res_tensor->type) return false; + + const bool is_same_shape = (ar_local->ne[0] == res_tensor->ne[0] && ar_local->ne[1] == res_tensor->ne[1] && + ar_local->ne[2] == res_tensor->ne[2] && ar_local->ne[3] == res_tensor->ne[3]); + const bool is_row_bcast = !is_same_shape && (ar_local->ne[0] == res_tensor->ne[0] && res_tensor->ne[1] == 1 && + res_tensor->ne[2] == 1 && res_tensor->ne[3] == 1); + + if (!is_same_shape && !is_row_bcast) return false; + + if (is_same_shape) { + if (ar_local->nb[1] != res_tensor->nb[1] || ar_local->nb[2] != res_tensor->nb[2] || + ar_local->nb[3] != res_tensor->nb[3]) { + return false; + } + if (ggml_is_contiguous(ar_local) != ggml_is_contiguous(res_tensor)) { + return false; + } + } + if (ggml_is_contiguous(ar_local) != ggml_is_contiguous(add_dst)) { + return false; + } + + for (uint32_t r = 0; r < n_ranks; r++) { + const ggml_tensor * ar_src = last_node.inputs[r]; + if (ggml_hexagon_tensors_overlap(add_dst, ar_src)) { + HEX_VERBOSE("ggml-hex: %s skip ALLREDUCE_ADD fusion: dst overlaps allreduce src %u\n", sess->c_name(), r); + return false; + } + } + + struct htp_allreduce_kernel_params new_kparams; + if (!ggml_hexagon_precompute_allreduce_params( + sess, add_dst, (uint32_t) ar_kparams->rank, (uint32_t) ar_kparams->n_ranks, true, is_row_bcast, &new_kparams + )) { + HEX_VERBOSE("ggml-hex: %s skip ALLREDUCE_ADD fusion: solver failed\n", sess->c_name()); + return false; + } + + if (!try_fuse_common(res_tensor, add_dst)) { + return false; + } + + last_node.opcode = HTP_OP_ALLREDUCE_ADD; + last_node.name = "ALLREDUCE+ADD"; + last_node.inputs.push_back(res_tensor); + last_node.outputs.clear(); + last_node.outputs.push_back(add_dst); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &new_kparams, sizeof(new_kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.opcode = HTP_OP_ALLREDUCE_ADD; + memcpy(o.kernel_params, &new_kparams, sizeof(new_kparams)); + + o.src[2 * n_ranks] = add_tensor(res_tensor); + o.dst[0] = add_tensor(add_dst); + for (uint32_t d = 1; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } + + HEX_VERBOSE("ggml-hex: %s fused ALLREDUCE+ADD (#%u)\n", sess->c_name(), n_ops - 1); + return true; + } + + bool try_fuse_rms_norm_mul(const htp_opnode & node) { + if (n_ops == 0) return false; + if (node.opcode != HTP_OP_MUL) return false; + + htp_opnode & last_node = ops[n_ops - 1]; + if (last_node.opcode != HTP_OP_RMS_NORM) return false; + + const ggml_tensor * mul_src0 = node.src0(); + const ggml_tensor * mul_src1 = node.src1(); + const ggml_tensor * rms_out = last_node.dst(); + + if (!ggml_hexagon_tensor_is_fuseable(rms_out)) return false; + + const ggml_tensor * weight; + if (mul_src0 == rms_out || mul_src0->data == rms_out->data) { + weight = mul_src1; + } else if (mul_src1 == rms_out || mul_src1->data == rms_out->data) { + weight = mul_src0; + } else { + return false; + } + + const ggml_tensor * src0 = last_node.src0(); + + if (src0->ne[0] != weight->ne[0] || src0->ne[0] != node.dst()->ne[0]) { + return false; + } + + const bool is_row_bcast = (weight->ne[1] == 1 && weight->ne[2] == 1 && weight->ne[3] == 1); + const bool is_same_shape = (src0->ne[0] == weight->ne[0] && src0->ne[1] == weight->ne[1] && + src0->ne[2] == weight->ne[2] && src0->ne[3] == weight->ne[3]); + if (!is_row_bcast && !is_same_shape) return false; + + if (!ggml_are_same_shape(src0, node.dst())) { + return false; + } + if (ggml_is_contiguous(src0) != ggml_is_contiguous(node.dst())) { + return false; + } + + struct htp_unary_kernel_params new_kparams; + ggml_hexagon_precompute_unary_params( + sess, HTP_OP_RMS_NORM_MUL, src0, weight, node.dst(), &new_kparams + ); + + if ((size_t) new_kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip RMS_NORM_MUL fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), new_kparams.vtcm_size, sess->vtcm_size); + return false; + } + + if (!try_fuse_common(weight, node.dst())) { + return false; + } + + last_node.opcode = HTP_OP_RMS_NORM_MUL; + last_node.name = "RMS_NORM+MUL"; + last_node.inputs.clear(); + last_node.inputs.push_back(src0); + last_node.inputs.push_back(weight); + last_node.outputs.clear(); + last_node.outputs.push_back(node.dst()); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &new_kparams, sizeof(new_kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.opcode = HTP_OP_RMS_NORM_MUL; + memcpy(o.kernel_params, &new_kparams, sizeof(new_kparams)); + + o.src[0] = add_tensor(src0); + o.src[1] = add_tensor(weight); + for (uint32_t s = 2; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + o.dst[0] = add_tensor(node.dst()); + for (uint32_t d = 1; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } + + HEX_VERBOSE("ggml-hex: %s fused RMS_NORM+MUL (#%u)\n", sess->c_name(), n_ops - 1); + return true; + } + + bool try_fuse_mul_mat_add(const htp_opnode & node) { + if (n_ops == 0) return false; + if (node.opcode != HTP_OP_ADD) return false; + + htp_opnode & last_node = ops[n_ops - 1]; + if (last_node.opcode != HTP_OP_MUL_MAT) return false; + + const ggml_tensor * add_src0 = node.src0(); + const ggml_tensor * add_src1 = node.src1(); + const ggml_tensor * mm_out = last_node.dst(); + + if (!ggml_hexagon_tensor_is_fuseable(mm_out)) return false; + + const ggml_tensor * src2; + if (add_src0 == mm_out || add_src0->data == mm_out->data) { + src2 = add_src1; + } else if (add_src1 == mm_out || add_src1->data == mm_out->data) { + src2 = add_src0; + } else { + return false; + } + + const ggml_tensor * src0 = last_node.src0(); + const ggml_tensor * src1 = last_node.src1(); + + if (src2->type != GGML_TYPE_F32) return false; + + const struct htp_mm_kernel_params * orig_kparams = (const struct htp_mm_kernel_params *) last_node.kernel_params; + struct htp_mm_kernel_params kparams; + ggml_hexagon_precompute_fused_matmul_add_params(sess, src0, src1, src2, node.dst(), &kparams); + if (kparams.kernel_type == HTP_MM_KERNEL_UNSUPPORTED) { + return false; + } + + const int src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; + const bool can_fuse = (kparams.n_hmx > 0) || (src1_nrows == 1); + if (!can_fuse) return false; + + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip MUL_MAT_ADD fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), kparams.vtcm_size, sess->vtcm_size); + return false; + } + + if (kparams.n_hmx > 0 && orig_kparams->n_hmx > 0) { + if (kparams.m_chunk < orig_kparams->m_chunk || + kparams.n_chunk < orig_kparams->n_chunk || + kparams.n_act_threads < orig_kparams->n_act_threads) { + HEX_VERBOSE("ggml-hex: %s skip MUL_MAT_ADD fusion: HMX efficiency reduced (m %d->%d, n %d->%d, th %d->%d)\n", + sess->c_name(), orig_kparams->m_chunk, kparams.m_chunk, + orig_kparams->n_chunk, kparams.n_chunk, + orig_kparams->n_act_threads, kparams.n_act_threads); + return false; + } + } + + if (!try_fuse_common(src2, node.dst())) { + return false; + } + + last_node.opcode = HTP_OP_MUL_MAT_ADD; + last_node.name = "MUL_MAT+ADD"; + last_node.inputs.clear(); + last_node.inputs.push_back(src0); + last_node.inputs.push_back(src1); + last_node.inputs.push_back(src2); + last_node.outputs.clear(); + last_node.outputs.push_back(node.dst()); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &kparams, sizeof(kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.opcode = HTP_OP_MUL_MAT_ADD; + memcpy(o.kernel_params, &kparams, sizeof(kparams)); + + o.src[0] = add_tensor(src0); + o.src[1] = add_tensor(src1); + o.src[2] = add_tensor(src2); + for (uint32_t s = 3; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + o.dst[0] = add_tensor(node.dst()); + for (uint32_t d = 1; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } + + HEX_VERBOSE("ggml-hex: %s fused MUL_MAT+ADD (#%u)\n", sess->c_name(), n_ops - 1); + return true; + } + + bool try_fuse_mul_mat_nx(const htp_opnode & node) { + if (n_ops == 0 || node.opcode != HTP_OP_MUL_MAT) return false; + if (!is_mergeable_mul_mat(node.node)) return false; + + const ggml_tensor * w_in = node.src0(); + const ggml_tensor * x_in = node.src1(); + const ggml_tensor * d_in = node.dst(); + + htp_opnode & last_node = ops[n_ops - 1]; + + // Case 1: last_node is already MUL_MAT_NX + if (last_node.opcode == HTP_OP_MUL_MAT_NX) { + const uint32_t curr_n = (uint32_t) last_node.outputs.size(); + if (curr_n >= HTP_OP_MAX_OUTPUTS || curr_n + 1 >= HTP_OP_MAX_INPUTS) { + return false; + } + + const ggml_tensor * w0 = last_node.inputs[0]; + const ggml_tensor * x = last_node.inputs[curr_n]; + + if (x_in != x || w_in->type != w0->type || w_in->ne[0] != w0->ne[0]) { + return false; + } + if (!last_node.fused.empty() && (mm_is_hmx_eligible(last_node.fused[0]) != mm_is_hmx_eligible(node.node))) { + return false; + } + + struct htp_mm_kernel_params kparams; + ggml_hexagon_precompute_fused_mmnx_params(sess, w0, x, curr_n + 1, &kparams); + if (!is_supported_mul_mat_nx_kernel(w0, &kparams)) { + return false; + } + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip NX fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), kparams.vtcm_size, sess->vtcm_size); + return false; + } + + if (!try_fuse_common(w_in, d_in)) { + return false; + } + + last_node.inputs[curr_n] = w_in; + last_node.inputs.push_back(x); + last_node.outputs.push_back(d_in); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &kparams, sizeof(kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + memcpy(o.kernel_params, &kparams, sizeof(kparams)); + + for (uint32_t s = 0; s <= curr_n + 1; s++) { + o.src[s] = add_tensor(last_node.inputs[s]); + } + for (uint32_t s = curr_n + 2; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + for (uint32_t d = 0; d <= curr_n; d++) { + o.dst[d] = add_tensor(last_node.outputs[d]); + } + for (uint32_t d = curr_n + 1; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } + + HEX_VERBOSE("ggml-hex: %s fused MUL_MAT_NX (N=%u, #%u)\n", sess->c_name(), curr_n + 1, n_ops - 1); + return true; + } + + // Case 2: last_node is single MUL_MAT + if (last_node.opcode == HTP_OP_MUL_MAT) { + if (!is_mergeable_mul_mat_pair(last_node.node, node.node)) { + return false; + } + + const ggml_tensor * w0 = last_node.src0(); + const ggml_tensor * x = last_node.src1(); + const ggml_tensor * w1 = node.src0(); + + struct htp_mm_kernel_params kparams; + ggml_hexagon_precompute_fused_mmnx_params(sess, w0, x, 2, &kparams); + if (!is_supported_mul_mat_nx_kernel(w0, &kparams)) { + return false; + } + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip NX fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), kparams.vtcm_size, sess->vtcm_size); + return false; + } + + if (!try_fuse_common(w1, node.dst())) { + return false; + } + + const ggml_tensor * dst_0 = last_node.dst(); + const ggml_tensor * dst_1 = node.dst(); + + last_node.opcode = HTP_OP_MUL_MAT_NX; + last_node.name = "MUL_MAT_NX"; + last_node.inputs.clear(); + last_node.inputs.push_back(w0); + last_node.inputs.push_back(w1); + last_node.inputs.push_back(x); + last_node.outputs.clear(); + last_node.outputs.push_back(dst_0); + last_node.outputs.push_back(dst_1); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &kparams, sizeof(kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.opcode = HTP_OP_MUL_MAT_NX; + memcpy(o.kernel_params, &kparams, sizeof(kparams)); + + o.src[0] = add_tensor(w0); + o.src[1] = add_tensor(w1); + o.src[2] = add_tensor(x); + for (uint32_t s = 3; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + o.dst[0] = add_tensor(dst_0); + o.dst[1] = add_tensor(dst_1); + for (uint32_t d = 2; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } + + HEX_VERBOSE("ggml-hex: %s fused MUL_MAT_NX (N=2, #%u)\n", sess->c_name(), n_ops - 1); + return true; + } + + return false; + } + + bool try_fuse_mul_mat_id_nx(const htp_opnode & node) { + if (n_ops == 0 || node.opcode != HTP_OP_MUL_MAT_ID) return false; + if (!is_mergeable_mul_mat_id(node.node)) return false; + + const ggml_tensor * w_in = node.src0(); + const ggml_tensor * x_in = node.src1(); + const ggml_tensor * ids_in = node.node->src[2]; + const ggml_tensor * d_in = node.dst(); + + htp_opnode & last_node = ops[n_ops - 1]; + + // Case 1: last_node is already MUL_MAT_ID_NX + if (last_node.opcode == HTP_OP_MUL_MAT_ID_NX) { + const uint32_t curr_n = (uint32_t) last_node.outputs.size(); + if (curr_n >= HTP_OP_MAX_OUTPUTS || curr_n + 2 >= HTP_OP_MAX_INPUTS) { + return false; + } + + const ggml_tensor * w0 = last_node.inputs[0]; + const ggml_tensor * x = last_node.inputs[curr_n]; + const ggml_tensor * ids = last_node.inputs[curr_n + 1]; + + if (x_in != x || ids_in != ids || w_in->type != w0->type || w_in->ne[0] != w0->ne[0] || w_in->ne[2] != w0->ne[2]) { + return false; + } + if (!last_node.fused.empty() && (mm_is_hmx_eligible(last_node.fused[0]) != mm_is_hmx_eligible(node.node))) { + return false; + } - htp_tensor &h = h_tens[ti]; - h.bi = add_buffer(sbuf); - h.ti = ti; - h.data = t_offset; - h.type = t->type; + struct htp_mm_kernel_params kparams; + ggml_hexagon_precompute_fused_mmidnx_params(sess, w0, x, d_in, curr_n + 1, &kparams); + if (!is_supported_mul_mat_id_nx_kernel(w0, &kparams)) { + return false; + } + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip ID NX fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), kparams.vtcm_size, sess->vtcm_size); + return false; + } - const bool is_repack = ggml_backend_buffer_is_hexagon_repack(t->buffer) && ggml_hexagon_is_repack_type(t->type); - if (is_repack) { - h.ne[0] = hex_round_up(t->ne[0], 32); - h.ne[1] = hex_round_up(t->ne[1], 32); - h.ne[2] = t->ne[2]; - h.ne[3] = t->ne[3]; + if (!try_fuse_common(w_in, d_in)) { + return false; + } - h.nb[0] = t->nb[0]; - h.nb[1] = ggml_row_size(t->type, h.ne[0]); - h.nb[2] = h.nb[1] * h.ne[1]; - h.nb[3] = h.nb[2] * h.ne[2]; - h.size = h.nb[3] * h.ne[3]; - t_size = h.size; - } else { - h.size = t_size; - h.ne[0] = t->ne[0]; h.ne[1] = t->ne[1]; h.ne[2] = t->ne[2]; h.ne[3] = t->ne[3]; - h.nb[0] = t->nb[0]; h.nb[1] = t->nb[1]; h.nb[2] = t->nb[2]; h.nb[3] = t->nb[3]; - } + last_node.inputs[curr_n] = w_in; + last_node.inputs[curr_n + 1] = x; + last_node.inputs.push_back(ids); + last_node.outputs.push_back(d_in); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &kparams, sizeof(kparams)); + htp_op_desc & o = h_ops[n_ops - 1]; + memcpy(o.kernel_params, &kparams, sizeof(kparams)); + for (uint32_t s = 0; s <= curr_n + 2; s++) { + o.src[s] = add_tensor(last_node.inputs[s]); + } + for (uint32_t s = curr_n + 3; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + for (uint32_t d = 0; d <= curr_n; d++) { + o.dst[d] = add_tensor(last_node.outputs[d]); + } + for (uint32_t d = curr_n + 1; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; + } - h.flags = 0; - if (ggml_backend_buffer_get_usage(t->buffer) != GGML_BACKEND_BUFFER_USAGE_WEIGHTS) { - h.flags |= HTP_TENSOR_COMPUTE; + HEX_VERBOSE("ggml-hex: %s fused MUL_MAT_ID_NX (N=%u, #%u)\n", sess->c_name(), curr_n + 1, n_ops - 1); + return true; } - HEX_VERBOSE("ggml-hex: %s add-tensor #%u %s : bi %d data %p offset %zu size %zu flags 0x%x : %zu:%zu:%zu:%zu\n", sess->c_name(), - ti, t->name, h.bi, (void*) t->data, (size_t) t_offset, t_size, h.flags, - (size_t) h.ne[0], (size_t) h.ne[1], (size_t) h.ne[2], (size_t) h.ne[3]); - - return ti; - } + // Case 2: last_node is single MUL_MAT_ID + if (last_node.opcode == HTP_OP_MUL_MAT_ID) { + if (!is_mergeable_mul_mat_id_pair(last_node.node, node.node)) { + return false; + } - bool fit_op(const htp_opnode & node) const { - if (n_ops >= n_ops_max ) return false; + const ggml_tensor * w0 = last_node.src0(); + const ggml_tensor * x = last_node.src1(); + const ggml_tensor * ids = last_node.node->src[2]; + const ggml_tensor * w1 = node.src0(); - // check how much extras we will need - size_t extra_bufs = 0; - size_t extra_vmem = 0; - size_t extra_tens = 0; + struct htp_mm_kernel_params kparams; + ggml_hexagon_precompute_fused_mmidnx_params(sess, w0, x, node.dst(), 2, &kparams); + if (!is_supported_mul_mat_id_nx_kernel(w0, &kparams)) { + return false; + } + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + HEX_VERBOSE("ggml-hex: %s skip ID NX fusion: VTCM needed (%d) > budget (%zu)\n", + sess->c_name(), kparams.vtcm_size, sess->vtcm_size); + return false; + } - auto fit_tensor = [&](const ggml_tensor *t) { - if (!t) return; - if (!t_map.count(t)) { - extra_tens++; + if (!try_fuse_common(w1, node.dst())) { + return false; + } - auto sbuf = static_cast(t->buffer->context); - if (!b_map.count(sbuf->fd)) { - extra_vmem += sbuf->size; - extra_bufs += 1; - } + const ggml_tensor * dst_0 = last_node.dst(); + const ggml_tensor * dst_1 = node.dst(); + + last_node.opcode = HTP_OP_MUL_MAT_ID_NX; + last_node.name = "MUL_MAT_ID_NX"; + last_node.inputs.clear(); + last_node.inputs.push_back(w0); + last_node.inputs.push_back(w1); + last_node.inputs.push_back(x); + last_node.inputs.push_back(ids); + last_node.outputs.clear(); + last_node.outputs.push_back(dst_0); + last_node.outputs.push_back(dst_1); + last_node.fused.push_back(node.node); + memcpy(last_node.kernel_params, &kparams, sizeof(kparams)); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.opcode = HTP_OP_MUL_MAT_ID_NX; + memcpy(o.kernel_params, &kparams, sizeof(kparams)); + + o.src[0] = add_tensor(w0); + o.src[1] = add_tensor(w1); + o.src[2] = add_tensor(x); + o.src[3] = add_tensor(ids); + for (uint32_t s = 4; s < HTP_OP_MAX_INPUTS; s++) { + o.src[s] = 0xffff; + } + o.dst[0] = add_tensor(dst_0); + o.dst[1] = add_tensor(dst_1); + for (uint32_t d = 2; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; } - }; - for (const auto * src : node.get_inputs()) { - fit_tensor(src); - } - for (const auto * output : node.get_outputs()) { - fit_tensor(output); + HEX_VERBOSE("ggml-hex: %s fused MUL_MAT_ID_NX (N=2, #%u)\n", sess->c_name(), n_ops - 1); + return true; } - if ((extra_bufs + n_bufs) > n_bufs_max) return false; - if ((extra_tens + n_tens) > n_tens_max) return false; - if ((extra_vmem + b_vmem) > b_vmem_max) return false; - - return true; + return false; } - // assumes that fit_op() was called first and returned true - void add_op(const htp_opnode & node) { - // Add new op + bool try_fuse_gdn_cpy(const htp_opnode & node) { + if (n_ops == 0 || node.opcode != HTP_OP_CPY) return false; - unsigned int n = n_ops++; - GGML_ASSERT(n_ops <= n_ops_max); + htp_opnode & last_node = ops[n_ops - 1]; + if (last_node.opcode != HTP_OP_GATED_DELTA_NET) return false; + if (last_node.outputs.size() != 1) return false; - ops[n] = node; + const ggml_tensor * gdn_out = last_node.dst(); + const ggml_tensor * cpy_node = node.node; + const ggml_tensor * cpy_src = node.src0(); + const ggml_tensor * cpy_dst = node.dst(); - htp_op_desc &o = h_ops[n]; - memcpy(o.params, node.node->op_params, sizeof(node.node->op_params)); - memcpy(o.kernel_params, node.kernel_params, sizeof(o.kernel_params)); - o.opcode = node.opcode; - o.flags = 0; + if (!cpy_src || !cpy_dst || !cpy_dst->data) return false; + if (gdn_out->type != GGML_TYPE_F32 || cpy_src->type != GGML_TYPE_F32 || cpy_dst->type != GGML_TYPE_F32) return false; + if ((gdn_out->flags & GGML_TENSOR_FLAG_OUTPUT) || (cpy_node->flags & GGML_TENSOR_FLAG_OUTPUT)) return false; + + const ggml_tensor * v = last_node.node->src[2]; + if (!v) return false; + + const int64_t S_v = v->ne[0]; + const int64_t H = v->ne[1]; + const int64_t n_tokens = v->ne[2]; + const int64_t n_seqs = v->ne[3]; + const int64_t K = ggml_get_op_params_i32(last_node.node, 0); + const size_t tail_off = (size_t) S_v * H * n_tokens * n_seqs * sizeof(float); - if (!(opt_opstage & HTP_OPSTAGE_COMPUTE)) { - o.flags |= HTP_OPFLAGS_SKIP_COMPUTE; + const int64_t D = S_v * S_v * H; + const int64_t n_written = std::min(n_tokens, K); + + if (cpy_src->op != GGML_OP_VIEW || (cpy_src->view_src != gdn_out && cpy_src->view_src->data != gdn_out->data) || + cpy_src->view_offs != tail_off || !ggml_is_contiguous(cpy_src)) { + return false; } - ggml_hexagon_dump_op_exec(sess->c_name(), ops[n], o.flags); + if (cpy_dst->ne[0] != D || cpy_dst->ne[1] != n_seqs || cpy_dst->nb[0] != sizeof(float)) { + return false; + } + if (n_seqs > 1 && cpy_dst->nb[1] != (size_t) D * sizeof(float)) { + return false; + } + if (n_written > 1) { + if (cpy_dst->ne[2] != n_written || cpy_dst->nb[2] != (size_t) D * n_seqs * sizeof(float)) { + return false; + } + } - auto inputs = node.get_inputs(); - for (unsigned int i=0; i < HTP_OP_MAX_INPUTS; i++) { - o.src[i] = (i < inputs.size() && inputs[i]) ? add_tensor(inputs[i]) : 0xffff; + if (!try_fuse_common({cpy_dst})) { + return false; } - auto outputs = node.get_outputs(); - for (unsigned int i=0; i < HTP_OP_MAX_OUTPUTS; i++) { - o.dst[i] = (i < outputs.size() && outputs[i]) ? add_tensor(outputs[i]) : 0xffff; + last_node.name += "+CPY"; + last_node.outputs.push_back(cpy_dst); + last_node.fused.push_back(node.node); + + htp_op_desc & o = h_ops[n_ops - 1]; + o.dst[1] = add_tensor(cpy_dst); + for (uint32_t d = 2; d < HTP_OP_MAX_OUTPUTS; d++) { + o.dst[d] = 0xffff; } + + HEX_VERBOSE("ggml-hex: %s fused GATED_DELTA_NET+CPY (#%u)\n", sess->c_name(), n_ops - 1); + return true; } - void finalize_ranges() { + bool try_fuse(const htp_opnode & node) { + if (!opt_opfusion) return false; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_ALLREDUCE_ADD) && try_fuse_allreduce_add(node)) return true; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_RMS_NORM_MUL) && try_fuse_rms_norm_mul(node)) return true; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_MUL_MAT_ADD) && try_fuse_mul_mat_add(node)) return true; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_MUL_MAT_NX) && try_fuse_mul_mat_nx(node)) return true; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_MUL_MAT_ID_NX) && try_fuse_mul_mat_id_nx(node)) return true; + if (ggml_hexagon_is_fusion_enabled(GGML_HEXAGON_FUSE_GDN_CPY) && try_fuse_gdn_cpy(node)) return true; + return false; } }; +struct ggml_hexagon_registry { + ggml_hexagon_registry(ggml_backend_reg_t reg); + ~ggml_hexagon_registry(); + + ggml_backend_device devices[GGML_HEXAGON_MAX_SESSIONS]; +}; + struct ggml_hexagon_opqueue { // Shared buffer for storing batches ggml_hexagon_shared_buffer *shm_buf; size_t shm_blk_size; + size_t depth; using opvec = std::vector; - std::queue done; // completed batch ids std::vector op_cache; // per batch op cache std::vector start_usec; // per batch start time - ggml_hexagon_opqueue(ggml_hexagon_session *sess, size_t batch_size, size_t depth) { + ggml_hexagon_opqueue(ggml_hexagon_session *sess, size_t batch_size, size_t depth) : depth(depth) { size_t n_bufs = HTP_OP_MAX_BUFS; size_t n_ops = batch_size; size_t n_tensors = n_ops * HTP_OP_MAX_OUTPUTS + n_ops * HTP_OP_MAX_INPUTS; @@ -1425,12 +3172,9 @@ struct ggml_hexagon_opqueue { op_cache.resize(depth); start_usec.resize(depth, 0); - // init done queue - for (unsigned int i = 0; i < depth; i++) { done.push(i); } - if (opt_verbose) { - GGML_LOG_INFO("ggml-hex: %s allocated op-queue : batch-size %zu depth %zu shm-size %zu shm-block-size %zu\n", - sess->c_name(), batch_size, depth, shm_buf->size, shm_blk_size); + GGML_LOG_INFO("ggml-hex: %s allocated opqueue : batch-size %zu depth %zu shm-size %zu shm-block-size %zu\n", + sess->c_name(), batch_size, depth, shm_buf->size(), shm_blk_size); } } @@ -1438,8 +3182,10 @@ struct ggml_hexagon_opqueue { delete shm_buf; } + size_t shm_size() const { return shm_buf ? shm_buf->size() : 0; } + // push new batch - bool push(htp_opbatch_req& req, dspqueue_buffer& dbuf, ggml_hexagon_opbatch* op_batch) { + bool push(htp_opbatch_req& req, dspqueue_buffer& dbuf, const ggml_hexagon_opbatch* op_batch, uint64_t seq) { static_assert(sizeof(htp_opbatch_req) % 8 == 0, "sizeof(htp_opbatch_req) must be multiple of 8"); static_assert(sizeof(htp_opbatch_rsp) % 8 == 0, "sizeof(htp_opbatch_rsp) must be multiple of 8"); static_assert(sizeof(htp_buf_desc) % 8 == 0, "sizeof(htp_buf_desc) must be multiple of 8"); @@ -1447,15 +3193,17 @@ struct ggml_hexagon_opqueue { static_assert(sizeof(htp_op_desc) % 8 == 0, "sizeof(htp_op_desc) must be multiple of 8"); static_assert(sizeof(htp_prof_desc) % 8 == 0, "sizeof(htp_prof_desc) must be multiple of 8"); - if (done.empty()) { return false; } + if (seq - shm_buf->sess->batch_rsp_seq > depth) { return false; } - req.id = done.front(); done.pop(); // batch id + const uint32_t slot = (uint32_t) ((seq - 1) % depth); + + req.seq = seq; req.n_bufs = op_batch->n_bufs; req.n_tensors = op_batch->n_tens; req.n_ops = op_batch->n_ops; - op_cache[req.id] = op_batch->ops; - start_usec[req.id] = ggml_time_us(); + op_cache[slot] = op_batch->ops; + start_usec[slot] = ggml_time_us(); const size_t b_size = sizeof(htp_buf_desc) * req.n_bufs; const size_t t_size = sizeof(htp_tensor) * req.n_tensors; @@ -1470,10 +3218,10 @@ struct ggml_hexagon_opqueue { req.n_traces = 0; } - dbuf.ptr = shm_buf->base + (req.id * shm_blk_size); - dbuf.fd = shm_buf->fd; + dbuf.ptr = shm_buf->base() + ((size_t) slot * shm_blk_size); + dbuf.fd = shm_buf->fd(); dbuf.flags = DSPQUEUE_BUFFER_FLAG_FLUSH_SENDER | DSPQUEUE_BUFFER_FLAG_INVALIDATE_RECIPIENT; - dbuf.offset = (uint8_t*) dbuf.ptr - (uint8_t*) shm_buf->base; + dbuf.offset = (uint8_t*) dbuf.ptr - (uint8_t*) shm_buf->base(); dbuf.size = b_size + t_size + o_size + p_size + tr_size; GGML_ASSERT(dbuf.size <= shm_blk_size); @@ -1487,12 +3235,10 @@ struct ggml_hexagon_opqueue { memcpy(t_ptr, (void *) op_batch->h_tens.data(), t_size); memcpy(o_ptr, (void *) op_batch->h_ops.data(), o_size); - HEX_VERBOSE("ggml-hex: %s op-queue push batch #%u : n-bufs %u n-tensors %u n-ops %u vmem %zu : b-size %zu t-size %zu o-size %zu m-size %zu\n", - shm_buf->sess->c_name(), req.id, req.n_bufs, req.n_tensors, req.n_ops, op_batch->b_vmem, + HEX_VERBOSE("ggml-hex: %s opqueue-push batch #%llu : n-bufs %u n-tensors %u n-ops %u vmem %zu : b-size %zu t-size %zu o-size %zu m-size %zu\n", + shm_buf->sess->c_name(), (unsigned long long) req.seq, req.n_bufs, req.n_tensors, req.n_ops, op_batch->b_vmem, b_size, t_size, o_size, (size_t) dbuf.size); - op_batch->reset(); - if (opt_verbose > 1) { htp_buf_desc *b = (htp_buf_desc*) b_ptr; for (unsigned int i=0; i < req.n_bufs; i++) { @@ -1501,8 +3247,8 @@ struct ggml_hexagon_opqueue { } htp_tensor *t = (htp_tensor*) t_ptr; for (unsigned int i=0; i < req.n_tensors; i++) { - GGML_LOG_DEBUG("ggml-hex: %s htp-tensor #%u : bi %u offset %u size %u : %zu:%zu:%zu:%zu\n", - shm_buf->sess->c_name(), i, t[i].bi, t[i].data, t[i].size, + GGML_LOG_DEBUG("ggml-hex: %s htp-tensor #%u : bi %u offset %llu size %u : %zu:%zu:%zu:%zu\n", + shm_buf->sess->c_name(), i, t[i].bi, (unsigned long long) t[i].data, t[i].size, (size_t) t[i].ne[0], (size_t) t[i].ne[1], (size_t) t[i].ne[2], (size_t) t[i].ne[3]); } } @@ -1511,9 +3257,7 @@ struct ggml_hexagon_opqueue { } void pop(htp_opbatch_rsp rsp, dspqueue_buffer dbuf) { - GGML_ASSERT(rsp.id < op_cache.size()); - - done.push(rsp.id); + const uint32_t slot = (uint32_t) ((rsp.seq - 1) % depth); const size_t b_size = sizeof(htp_buf_desc) * rsp.n_bufs; const size_t t_size = sizeof(htp_tensor) * rsp.n_tensors; @@ -1530,40 +3274,72 @@ struct ggml_hexagon_opqueue { const size_t m_size = b_size + t_size + o_size + p_size + tr_size; GGML_ASSERT(m_size <= shm_blk_size); - HEX_VERBOSE("ggml-hex: %s op-queue pop batch #%u : n-bufs %u n-tensors %u n-ops %u : m-size %zu b-size %zu t-size %zu o-size %zu\n", - shm_buf->sess->c_name(), rsp.id, rsp.n_bufs, rsp.n_tensors, rsp.n_ops, + HEX_VERBOSE("ggml-hex: %s opqueue-pop batch #%llu : n-bufs %u n-tensors %u n-ops %u : m-size %zu b-size %zu t-size %zu o-size %zu\n", + shm_buf->sess->c_name(), (unsigned long long) rsp.seq, rsp.n_bufs, rsp.n_tensors, rsp.n_ops, (size_t) dbuf.size, b_size, t_size, o_size); uint8_t * m_ptr = (uint8_t*) dbuf.ptr; uint8_t * p_ptr = m_ptr + (b_size + t_size + o_size); - if (opt_profile && rsp.n_ops > 0) { - auto & ops = op_cache[rsp.id]; - + if (rsp.n_ops > 0) { + auto & ops = op_cache[slot]; GGML_ASSERT(rsp.n_ops <= ops.size()); const htp_prof_desc * pd = (const htp_prof_desc *) p_ptr; - const htp_trace_desc * trace_events = nullptr; - if (opt_profile == 3) { trace_events = (const htp_trace_desc *) (p_ptr + p_size); } - ggml_hexagon_dump_batch_prof(shm_buf->sess->name, rsp); + if (opt_profile) { + ggml_hexagon_dump_batch_prof(shm_buf->sess->name, rsp); + } for (uint32_t i = 0; i < rsp.n_ops; i++) { - ggml_hexagon_dump_op_prof(shm_buf->sess->name, ops[i], pd[i]); + if (opt_profile) { + ggml_hexagon_dump_op_prof(shm_buf->sess->name, ops[i], pd[i]); + } } - ggml_hexagon_dump_trace_events(shm_buf->sess->name, rsp, trace_events, n_traces); + if (opt_profile) { + ggml_hexagon_dump_trace_events(shm_buf->sess->name, rsp, trace_events, n_traces); + } } } }; -// Flush HTP response queue i.e wait for all outstanding requests to complete +void ggml_hexagon_session::flush_peers() { + auto vpeers = std::move(virt_peers); + virt_peers.clear(); + for (auto * peer : vpeers) { + peer->flush_sync(); + } + + auto ppeers = std::move(phys_peers); + phys_peers.clear(); + for (auto * peer : ppeers) { + peer->flush_async(); + } + + for (auto & sub : this->mdev.sessions) { + sub->flush_peers(); + } +} + +void ggml_hexagon_session::flush_async() { + flush_peers(); + flush_batch(); +} + void ggml_hexagon_session::flush_pending(bool all) { - while (this->op_pending) { + for (auto & sub : this->mdev.sessions) { + sub->flush_pending(all); + if (sub->last_error > HTP_STATUS_OK) { + this->last_error = sub->last_error.load(); + } + } + + while (this->batch_rsp_seq < this->batch_req_seq) { struct htp_opbatch_rsp rsp; uint32_t rsp_size; uint32_t flags; @@ -1588,34 +3364,70 @@ void ggml_hexagon_session::flush_pending(bool all) { GGML_ABORT("ggml-hex: %s dspcall : bad response : size %u dspbufs %u\n", this->c_name(), rsp_size, n_dbufs); } - if (rsp.status != HTP_STATUS_OK) { - GGML_LOG_ERROR("ggml-hex: %s dspcall : dsp-rsp: %s\n", this->c_name(), status_to_str(rsp.status)); - // TODO: handle errors + if (rsp.status > HTP_STATUS_OK) { + GGML_LOG_ERROR("ggml-hex: %s dspcall : dsp-rsp %s\n", this->c_name(), status_to_str(rsp.status)); + this->last_error = rsp.status; + for (auto & sub : this->mdev.sessions) { + sub->last_error = rsp.status; + } } op_queue->pop(rsp, dbuf); - this->op_pending--; // atomic dec + GGML_ASSERT(rsp.seq == this->batch_rsp_seq + 1); + this->batch_rsp_seq = rsp.seq; if (!all) break; } } -void ggml_hexagon_session::flush_batch() { - if (op_batch->empty()) { return; } +void ggml_hexagon_session::flush_sync(bool all) { + flush_async(); + flush_pending(all); +} + +void ggml_hexagon_session::start_batch() { + if (this->mdev.count > 1) { + enqueue_mdev_group(); + } +} + +void ggml_hexagon_session::flush_batch(size_t min_ops) { + if (op_batch->n_ops < min_ops) { return; } - op_batch->finalize_ranges(); + op_batch->sort_buffers(); htp_opbatch_req req {}; dspqueue_buffer dbuf{}; - if (!op_queue->push(req, dbuf, op_batch)) { + const uint64_t seq = ++this->batch_req_seq; + + op_batch->update_mdev_group(this->mdev.idx); + + if (!op_queue->push(req, dbuf, op_batch, seq)) { flush_pending(false); - op_queue->push(req, dbuf, op_batch); + op_queue->push(req, dbuf, op_batch, seq); } - // Bump pending flag (cleared in the session::flush once we get the response) - this->op_pending++; // atomic inc + for (auto & sub : this->mdev.sessions) { + htp_opbatch_req sub_req {}; + dspqueue_buffer sub_dbuf{}; + + sub->batch_req_seq = seq; + op_batch->update_mdev_group(sub->mdev.idx); + + if (!sub->op_queue->push(sub_req, sub_dbuf, op_batch, seq)) { + sub->flush_pending(false); + sub->op_queue->push(sub_req, sub_dbuf, op_batch, seq); + } + + HEX_VERBOSE("ggml-hex: %s queue-opbatch: %p size %u\n", sub->c_name(), sub_dbuf.ptr, sub_dbuf.size); + + int err = dspqueue_write(sub->queue, 0, 1, &sub_dbuf, sizeof(sub_req), (const uint8_t*) &sub_req, DSPQUEUE_TIMEOUT); + if (err != 0) { + GGML_ABORT("ggml-hex: %s dspqueue_write failed: 0x%08x\n", sub->c_name(), (unsigned) err); + } + } HEX_VERBOSE("ggml-hex: %s queue-opbatch: %p size %u\n", this->c_name(), dbuf.ptr, dbuf.size); @@ -1623,19 +3435,350 @@ void ggml_hexagon_session::flush_batch() { if (err != 0) { GGML_ABORT("ggml-hex: %s dspqueue_write failed: 0x%08x\n", this->c_name(), (unsigned) err); } + + op_batch->reset(); } void ggml_hexagon_session::enqueue_op(const htp_opnode & node) { + auto clone_tensor_buffer = [this](const ggml_tensor * t) { + auto sbuf = this->mmap_tensor(t); + if (!sbuf) return; + if (sbuf->sess != this) { + this->clone_buffer(sbuf); + } + for (auto & sub : this->mdev.sessions) { + sub->clone_buffer(sbuf); + } + }; + + for (auto t : node.get_inputs()) { + clone_tensor_buffer(t); + } + for (auto t : node.get_outputs()) { + clone_tensor_buffer(t); + } + + if (opt_opfusion && op_batch->try_fuse(node)) { + return; + } + + if (!op_batch->fit_op(node)) { + flush_async(); + } + + if (op_batch->empty()) { + start_batch(); + } + if (!op_batch->fit_op(node)) { - flush_batch(); + GGML_ABORT("ggml-hex: %s op does not fit into empty batch (vmem/tensor/buffer limit exceeded)\n", + c_name()); + } + + op_batch->add_op(node); +} + +void ggml_hexagon_session::enqueue_mdev_group() { + htp_opnode group_node(HTP_OP_MDEV_GROUP); + + uint8_t * fence_slot = this->mdev_fence_slot; + + static ggml_hexagon_tensor_extra fence_extra { {}, 0, GGML_HEXAGON_TENSOR_FENCE }; + ggml_tensor dummy_t {}; + dummy_t.buffer = &this->fence_buf->backend_buffer; + dummy_t.extra = &fence_extra; + dummy_t.data = (void *) fence_slot; + dummy_t.type = GGML_TYPE_I8; + dummy_t.ne[0] = HTP_FENCE_SLOT_SIZE; + dummy_t.ne[1] = (int64_t) this->mdev.count; + dummy_t.ne[2] = 1; + dummy_t.ne[3] = 1; + dummy_t.nb[0] = 1; + dummy_t.nb[1] = HTP_FENCE_SLOT_SIZE; + dummy_t.nb[2] = dummy_t.nb[1] * dummy_t.ne[1]; + dummy_t.nb[3] = dummy_t.nb[2]; + dummy_t.op = GGML_OP_NONE; + dummy_t.op_params[0] = (int32_t) this->mdev.idx; + + ggml_tensor * node = group_node.add_dummy(dummy_t); + node->src[0] = node; + group_node.init(node); + group_node.outputs.clear(); + group_node.name = "MDEV_GROUP"; + + for (auto & sub : this->mdev.sessions) { + sub->clone_buffer(this->fence_buf); + } + + op_batch->add_op(group_node); +} + +void ggml_hexagon_session::enqueue_cpy(const ggml_tensor * src, ggml_tensor * dst, const ggml_tensor * sync_tensor, uint32_t fence_seq) { + const bool with_fence = sync_tensor != nullptr; + htp_opnode cpy_node(with_fence ? HTP_OP_CPY_FENCE : HTP_OP_CPY); + + ggml_tensor* node = cpy_node.add_dummy(*dst); + node->op = GGML_OP_CPY; + node->src[0] = const_cast(src); + node->src[1] = with_fence ? cpy_node.add_dummy(*sync_tensor) : nullptr; + if (with_fence) { + node->op_params[0] = (int32_t) fence_seq; + } + + cpy_node.init(node); + if (with_fence) { + cpy_node.name = "CPY+FENCE"; + } + this->enqueue_op(cpy_node); +} + +void ggml_hexagon_session::enqueue_fence(const ggml_tensor * sync_tensor, uint32_t fence_seq, bool wait) { + htp_opnode sync_node(HTP_OP_FENCE); + + ggml_tensor* node = sync_node.add_dummy(*sync_tensor); + node->op = GGML_OP_NONE; + node->src[0] = node; + node->op_params[0] = (int32_t) fence_seq; + node->op_params[1] = wait ? 0 : 1; + + sync_node.init(node); + sync_node.name = wait ? "FENCE_WAIT" : "FENCE_SIGNAL"; + this->enqueue_op(sync_node); +} + +static bool ggml_hexagon_precompute_allreduce_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * dst, + uint32_t rank, + uint32_t n_ranks, + bool has_add, + bool is_row_bcast, + struct htp_allreduce_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + kparams->rank = (int32_t) rank; + kparams->n_ranks = (int32_t) n_ranks; + kparams->is_row_bcast = (has_add && is_row_bcast) ? 1 : 0; + + const uint32_t nelem = (uint32_t) ggml_nelements(dst); + const uint32_t elem_size = (dst->type == GGML_TYPE_F16) ? sizeof(ggml_fp16_t) : sizeof(float); + const bool is_contiguous = ggml_is_contiguous(dst); + + const uint32_t ne0 = (uint32_t) dst->ne[0]; + const uint32_t ne1 = (uint32_t) (dst->ne[1] * dst->ne[2] * dst->ne[3]); + kparams->ne0 = (int32_t) ne0; + kparams->ne1 = (int32_t) ne1; + + const bool use_1d = is_contiguous && !(has_add && is_row_bcast && ne1 > 1); + + if (has_add) { + kparams->n_dsts = 1; + if (use_1d) { + kparams->rank_elem_start = 0; + kparams->rank_nelem = (int32_t) nelem; + } else { + kparams->rank_elem_start = 0; + kparams->rank_nelem = (int32_t) ne1; + } + } else { + kparams->n_dsts = (int32_t) n_ranks; + if (use_1d) { + const uint32_t rank_chunk_elems = hex_round_up((nelem + n_ranks - 1) / n_ranks, 128); + const uint32_t rank_elem_start = (std::min)(rank * rank_chunk_elems, nelem); + const uint32_t rank_elem_end = (std::min)(rank_elem_start + rank_chunk_elems, nelem); + const uint32_t rank_nelem = rank_elem_end - rank_elem_start; + kparams->rank_elem_start = (int32_t) rank_elem_start; + kparams->rank_nelem = (int32_t) rank_nelem; + } else { + const uint32_t rank_chunk_rows = (ne1 + n_ranks - 1) / n_ranks; + const uint32_t rank_r0 = (std::min)(rank * rank_chunk_rows, ne1); + const uint32_t rank_r1 = (std::min)(rank_r0 + rank_chunk_rows, ne1); + const uint32_t rank_nrows = rank_r1 - rank_r0; + kparams->rank_elem_start = (int32_t) rank_r0; + kparams->rank_nelem = (int32_t) rank_nrows; + } + } + + if (use_1d) { + const uint32_t rank_nelem = (uint32_t) kparams->rank_nelem; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, (std::max)(1u, rank_nelem / 128)); + kparams->n_threads = n_threads; + const size_t n_vtcm_buffers = htp_allreduce_vtcm_buffer_count(n_ranks, n_threads, has_add, is_row_bcast); + + uint32_t block_elems = 65536; + if (block_elems > rank_nelem / n_threads && rank_nelem / n_threads > 128) { + block_elems = hex_round_up(rank_nelem / (n_threads * 2), 128); + } + block_elems = (std::max)(128u, block_elems); + + kparams->block_elems = block_elems; + kparams->vtcm_size_per_thread = 2 * block_elems * elem_size; + kparams->vtcm_size = n_vtcm_buffers * kparams->vtcm_size_per_thread; + + while ((size_t) kparams->vtcm_size > sess->vtcm_size && block_elems > 128) { + const size_t max_bytes_per_buf = sess->vtcm_size / (n_vtcm_buffers * 2); + block_elems = (uint32_t) hex_align_down((size_t) (max_bytes_per_buf / elem_size), 128); + if (block_elems < 128) break; + kparams->block_elems = block_elems; + kparams->vtcm_size_per_thread = 2 * block_elems * elem_size; + kparams->vtcm_size = n_vtcm_buffers * kparams->vtcm_size_per_thread; + } + + if (sess->vtcm_size < (size_t) kparams->vtcm_size || block_elems < 128) { + HEX_VERBOSE("ggml-hex: %s allreduce 1D solver failed to fit VTCM (%d > %zu)\n", + sess->c_name(), kparams->vtcm_size, sess->vtcm_size); + return false; + } + + kparams->elems_per_thread = hex_round_up((rank_nelem + n_threads - 1) / n_threads, block_elems); + kparams->kernel_type = HTP_ALLREDUCE_KERNEL_DMA_1D; + return true; + } else { + const uint32_t rank_nrows = (uint32_t) kparams->rank_nelem; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, (std::max)(1u, rank_nrows)); + kparams->n_threads = n_threads; + const size_t n_vtcm_buffers = htp_allreduce_vtcm_buffer_count(n_ranks, n_threads, has_add, is_row_bcast); + + const uint32_t row_bytes = ne0 * elem_size; + const uint32_t row_size_aligned = (uint32_t) hex_align_up(row_bytes, 128); + kparams->row_size_aligned = row_size_aligned; + + const uint32_t nrows_per_thread = (rank_nrows + n_threads - 1) / n_threads; + uint32_t block_rows = (std::min)(128u, nrows_per_thread); + block_rows = (std::max)(1u, block_rows); + kparams->block_elems = block_rows; + + kparams->vtcm_size_per_thread = 2 * (block_rows * row_size_aligned); + kparams->vtcm_size = n_vtcm_buffers * kparams->vtcm_size_per_thread; + + while ((size_t) kparams->vtcm_size > sess->vtcm_size && block_rows > 1) { + const size_t max_rows_per_buf = sess->vtcm_size / (n_vtcm_buffers * 2 * row_size_aligned); + block_rows = (std::max)(1u, (uint32_t) max_rows_per_buf); + kparams->block_elems = block_rows; + kparams->vtcm_size_per_thread = 2 * (block_rows * row_size_aligned); + kparams->vtcm_size = n_vtcm_buffers * kparams->vtcm_size_per_thread; + if (max_rows_per_buf == 0) break; + } + + if (sess->vtcm_size < (size_t) kparams->vtcm_size || block_rows < 1) { + HEX_VERBOSE("ggml-hex: %s allreduce 2D solver failed to fit VTCM (%d > %zu)\n", + sess->c_name(), kparams->vtcm_size, sess->vtcm_size); + return false; + } + + kparams->elems_per_thread = nrows_per_thread; + kparams->kernel_type = HTP_ALLREDUCE_KERNEL_DMA_2D; + return true; + } +} + +void ggml_hexagon_session::enqueue_allreduce( + const ggml_tensor * dst, + const std::vector & src_tensors, + const std::vector & sync_tensors, + uint32_t rank, + uint32_t n_ranks, + uint32_t fence_seq_entry, + uint32_t fence_seq_exit +) { + htp_opnode ar_node(HTP_OP_ALLREDUCE); + + ggml_tensor* node = ar_node.add_dummy(*dst); + node->op = GGML_OP_NONE; + node->op_params[0] = (int32_t) fence_seq_entry; + node->op_params[1] = (int32_t) fence_seq_exit; + + ar_node.init(node); + + ar_node.inputs.clear(); + for (size_t i = 0; i < src_tensors.size(); i++) { + ar_node.inputs.push_back(src_tensors[i]); + } + for (size_t i = 0; i < sync_tensors.size(); i++) { + ar_node.inputs.push_back(ar_node.add_dummy(*sync_tensors[i])); + } + + ar_node.outputs.clear(); + for (size_t i = 0; i < src_tensors.size(); i++) { + ar_node.outputs.push_back(src_tensors[i]); + } + + ggml_hexagon_precompute_allreduce_params( + this, dst, rank, n_ranks, false, false, + (struct htp_allreduce_kernel_params *) ar_node.kernel_params + ); + + ar_node.name = "ALLREDUCE"; + this->enqueue_op(ar_node); +} + +ggml_hexagon_shared_buffer * ggml_hexagon_session::mmap_tensor(const ggml_tensor * t) { + if (!t) return nullptr; + + auto sbuf = static_cast(t->buffer->context); + if (!sbuf->mapped) { + const bool is_weight = ggml_backend_buffer_get_usage(t->buffer) == GGML_BACKEND_BUFFER_USAGE_WEIGHTS; + const bool extended = opt_dma64 && is_weight; + sbuf->mmap(extended); + } + return sbuf; +} + +bool ggml_hexagon_session::clone_buffer(const ggml_hexagon_shared_buffer * sbuf) +{ + GGML_ASSERT(sbuf && sbuf->mem); + if (sbuf->sess == this) return true; + + auto mem = sbuf->mem; + int fd = mem->fd; + + GGML_ASSERT(fd >= 0); + + if (this->cloned_buffers.find(fd) != this->cloned_buffers.end()) return true; + + GGML_ASSERT(sbuf->mapped); + + HEX_VERBOSE("ggml-hex: %s clone-buffer: %s base %p size %zu fd %d\n", this->name.c_str(), + sbuf->c_name(), sbuf->base(), sbuf->size(), fd); + + auto clone = std::make_unique(this, *sbuf); + try { + clone->mmap(sbuf->extended); + } catch (const std::exception & exc) { + GGML_LOG_ERROR("ggml-hex: %s lazy mapping of buffer context failed: %s\n", this->c_name(), exc.what()); + return false; + } + + this->cloned_buffers[fd] = std::move(clone); + mem->mapped_clones.insert(this); + return true; +} + +void ggml_hexagon_session::release_buffer(const ggml_hexagon_shared_buffer * sbuf) { + GGML_ASSERT(sbuf && sbuf->mem); + + auto mem = sbuf->mem; + int fd = mem->fd; + + GGML_ASSERT(fd >= 0); + + auto it = this->cloned_buffers.find(fd); + if (it != this->cloned_buffers.end()) { + auto clone = std::move(it->second); + this->cloned_buffers.erase(it); } - op_batch->add_op(node); + mem->mapped_clones.erase(this); } -// Flush HTP response queue i.e wait for all outstanding requests to complete -void ggml_hexagon_session::flush(bool all) { - flush_batch(); - flush_pending(all); +void ggml_hexagon_session::unclone_buffer(const ggml_hexagon_shared_buffer * sbuf) { + GGML_ASSERT(sbuf && sbuf->mem); + + auto mem = sbuf->mem; + std::vector sessions(mem->mapped_clones.begin(), mem->mapped_clones.end()); + + for (auto * sess : sessions) { + sess->release_buffer(sbuf); + } } static size_t ggml_hexagon_measure_max_vmem(ggml_hexagon_session *sess) { @@ -1667,38 +3810,46 @@ static size_t ggml_hexagon_measure_max_vmem(ggml_hexagon_session *sess) { return vmem - step; // backoff to account for overhead from internal mappings } -void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { +void ggml_hexagon_session::allocate(const ggml_hexagon_device_config & config) noexcept(false) { + int phys_idx = config.physical_idx; + int virt_idx = config.virtual_idx; + this->valid_session = false; this->valid_handle = false; this->valid_queue = false; this->valid_iface = false; - this->domain_id = 3; // Default for CDSP, updated after the session is created - this->session_id = 0; // Default for CDSP, updated after the session is created - this->dev_id = dev_id; - this->name = std::string("HTP") + std::to_string(dev_id); + this->name = config.name; + this->phys_idx = phys_idx; + this->virt_idx = virt_idx; + this->domain_id = config.domain_id; + this->session_id = 0; + this->batch_req_seq = 0; + this->batch_rsp_seq = 0; + this->last_error = HTP_STATUS_OK; - this->op_pending = 0; + GGML_LOG_DEBUG("ggml-hex: %s allocating new session : domain %u phys-idx %u virt-idx %u\n", this->name.c_str(), this->domain_id, phys_idx, virt_idx); - GGML_LOG_DEBUG("ggml-hex: %s allocating new session\n", this->name.c_str()); - - domain * my_domain = htpdrv_get_domain(this->domain_id); - if (my_domain == NULL) { - GGML_LOG_ERROR("ggml-hex: unable to get domain struct for CDSP\n"); - throw std::runtime_error("ggml-hex: failed to get CDSP domain (see log for details)"); + if (config.domain_id < 0 || config.domain_name.empty()) { + GGML_LOG_ERROR("ggml-hex: %s: invalid physical CDSP core %d\n", config.name.c_str(), config.physical_idx); + throw std::runtime_error("ggml-hex: invalid physical CDSP core"); } - // Create new session - if (dev_id != 0) { - struct remote_rpc_reserve_new_session n; - n.domain_name_len = strlen(CDSP_DOMAIN_NAME); - n.domain_name = const_cast(CDSP_DOMAIN_NAME); + const std::string & dom_name = config.domain_name; + + // Create new session if virtual_idx > 0 + if (virt_idx > 0) { + struct remote_rpc_reserve_new_session n {}; + n.domain_name_len = dom_name.size(); + n.domain_name = const_cast(dom_name.c_str()); n.session_name = const_cast(this->name.c_str()); n.session_name_len = this->name.size(); + n.session_id = virt_idx; int err = remote_session_control(FASTRPC_RESERVE_NEW_SESSION, (void *) &n, sizeof(n)); if (err != AEE_SUCCESS) { - GGML_LOG_ERROR("ggml-hex: failed to reserve new session %d : error 0x%x\n", dev_id, err); + GGML_LOG_ERROR("ggml-hex: %s failed to reserve new session (physical %d, virtual %d) : error 0x%x\n", + this->c_name(), phys_idx, virt_idx, err); throw std::runtime_error("ggml-hex: remote_session_control(new-sess) failed (see log for details)"); } @@ -1706,9 +3857,32 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { this->session_id = n.session_id; this->domain_id = n.effective_domain_id; this->valid_session = true; + } else { + struct remote_rpc_effective_domain_id eff {}; + eff.domain_name = const_cast(dom_name.c_str()); + eff.domain_name_len = dom_name.size(); + eff.session_id = 0; + + int err = remote_session_control(FASTRPC_GET_EFFECTIVE_DOMAIN_ID, (void *) &eff, sizeof(eff)); + if (err == AEE_SUCCESS) { + this->domain_id = eff.effective_domain_id; + } else { + GGML_LOG_DEBUG("ggml-hex: %s FASTRPC_GET_EFFECTIVE_DOMAIN_ID returned 0x%x, using domain_id %d\n", + this->name.c_str(), err, this->domain_id); + } } - // Get session URI + // Enable unsigned modules + { + struct remote_rpc_control_unsigned_module u; + u.domain = this->domain_id; + u.enable = 1; + int err = remote_session_control(DSPRPC_CONTROL_UNSIGNED_MODULE, (void *) &u, sizeof(u)); + if (err != AEE_SUCCESS) { + GGML_LOG_ERROR("ggml-hex: %s failed to enable unsigned PD : error 0x%x\n", this->c_name(), err); + throw std::runtime_error("ggml-hex: remote_session_control(unsign) failed (see log for details)"); + } + } char session_uri[256]; { @@ -1717,8 +3891,8 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { struct remote_rpc_get_uri u = {}; u.session_id = this->session_id; - u.domain_name = const_cast(CDSP_DOMAIN_NAME); - u.domain_name_len = strlen(CDSP_DOMAIN_NAME); + u.domain_name = const_cast(dom_name.c_str()); + u.domain_name_len = dom_name.size(); u.module_uri = const_cast(htp_uri); u.module_uri_len = strlen(htp_uri); u.uri = session_uri; @@ -1726,38 +3900,24 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { int err = remote_session_control(FASTRPC_GET_URI, (void *) &u, sizeof(u)); if (err != AEE_SUCCESS) { - // fallback to single session uris - int htp_URI_domain_len = strlen(htp_uri) + MAX_DOMAIN_NAMELEN; - - snprintf(session_uri, htp_URI_domain_len, "%s%s", htp_uri, my_domain->uri); - - GGML_LOG_WARN("ggml-hex: failed to get URI for session %d : error 0x%x. Falling back to single session URI: %s\n", dev_id, err, session_uri); - } - } + snprintf(session_uri, sizeof(session_uri), "%s&_dom=%s&_session=%u", + htp_uri, dom_name.c_str(), this->session_id); - // Enable Unsigned PD - { - struct remote_rpc_control_unsigned_module u; - u.domain = this->domain_id; - u.enable = 1; - int err = remote_session_control(DSPRPC_CONTROL_UNSIGNED_MODULE, (void *) &u, sizeof(u)); - if (err != AEE_SUCCESS) { - GGML_LOG_ERROR("ggml-hex: failed to enable unsigned PD for session %d : error 0x%x\n", dev_id, err); - throw std::runtime_error("ggml-hex: remote_session_control(unsign) failed (see log for details)"); + GGML_LOG_WARN("ggml-hex: %s failed to get URI (physical %d, virtual %d) : error 0x%x. Falling back to single session URI: %s\n", + this->c_name(), phys_idx, virt_idx, err, session_uri); } } // Open session int err = htp_iface_open(session_uri, &this->handle); if (err != AEE_SUCCESS) { - GGML_LOG_ERROR("ggml-hex: failed to open session %d : error 0x%x\n", dev_id, err); + GGML_LOG_ERROR("ggml-hex: %s failed to open session : uri %s error 0x%x\n", this->c_name(), session_uri, err); throw std::runtime_error("ggml-hex: failed to open session (see log for details)"); } this->valid_handle = true; // Query HW info and resolve session options - this->max_bufsize = opt_mbuf; { unsigned int hw_n_threads = 0; unsigned int hw_n_hvx = 0; @@ -1765,8 +3925,9 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { unsigned long long hw_vtcm_size = 0; int hw_err = htp_iface_hwinfo(this->handle, &hw_n_threads, &hw_n_hvx, &hw_n_hmx, &hw_vtcm_size); if (hw_err == 0) { - this->n_threads = opt_nhvx > 0 ? (uint32_t)opt_nhvx : (uint32_t)hw_n_threads; - this->n_hvx = opt_nhvx > 0 ? (uint32_t)opt_nhvx : (uint32_t)hw_n_hvx; + const uint32_t max_n_threads = (std::min)((uint32_t) HTP_MAX_NTHREADS, (uint32_t) hw_n_threads); + this->n_threads = opt_nhvx > 0 ? (uint32_t) (std::min)(opt_nhvx, (size_t) max_n_threads) : max_n_threads; + this->n_hvx = this->n_threads; this->n_hmx = (opt_nhmx != 0) ? (uint32_t)hw_n_hmx : 0; this->vtcm_size = (uint64_t)hw_vtcm_size; GGML_LOG_INFO("ggml-hex: %s hwinfo: threads %u, hvx %u, hmx %u, vtcm %llu MB\n", @@ -1774,8 +3935,9 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { (unsigned long long)(this->vtcm_size / (1024 * 1024))); } else { GGML_LOG_WARN("ggml-hex: %s failed to query hwinfo (0x%x), using defaults\n", this->c_name(), hw_err); - this->n_threads = opt_nhvx > 0 ? (uint32_t)opt_nhvx : 8; - this->n_hvx = opt_nhvx > 0 ? (uint32_t)opt_nhvx : 8; + const uint32_t default_n_threads = (std::min)(8u, (uint32_t) HTP_MAX_NTHREADS); + this->n_threads = opt_nhvx > 0 ? (uint32_t) (std::min)(opt_nhvx, (size_t) HTP_MAX_NTHREADS) : default_n_threads; + this->n_hvx = this->n_threads; this->n_hmx = (opt_nhmx != 0) ? 1 : 0; this->vtcm_size = 8 * 1024 * 1024; } @@ -1831,16 +3993,22 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { // Allocate buffers and state for op batching this->op_queue = new ggml_hexagon_opqueue(this, opt_opbatch, opt_opqueue); + this->fence_buf = new ggml_hexagon_fence_buffer(this, &dev_ctx->fence_buffer_type, 64 * 1024); + if (this->mdev.count > 1) { + this->mdev_fence_slot = this->alloc_fence(this->mdev.count); + } + if (!opt_vmem) { opt_vmem = ggml_hexagon_measure_max_vmem(this); GGML_LOG_INFO("ggml-hex: %s measured max vmem %zu\n", this->c_name(), opt_vmem); } - this->max_vmem = opt_vmem; + const size_t shm_size = this->op_queue->shm_size(); + this->max_vmem = (opt_vmem > shm_size) ? (opt_vmem - shm_size) : opt_vmem; this->op_batch = new ggml_hexagon_opbatch(this, opt_opbatch, this->max_vmem); // Start dspqueue/opbatch processing - err = htp_iface_start(this->handle, dev_id, this->queue_id, opt_nhvx, opt_nhmx, this->max_vmem); + err = htp_iface_start(this->handle, this->session_id, this->queue_id, this->n_threads, opt_nhmx, this->max_vmem); if (err != 0) { GGML_LOG_ERROR("ggml-hex: %s failed to start session: 0x%08x\n", this->c_name(), (unsigned) err); throw std::runtime_error("ggml-hex: iface start failed (see log for details)"); @@ -1861,6 +4029,8 @@ void ggml_hexagon_session::allocate(int dev_id) noexcept(false) { void ggml_hexagon_session::release() noexcept(true) { GGML_LOG_INFO("ggml-hex: releasing session: %s\n", this->name.c_str()); + this->mdev.sessions.clear(); + int err; if (this->valid_iface) { @@ -1873,6 +4043,19 @@ void ggml_hexagon_session::release() noexcept(true) { delete this->op_batch; delete this->op_queue; + for (auto & it : this->cpy_fence_slots) { + free_fence((void *) it.second, 1); + } + this->cpy_fence_slots.clear(); + + if (this->fence_buf) { + unclone_buffer(this->fence_buf); + delete this->fence_buf; + this->fence_buf = nullptr; + } + while (!this->cloned_buffers.empty()) { + release_buffer(this->cloned_buffers.begin()->second.get()); + } if (opt_etm) { err = htp_iface_etm(this->handle, 0); @@ -1901,21 +4084,24 @@ void ggml_hexagon_session::release() noexcept(true) { } } -ggml_hexagon_session::ggml_hexagon_session(int dev_id, ggml_backend_dev_t dev) noexcept(false) { - buffer_type.device = dev; - repack_buffer_type.device = dev; - - op_batch = nullptr; - op_queue = nullptr; +ggml_hexagon_session::ggml_hexagon_session(const ggml_hexagon_device_config & config, ggml_backend_dev_t dev, uint32_t mdev_idx, uint32_t mdev_count) noexcept(false) { + this->dev = dev; + this->dev_ctx = static_cast(dev->context); + this->mdev.idx = mdev_idx; + this->mdev.count = mdev_count > 0 ? mdev_count : (uint32_t) (1 + config.mdev_group.size()); + op_batch = nullptr; + op_queue = nullptr; + fence_buf = nullptr; + fence_seq = ((uintptr_t)this) & 0xFFFF; try { - allocate(dev_id); - - buffer_type.iface = ggml_backend_hexagon_buffer_type_interface; - buffer_type.context = new ggml_backend_hexagon_buffer_type_context(this->name, this); - - repack_buffer_type.iface = ggml_backend_hexagon_repack_buffer_type_interface; - repack_buffer_type.context = new ggml_backend_hexagon_buffer_type_context(this->name + "-REPACK", this); + allocate(config); + if (this->mdev.idx == 0 && !config.mdev_group.empty()) { + for (size_t i = 0; i < config.mdev_group.size(); i++) { + this->mdev.sessions.push_back(std::make_unique( + config.mdev_group[i], this->dev, (uint32_t) (i + 1), this->mdev.count)); + } + } } catch (const std::exception & exc) { release(); throw; @@ -1924,9 +4110,6 @@ ggml_hexagon_session::ggml_hexagon_session(int dev_id, ggml_backend_dev_t dev) n ggml_hexagon_session::~ggml_hexagon_session() noexcept(true) { release(); - - delete static_cast(buffer_type.context); - delete static_cast(repack_buffer_type.context); } // ** backend interface @@ -1946,14 +4129,17 @@ static bool ggml_hexagon_flash_attn_is_hmx_eligible( return false; } - if (k->type != GGML_TYPE_F16 || v->type != GGML_TYPE_F16) { + if ((k->type != GGML_TYPE_F16 && k->type != GGML_TYPE_Q8_0) || + (v->type != GGML_TYPE_F16 && v->type != GGML_TYPE_Q8_0)) { return false; } const uint32_t DK = q->ne[0]; const uint32_t DV = v->ne[0]; - if (DK % 64 != 0 || DV % 64 != 0) { + // Head dims that are not multiples of 64 are handled by internally padding to + // DK_pad/DV_pad = round_up(.,64) and zero-filling the tail lanes. + if (DK % 8 != 0 || DV % 8 != 0) { return false; } @@ -2026,8 +4212,13 @@ static bool ggml_hexagon_precompute_flash_attn_params( // Check HMX eligibility const struct ggml_tensor * sinks = op->src[4]; if (ggml_hexagon_flash_attn_is_hmx_eligible(sess, q, k, v, sinks)) { + // HMX tiles head_dim in units of 64; when DK/DV are not 64-aligned the kernel + // operates on padded dims with zero-filled tail lanes. VTCM budget and chunk-size + // are sized for the padded tiles. + const uint32_t DK_pad = hex_round_up(DK, 64); + const uint32_t DV_pad = hex_round_up(DV, 64); size_t Br = 0, Bc = 0; - int ret = hmx_fa_find_chunk_size(&Br, &Bc, G, DK, DV, neq1, nek1, sess->vtcm_size, sess->n_threads, kparams->is_q_fp32 != 0); + int ret = hmx_fa_find_chunk_size(&Br, &Bc, G, DK_pad, DV_pad, neq1, nek1, sess->vtcm_size, sess->n_threads, kparams->is_q_fp32 != 0, sinks != nullptr, n_head); if (ret == 0) { kparams->kernel_type = HTP_FA_KERNEL_HMX; kparams->Br = Br; @@ -2037,7 +4228,7 @@ static bool ggml_hexagon_precompute_flash_attn_params( kparams->u.hmx.g_br = hex_align_up(G * Br, 32); kparams->u.hmx.pipeline = (kparams->n_kv_blocks >= 3 && sess->n_threads >= 2) ? 1 : 0; - kparams->vtcm_size = hmx_fa_compute_vtcm_usage(G, DK, DV, Br, Bc, kparams->n_threads, kparams->u.hmx.pipeline != 0, kparams->is_q_fp32 != 0); + kparams->vtcm_size = hmx_fa_compute_vtcm_usage(G, DK_pad, DV_pad, Br, Bc, kparams->n_threads, kparams->u.hmx.pipeline != 0, kparams->is_q_fp32 != 0, sinks != nullptr, n_head); const size_t row_vec_bytes = hex_align_up(Bc * sizeof(uint16_t), 256); kparams->u.hmx.row_buf_stride = row_vec_bytes / 128; // HVX vector is 128 bytes @@ -2068,7 +4259,7 @@ static bool ggml_hexagon_precompute_flash_attn_params( const size_t size_k_row_padded = hex_round_up(k->ne[0] * 2, 128); const size_t size_v_row_padded = hex_round_up(v->ne[0] * 2, 128); - kparams->vtcm_size = hvx_fa_compute_vtcm_usage(DK, DV, kparams->is_q_fp32 != 0, mask != nullptr, sess->n_threads); + kparams->vtcm_size = hvx_fa_compute_vtcm_usage(DK, DV, kparams->is_q_fp32 != 0, mask != nullptr, sinks != nullptr, n_head, sess->n_threads); kparams->u.hvx.size_q_row_padded = size_q_row_padded; kparams->u.hvx.size_k_row_padded = size_k_row_padded; @@ -2098,8 +4289,10 @@ static bool ggml_hexagon_supported_flash_attn_ext(const struct ggml_hexagon_sess const struct ggml_tensor * src4 = op->src[4]; const struct ggml_tensor * dst = op; - // Check for F16 support only as requested - if ((src0->type != GGML_TYPE_F16 && src0->type != GGML_TYPE_F32) || src1->type != GGML_TYPE_F16 || src2->type != GGML_TYPE_F16) { + // Check for F16/Q8_0 support + if ((src0->type != GGML_TYPE_F16 && src0->type != GGML_TYPE_F32) || + (src1->type != GGML_TYPE_F16 && src1->type != GGML_TYPE_Q8_0) || + (src2->type != GGML_TYPE_F16 && src2->type != GGML_TYPE_Q8_0)) { return false; } @@ -2137,6 +4330,10 @@ static bool ggml_hexagon_supported_flash_attn_ext(const struct ggml_hexagon_sess } static bool ggml_hexagon_supported_gated_delta_net(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { + if (opt_gdn_select < 1) { + return false; + } + const struct ggml_tensor * q = op->src[0]; const struct ggml_tensor * k = op->src[1]; const struct ggml_tensor * v = op->src[2]; @@ -2145,10 +4342,6 @@ static bool ggml_hexagon_supported_gated_delta_net(const struct ggml_hexagon_ses const struct ggml_tensor * state = op->src[5]; const struct ggml_tensor * dst = op; - if (!q || !k || !v || !g || !beta || !state) { - return false; - } - if (q->type != GGML_TYPE_F32 || k->type != GGML_TYPE_F32 || v->type != GGML_TYPE_F32 || g->type != GGML_TYPE_F32 || beta->type != GGML_TYPE_F32 || state->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { @@ -2186,9 +4379,31 @@ static bool ggml_hexagon_supported_gated_delta_net(const struct ggml_hexagon_ses return false; } - return true; + const uint32_t total_rows = (uint32_t) (H * n_seqs); + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, total_rows); - GGML_UNUSED(sess); + const bool can_use_hmx = (opt_gdn_select >= 2) && + (sess->n_hmx > 0) && + (S_v % 64 == 0) && + (n_tokens >= HTP_GDN_MIN_TOKENS) && + (g->ne[0] == 1) && + (K == 1); + + if (can_use_hmx) { + struct htp_gdn_hmx_vtcm_layout layout; + uint32_t n_heads_batch = 0; + if (!htp_gdn_hmx_solve_layout(&layout, (uint32_t) S_v, HTP_GDN_CHUNK_SIZE, total_rows, sess->vtcm_size, n_threads, true, &n_heads_batch)) { + return false; + } + } else { + struct htp_gdn_vtcm_layout layout; + htp_gdn_vtcm_layout_build(&layout, (uint32_t) S_v, n_threads); + if (layout.total_bytes > sess->vtcm_size) { + return false; + } + } + + return true; } static bool ggml_hexagon_matmul_is_hmx_eligible( @@ -2199,6 +4414,10 @@ static bool ggml_hexagon_matmul_is_hmx_eligible( bool is_matmul_id, bool is_batched ) { + if (src1->type != GGML_TYPE_F32) { + return false; + } + const int ne00 = src0->ne[0]; const int ne11 = src1->ne[1]; const int ne12 = src1->ne[2]; @@ -2229,7 +4448,8 @@ static bool ggml_hexagon_matmul_is_hmx_eligible( return false; } - // M alignment: Use HMX when M > HTP_MM_HMX_MIN_NROWS + // M alignment: Use HMX when M > HTP_MM_HMX_MIN_NROWS. + // For MUL_MAT_ID, src1 shape is [K, n_expert_used, n_tokens, 1], so n_tokens is ne12. const int m = is_matmul_id ? ne12 : ne11; if (m <= HTP_MM_HMX_MIN_NROWS) { return false; @@ -2254,6 +4474,7 @@ static bool ggml_hexagon_precompute_hmx_mm_params( int ne11_padded, bool is_matmul_id, bool is_batched, + size_t src2_size, size_t vtcm_budget, struct htp_mm_kernel_params * kparams ) { @@ -2274,15 +4495,15 @@ static bool ggml_hexagon_precompute_hmx_mm_params( if (is_batched_val && wtype == GGML_TYPE_F16 && group_size > 1) { // Try grouped path first const bool use_dma_activation = (src1->nb[1]/sizeof(float) > (size_t)ne00_padded); - if (htp_mm_hmx_solve_batched_params(wtype, ne00_padded, ne01_padded, ne11, group_size, use_dma_activation, n_threads, pipeline, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) { + if (htp_mm_hmx_solve_batched_params(wtype, ne00_padded, ne01_padded, ne11, group_size, use_dma_activation, n_threads, pipeline, src2_size, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) { use_grouped = true; } } if (!use_grouped) { // Fallback to simple 2D path (group_size = 1) - const int m_id_rows = (int) ((size_t) dst->ne[1] * dst->ne[2]); - if (!htp_mm_hmx_solve_2d_params(wtype, ne00_padded, m_id_rows, ne01_padded, ne11_padded, ne11, n_threads, pipeline, is_matmul_id, aligned_tile_size, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) { + const int m_id_rows = (dst && is_matmul_id) ? (int) ((size_t) dst->ne[1] * dst->ne[2]) : 0; + if (!htp_mm_hmx_solve_2d_params(wtype, ne00_padded, m_id_rows, ne01_padded, ne11_padded, ne11, n_threads, pipeline, is_matmul_id, aligned_tile_size, src2_size, vtcm_budget, &m_chunk, &n_chunk, &act_threads_selected, &vtcm_size)) { return false; } } @@ -2295,12 +4516,13 @@ static bool ggml_hexagon_precompute_hmx_mm_params( kparams->n_act_threads = act_threads_selected; kparams->tile_size = htp_mm_get_weight_tile_size(wtype); kparams->aligned_tile_size = aligned_tile_size; - kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1 || wtype == GGML_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); kparams->vtcm_size = vtcm_size; kparams->vtcm_src0_size = 0; kparams->div_n_act_threads = init_fastdiv_values(act_threads_selected); kparams->div_ne00_padded = init_fastdiv_values(ne00_padded); kparams->vtcm_src1_size = 0; + kparams->vtcm_src2_size = (int32_t) src2_size; kparams->vtcm_dst_size = 0; if (is_batched && !is_matmul_id) { @@ -2330,7 +4552,13 @@ static void ggml_hexagon_precompute_hvx_mm_params( size_t vtcm_budget, struct htp_mm_kernel_params * kparams ) { + if (opt_mm_select < 1) { + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; + } + kparams->n_hmx = 0; + kparams->n_threads = sess->n_threads; const bool is_quant = (wtype != GGML_TYPE_F16 && wtype != GGML_TYPE_F32); const int src1_nrows = ne11 * ne12 * ne13; @@ -2344,7 +4572,7 @@ static void ggml_hexagon_precompute_hvx_mm_params( if (is_matmul_id) { kparams->kernel_type = (src1_nrows < (int) sess->n_threads) ? HTP_MM_KERNEL_HVX_QUANT_BLOCK : HTP_MM_KERNEL_HVX_QUANT_ROW; - kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1 || wtype == GGML_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); struct htp_mm_hvx_vtcm_layout L; uint32_t max_prefetch = (src1_nrows > HTP_MM_HMX_MIN_NROWS) ? 2 : 16; @@ -2352,29 +4580,30 @@ static void ggml_hexagon_precompute_hvx_mm_params( for (uint32_t d = max_prefetch; d >= 2; d /= 2) { htp_mm_hvx_vtcm_layout_build( &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0->nb[1], 0, src2_row_size, d, true, false, false + 0, src0->nb[1], kparams->src1_row_size, 0, d, true, false ); if (L.total_bytes <= vtcm_budget) { best_n_prefetch = d; break; } } - if (best_n_prefetch == 2 && L.total_bytes > vtcm_budget) { - htp_mm_hvx_vtcm_layout_build( - &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0->nb[1], 0, src2_row_size, 2, true, false, false - ); + if (L.total_bytes > vtcm_budget) { + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; } - kparams->n_prefetch = best_n_prefetch; + kparams->n_prefetch = best_n_prefetch; kparams->vtcm_size = L.total_bytes; kparams->vtcm_src0_size = L.src0_bytes; kparams->vtcm_src1_size = L.src1_bytes; kparams->vtcm_dst_size = L.dst_bytes; + goto done_quant; } else { - bool try_tiled = (k_align && opt_mm_select >= 2); + bool try_tiled = (k_align && opt_mm_select >= 1); if (try_tiled) { - kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); - if (src1_nrows < (int)sess->n_threads) { + kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1 || wtype == GGML_TYPE_Q4_K) + ? htp_mm_q8_1_tiled_row_size(ne10) + : htp_mm_q8_0_tiled_row_size(ne10); + if (src1_nrows < (int) sess->n_threads) { kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_BLOCK; } else { kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW; @@ -2386,120 +4615,79 @@ static void ggml_hexagon_precompute_hvx_mm_params( for (uint32_t d = max_prefetch; d >= 2; d /= 2) { htp_mm_hvx_vtcm_layout_build( &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, d, false, false, false + dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, d, false, false ); if (L.total_bytes <= vtcm_budget) { best_n_prefetch = d; break; } } - if (best_n_prefetch == 2 && L.total_bytes > vtcm_budget) { - htp_mm_hvx_vtcm_layout_build( - &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 2, false, false, false - ); - } - kparams->n_prefetch = best_n_prefetch; - - if (L.total_bytes <= vtcm_budget) { - kparams->vtcm_size = L.total_bytes; + uint32_t m_chunk = 0; + if (htp_mm_hvx_solve_vtcm_params( + kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, + dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, best_n_prefetch, vtcm_budget, + &L, &m_chunk)) { + kparams->n_prefetch = best_n_prefetch; + kparams->m_chunk = (m_chunk < (uint32_t) src1_nrows) ? m_chunk : 0; + kparams->vtcm_size = L.total_bytes; kparams->vtcm_src0_size = L.src0_bytes; kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_dst_size = L.dst_bytes; + kparams->vtcm_src2_size = L.src2_bytes; + kparams->vtcm_dst_size = L.dst_bytes; goto done_quant; } - HEX_VERBOSE("ggml-hex: %s HVX tiled path VTCM size needed (%zu) > budget (%zu), falling back to HVX flat\n", sess->name.c_str(), L.total_bytes, vtcm_budget); } - // Flat HVX fallback - { - kparams->src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); - kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT; - - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build( - &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, false, false, false - ); - - kparams->n_prefetch = 16; - kparams->vtcm_size = L.total_bytes; - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - } + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; } done_quant:; } else if (wtype == GGML_TYPE_F16) { // F16 HVX - const bool is_batched = (ne02 > 1) || (ne03 > 1); - const bool is_permuted = ggml_is_permuted(src0) || ggml_is_permuted(src1); - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_F16_F16_VTCM, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, false, false, false - ); - - if (!is_batched && !is_permuted && L.total_bytes <= vtcm_budget) { + uint32_t m_chunk = 0; + if (htp_mm_hvx_solve_vtcm_params( + HTP_MM_KERNEL_HVX_F16_F16_VTCM, wtype, ne10, src1_nrows, sess->n_threads, + dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, vtcm_budget, + &L, &m_chunk)) { kparams->kernel_type = HTP_MM_KERNEL_HVX_F16_F16_VTCM; + kparams->m_chunk = (m_chunk < (uint32_t) src1_nrows) ? m_chunk : 0; kparams->src1_row_size = hex_round_up(ne10 * 2, 128); kparams->vtcm_size = L.total_bytes; kparams->vtcm_src0_size = L.src0_bytes; kparams->vtcm_src1_size = L.src1_bytes; + kparams->vtcm_src2_size = L.src2_bytes; kparams->vtcm_dst_size = L.dst_bytes; kparams->n_prefetch = 16; - } else { - if (src1->type == GGML_TYPE_F32) { - kparams->kernel_type = HTP_MM_KERNEL_HVX_F16_F32_DDR; - } else { - kparams->kernel_type = HTP_MM_KERNEL_HVX_F16_F16_DDR; - } - kparams->src1_row_size = src1->nb[1]; - htp_mm_hvx_vtcm_layout_build( - &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, false, false, false - ); - kparams->vtcm_size = L.total_bytes; - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->n_prefetch = 16; + return; } + + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; } else { // F32 HVX - const bool is_batched = (ne02 > 1) || (ne03 > 1); - const bool is_permuted = ggml_is_permuted(src0) || ggml_is_permuted(src1); - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_F32_F32_VTCM, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, false, false, false - ); - - if (!is_batched && !is_permuted && L.total_bytes <= vtcm_budget) { + uint32_t m_chunk = 0; + if (htp_mm_hvx_solve_vtcm_params( + HTP_MM_KERNEL_HVX_F32_F32_VTCM, wtype, ne10, src1_nrows, sess->n_threads, + dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, vtcm_budget, + &L, &m_chunk)) { kparams->kernel_type = HTP_MM_KERNEL_HVX_F32_F32_VTCM; + kparams->m_chunk = (m_chunk < (uint32_t) src1_nrows) ? m_chunk : 0; kparams->src1_row_size = hex_round_up(ne10 * 4, 128); kparams->vtcm_size = L.total_bytes; kparams->vtcm_src0_size = L.src0_bytes; kparams->vtcm_src1_size = L.src1_bytes; + kparams->vtcm_src2_size = L.src2_bytes; kparams->vtcm_dst_size = L.dst_bytes; kparams->n_prefetch = 16; - } else { - kparams->kernel_type = HTP_MM_KERNEL_HVX_F32_F32_DDR; - kparams->src1_row_size = src1->nb[1]; - htp_mm_hvx_vtcm_layout_build( - &L, kparams->kernel_type, wtype, ne10, src1_nrows, sess->n_threads, - dst->nb[1], src0->nb[1], src1->nb[1], src2_row_size, 16, false, false, false - ); - kparams->vtcm_size = L.total_bytes; - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->n_prefetch = 16; + return; } + + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; } } @@ -2509,6 +4697,7 @@ static void ggml_hexagon_precompute_matmul_params_impl( const struct ggml_tensor * src1, const struct ggml_tensor * dst, const size_t src2_row_size, + const size_t src2_size, struct htp_mm_kernel_params * kparams ) { memset(kparams, 0, sizeof(*kparams)); @@ -2535,9 +4724,9 @@ static void ggml_hexagon_precompute_matmul_params_impl( const size_t vtcm_budget = sess->vtcm_size; // Check HMX eligibility and try precomputing HMX parameters - bool hmx_enabled = (sess->n_hmx > 0) && (opt_mm_select >= 3); + bool hmx_enabled = (sess->n_hmx > 0) && (opt_mm_select >= 2); if (hmx_enabled && ggml_hexagon_matmul_is_hmx_eligible(src0, src1, dst, ne01_padded, is_matmul_id, is_batched)) { - if (ggml_hexagon_precompute_hmx_mm_params(sess, src0, src1, dst, wtype, ne00_padded, ne01_padded, ne02, ne11, ne12, ne11_padded, is_matmul_id, is_batched, vtcm_budget, kparams)) { + if (ggml_hexagon_precompute_hmx_mm_params(sess, src0, src1, dst, wtype, ne00_padded, ne01_padded, ne02, ne11, ne12, ne11_padded, is_matmul_id, is_batched, src2_size, vtcm_budget, kparams)) { goto finalize; } } @@ -2550,7 +4739,7 @@ static void ggml_hexagon_precompute_matmul_params_impl( kparams->div_ne1 = init_fastdiv_values(ne11); kparams->div_r2 = init_fastdiv_values(ne02 > 0 ? ne12 / ne02 : 1); kparams->div_r3 = init_fastdiv_values(ne03 > 0 ? ne13 / ne03 : 1); - kparams->div_ne11 = init_fastdiv_values(ne11); + kparams->div_ne12 = init_fastdiv_values(ne12); } static void ggml_hexagon_precompute_matmul_params( @@ -2560,224 +4749,640 @@ static void ggml_hexagon_precompute_matmul_params( const struct ggml_tensor * dst, struct htp_mm_kernel_params * kparams ) { - ggml_hexagon_precompute_matmul_params_impl(sess, src0, src1, dst, 0, kparams); + ggml_hexagon_precompute_matmul_params_impl(sess, src0, src1, dst, 0, 0, kparams); +} + +static void ggml_hexagon_precompute_fused_matmul_add_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * src2, + const struct ggml_tensor * dst, + struct htp_mm_kernel_params * kparams +) { + const size_t src2_size = src2 ? hex_round_up(ggml_nbytes(src2), 128) : 0; + ggml_hexagon_precompute_matmul_params_impl(sess, src0, src1, dst, src2 ? src2->nb[1] : 0, src2_size, kparams); +} + +static bool ggml_hexagon_precompute_binary_params( + const struct ggml_hexagon_session * sess, + uint32_t op, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_binary_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + + const size_t elem_size = ggml_type_size(src0->type); + const size_t src0_row_size = src0->ne[0] * elem_size; + const size_t src1_row_size = src1->ne[0] * elem_size; + const size_t dst_row_size = dst->ne[0] * elem_size; + + const size_t src0_row_size_aligned = hex_round_up(src0_row_size, 128); + const size_t src1_row_size_aligned = hex_round_up(src1_row_size, 128); + const size_t dst_row_size_aligned = hex_round_up(dst_row_size, 128); + + const bool is_add_id = op == HTP_OP_ADD_ID; + const bool is_scalar = !is_add_id && src1->ne[0] == 1; + const bool is_transposed = src0->nb[1] < src0_row_size || src1->nb[1] < src1_row_size || dst->nb[1] < dst_row_size; + const bool is_same_shape = !is_add_id && !is_scalar && !is_transposed && + src1->ne[0] == src0->ne[0] && + (src1->ne[1] == src0->ne[1] || src1->ne[1] == 1) && + (src1->ne[2] == src0->ne[2] || src1->ne[2] == 1) && + (src1->ne[3] == src0->ne[3] || src1->ne[3] == 1); + const bool is_row_bcast = is_same_shape && src1->ne[1] == 1 && src1->ne[2] == 1 && src1->ne[3] == 1; + const bool is_complex = !is_add_id && !is_scalar && !is_same_shape && (src1->ne[0] == src0->ne[0]); + + enum htp_binary_kernel_type kernel_type; + size_t src1_size = 0; + + if (is_add_id) { + kernel_type = HTP_BINARY_KERNEL_ADD_ID; + src1_size = hex_round_up(src1->ne[1] * src1_row_size_aligned, 128); + } else if (is_row_bcast) { + kernel_type = HTP_BINARY_KERNEL_ROW_BCAST; + src1_size = src1_row_size_aligned; + } else if (is_scalar) { + const bool is_scalar_static = (src1->ne[2] == 1 && src1->ne[3] == 1) && + (src1->ne[1] == 1 || src1->nb[1] == elem_size); + if (is_scalar_static) { + kernel_type = HTP_BINARY_KERNEL_SCALAR_DMA; + src1_size = hex_round_up(src1->ne[1] * elem_size, 128); + } else { + kernel_type = HTP_BINARY_KERNEL_SCALAR; + } + } else if (is_same_shape) { + kernel_type = HTP_BINARY_KERNEL_SAME_SHAPE; + } else if (is_complex) { + kernel_type = HTP_BINARY_KERNEL_COMPLEX; + } else { + kernel_type = HTP_BINARY_KERNEL_REPEAT; + } + + kparams->kernel_type = kernel_type; + kparams->n_threads = sess->n_threads; + kparams->src0_row_size_aligned = src0_row_size_aligned; + kparams->src1_row_size_aligned = src1_row_size_aligned; + kparams->dst_row_size_aligned = dst_row_size_aligned; + kparams->src1_size = src1_size; + + struct htp_binary_vtcm_layout L; + htp_binary_vtcm_layout_build(&L, kparams, sess->vtcm_size); + if (L.rows_per_buffer == 0 || L.total_bytes > sess->vtcm_size) { + return false; + } + + kparams->rows_per_buffer = L.rows_per_buffer; + kparams->vtcm_size = L.total_bytes; + + return true; +} + +static void ggml_hexagon_precompute_unary_params( + const struct ggml_hexagon_session * sess, + uint32_t op, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_unary_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const uint32_t n_threads = (std::min)((uint32_t)sess->n_threads, src0_nrows); + + kparams->n_threads = n_threads; + + const size_t elem_size = ggml_type_size(src0->type); + + const size_t src0_data_row_size = src0->ne[0] * elem_size; + const size_t dst_data_row_size = dst->ne[0] * ggml_type_size(dst->type); + + const size_t src0_row_size_aligned = hex_round_up(src0_data_row_size, 128); + const size_t dst_row_size_aligned = hex_round_up(dst_data_row_size, 128); + + kparams->src0_row_size_aligned = src0_row_size_aligned; + kparams->dst_row_size_aligned = dst_row_size_aligned; + + size_t src1_data_row_size = 0; + size_t src1_row_size_aligned = 0; + bool broadcast_weight = false; + + if (op == HTP_OP_RMS_NORM_MUL) { + GGML_ASSERT(src1 != nullptr); + src1_data_row_size = src1->ne[0] * ggml_type_size(src1->type); + src1_row_size_aligned = hex_round_up(src1_data_row_size, 128); + broadcast_weight = (src1->ne[1] * src1->ne[2] * src1->ne[3] == 1); + } + + kparams->src1_row_size_aligned = src1_row_size_aligned; + kparams->broadcast_weight = broadcast_weight; + + struct htp_unary_vtcm_layout L; + uint32_t col_tile = 0; + uint32_t vtcm_row_per_thread = 0; + + htp_unary_vtcm_layout_build(&L, op, src0->ne[0], dst->ne[0], + op == HTP_OP_RMS_NORM_MUL ? src1->ne[0] : 0, + broadcast_weight, n_threads, sess->vtcm_size, elem_size, + &col_tile, &vtcm_row_per_thread); + + kparams->col_tile = col_tile; + kparams->vtcm_row_per_thread = vtcm_row_per_thread; + kparams->vtcm_size = L.total_bytes; + + kparams->vtcm_src0_size_per_thread = L.src0_bytes; + kparams->vtcm_src1_size_per_thread = L.src1_bytes; + kparams->vtcm_dst_size_per_thread = L.dst_bytes; + + kparams->vtcm_src0_size = L.src0_bytes * n_threads; + kparams->vtcm_src1_size = L.src1_bytes * n_threads; + kparams->vtcm_dst_size = L.dst_bytes * n_threads; + + kparams->block = col_tile ? 0 : ((L.src0_bytes / 2) / src0_row_size_aligned); + + const uint32_t tiles_per_row = col_tile > 0 ? (src0->ne[0] + col_tile - 1) / col_tile : 1; + kparams->div_ne01 = init_fastdiv_values(src0->ne[1]); + kparams->div_ne02 = init_fastdiv_values(src0->ne[2]); + kparams->div_ne012 = init_fastdiv_values(src0->ne[1] * src0->ne[2]); + kparams->div_tpr = init_fastdiv_values(tiles_per_row); +} + +static void ggml_hexagon_precompute_get_rows_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, + const struct ggml_tensor * src1, + const struct ggml_tensor * dst, + struct htp_get_rows_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + + const uint32_t ne00 = src0->ne[0]; + const uint32_t ne02 = src0->ne[2]; + const uint32_t ne03 = src0->ne[3]; + + const uint32_t ne10 = src1->ne[0]; + const uint32_t ne11 = src1->ne[1]; + const uint32_t ne12 = src1->ne[2]; + const uint32_t nr = ne10 * ne11 * ne12; + + const size_t nb01 = src0->nb[1]; + const size_t nb1 = dst->nb[1]; + + const bool can_use_dma = (src0->type == dst->type) && (nb01 == nb1); + const bool use_dma = can_use_dma && (ne00 >= 2048); + + kparams->use_dma = use_dma ? 1 : 0; + + uint32_t chunks_per_row = 1; + uint32_t chunk_size = ne00; + uint32_t total_tasks = nr; + + if (use_dma) { + kparams->n_threads = (std::min)((uint32_t)sess->n_threads, nr); + kparams->tasks_per_thread = (nr + kparams->n_threads - 1) / kparams->n_threads; + } else { + if (src0->type == GGML_TYPE_F32 && nr < sess->n_threads) { + const uint32_t min_chunk_size = 1024; + uint32_t max_chunks = ne00 / min_chunk_size; + if (max_chunks == 0) { + max_chunks = 1; + } + chunks_per_row = (std::min)((sess->n_threads + nr - 1) / nr, max_chunks); + chunk_size = (ne00 + chunks_per_row - 1) / chunks_per_row; + total_tasks = nr * chunks_per_row; + } + kparams->n_threads = (std::min)(total_tasks, (uint32_t)sess->n_threads); + kparams->tasks_per_thread = (total_tasks + kparams->n_threads - 1) / kparams->n_threads; + } + + kparams->chunks_per_row = chunks_per_row; + kparams->chunk_size = chunk_size; + kparams->total_tasks = total_tasks; + + kparams->div_ne10 = init_fastdiv_values(ne10); + kparams->div_ne10_ne11 = init_fastdiv_values(ne10 * ne11); + kparams->div_chunks_per_row = init_fastdiv_values(chunks_per_row); + kparams->div_ne02 = init_fastdiv_values(ne02); + kparams->div_ne03 = init_fastdiv_values(ne03); + + struct htp_get_rows_vtcm_layout vtcm_layout; + htp_get_rows_vtcm_layout_build(&vtcm_layout, src0->type, ne00, kparams->n_threads); + kparams->vtcm_size = vtcm_layout.total_bytes; +} + +static void ggml_hexagon_precompute_set_rows_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, // values + const struct ggml_tensor * src1, // indices + const struct ggml_tensor * dst, // destination + struct htp_set_rows_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + + const uint32_t nr = src0->ne[1]; + + kparams->n_threads = (std::min)((uint32_t)sess->n_threads, nr); + kparams->tasks_per_thread = (nr + kparams->n_threads - 1) / kparams->n_threads; + kparams->total_tasks = nr; + + kparams->div_ne11 = init_fastdiv_values(src1->ne[1]); + kparams->div_ne12 = init_fastdiv_values(src1->ne[2]); + kparams->div_tasks_per_thread = init_fastdiv_values(kparams->tasks_per_thread); + kparams->div_ne02 = init_fastdiv_values(src0->ne[2]); + + struct htp_set_rows_vtcm_layout vtcm_layout; + htp_set_rows_vtcm_layout_build(&vtcm_layout, dst->type, src0->ne[0], kparams->n_threads); + kparams->vtcm_size = vtcm_layout.total_bytes; +} + +static void ggml_hexagon_precompute_softmax_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * op, + struct htp_softmax_kernel_params * kparams +) { + memset(kparams, 0, sizeof(*kparams)); + + const struct ggml_tensor * src0 = op->src[0]; + const struct ggml_tensor * src1 = op->src[1]; + + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, src0_nrows); + + float scale = 1.0f; + float max_bias = 0.0f; + memcpy(&scale, &op->op_params[0], sizeof(float)); + memcpy(&max_bias, &op->op_params[1], sizeof(float)); + + kparams->scale = scale; + kparams->max_bias = max_bias; + + const uint32_t n_head = src0->ne[2]; + const uint32_t n_head_log2 = 1u << (uint32_t) floor(log2(n_head)); + kparams->n_head = n_head; + kparams->n_head_log2 = n_head_log2; + + if (max_bias > 0.0f && n_head_log2 > 0) { + kparams->m0 = powf(2.0f, -(max_bias) / n_head_log2); + kparams->m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2); + } else { + kparams->m0 = 1.0f; + kparams->m1 = 1.0f; + } + + kparams->use_src1 = (src1 != nullptr) ? 1 : 0; + kparams->use_f16 = (src1 != nullptr && src1->type == GGML_TYPE_F16) ? 1 : 0; + + const uint32_t ne00 = src0->ne[0]; + const uint32_t ne10 = src1 ? src1->ne[0] : 1; + + struct htp_softmax_vtcm_layout layout; + htp_softmax_vtcm_layout_build(&layout, ne00, ne10, kparams->use_src1 != 0, kparams->use_f16 != 0, n_threads); + + kparams->n_threads = n_threads; + kparams->src0_nrows = src0_nrows; + kparams->src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; + kparams->vtcm_size = (uint32_t) layout.total_bytes; + kparams->vtcm_src0_size_per_thread = (uint32_t) layout.src0_bytes_per_thread; + kparams->vtcm_src1_size_per_thread = (uint32_t) layout.src1_bytes_per_thread; + kparams->vtcm_dst_size_per_thread = (uint32_t) layout.dst_bytes_per_thread; + kparams->src0_row_size_aligned = (uint32_t) layout.src0_spad_half_size; + kparams->src1_row_size_aligned = (uint32_t) layout.src1_spad_half_size; + kparams->dst_row_size_aligned = (uint32_t) layout.dst_spad_half_size; + kparams->src0_spad_half_size = (uint32_t) layout.src0_spad_half_size; + kparams->src1_spad_half_size = (uint32_t) layout.src1_spad_half_size; + kparams->dst_spad_half_size = (uint32_t) layout.dst_spad_half_size; + if (!kparams->use_src1) { + kparams->kernel_id = HTP_SOFTMAX_KERNEL_NOMASK; + } else if (kparams->use_f16) { + kparams->kernel_id = HTP_SOFTMAX_KERNEL_MASK_F16; + } else { + kparams->kernel_id = HTP_SOFTMAX_KERNEL_MASK_F32; + } + + if (src0->ne[1] > 0) kparams->div_ne01 = init_fastdiv_values(src0->ne[1]); + if (src0->ne[2] > 0) kparams->div_ne02 = init_fastdiv_values(src0->ne[2]); + const uint32_t ne12 = src1 ? src1->ne[2] : 1; + const uint32_t ne13 = src1 ? src1->ne[3] : 1; + if (ne12 > 0) kparams->div_ne12 = init_fastdiv_values(ne12); + if (ne13 > 0) kparams->div_ne13 = init_fastdiv_values(ne13); } -static void ggml_hexagon_precompute_fused_matmul_add_params( +static void ggml_hexagon_precompute_rope_params( const struct ggml_hexagon_session * sess, - const struct ggml_tensor * src0, - const struct ggml_tensor * src1, - const struct ggml_tensor * src2, - const struct ggml_tensor * dst, - struct htp_mm_kernel_params * kparams + const struct ggml_tensor * op, + struct htp_rope_kernel_params * kparams ) { - ggml_hexagon_precompute_matmul_params_impl(sess, src0, src1, dst, src2->nb[1], kparams); + memset(kparams, 0, sizeof(*kparams)); + + const struct ggml_tensor * src0 = op->src[0]; + const struct ggml_tensor * src2 = op->src[2]; + const struct ggml_tensor * dst = op; + + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, src0_nrows); + const uint32_t n_freq_factors = src2 ? (uint32_t) src2->ne[0] : 0; + + struct htp_rope_vtcm_layout layout; + htp_rope_vtcm_layout_build(&layout, src0->ne[0], n_threads, n_freq_factors); + + kparams->n_threads = n_threads; + kparams->src0_nrows = src0_nrows; + kparams->src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; + kparams->vtcm_size = (uint32_t) layout.total_bytes; + kparams->spad_per_thread = (uint32_t) layout.bytes_per_thread; + kparams->theta_cache_offset = (uint32_t) layout.theta_cache_size_aligned; + kparams->src0_row_size_aligned = (uint32_t) layout.src0_row_size_aligned; + kparams->freq_factors_offset = (uint32_t) (layout.bytes_per_thread * n_threads); + kparams->freq_factors_size = (uint32_t) layout.freq_factors_size_aligned; + + if (src0_nrows > 0) { + kparams->div_ne2_ne1 = init_fastdiv_values(dst->ne[2] * dst->ne[1]); + kparams->div_ne1 = init_fastdiv_values(dst->ne[1]); + } } -static void ggml_hexagon_precompute_unary_params( +static void ggml_hexagon_precompute_ssm_conv_params( const struct ggml_hexagon_session * sess, - uint32_t op, const struct ggml_tensor * src0, const struct ggml_tensor * src1, const struct ggml_tensor * dst, - struct htp_unary_kernel_params * kparams + struct htp_ssm_conv_kernel_params * kparams ) { memset(kparams, 0, sizeof(*kparams)); - const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = (std::min)((uint32_t)sess->n_threads, src0_nrows); + const uint32_t d_conv = (uint32_t) src1->ne[0]; + const uint32_t d_inner = (uint32_t) src0->ne[1]; + const uint32_t n_t = (uint32_t) dst->ne[1]; + const uint32_t n_s = (uint32_t) dst->ne[2]; + const uint32_t ncs = (uint32_t) src0->ne[0]; + + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, (d_inner + 31) / 32); kparams->n_threads = n_threads; + kparams->d_conv = d_conv; + kparams->d_inner = d_inner; + kparams->n_t = n_t; + kparams->n_s = n_s; - const size_t src0_data_row_size = src0->ne[0] * sizeof(float); - const size_t dst_data_row_size = dst->ne[0] * sizeof(float); + const uint32_t raw_rpt = (d_inner + n_threads - 1) / n_threads; + const uint32_t d_inner_per_thread = hex_round_up(raw_rpt, 32); + kparams->d_inner_per_thread = d_inner_per_thread; - const size_t src0_row_size_aligned = hex_round_up(src0_data_row_size, 128); - const size_t dst_row_size_aligned = hex_round_up(dst_data_row_size, 128); + kparams->src0_row_size_aligned = hex_round_up(ncs * sizeof(float), 128); + kparams->src1_row_size_aligned = hex_round_up(d_conv * sizeof(float), 128); + kparams->dst_row_size_aligned = hex_round_up(d_inner * sizeof(float), 128); - kparams->src0_row_size_aligned = src0_row_size_aligned; - kparams->dst_row_size_aligned = dst_row_size_aligned; + if (n_t == 1) { + kparams->d_inner_tile = d_inner_per_thread; - size_t src1_data_row_size = 0; - size_t src1_row_size_aligned = 0; - bool broadcast_weight = false; + const uint32_t src1_raw_bytes = hex_round_up(d_inner_per_thread * d_conv * sizeof(float), 128) + 128; + const uint32_t src1_T_bytes = hex_round_up(d_conv * d_inner_per_thread * sizeof(float), 128); + const uint32_t vtcm_src1_per_thread = src1_raw_bytes + src1_T_bytes; - if (op == HTP_OP_RMS_NORM_MUL) { - GGML_ASSERT(src1 != nullptr); - src1_data_row_size = src1->ne[0] * sizeof(float); - src1_row_size_aligned = hex_round_up(src1_data_row_size, 128); - broadcast_weight = (src1->ne[1] * src1->ne[2] * src1->ne[3] == 1); - } + const uint32_t src0_raw_bytes = hex_round_up(d_inner_per_thread * d_conv * sizeof(float), 128) + 128; + const uint32_t src0_T_bytes = hex_round_up(d_conv * d_inner_per_thread * sizeof(float), 128); + const uint32_t vtcm_src0_per_thread = src0_raw_bytes + src0_T_bytes; - kparams->src1_row_size_aligned = src1_row_size_aligned; - kparams->broadcast_weight = broadcast_weight; + const uint32_t vtcm_dst_per_thread = hex_round_up(d_inner_per_thread * sizeof(float), 128); - struct htp_unary_vtcm_layout L; - uint32_t col_tile = 0; - uint32_t vtcm_row_per_thread = 0; + kparams->vtcm_src0_size_per_thread = vtcm_src0_per_thread; + kparams->vtcm_src1_size_per_thread = vtcm_src1_per_thread; + kparams->vtcm_dst_size_per_thread = vtcm_dst_per_thread; - htp_unary_vtcm_layout_build(&L, op, src0->ne[0], dst->ne[0], - op == HTP_OP_RMS_NORM_MUL ? src1->ne[0] : 0, - broadcast_weight, n_threads, sess->vtcm_size, - &col_tile, &vtcm_row_per_thread); + kparams->vtcm_src0_size = vtcm_src0_per_thread * n_threads; + kparams->vtcm_src1_size = vtcm_src1_per_thread * n_threads; + kparams->vtcm_dst_size = vtcm_dst_per_thread * n_threads; + kparams->vtcm_size = kparams->vtcm_src0_size + kparams->vtcm_src1_size + kparams->vtcm_dst_size; + } else { + const uint32_t src1_raw_bytes = hex_round_up(d_inner_per_thread * d_conv * sizeof(float), 128) + 128; + const uint32_t src1_T_bytes = hex_round_up(d_conv * d_inner_per_thread * sizeof(float), 128); + const uint32_t vtcm_src1_per_thread = src1_raw_bytes + src1_T_bytes; - kparams->col_tile = col_tile; - kparams->vtcm_row_per_thread = vtcm_row_per_thread; - kparams->vtcm_size = L.total_bytes; + const size_t vtcm_budget = (sess->vtcm_size > 0 ? sess->vtcm_size / n_threads : (1024 * 1024)); + const size_t avail_for_src0 = vtcm_budget > vtcm_src1_per_thread ? vtcm_budget - vtcm_src1_per_thread : (128 * 1024); - kparams->vtcm_src0_size_per_thread = L.src0_bytes; - kparams->vtcm_src1_size_per_thread = L.src1_bytes; - kparams->vtcm_dst_size_per_thread = L.dst_bytes; + uint32_t d_inner_tile = (uint32_t)((avail_for_src0 / 2) / (ncs * sizeof(float) + n_t * sizeof(float) + 1)); + d_inner_tile = (d_inner_tile / 32) * 32; + if (d_inner_tile == 0) { + d_inner_tile = 32; + } + if (d_inner_tile > d_inner_per_thread) { + d_inner_tile = d_inner_per_thread; + } + kparams->d_inner_tile = d_inner_tile; - kparams->vtcm_src0_size = L.src0_bytes * n_threads; - kparams->vtcm_src1_size = L.src1_bytes * n_threads; - kparams->vtcm_dst_size = L.dst_bytes * n_threads; + const uint32_t src0_tile_raw = hex_round_up(d_inner_tile * ncs * sizeof(float), 128) + 128; + const uint32_t src0_tile_T = hex_round_up(ncs * d_inner_tile * sizeof(float), 128); + const uint32_t vtcm_src0_per_thread = src0_tile_raw + src0_tile_T; - kparams->block = col_tile ? 0 : ((L.src0_bytes / 2) / src0_row_size_aligned); + const uint32_t vtcm_dst_per_thread = hex_round_up(d_inner_tile * n_t * sizeof(float), 128); - const uint32_t tiles_per_row = col_tile > 0 ? (src0->ne[0] + col_tile - 1) / col_tile : 1; - kparams->div_ne01 = init_fastdiv_values(src0->ne[1]); - kparams->div_ne02 = init_fastdiv_values(src0->ne[2]); - kparams->div_ne012 = init_fastdiv_values(src0->ne[1] * src0->ne[2]); - kparams->div_tpr = init_fastdiv_values(tiles_per_row); + kparams->vtcm_src0_size_per_thread = vtcm_src0_per_thread; + kparams->vtcm_src1_size_per_thread = vtcm_src1_per_thread; + kparams->vtcm_dst_size_per_thread = vtcm_dst_per_thread; + + kparams->vtcm_src0_size = vtcm_src0_per_thread * n_threads; + kparams->vtcm_src1_size = vtcm_src1_per_thread * n_threads; + kparams->vtcm_dst_size = vtcm_dst_per_thread * n_threads; + kparams->vtcm_size = kparams->vtcm_src0_size + kparams->vtcm_src1_size + kparams->vtcm_dst_size; + } + + kparams->div_n_threads = init_fastdiv_values(n_threads); } -static void ggml_hexagon_precompute_fused_qkv_params( +static void ggml_hexagon_precompute_gated_delta_net_params( const struct ggml_hexagon_session * sess, - const struct ggml_tensor * src0, // Wk - const struct ggml_tensor * src1, // x - struct htp_mm_kernel_params * kparams + const struct ggml_tensor * op, + struct htp_gdn_kernel_params * kparams ) { memset(kparams, 0, sizeof(*kparams)); - const int wtype = src0->type; - const bool is_repack = ggml_hexagon_is_repack_type((ggml_type) wtype); - - const int ne10 = src1->ne[0]; - const int src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; - const size_t src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); - const size_t src0_row_size = src0->nb[1]; - - uint32_t best_n_prefetch = 16; - - if (is_repack) { - const uint32_t max_prefetch = (src1_nrows > HTP_MM_HMX_MIN_NROWS) ? 2 : 16; - best_n_prefetch = 2; - for (uint32_t d = max_prefetch; d >= 2; d /= 2) { - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, src1_row_size, 0, d, false, true, false - ); - if (L.total_bytes <= sess->vtcm_size) { - best_n_prefetch = d; - break; - } - } - } - - struct htp_mm_hvx_vtcm_layout L; - bool try_tiled = (opt_mm_select >= 2); - - // Test tiled first - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, src1_row_size, 0, best_n_prefetch, false, true, false - ); + const struct ggml_tensor * q = op->src[0]; + const struct ggml_tensor * k = op->src[1]; + const struct ggml_tensor * v = op->src[2]; + const struct ggml_tensor * g = op->src[3]; + const struct ggml_tensor * state = op->src[5]; - if (try_tiled && L.total_bytes <= sess->vtcm_size) { - kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW; - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_src2_size = L.src2_bytes; - kparams->vtcm_src3_size = L.src3_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->vtcm_size = L.total_bytes; - kparams->n_prefetch = best_n_prefetch; + const uint32_t S_v = (uint32_t) v->ne[0]; + const uint32_t H = (uint32_t) v->ne[1]; + const uint32_t n_tokens = (uint32_t) v->ne[2]; + const uint32_t n_seqs = (uint32_t) v->ne[3]; + const uint32_t K = (uint32_t) ggml_get_op_params_i32(op, 0); + + const uint32_t rq3 = (uint32_t) (n_seqs / q->ne[3]); + const uint32_t rk3 = (uint32_t) (n_seqs / k->ne[3]); + const uint32_t total_rows = H * n_seqs; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, total_rows); + + const bool can_use_hmx = (opt_gdn_select >= 2) && + (sess->n_hmx > 0) && + (S_v % 64 == 0) && + (n_tokens >= HTP_GDN_MIN_TOKENS) && + (g->ne[0] == 1) && + (K == 1); + + struct htp_gdn_hmx_vtcm_layout hmx_layout; + struct htp_gdn_vtcm_layout hvx_layout; + uint32_t n_heads_batch = 1; + + if (can_use_hmx && htp_gdn_hmx_solve_layout(&hmx_layout, S_v, HTP_GDN_CHUNK_SIZE, total_rows, sess->vtcm_size, n_threads, true, &n_heads_batch)) { + kparams->kernel_type = HTP_GDN_KERNEL_HMX_CHUNKED; + kparams->pipeline = hmx_layout.pipeline ? 1 : 0; + kparams->chunk_size = HTP_GDN_CHUNK_SIZE; + kparams->n_chunks = (n_tokens + HTP_GDN_CHUNK_SIZE - 1) / HTP_GDN_CHUNK_SIZE; + kparams->n_heads_batch = (uint16_t) n_heads_batch; + kparams->vtcm_size = (uint32_t) hmx_layout.total_bytes; + kparams->state_aligned = (uint32_t) hmx_layout.state_f32_bytes; + kparams->vtcm_per_thread = (uint32_t) (hmx_layout.total_bytes / (n_threads > 0 ? n_threads : 1)); } else { - kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT; - size_t flat_src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); - - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, flat_src1_row_size, 0, best_n_prefetch, false, true, false - ); - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_src2_size = L.src2_bytes; - kparams->vtcm_src3_size = L.src3_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->vtcm_size = L.total_bytes; - kparams->n_prefetch = best_n_prefetch; - } + htp_gdn_vtcm_layout_build(&hvx_layout, S_v, n_threads); + kparams->kernel_type = HTP_GDN_KERNEL_HVX_RECURRENT; + kparams->pipeline = 0; + kparams->n_heads_batch = 1; + kparams->state_aligned = (uint32_t) hvx_layout.state_aligned; + kparams->vtcm_per_thread = (uint32_t) hvx_layout.bytes_per_thread; + kparams->vtcm_size = (uint32_t) hvx_layout.total_bytes; + } + + kparams->n_threads = n_threads; + kparams->S_v = S_v; + kparams->H = H; + kparams->n_tokens = n_tokens; + kparams->n_seqs = n_seqs; + kparams->K = K; + kparams->total_rows = total_rows; + kparams->rows_per_thread = (total_rows + kparams->n_threads - 1) / kparams->n_threads; + kparams->kda = (g->ne[0] == S_v) ? 1 : 0; + kparams->state_seq_stride = (uint32_t) (state->nb[3] / sizeof(float)); + kparams->state_size_per_snap = S_v * S_v * H * n_seqs; + kparams->scale = 1.0f / sqrtf((float) S_v); + + if (H > 0) kparams->div_H = init_fastdiv_values(H); + if (q->ne[1] > 0) kparams->div_q1 = init_fastdiv_values((uint32_t) q->ne[1]); + if (k->ne[1] > 0) kparams->div_k1 = init_fastdiv_values((uint32_t) k->ne[1]); + if (rq3 > 0) kparams->div_rq3 = init_fastdiv_values(rq3); + if (rk3 > 0) kparams->div_rk3 = init_fastdiv_values(rk3); + if (kparams->n_threads > 0) kparams->div_n_threads = init_fastdiv_values(kparams->n_threads); } -static void ggml_hexagon_precompute_fused_ffn_params( +static void ggml_hexagon_precompute_fused_mmnx_params( const struct ggml_hexagon_session * sess, - const struct ggml_tensor * src0, // Wgate - const struct ggml_tensor * src1, // y + const struct ggml_tensor * src0, // W0 + const struct ggml_tensor * src1, // x + int32_t n_weights, struct htp_mm_kernel_params * kparams ) { memset(kparams, 0, sizeof(*kparams)); + kparams->n_threads = sess->n_threads; - const int wtype = src0->type; - const bool is_repack = ggml_hexagon_is_repack_type((ggml_type) wtype); + const int ne00 = src0->ne[0]; + const int ne01 = src0->ne[1]; + const int ne02 = src0->ne[2]; + const int ne03 = src0->ne[3]; const int ne10 = src1->ne[0]; - const int src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; - const size_t src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); - const size_t src0_row_size = src0->nb[1]; + const int ne11 = src1->ne[1]; + const int ne12 = src1->ne[2]; + const int ne13 = src1->ne[3]; - uint32_t best_n_prefetch = 16; + const int wtype = src0->type; + const bool is_repack = ggml_hexagon_is_repack_type((ggml_type) wtype); + const int ne00_padded = is_repack ? hex_round_up(ne00, 32) : ne00; + const int ne01_padded = is_repack ? hex_round_up(ne01, 32) : ne01; + const int ne11_padded = hex_round_up(ne11, 32); - if (is_repack) { - const uint32_t max_prefetch = (src1_nrows > HTP_MM_HMX_MIN_NROWS) ? 2 : 16; - best_n_prefetch = 2; - for (uint32_t d = max_prefetch; d >= 2; d /= 2) { - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, src1_row_size, 0, d, false, false, true - ); - if (L.total_bytes <= sess->vtcm_size) { - best_n_prefetch = d; - break; - } + const size_t vtcm_budget = sess->vtcm_size; + const bool is_batched = (ne02 * ne03 > 1 || ne12 * ne13 > 1); + + bool hmx_enabled = (sess->n_hmx > 0) && (opt_mm_select >= 2); + if (hmx_enabled && ggml_hexagon_matmul_is_hmx_eligible(src0, src1, nullptr, ne01_padded, false, is_batched)) { + if (ggml_hexagon_precompute_hmx_mm_params(sess, src0, src1, nullptr, wtype, ne00_padded, ne01_padded, ne02, ne11, ne12, ne11_padded, false, is_batched, 0, vtcm_budget, kparams)) { + kparams->n_weights = n_weights; + goto finalize; } } - struct htp_mm_hvx_vtcm_layout L; - bool try_tiled = (opt_mm_select >= 2); + if (!is_repack) { + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; + } - // Test tiled first - htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, src1_row_size, 0, best_n_prefetch, false, false, true - ); + { + const int src1_nrows = ne11 * ne12 * ne13; + const size_t src1_row_size = (wtype == GGML_TYPE_Q4_1 || wtype == GGML_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + const size_t src0_row_size = src0->nb[1]; - if (try_tiled && L.total_bytes <= sess->vtcm_size) { - kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW; - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_src2_size = L.src2_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->vtcm_size = L.total_bytes; - kparams->n_prefetch = best_n_prefetch; - } else { - kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT; - size_t flat_src1_row_size = (wtype == GGML_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); + uint32_t best_n_prefetch = 16; + + if (is_repack) { + const uint32_t max_prefetch = (src1_nrows > HTP_MM_HMX_MIN_NROWS) ? 2 : 16; + best_n_prefetch = 2; + for (uint32_t d = max_prefetch; d >= 2; d /= 2) { + struct htp_mm_hvx_vtcm_layout L; + htp_mm_hvx_vtcm_layout_build( + &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, + 0, src0_row_size, src1_row_size, 0, d, false, true + ); + if (L.total_bytes <= sess->vtcm_size) { + best_n_prefetch = d; + break; + } + } + } + + struct htp_mm_hvx_vtcm_layout L; + bool try_tiled = (opt_mm_select >= 1); + // Test tiled first htp_mm_hvx_vtcm_layout_build( - &L, HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT, wtype, ne10, src1_nrows, sess->n_threads, - 0, src0_row_size, flat_src1_row_size, 0, best_n_prefetch, false, false, true + &L, HTP_MM_KERNEL_HVX_QUANT_ROW, wtype, ne10, src1_nrows, sess->n_threads, + 0, src0_row_size, src1_row_size, 0, best_n_prefetch, false, true ); - kparams->vtcm_src0_size = L.src0_bytes; - kparams->vtcm_src1_size = L.src1_bytes; - kparams->vtcm_src2_size = L.src2_bytes; - kparams->vtcm_dst_size = L.dst_bytes; - kparams->vtcm_size = L.total_bytes; - kparams->n_prefetch = best_n_prefetch; + + if (try_tiled && L.total_bytes <= sess->vtcm_size) { + kparams->kernel_type = HTP_MM_KERNEL_HVX_QUANT_ROW; + kparams->vtcm_src0_size = L.src0_bytes; + kparams->vtcm_src1_size = L.src1_bytes; + kparams->vtcm_dst_size = L.dst_bytes; + kparams->vtcm_size = L.total_bytes; + kparams->n_prefetch = best_n_prefetch; + kparams->n_weights = n_weights; + } else { + kparams->kernel_type = HTP_MM_KERNEL_UNSUPPORTED; + return; + } } + +finalize: + kparams->div_ne12_ne1 = init_fastdiv_values(ne12 * ne11); + kparams->div_ne1 = init_fastdiv_values(ne11); + kparams->div_r2 = init_fastdiv_values(ne02 > 0 ? ne12 / ne02 : 1); + kparams->div_r3 = init_fastdiv_values(ne03 > 0 ? ne13 / ne03 : 1); + kparams->div_ne12 = init_fastdiv_values(ne12); +} + +static void ggml_hexagon_precompute_fused_mmidnx_params( + const struct ggml_hexagon_session * sess, + const struct ggml_tensor * src0, // W0 + const struct ggml_tensor * src1, // x + const struct ggml_tensor * dst, // dst0 + int32_t n_weights, + struct htp_mm_kernel_params * kparams +) { + ggml_hexagon_precompute_matmul_params_impl(sess, src0, src1, dst, 0, 0, kparams); + kparams->n_weights = n_weights; +} + +static bool ggml_hexagon_tensor_is_host(const struct ggml_hexagon_session * sess, const struct ggml_tensor * t) { + return t && t->buffer && ggml_backend_buft_is_host(t->buffer->buft); + GGML_UNUSED(sess); +} + +static bool ggml_hexagon_tensor_is_non_host(const struct ggml_hexagon_session * sess, const struct ggml_tensor * t) { + return t && t->buffer && !ggml_backend_buft_is_host(t->buffer->buft); + GGML_UNUSED(sess); } static bool ggml_hexagon_supported_mul_mat(const struct ggml_hexagon_session * sess, const struct ggml_tensor * dst) { @@ -2798,23 +5403,26 @@ static bool ggml_hexagon_supported_mul_mat(const struct ggml_hexagon_session * s case GGML_TYPE_Q8_0: case GGML_TYPE_IQ4_NL: case GGML_TYPE_MXFP4: - if (src0->ne[0] % 32) { + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q6_K: + if (!ggml_is_contiguous(src0) || ggml_is_permuted(src0)) { return false; } - // hardcoded limit to refuse the lm-head for now - if (src0->ne[1] > 32768) { + if (src0->ne[0] % ((src0->type == GGML_TYPE_Q6_K || src0->type == GGML_TYPE_Q4_K) ? QK_K : 32)) { return false; } - if (src1->ne[2] != 1 || src1->ne[3] != 1) { - return false; // no broadcasting (for now) + if (src1->ne[2] < src0->ne[2] || src1->ne[3] < src0->ne[3]) { + return false; } - - // src0 (weights) must be repacked - if (src0->buffer && !ggml_backend_buffer_is_hexagon_repack(src0->buffer)) { + if (src1->ne[2] % src0->ne[2] != 0 || src1->ne[3] % src0->ne[3] != 0) { return false; } + + if (!src0->buffer) { + sess->needs_repack.insert(src0); + } break; case GGML_TYPE_F16: @@ -2824,6 +5432,9 @@ static bool ggml_hexagon_supported_mul_mat(const struct ggml_hexagon_session * s if (src1->ne[2] < src0->ne[2] || src1->ne[3] < src0->ne[3]) { return false; } + if (src1->ne[2] % src0->ne[2] != 0 || src1->ne[3] % src0->ne[3] != 0) { + return false; + } break; case GGML_TYPE_F32: @@ -2836,6 +5447,9 @@ static bool ggml_hexagon_supported_mul_mat(const struct ggml_hexagon_session * s if (src1->ne[2] < src0->ne[2] || src1->ne[3] < src0->ne[3]) { return false; } + if (src1->ne[2] % src0->ne[2] != 0 || src1->ne[3] % src0->ne[3] != 0) { + return false; + } break; default: @@ -2844,7 +5458,7 @@ static bool ggml_hexagon_supported_mul_mat(const struct ggml_hexagon_session * s struct htp_mm_kernel_params kparams; ggml_hexagon_precompute_matmul_params(sess, src0, src1, dst, &kparams); - if ((size_t)kparams.vtcm_size > sess->vtcm_size) { + if (kparams.kernel_type == HTP_MM_KERNEL_UNSUPPORTED || (size_t) kparams.vtcm_size > sess->vtcm_size) { HEX_VERBOSE("ggml-hex: %s supported MUL_MAT VTCM size needed (%d) > budget (%zu)\n", sess->c_name(), kparams.vtcm_size, sess->vtcm_size); return false; } @@ -2868,14 +5482,19 @@ static bool ggml_hexagon_supported_mul_mat_id(const struct ggml_hexagon_session case GGML_TYPE_Q8_0: case GGML_TYPE_IQ4_NL: case GGML_TYPE_MXFP4: - if ((src0->ne[0] % 32)) { + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q6_K: + if (!ggml_is_contiguous(src0) || ggml_is_permuted(src0)) { return false; } - // src0 (weights) must be repacked - if (src0->buffer && !ggml_backend_buffer_is_hexagon_repack(src0->buffer)) { + if (src0->ne[0] % ((src0->type == GGML_TYPE_Q6_K || src0->type == GGML_TYPE_Q4_K) ? QK_K : 32)) { return false; } + + if (!src0->buffer) { + sess->needs_repack.insert(src0); + } break; default: @@ -2884,7 +5503,7 @@ static bool ggml_hexagon_supported_mul_mat_id(const struct ggml_hexagon_session struct htp_mm_kernel_params kparams; ggml_hexagon_precompute_matmul_params(sess, src0, src1, dst, &kparams); - if ((size_t)kparams.vtcm_size > sess->vtcm_size) { + if (kparams.kernel_type == HTP_MM_KERNEL_UNSUPPORTED || (size_t) kparams.vtcm_size > sess->vtcm_size) { HEX_VERBOSE("ggml-hex: %s supported MUL_MAT_ID VTCM size needed (%d) > budget (%zu)\n", sess->c_name(), kparams.vtcm_size, sess->vtcm_size); return false; } @@ -2927,52 +5546,81 @@ static bool ggml_hexagon_supported_binary(const struct ggml_hexagon_session * se return false; } - return true; - - GGML_UNUSED(sess); + struct htp_binary_kernel_params kparams; + return ggml_hexagon_precompute_binary_params(sess, op_remap_to_htp(op), src0, src1, dst, &kparams); } static bool ggml_hexagon_supported_add_id(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { const struct ggml_tensor * src0 = op->src[0]; const struct ggml_tensor * src1 = op->src[1]; + const struct ggml_tensor * src2 = op->src[2]; const struct ggml_tensor * dst = op; - if (src0->type != GGML_TYPE_F32) { + if (!src2) { return false; } - if (src1->type != GGML_TYPE_F32) { + if (src0->type != GGML_TYPE_F32 || src1->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32 || src2->type != GGML_TYPE_I32) { return false; } - if (dst->type != GGML_TYPE_F32) { + if (!ggml_are_same_shape(src0, dst)) { return false; } - if (!ggml_are_same_shape(src0, dst)) { + if (src1->ne[0] != src0->ne[0] || src1->ne[2] != 1 || src1->ne[3] != 1) { + return false; + } + if (src2->ne[0] != src0->ne[1] || src2->ne[1] != src0->ne[2]) { + return false; + } + if (src0->nb[0] != sizeof(float) || src1->nb[0] != sizeof(float) || dst->nb[0] != sizeof(float) || src2->nb[0] != sizeof(int32_t)) { return false; } - // REVISIT: add support for non-contigiuos tensors + // REVISIT: add support for non-contiguous tensors if (!ggml_is_contiguous(src0) || !ggml_is_contiguous(src1) || !ggml_is_contiguous(dst)) { return false; } - return true; - - GGML_UNUSED(sess); + struct htp_binary_kernel_params kparams; + return ggml_hexagon_precompute_binary_params(sess, HTP_OP_ADD_ID, src0, src1, dst, &kparams); } static bool ggml_hexagon_supported_unary(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { const struct ggml_tensor * src0 = op->src[0]; const struct ggml_tensor * dst = op; - if (src0->type != GGML_TYPE_F32) { + if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) { return false; } - if (dst->type != GGML_TYPE_F32) { + if (dst->type != src0->type) { return false; } - if (ggml_is_permuted(src0)) { + if (!ggml_is_contiguous_rows(src0)) { return false; } + + // F16 device kernels only cover this explicit whitelist (must stay in sync with + // the is_f16 whitelist in execute_op_unary(), unary-ops.c). + if (src0->type == GGML_TYPE_F16) { + switch (op->op) { + case GGML_OP_NORM: + case GGML_OP_RMS_NORM: + case GGML_OP_L2_NORM: + case GGML_OP_SCALE: + case GGML_OP_CLAMP: + case GGML_OP_SQR: + case GGML_OP_SQRT: + case GGML_OP_LOG: + break; + case GGML_OP_UNARY: + if (ggml_get_unary_op(op) != GGML_UNARY_OP_ABS) { + return false; + } + break; + default: + return false; + } + } + if (!ggml_are_same_shape(src0, dst)) { return false; } @@ -3061,6 +5709,10 @@ static bool ggml_hexagon_supported_softmax(const struct ggml_hexagon_session * s return false; } + if (src0->ne[2] > 512) { + return false; + } + if (src1) { if (src1->type != GGML_TYPE_F32 && src1->type != GGML_TYPE_F16) { return false; @@ -3106,6 +5758,14 @@ static bool ggml_hexagon_supported_softmax(const struct ggml_hexagon_session * s return false; } + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, src0_nrows); + struct htp_softmax_vtcm_layout layout; + htp_softmax_vtcm_layout_build(&layout, src0->ne[0], src1 ? src1->ne[0] : 1, src1 != nullptr, src1 && src1->type == GGML_TYPE_F16, n_threads); + if (layout.total_bytes > sess->vtcm_size) { + return false; + } + return true; GGML_UNUSED(sess); @@ -3114,7 +5774,11 @@ static bool ggml_hexagon_supported_softmax(const struct ggml_hexagon_session * s static bool ggml_hexagon_supported_set_rows(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { const struct ggml_tensor * src0 = op->src[0]; // values const struct ggml_tensor * src1 = op->src[1]; // indices - const struct ggml_tensor * dst = op; + const struct ggml_tensor * dst = op->src[2] ? op->src[2] : op; + + if (dst->type == GGML_TYPE_Q8_0 && src0->ne[0] < 32) { + return false; + } if (src0->type != GGML_TYPE_F32) { return false; @@ -3124,7 +5788,7 @@ static bool ggml_hexagon_supported_set_rows(const struct ggml_hexagon_session * return false; } - if (dst->type != GGML_TYPE_F16) { + if (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16 && dst->type != GGML_TYPE_Q8_0) { return false; } @@ -3138,7 +5802,19 @@ static bool ggml_hexagon_supported_get_rows(const struct ggml_hexagon_session * const struct ggml_tensor * src1 = op->src[1]; // indices const struct ggml_tensor * dst = op; - if (src0->type != GGML_TYPE_F32) { + if (src0->extra) { + const auto * extra = (const ggml_hexagon_tensor_extra *) src0->extra; + if (extra->flags & GGML_HEXAGON_TENSOR_REPACK) { + return false; + } + } + + if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_I32 && src0->ne[0] < 32) { + return false; + } + + if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16 && + src0->type != GGML_TYPE_Q8_0 && src0->type != GGML_TYPE_I32) { return false; } @@ -3146,7 +5822,12 @@ static bool ggml_hexagon_supported_get_rows(const struct ggml_hexagon_session * return false; } - if (dst->type != GGML_TYPE_F32) { + if (src0->type == GGML_TYPE_I32) { + if (dst->type != GGML_TYPE_I32) { + return false; + } + } + else if (dst->type != GGML_TYPE_F32) { return false; } @@ -3177,52 +5858,109 @@ static bool ggml_hexagon_supported_argsort(const struct ggml_hexagon_session * s GGML_UNUSED(sess); } -static bool ggml_hexagon_supported_rope(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { - const int32_t * op_params = &op->op_params[0]; +static bool ggml_hexagon_supported_top_k(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { + const struct ggml_tensor * src0 = op->src[0]; // values + const struct ggml_tensor * dst = op; // indices - int mode = op_params[2]; + if (src0->type != GGML_TYPE_F32) { + return false; + } - // n_dims == ne0/2, so the rotation spans the full row - if (mode == GGML_ROPE_TYPE_VISION) { - const int n_dims = op_params[1]; - if (n_dims != (int) (op->src[0]->ne[0] / 2)) { - return false; - } + if (dst->type != GGML_TYPE_I32) { + return false; } - if (mode & 1) { + + // Single row uses the threaded chunk+merge path. Multi-row uses one full + // buffer per thread, so it keeps the tighter 64K cap. + const bool single_row = (src0->ne[1] == 1 && src0->ne[2] == 1 && src0->ne[3] == 1); + const int64_t max_ne00 = single_row ? (256*1024) : (64*1024); + + if (src0->ne[0] > max_ne00) { return false; } + return true; +} + +static bool ggml_hexagon_supported_rope(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { const struct ggml_tensor * src0 = op->src[0]; const struct ggml_tensor * src1 = op->src[1]; const struct ggml_tensor * src2 = op->src[2]; const struct ggml_tensor * dst = op; - if (src0->type != GGML_TYPE_F32) { - return false; // FIXME: add support for GGML_TYPE_F16 for src0 + if (!ggml_are_same_shape(src0, dst)) { + return false; + } + + if (src0->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32 || src1->type != GGML_TYPE_I32) { + return false; + } + + if (src0->ne[0] <= 0) { + return false; + } + + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + if (src0_nrows == 0) { + return false; + } + + const int32_t * op_params = &op->op_params[0]; + const int n_dims = op_params[1]; + const int mode = op_params[2]; + const int n_offs = op_params[15]; + + // llama probes weight placement with a dummy rope where every param is 0 (llama-model-loader.cpp). + // Rejecting it puts rope_freqs on the CPU, which then splits the graph at every full-attention layer. + if (n_dims < 0 || n_dims % 2 != 0) { + return false; + } + + // ggml_rope_set_offset: HVX kernels need a VLEN-aligned window start (32 f32 elems) + if (n_offs < 0 || (n_offs % 32 != 0) || (n_offs + n_dims > src0->ne[0])) { + return false; } - if (dst->type != GGML_TYPE_F32) { + + float freq_base; + memcpy(&freq_base, op_params + 5, sizeof(float)); + if (freq_base < 0.0f) { return false; } - if (src1->type != GGML_TYPE_I32) { + + if (mode != GGML_ROPE_TYPE_NORMAL && + mode != GGML_ROPE_TYPE_NEOX && + mode != GGML_ROPE_TYPE_MROPE && + mode != GGML_ROPE_TYPE_VISION && + mode != GGML_ROPE_TYPE_IMROPE) { return false; } - if (src2) { - if (src2->type != GGML_TYPE_F32) { + + const bool is_mrope = (mode & GGML_ROPE_TYPE_MROPE) != 0; + + // n_dims == ne0/2, so the rotation spans the full row + if (mode == GGML_ROPE_TYPE_VISION) { + if (n_dims != (int) (src0->ne[0] / 2) || n_offs != 0) { return false; } - int n_dims = op_params[1]; - if (src2->ne[0] < (n_dims / 2)) { + } + + if (is_mrope) { + const int32_t * sections = op_params + 11; + if (sections[0] <= 0 && sections[1] <= 0 && sections[2] <= 0) { return false; } } + const int64_t min_pos_len = (is_mrope || mode == GGML_ROPE_TYPE_VISION) ? src0->ne[2] * 4 : src0->ne[2]; + if (src1->ne[0] < min_pos_len || !ggml_is_contiguous(src1)) { + return false; + } + if (src2) { - if (!ggml_is_contiguous(src1) || !ggml_is_contiguous(src2)) { + if (src2->type != GGML_TYPE_F32 || !ggml_is_contiguous(src2)) { return false; } - } else { - if (!ggml_is_contiguous(src1)) { + if (src2->ne[0] < (n_dims / 2)) { return false; } } @@ -3235,9 +5973,17 @@ static bool ggml_hexagon_supported_rope(const struct ggml_hexagon_session * sess if (src0->nb[1] < src0->ne[0] * sizeof(float) || dst->nb[1] < dst->ne[0] * sizeof(float)) { return false; } - return true; - GGML_UNUSED(sess); + const uint32_t n_threads = (std::min)((uint32_t) sess->n_threads, src0_nrows); + const uint32_t n_freq_factors = src2 ? (uint32_t) src2->ne[0] : 0; + + struct htp_rope_vtcm_layout layout; + htp_rope_vtcm_layout_build(&layout, src0->ne[0], n_threads, n_freq_factors); + if (layout.total_bytes > sess->vtcm_size) { + return false; + } + + return true; } static bool ggml_hexagon_supported_ssm_conv(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { @@ -3255,11 +6001,14 @@ static bool ggml_hexagon_supported_ssm_conv(const struct ggml_hexagon_session * return false; // src0 should be effectively 3D } - const int d_conv = src1->ne[0]; + const int d_conv = src1->ne[0]; const int d_inner = src0->ne[1]; const int n_t = dst->ne[1]; const int n_s = dst->ne[2]; + if (d_conv == 0 || d_conv > 32 || d_inner == 0) { + return false; + } if (src0->ne[0] != d_conv - 1 + n_t || src0->ne[1] != d_inner || src0->ne[2] != n_s) { return false; } @@ -3276,20 +6025,19 @@ static bool ggml_hexagon_supported_ssm_conv(const struct ggml_hexagon_session * return false; } - return true; + struct htp_ssm_conv_kernel_params kparams; + ggml_hexagon_precompute_ssm_conv_params(sess, src0, src1, dst, &kparams); + if ((size_t) kparams.vtcm_size > sess->vtcm_size) { + return false; + } - GGML_UNUSED(sess); + return true; } static bool ggml_hexagon_supported_im2col(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { const struct ggml_tensor * src1 = op->src[1]; const struct ggml_tensor * dst = op; - const bool is_2D = ((const int32_t *) op->op_params)[6] == 1; - if (!is_2D) { - return false; - } - // For now support F32->F32 and F32->F16 only. if (src1->type != GGML_TYPE_F32 || (dst->type != GGML_TYPE_F16 && dst->type != GGML_TYPE_F32)) { return false; @@ -3299,13 +6047,6 @@ static bool ggml_hexagon_supported_im2col(const struct ggml_hexagon_session * se return false; } - // For now keep padded OPs on CPU. Will revisit once we expand coverage past patch-embed OPs. - const int32_t p0 = ((const int32_t *) op->op_params)[2]; - const int32_t p1 = ((const int32_t *) op->op_params)[3]; - if (p0 != 0 || p1 != 0) { - return false; - } - GGML_UNUSED(sess); return true; } @@ -3318,6 +6059,14 @@ static bool ggml_hexagon_supported_pad(const struct ggml_hexagon_session * sess, return false; } + const int32_t lp0 = ((const int32_t *) op->op_params)[0]; + const int32_t rp0 = ((const int32_t *) op->op_params)[1]; + const int32_t circular = ((const int32_t *) op->op_params)[8]; + + if (circular && (lp0 > src0->ne[0] || rp0 > src0->ne[0])) { + return false; + } + return true; GGML_UNUSED(sess); @@ -3369,10 +6118,6 @@ static bool ggml_hexagon_supported_solve_tri(const struct ggml_hexagon_session * const struct ggml_tensor * src1 = op->src[1]; // B const struct ggml_tensor * dst = op; // X - if (!src0 || !src1) { - return false; - } - if (src0->type != GGML_TYPE_F32 || src1->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { return false; } @@ -3440,14 +6185,17 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) { case GGML_OP_SET_ROWS: return HTP_OP_SET_ROWS; case GGML_OP_SUM_ROWS: return HTP_OP_SUM_ROWS; case GGML_OP_ARGSORT: return HTP_OP_ARGSORT; + case GGML_OP_TOP_K: return HTP_OP_TOP_K; case GGML_OP_NORM: return HTP_OP_NORM; case GGML_OP_L2_NORM: return HTP_OP_L2_NORM; case GGML_OP_RMS_NORM: return HTP_OP_RMS_NORM; case GGML_OP_CONCAT: return HTP_OP_CONCAT; case GGML_OP_SCALE: return HTP_OP_SCALE; case GGML_OP_CLAMP: return HTP_OP_CLAMP; + case GGML_OP_LEAKY_RELU: return HTP_OP_LEAKY_RELU; case GGML_OP_SQR: return HTP_OP_SQR; case GGML_OP_SQRT: return HTP_OP_SQRT; + case GGML_OP_LOG: return HTP_OP_UNARY_LOG; case GGML_OP_SOFT_MAX: return HTP_OP_SOFTMAX; case GGML_OP_SSM_CONV: return HTP_OP_SSM_CONV; case GGML_OP_GATED_DELTA_NET: return HTP_OP_GATED_DELTA_NET; @@ -3460,6 +6208,7 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) { case GGML_OP_TRI: return HTP_OP_TRI; case GGML_OP_PAD: return HTP_OP_PAD; case GGML_OP_IM2COL: return HTP_OP_IM2COL; + case GGML_OP_ROLL: return HTP_OP_ROLL; case GGML_OP_UNARY: switch (ggml_get_unary_op(t)) { @@ -3471,6 +6220,8 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) { case GGML_UNARY_OP_EXP: return HTP_OP_UNARY_EXP; case GGML_UNARY_OP_SOFTPLUS: return HTP_OP_UNARY_SOFTPLUS; case GGML_UNARY_OP_TANH: return HTP_OP_UNARY_TANH; + case GGML_UNARY_OP_ABS: return HTP_OP_UNARY_ABS; + case GGML_UNARY_OP_RELU: return HTP_OP_UNARY_RELU; default: break; } @@ -3480,7 +6231,9 @@ static htp_op_code op_remap_to_htp(const ggml_tensor * t) { switch (ggml_get_glu_op(t)) { case GGML_GLU_OP_SWIGLU: return HTP_OP_GLU_SWIGLU; case GGML_GLU_OP_SWIGLU_OAI: return HTP_OP_GLU_SWIGLU_OAI; + case GGML_GLU_OP_SWIGLU_CLAMP: return HTP_OP_GLU_SWIGLU_CLAMP; case GGML_GLU_OP_GEGLU: return HTP_OP_GLU_GEGLU; + case GGML_GLU_OP_GEGLU_QUICK: return HTP_OP_GLU_GEGLU_QUICK; default: break; } break; @@ -3512,10 +6265,43 @@ static bool mm_is_hmx_eligible(const ggml_tensor * t) { return ggml_hexagon_matmul_is_hmx_eligible(src0, src1, t, ne01_padded, is_matmul_id, is_batched); } +static bool is_supported_mul_mat_nx_kernel(const ggml_tensor * src0, const struct htp_mm_kernel_params * kparams) { + if (kparams->n_hmx) { + return kparams->kernel_type == HTP_MM_KERNEL_HMX_2D; + } + + if (!ggml_hexagon_is_repack_type(src0->type) || src0->type == GGML_TYPE_Q6_K) { + return false; // Q6_K has no fused HVX kernel + } + + return kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW; +} + +static bool is_supported_mul_mat_id_nx_kernel(const ggml_tensor * src0, const struct htp_mm_kernel_params * kparams) { + if (kparams->n_hmx) { + return kparams->kernel_type == HTP_MM_KERNEL_HMX_2D; + } + + if (!ggml_hexagon_is_repack_type(src0->type)) { + return false; + } + + return kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW || kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_BLOCK; +} + static bool is_mergeable_mul_mat(const ggml_tensor * t) { - if (!t || t->op != GGML_OP_MUL_MAT) return false; - if (t->src[1]->type != GGML_TYPE_F32) return false; - return ggml_is_quantized(t->src[0]->type) && !mm_is_hmx_eligible(t); + if (t->op != GGML_OP_MUL_MAT) return false; + + const ggml_tensor * src0 = t->src[0]; + const ggml_tensor * src1 = t->src[1]; + if (src1->type != GGML_TYPE_F32) return false; + if (src0->ne[2] != 1 || src0->ne[3] != 1) return false; + + if (mm_is_hmx_eligible(t)) { + return ggml_hexagon_is_hmx_weight_type(src0->type); + } + + return ggml_hexagon_is_repack_type(src0->type) && src0->type != GGML_TYPE_Q6_K; } static bool is_mergeable_mul_mat_pair(const ggml_tensor * n1, const ggml_tensor * n2) { @@ -3525,159 +6311,104 @@ static bool is_mergeable_mul_mat_pair(const ggml_tensor * n1, const ggml_tensor if (n1->src[1] != n2->src[1]) { return false; } - if (n1->src[0]->ne[0] != n2->src[0]->ne[0] || - n1->src[0]->ne[1] != n2->src[0]->ne[1]) { + if (n1->src[0]->ne[0] != n2->src[0]->ne[0]) { return false; } if (n1->src[0]->type != n2->src[0]->type) { return false; } + if (mm_is_hmx_eligible(n1) != mm_is_hmx_eligible(n2)) { + return false; + } return true; } -static bool is_qkv_mergeable(const ggml_tensor * n_q, const ggml_tensor * n_k, const ggml_tensor * n_v) { - if (!is_mergeable_mul_mat(n_q) || !is_mergeable_mul_mat(n_k) || !is_mergeable_mul_mat(n_v)) { +static bool is_mergeable_mul_mat_id(const ggml_tensor * t) { + if (t->op != GGML_OP_MUL_MAT_ID) return false; + + const ggml_tensor * src0 = t->src[0]; + return ggml_hexagon_is_repack_type(src0->type); +} + +static bool is_mergeable_mul_mat_id_pair(const ggml_tensor * n1, const ggml_tensor * n2) { + if (!is_mergeable_mul_mat_id(n1) || !is_mergeable_mul_mat_id(n2)) { return false; } - if (n_q->src[1] != n_k->src[1] || n_q->src[1] != n_v->src[1]) { + if (n1->src[1] != n2->src[1]) { return false; } - if (n_q->src[0]->type != n_k->src[0]->type || n_q->src[0]->type != n_v->src[0]->type) { + if (n1->src[2] != n2->src[2]) { return false; } - if (n_k->src[0]->ne[0] != n_v->src[0]->ne[0] || - n_k->src[0]->ne[1] != n_v->src[0]->ne[1]) { + if (n1->src[0]->ne[0] != n2->src[0]->ne[0]) { return false; } - if (n_q->src[0]->ne[0] != n_k->src[0]->ne[0]) { + if (n1->src[0]->ne[2] != n2->src[0]->ne[2]) { return false; } - return true; -} - -static bool try_fuse_node(const ggml_hexagon_session * sess, const ggml_cgraph * graph, int & i, std::vector & nodes) { - if (!opt_opfusion) { + if (n1->src[0]->type != n2->src[0]->type) { return false; } - - ggml_tensor * n = graph->nodes[i]; - ggml_tensor * next_node = (i + 1 < graph->n_nodes) ? graph->nodes[i + 1] : nullptr; - - if (n->op == GGML_OP_RMS_NORM && next_node) { - if (next_node->op == GGML_OP_MUL && op_is_compute(next_node) && ggml_can_fuse(graph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) { - htp_opnode node(n, {}, HTP_OP_RMS_NORM_MUL); - node.add_fused(next_node); - - auto inputs = node.get_inputs(); - const struct ggml_tensor * src0 = inputs[0]; - const struct ggml_tensor * src1 = inputs.size() > 1 ? inputs[1] : nullptr; - ggml_hexagon_precompute_unary_params(sess, - node.opcode, src0, src1, node.dst(), - (struct htp_unary_kernel_params *)node.kernel_params - ); - - nodes.push_back(std::move(node)); - i++; // skip the fused MUL node - return true; - } - } - - if (is_mergeable_mul_mat(n)) { - ggml_tensor * n1 = (i + 1 < graph->n_nodes) ? graph->nodes[i + 1] : nullptr; - ggml_tensor * n2 = (i + 2 < graph->n_nodes) ? graph->nodes[i + 2] : nullptr; - if (is_qkv_mergeable(n, n1, n2)) { - struct htp_mm_kernel_params kparams; - ggml_hexagon_precompute_fused_qkv_params(sess, n1->src[0], n1->src[1], &kparams); - if ((size_t)kparams.vtcm_size <= sess->vtcm_size) { - // Reorder to KVQ: K (n1), V (n2), Q (n) - htp_opnode node(n1, {}, HTP_OP_MUL_MAT_QKV); - node.add_fused(n2, true); - node.add_fused(n, true); - memcpy(node.kernel_params, &kparams, sizeof(kparams)); - nodes.push_back(std::move(node)); - i += 2; - return true; - } else { - HEX_VERBOSE("ggml-hex: skip QKV fusion because VTCM needed (%d) > budget (%zu)\n", - kparams.vtcm_size, sess->vtcm_size); - } - } - if (is_mergeable_mul_mat_pair(n, n1)) { - struct htp_mm_kernel_params kparams; - ggml_hexagon_precompute_fused_ffn_params(sess, n->src[0], n->src[1], &kparams); - if ((size_t)kparams.vtcm_size <= sess->vtcm_size) { - htp_opnode node(n, {}, HTP_OP_MUL_MAT_FFN); - node.add_fused(n1, true); - memcpy(node.kernel_params, &kparams, sizeof(kparams)); - nodes.push_back(std::move(node)); - i += 1; - return true; - } else { - HEX_VERBOSE("ggml-hex: skip FFN fusion because VTCM needed (%d) > budget (%zu)\n", - kparams.vtcm_size, sess->vtcm_size); - } - } - } - - if (n->op == GGML_OP_MUL_MAT && next_node) { - if (next_node->op == GGML_OP_ADD && op_is_compute(next_node) && ggml_can_fuse(graph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) { - if (next_node->src[0] == n || next_node->src[1] == n) { - const struct ggml_tensor * src2 = (next_node->src[0] == n) ? next_node->src[1] : next_node->src[0]; - struct htp_mm_kernel_params kparams; - ggml_hexagon_precompute_fused_matmul_add_params(sess, n->src[0], n->src[1], src2, next_node, &kparams); - const int src1_nrows = n->src[1]->ne[1] * n->src[1]->ne[2] * n->src[1]->ne[3]; - const bool can_fuse = (kparams.n_hmx > 0) || (src1_nrows == 1); - if (can_fuse && (size_t)kparams.vtcm_size <= sess->vtcm_size) { - htp_opnode node(n, {}, HTP_OP_MUL_MAT_ADD); - node.add_fused(next_node); - memcpy(node.kernel_params, &kparams, sizeof(kparams)); - nodes.push_back(std::move(node)); - i += 1; - return true; - } else if (can_fuse) { - HEX_VERBOSE("ggml-hex: skip MUL_MAT_ADD fusion because VTCM needed (%d) > budget (%zu)\n", - kparams.vtcm_size, sess->vtcm_size); - } - } - } + if (mm_is_hmx_eligible(n1) != mm_is_hmx_eligible(n2)) { + return false; } - - return false; + return true; } static ggml_status ggml_backend_hexagon_graph_compute(ggml_backend_t backend, ggml_cgraph * graph) { auto sess = static_cast(backend->context); + if (sess->last_error > HTP_STATUS_OK) { + return GGML_STATUS_FAILED; + } + HEX_VERBOSE("ggml-hex: %s graph-compute n_nodes %d\n", sess->c_name(), graph->n_nodes); const std::vector * nodes_ptr = nullptr; std::vector computed_nodes; // Check for cache hit - bool cache_hit = (graph->uid != 0 && sess->cached_graph.uid == graph->uid); + bool cache_hit = (graph->uid != 0 && sess->cached_uid == graph->uid); if (cache_hit) { - nodes_ptr = &sess->cached_graph.htp_nodes; + nodes_ptr = &sess->cached_nodes; } else { + // Tag fusable tensors in graph + for (int i = 0; i < graph->n_nodes; i++) { + auto * extra = (ggml_hexagon_tensor_extra *) graph->nodes[i]->extra; + if (!extra) continue; + + extra->flags &= ~GGML_HEXAGON_TENSOR_FUSEABLE; + + if (graph->nodes[i]->op == GGML_OP_RMS_NORM && ggml_can_fuse(graph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) { + extra->flags |= GGML_HEXAGON_TENSOR_FUSEABLE; + } else if (graph->nodes[i]->op == GGML_OP_MUL_MAT || graph->nodes[i]->op == GGML_OP_MUL_MAT_ID) { + if ((i + 1 < graph->n_nodes && graph->nodes[i + 1]->op == GGML_OP_ADD && ggml_can_fuse(graph, i, { graph->nodes[i]->op, GGML_OP_ADD })) || + ggml_node_has_n_uses(graph, i, 1)) { + extra->flags |= GGML_HEXAGON_TENSOR_FUSEABLE; + } + } + } + computed_nodes.reserve(graph->n_nodes); - // Fuse and finalize for (int i = 0; i < graph->n_nodes; ++i) { ggml_tensor * n = graph->nodes[i]; if (!op_is_compute(n)) { continue; } - if (try_fuse_node(sess, graph, i, computed_nodes)) { - continue; - } - - htp_opnode node(n, {}, HTP_OP_INVALID); + htp_opnode node(HTP_OP_INVALID, n); node.opcode = op_remap_to_htp(n); if (node.opcode == HTP_OP_MUL_MAT || node.opcode == HTP_OP_MUL_MAT_ID) { ggml_hexagon_precompute_matmul_params(sess, node.node->src[0], node.node->src[1], node.node, (struct htp_mm_kernel_params *)node.kernel_params ); + } else if (node.opcode == HTP_OP_MUL || node.opcode == HTP_OP_ADD || node.opcode == HTP_OP_ADD_ID || node.opcode == HTP_OP_SUB || node.opcode == HTP_OP_DIV) { + const ggml_tensor * src1 = node.node->src[1]; + GGML_ASSERT(ggml_hexagon_precompute_binary_params(sess, + node.opcode, node.node->src[0], src1, node.node, + (struct htp_binary_kernel_params *) node.kernel_params)); } else if (node.opcode == HTP_OP_FLASH_ATTN_EXT) { ggml_hexagon_precompute_flash_attn_params(sess, node.node, @@ -3691,28 +6422,57 @@ static ggml_status ggml_backend_hexagon_graph_compute(ggml_backend_t backend, gg node.opcode, src0, src1, node.dst(), (struct htp_unary_kernel_params *)node.kernel_params ); + } else if (node.opcode == HTP_OP_GET_ROWS) { + ggml_hexagon_precompute_get_rows_params(sess, + node.node->src[0], node.node->src[1], node.dst(), + (struct htp_get_rows_kernel_params *)node.kernel_params + ); + } else if (node.opcode == HTP_OP_SET_ROWS) { + ggml_hexagon_precompute_set_rows_params(sess, + node.node->src[0], node.node->src[1], node.dst(), + (struct htp_set_rows_kernel_params *)node.kernel_params + ); + } else if (node.opcode == HTP_OP_ROPE) { + ggml_hexagon_precompute_rope_params(sess, + node.node, + (struct htp_rope_kernel_params *)node.kernel_params + ); + } else if (node.opcode == HTP_OP_SSM_CONV) { + ggml_hexagon_precompute_ssm_conv_params(sess, + node.node->src[0], node.node->src[1], node.dst(), + (struct htp_ssm_conv_kernel_params *)node.kernel_params + ); + } else if (node.opcode == HTP_OP_SOFTMAX) { + ggml_hexagon_precompute_softmax_params(sess, + node.node, + (struct htp_softmax_kernel_params *)node.kernel_params + ); + } else if (node.opcode == HTP_OP_GATED_DELTA_NET) { + ggml_hexagon_precompute_gated_delta_net_params(sess, + node.node, + (struct htp_gdn_kernel_params *)node.kernel_params + ); } computed_nodes.push_back(std::move(node)); } if (graph->uid != 0) { - sess->cached_graph.uid = graph->uid; - sess->cached_graph.htp_nodes = std::move(computed_nodes); - nodes_ptr = &sess->cached_graph.htp_nodes; + sess->cached_uid = graph->uid; + sess->cached_nodes = std::move(computed_nodes); + nodes_ptr = &sess->cached_nodes; } else { nodes_ptr = &computed_nodes; } } // Queue and execute - if (opt_opstage & HTP_OPSTAGE_QUEUE) { - for (const auto & node : *nodes_ptr) { - sess->enqueue_op(node); - } + for (const auto & node : *nodes_ptr) { + sess->enqueue_op(node); } - // Wait until all pending ops complete - sess->flush(); + if (sess->last_error > HTP_STATUS_OK) { + return GGML_STATUS_FAILED; + } return GGML_STATUS_SUCCESS; } @@ -3723,7 +6483,110 @@ static void ggml_backend_hexagon_synchronize(ggml_backend_t backend) { HEX_VERBOSE("ggml-hex: %s synchronize\n", sess->c_name()); // Wait until all pending ops complete - sess->flush(); + sess->flush_sync(); + if (sess->last_error > HTP_STATUS_OK) { + GGML_ABORT("ggml-hex: %s synchronize failed : dsp-error %s\n", sess->c_name(), status_to_str(sess->last_error)); + } +} + +enum ggml_hexagon_mem_range_type { + HEXAGON_MEM_RANGE_TYPE_SRC, + HEXAGON_MEM_RANGE_TYPE_DST, +}; + +struct ggml_hexagon_mem_range { + uint64_t pb; + uint64_t p0; + uint64_t p1; + ggml_hexagon_mem_range_type pt; +}; + +struct ggml_hexagon_mem_ranges { + std::vector ranges; + + void reset() { + ranges.clear(); + } + + void add(const ggml_hexagon_mem_range & mr) { + ranges.push_back(mr); + } + + bool check(const ggml_hexagon_mem_range & mr) const { + for (const auto & cmp : ranges) { + if (mr.pb != cmp.pb) { + continue; + } + if (mr.pt == HEXAGON_MEM_RANGE_TYPE_SRC && cmp.pt == HEXAGON_MEM_RANGE_TYPE_SRC) { + continue; + } + if (mr.p0 < cmp.p1 && mr.p1 > cmp.p0) { + return false; + } + } + return true; + } +}; + +static ggml_hexagon_mem_range ggml_hexagon_mem_range_from_tensor(const ggml_tensor * tensor, ggml_hexagon_mem_range_type pt) { + const ggml_tensor * base = tensor->view_src ? tensor->view_src : tensor; + ggml_hexagon_mem_range mr; + if (tensor->buffer) { + mr = { + /*.pb =*/ (uint64_t) tensor->buffer, + /*.p0 =*/ (uint64_t) tensor->data, + /*.p1 =*/ (uint64_t) tensor->data + ggml_backend_buft_get_alloc_size(tensor->buffer->buft, tensor), + /*.pt =*/ pt, + }; + } else { + mr = { + /*.pb =*/ (uint64_t) base, + /*.p0 =*/ 0, + /*.p1 =*/ 1024, + /*.pt =*/ pt, + }; + } + return mr; +} + +static void ggml_hexagon_mem_ranges_add_node(ggml_hexagon_mem_ranges & mrs, const htp_opnode & node) { + if (node.is_empty()) return; + + for (int i = 0; i < GGML_MAX_SRC; i++) { + if (node.node->src[i]) { + mrs.add(ggml_hexagon_mem_range_from_tensor(node.node->src[i], HEXAGON_MEM_RANGE_TYPE_SRC)); + } + } + for (const auto * fused : node.fused) { + for (int i = 0; i < GGML_MAX_SRC; i++) { + if (fused->src[i]) { + mrs.add(ggml_hexagon_mem_range_from_tensor(fused->src[i], HEXAGON_MEM_RANGE_TYPE_SRC)); + } + } + } + mrs.add(ggml_hexagon_mem_range_from_tensor(node.dst(), HEXAGON_MEM_RANGE_TYPE_DST)); +} + +static bool ggml_hexagon_mem_ranges_check_node(const ggml_hexagon_mem_ranges & mrs, const htp_opnode & node) { + if (node.is_empty()) return true; + + for (int i = 0; i < GGML_MAX_SRC; i++) { + if (node.node->src[i]) { + if (!mrs.check(ggml_hexagon_mem_range_from_tensor(node.node->src[i], HEXAGON_MEM_RANGE_TYPE_SRC))) { + return false; + } + } + } + for (const auto * fused : node.fused) { + for (int i = 0; i < GGML_MAX_SRC; i++) { + if (fused->src[i]) { + if (!mrs.check(ggml_hexagon_mem_range_from_tensor(fused->src[i], HEXAGON_MEM_RANGE_TYPE_SRC))) { + return false; + } + } + } + } + return mrs.check(ggml_hexagon_mem_range_from_tensor(node.dst(), HEXAGON_MEM_RANGE_TYPE_DST)); } static std::vector ggml_hexagon_graph_optimize_reorder(const std::vector & nodes) { @@ -3734,28 +6597,32 @@ static std::vector ggml_hexagon_graph_optimize_reorder(const std::vector used(n, false); - // The main goal here is to stack the MUL_MAT ops with the same src1 input. - // This allows use to reuse dynamically quantized src1 in VTCM. + ggml_hexagon_mem_ranges mrs; - // TODO: the current version might do incorrect reordering in cases where quantized src0 - // input is an output of another Op. + // The main goal here is to stack the MUL_MAT ops with the same src1 input. + // This allows us to reuse dynamically quantized src1 in VTCM. for (int i0 = 0; i0 < n; i0++) { if (used[i0]) { continue; } - res.push_back(i0); - const auto & node0 = nodes[i0]; if (!node0.stackable()) { + res.push_back(i0); + used[i0] = true; continue; } // that many nodes forward to search for stackable nodes that can reuse VTCM constexpr int N_FORWARD = 16; + std::vector stack; + stack.push_back(i0); + + mrs.reset(); + for (int i1 = i0 + 1; i1 < i0 + N_FORWARD && i1 < n; i1++) { if (used[i1]) { continue; @@ -3763,17 +6630,25 @@ static std::vector ggml_hexagon_graph_optimize_reorder(const std::vectorn_nodes; constexpr int MAX_FUSE = 16; @@ -3783,14 +6658,9 @@ static void ggml_backend_hexagon_graph_optimize(ggml_backend_t backend, ggml_cgr std::vector nodes; nodes.reserve(gf->n_nodes); - // fuse nodes: - // we don't want to make reorders that break fusing, so we first pack all fusable tensors - // and perform the reorder over the fused nodes. after the reorder is done, we unfuse + // Pack nodes for reordering for (int i = 0; i < n; i++) { - htp_opnode node = { - /*.node =*/gf->nodes[i], - /*.fused =*/{}, - }; + htp_opnode node(HTP_OP_INVALID, gf->nodes[i]); // fuse only ops that start with these operations // can be expanded when needed @@ -3813,60 +6683,321 @@ static void ggml_backend_hexagon_graph_optimize(ggml_backend_t backend, ggml_cgr f++; } - f -= i; - for (; f > 1; f--) { - if (ggml_can_fuse(gf, i, ops, f)) { - break; - } - } + f -= i; + for (; f > 1; f--) { + if (ggml_can_fuse(gf, i, ops, f)) { + break; + } + } + + // add the fused tensors into the node info so we can unfuse them later + for (int k = 1; k < f; k++) { + ++i; + + // the .dst() becomes the last fused tensor + node.add_fused(gf->nodes[i]); + } + } + + nodes.push_back(std::move(node)); + } + + const auto order = ggml_hexagon_graph_optimize_reorder(nodes); + + // unfuse + { + int j = 0; + for (const auto i : order) { + const auto & node = nodes[i]; + + gf->nodes[j++] = node.node; + + for (auto * fused : node.fused) { + gf->nodes[j++] = fused; + } + } + } + + GGML_UNUSED(backend); +} + +static uint64_t ggml_hexagon_session_key(const ggml_hexagon_session * sess) { + return ((uint64_t) (uint32_t) sess->phys_idx << 32) | (uint32_t) sess->virt_idx; +} + +static bool ggml_hexagon_cpy_tensor_async_phys(ggml_backend_t backend_src, ggml_backend_t backend_dst, const ggml_tensor * src, ggml_tensor * dst) { + auto sess_src = static_cast(backend_src->context); + auto sess_dst = static_cast(backend_dst->context); + + sess_src->mmap_tensor(src); + auto sbuf_dst = sess_dst->mmap_tensor(dst); + + if (!sess_src->clone_buffer(sbuf_dst)) { return false; } + + const uint64_t src_key = ggml_hexagon_session_key(sess_src); + auto & fence_slot = sess_dst->cpy_fence_slots[src_key]; + if (!fence_slot) { + fence_slot = (volatile uint32_t *) sess_dst->alloc_fence(1); + } + + if (!sess_src->clone_buffer(sess_dst->fence_buf)) { return false; } + + if (++sess_dst->fence_seq == 0) sess_dst->fence_seq = 1; + uint32_t fence_seq = sess_dst->fence_seq; + + HEX_VERBOSE("ggml-hex: %s cpy-tensor-async %s -> %s size %zu : seq 0x%x\n", + sess_dst->name.c_str(), src->name, dst->name, ggml_nbytes(src), fence_seq); + + // dummy fence extra (must be static) + static ggml_hexagon_tensor_extra fence_extra { {}, 0, GGML_HEXAGON_TENSOR_FENCE }; + + ggml_tensor fence_tensor {}; + fence_tensor.buffer = &sess_dst->fence_buf->backend_buffer; + fence_tensor.extra = &fence_extra; + fence_tensor.data = (void *) fence_slot; + fence_tensor.type = GGML_TYPE_I32; + fence_tensor.ne[0] = 1; + fence_tensor.ne[1] = 1; + fence_tensor.ne[2] = 1; + fence_tensor.ne[3] = 1; + fence_tensor.nb[0] = sizeof(int32_t); + fence_tensor.nb[1] = sizeof(int32_t); + fence_tensor.nb[2] = sizeof(int32_t); + fence_tensor.nb[3] = sizeof(int32_t); + fence_tensor.op = GGML_OP_NONE; + + sess_src->enqueue_cpy(src, dst, &fence_tensor, fence_seq); + sess_dst->enqueue_fence(&fence_tensor, fence_seq, /* wait = */ true); + + sess_dst->add_peer(sess_src); + + return true; +} + +static bool ggml_hexagon_cpy_tensor_async_virt(ggml_backend_t backend_src, ggml_backend_t backend_dst, const ggml_tensor * src, ggml_tensor * dst) { + auto sess_src = static_cast(backend_src->context); + auto sess_dst = static_cast(backend_dst->context); + + auto sbuf_src = sess_src->mmap_tensor(src); + sess_dst->mmap_tensor(dst); + + if (!sess_dst->clone_buffer(sbuf_src)) { return false; } + + HEX_VERBOSE("ggml-hex: %s cpy-tensor-async %s -> %s size %zu\n", + sess_dst->name.c_str(), src->name, dst->name, ggml_nbytes(src)); + + sess_dst->enqueue_cpy(src, dst); + sess_dst->add_peer(sess_src); + + return true; +} + +static bool ggml_backend_hexagon_cpy_tensor_async(ggml_backend_t backend_src, ggml_backend_t backend_dst, const ggml_tensor * src, ggml_tensor * dst) { + if (!ggml_backend_is_hexagon(backend_src) || !ggml_backend_is_hexagon(backend_dst)) { + return false; + } + + // FIXME: ggml-meta needs to call init_tensor on auxiliary tensors + if (!dst->extra) { + ggml_backend_buffer_init_tensor(dst->buffer, dst); + } + + auto * dst_extra = static_cast(dst->extra); + const auto * src_extra = static_cast(src->extra); + dst_extra->flags = src_extra->flags & ~GGML_HEXAGON_TENSOR_FUSEABLE; + + auto sess_src = static_cast(backend_src->context); + auto sess_dst = static_cast(backend_dst->context); + + if (sess_src == sess_dst) { + HEX_VERBOSE("ggml-hex: %s cpy-tensor-async %s -> %s size %zu\n", sess_dst->name.c_str(), src->name, dst->name, ggml_nbytes(src)); + sess_src->enqueue_cpy(src, dst); + return true; + } + + if (sess_src->phys_idx != sess_dst->phys_idx) + return ggml_hexagon_cpy_tensor_async_phys(backend_src, backend_dst, src, dst); - // add the fused tensors into the node info so we can unfuse them later - for (int k = 1; k < f; k++) { - ++i; + return ggml_hexagon_cpy_tensor_async_virt(backend_src, backend_dst, src, dst); +} - // the .dst() becomes the last fused tensor - node.add_fused(gf->nodes[i]); - } - } +static ggml_backend_event_t ggml_backend_hexagon_device_event_new(ggml_backend_dev_t dev) { + auto dev_ctx = static_cast(dev->context); + auto sess = dev_ctx->session(); + + ggml_hexagon_event * hex_event = new ggml_hexagon_event(); + hex_event->fence_sess = sess; + hex_event->sess = sess; + hex_event->fence_slot = (volatile uint32_t *) sess->alloc_fence(1); + + static ggml_hexagon_tensor_extra fence_extra { {}, 0, GGML_HEXAGON_TENSOR_FENCE }; + hex_event->fence_tensor.buffer = &sess->fence_buf->backend_buffer; + hex_event->fence_tensor.extra = &fence_extra; + hex_event->fence_tensor.data = (void *) hex_event->fence_slot; + hex_event->fence_tensor.type = GGML_TYPE_I32; + hex_event->fence_tensor.ne[0] = 1; + hex_event->fence_tensor.ne[1] = 1; + hex_event->fence_tensor.ne[2] = 1; + hex_event->fence_tensor.ne[3] = 1; + hex_event->fence_tensor.nb[0] = sizeof(int32_t); + hex_event->fence_tensor.nb[1] = sizeof(int32_t); + hex_event->fence_tensor.nb[2] = sizeof(int32_t); + hex_event->fence_tensor.nb[3] = sizeof(int32_t); + hex_event->fence_tensor.op = GGML_OP_NONE; + + HEX_VERBOSE("ggml-hex: %s event-new : event %p fence %p\n", ggml_backend_dev_name(dev), (void *)hex_event, (void *)hex_event->fence_slot); + + return new ggml_backend_event { + /* .device = */ dev, + /* .context = */ hex_event, + }; +} - nodes.push_back(std::move(node)); +static void ggml_hexagon_event_synchronize(ggml_backend_dev_t dev, ggml_hexagon_event * hex_event) { + if (hex_event->seq == 0) { + return; } - const auto order = ggml_hexagon_graph_optimize_reorder(nodes); + HEX_VERBOSE("ggml-hex: %s event-synchronize : event %p seq 0x%x fence %p\n", + ggml_backend_dev_name(dev), (void *)hex_event, hex_event->seq, (void *)hex_event->fence_slot); - // unfuse - { - int j = 0; - for (const auto i : order) { - const auto & node = nodes[i]; + auto * fence = reinterpret_cast *>(hex_event->fence_slot); - gf->nodes[j++] = node.node; + if ((int32_t)(fence[0].load(std::memory_order_relaxed) - hex_event->seq) < 0) { + hex_event->sess->flush_async(); + } - for (auto * fused : node.fused) { - gf->nodes[j++] = fused; + while (true) { + if ((int32_t)(fence[0].load(std::memory_order_acquire) - hex_event->seq) >= 0) { + uint32_t status = fence[1].load(std::memory_order_acquire); + if (status > HTP_STATUS_OK) { + GGML_ABORT("ggml-hex: %s event-synchronize failed : dsp-error %s\n", + hex_event->sess->c_name(), status_to_str(status)); } + break; } + std::this_thread::yield(); } +} - GGML_UNUSED(backend); +static void ggml_backend_hexagon_device_event_free(ggml_backend_dev_t dev, ggml_backend_event_t event) { + auto * hex_event = static_cast(event->context); + ggml_hexagon_event_synchronize(dev, hex_event); + HEX_VERBOSE("ggml-hex: %s event-free : event %p\n", ggml_backend_dev_name(dev), (void *)hex_event); + hex_event->fence_sess->free_fence((void *) hex_event->fence_slot, 1); + delete hex_event; + delete event; +} + +static void ggml_backend_hexagon_device_event_synchronize(ggml_backend_dev_t dev, ggml_backend_event_t event) { + auto * hex_event = static_cast(event->context); + ggml_hexagon_event_synchronize(dev, hex_event); +} + +static void ggml_backend_hexagon_event_record(ggml_backend_t backend, ggml_backend_event_t event) { + auto sess = static_cast(backend->context); + auto hex_event = static_cast(event->context); + + if (++sess->fence_seq == 0) sess->fence_seq = 1; + hex_event->sess = sess; + hex_event->seq = sess->fence_seq; + + sess->enqueue_fence(&hex_event->fence_tensor, hex_event->seq, /* wait = */ false); + + HEX_VERBOSE("ggml-hex: %s event-record : event %p seq 0x%x fence %p\n", + sess->c_name(), (void *)hex_event, hex_event->seq, (void *)hex_event->fence_slot); +} + +static void ggml_backend_hexagon_event_wait(ggml_backend_t backend, ggml_backend_event_t event) { + auto sess = static_cast(backend->context); + auto hex_event = static_cast(event->context); + + if (hex_event->seq == 0) { + return; + } + + HEX_VERBOSE("ggml-hex: %s event-wait : event %p seq 0x%x fence %p\n", + sess->c_name(), (void *)hex_event, hex_event->seq, (void *)hex_event->fence_slot); + + // same physical NPU runs sequentially in FIFO order + if (sess->phys_idx == hex_event->sess->phys_idx) { + if (sess != hex_event->sess) { + sess->add_peer(hex_event->sess); + } + return; + } + + sess->clone_buffer(hex_event->fence_sess->fence_buf); + sess->add_peer(hex_event->sess); + sess->enqueue_fence(&hex_event->fence_tensor, hex_event->seq, /* wait = */ true); +} + +static void ggml_backend_hexagon_set_tensor_async(ggml_backend_t backend, struct ggml_tensor * tensor, const void * data, size_t offset, size_t size) { + auto sess = static_cast(backend->context); + HEX_VERBOSE("ggml-hex: %s set-tensor-async %s : data %p offset %zu size %zu usage %d\n", + sess->c_name(), tensor->name, data, offset, size, tensor->buffer ? (int) tensor->buffer->usage : -1); + ggml_backend_tensor_set(tensor, data, offset, size); +} + +static void ggml_backend_hexagon_get_tensor_async(ggml_backend_t backend, const struct ggml_tensor * tensor, void * data, size_t offset, size_t size) { + auto sess = static_cast(backend->context); + HEX_VERBOSE("ggml-hex: %s get-tensor-async %s : data %p offset %zu size %zu usage %d\n", + sess->c_name(), tensor->name, data, offset, size, tensor->buffer ? (int) tensor->buffer->usage : -1); + sess->flush_sync(); + if (sess->last_error > HTP_STATUS_OK) { + GGML_ABORT("ggml-hex: %s get-tensor-async failed : dsp-error %s\n", sess->c_name(), status_to_str(sess->last_error)); + } + ggml_backend_tensor_get(tensor, data, offset, size); +} + +static void ggml_backend_hexagon_set_tensor_2d_async(ggml_backend_t backend, + struct ggml_tensor * tensor, + const void * data, + size_t offset, + size_t size, + size_t n_copies, + size_t stride_tensor, + size_t stride_data) { + auto sess = static_cast(backend->context); + HEX_VERBOSE("ggml-hex: %s set-tensor-2d-async %s : data %p offset %zu size %zu n_copies %zu stride_tensor %zu stride_data %zu usage %d\n", + sess->c_name(), tensor->name, data, offset, size, n_copies, stride_tensor, stride_data, tensor->buffer ? (int) tensor->buffer->usage : -1); + ggml_backend_tensor_set_2d(tensor, data, offset, size, n_copies, stride_tensor, stride_data); +} + +static void ggml_backend_hexagon_get_tensor_2d_async(ggml_backend_t backend, + const struct ggml_tensor * tensor, + void * data, + size_t offset, + size_t size, + size_t n_copies, + size_t stride_tensor, + size_t stride_data) { + auto sess = static_cast(backend->context); + HEX_VERBOSE("ggml-hex: %s get-tensor-2d-async %s : data %p offset %zu size %zu n_copies %zu stride_tensor %zu stride_data %zu usage %d\n", + sess->c_name(), tensor->name, data, offset, size, n_copies, stride_tensor, stride_data, tensor->buffer ? (int) tensor->buffer->usage : -1); + sess->flush_sync(); + if (sess->last_error > HTP_STATUS_OK) { + GGML_ABORT("ggml-hex: %s get-tensor-2d-async failed : dsp-error %s\n", sess->c_name(), status_to_str(sess->last_error)); + } + ggml_backend_tensor_get_2d(tensor, data, offset, size, n_copies, stride_tensor, stride_data); } static struct ggml_backend_i hexagon_backend_i = { /* .get_name = */ ggml_backend_hexagon_name, /* .free = */ ggml_backend_hexagon_free, - /* .set_tensor_async = */ NULL, - /* .get_tensor_async = */ NULL, - /* .set_tensor_2d_async = */ NULL, - /* .get_tensor_2d_async = */ NULL, - /* .cpy_tensor_async = */ NULL, + /* .set_tensor_async = */ ggml_backend_hexagon_set_tensor_async, + /* .get_tensor_async = */ ggml_backend_hexagon_get_tensor_async, + /* .set_tensor_2d_async = */ ggml_backend_hexagon_set_tensor_2d_async, + /* .get_tensor_2d_async = */ ggml_backend_hexagon_get_tensor_2d_async, + /* .cpy_tensor_async = */ ggml_backend_hexagon_cpy_tensor_async, /* .synchronize = */ ggml_backend_hexagon_synchronize, /* .graph_plan_create = */ NULL, /* .graph_plan_free = */ NULL, /* .graph_plan_update = */ NULL, /* .graph_plan_compute = */ NULL, /* .graph_compute = */ ggml_backend_hexagon_graph_compute, - /* .event_record = */ NULL, - /* .event_wait = */ NULL, + /* .event_record = */ ggml_backend_hexagon_event_record, + /* .event_wait = */ ggml_backend_hexagon_event_wait, /* .graph_optimize = */ ggml_backend_hexagon_graph_optimize, }; @@ -3883,7 +7014,8 @@ bool ggml_backend_is_hexagon(ggml_backend_t backend) { // device interface static ggml_backend_t ggml_backend_hexagon_device_init(ggml_backend_dev_t dev, const char * params) { - auto sess = static_cast(dev->context); + auto dev_ctx = static_cast(dev->context); + auto sess = dev_ctx->session(); return new ggml_backend{ /* .guid = */ ggml_backend_hexagon_guid(), @@ -3896,8 +7028,8 @@ static ggml_backend_t ggml_backend_hexagon_device_init(ggml_backend_dev_t dev, c } static const char * ggml_backend_hexagon_device_get_name(ggml_backend_dev_t dev) { - auto sess = static_cast(dev->context); - return sess->c_name(); + auto dev_ctx = static_cast(dev->context); + return dev_ctx->c_name(); GGML_UNUSED(dev); } @@ -3927,44 +7059,24 @@ static void ggml_backend_hexagon_device_get_props(ggml_backend_dev_t dev, struct ggml_backend_hexagon_device_get_memory(dev, &props->memory_free, &props->memory_total); props->caps = { /* .async = */ true, - /* .host_buffer = */ (bool) opt_hostbuf, + /* .host_buffer = */ false, /* .buffer_from_host_ptr = */ false, - /* .events = */ false, + /* .events = */ true, /* .mmap_support = */ false, }; } static ggml_backend_buffer_type_t ggml_backend_hexagon_device_get_buffer_type(ggml_backend_dev_t dev) { - auto sess = static_cast(dev->context); - return &sess->buffer_type; -} - -static ggml_backend_buffer_type_t ggml_backend_hexagon_device_get_repack_buffer_type(ggml_backend_dev_t dev) { - auto sess = static_cast(dev->context); - return &sess->repack_buffer_type; -} - -static bool ggml_hexagon_supported_buffer(ggml_hexagon_session *sess, const struct ggml_tensor * t) { - if (t && t->buffer) { - if (ggml_backend_buffer_is_hexagon(t->buffer) == false) return false; // not our buffer - if (ggml_backend_hexagon_buffer_get_sess(t->buffer) != sess) return false; // wrong session - } - return true; + auto dev_ctx = static_cast(dev->context); + return &dev_ctx->buffer_type; } -static bool ggml_hexagon_supported_buffers(ggml_hexagon_session *sess, const struct ggml_tensor * t) { - // all srcs & dsts must be mapped to the same session - if (!ggml_hexagon_supported_buffer(sess, t)) { - return false; - } - - for (int i = 0; i < GGML_MAX_SRC; i++) { - if (!ggml_hexagon_supported_buffer(sess, t->src[i])) { - return false; - } +static ggml_backend_buffer_type_t ggml_backend_hexagon_device_get_host_buffer_type(ggml_backend_dev_t dev) { + if (!opt_hostbuf) { + return NULL; } - - return true; + auto dev_ctx = static_cast(dev->context); + return &dev_ctx->host_buffer_type; } static bool ggml_hexagon_supported_cpy(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { @@ -4054,17 +7166,37 @@ static bool ggml_hexagon_supported_fill(const struct ggml_hexagon_session * sess GGML_UNUSED(sess); } -static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, const struct ggml_tensor * op) { - auto sess = static_cast(dev->context); +static bool ggml_hexagon_supported_roll(const struct ggml_hexagon_session * sess, const struct ggml_tensor * op) { + GGML_UNUSED(sess); - // reject ops that match the filter - if (opt_opfilter && std::regex_match(ggml_op_desc(op), *opt_opfilter)) { + const struct ggml_tensor * src0 = op->src[0]; + const struct ggml_tensor * dst = op; + + if (src0->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { + return false; + } + + if (!ggml_are_same_shape(src0, dst)) { + return false; + } + + if (src0->nb[0] != ggml_type_size(src0->type) || dst->nb[0] != ggml_type_size(dst->type)) { + return false; + } + + if (!ggml_is_contiguous(dst)) { return false; } - // all srcs & dsts must be mapped to the same session - if (!ggml_hexagon_supported_buffers(sess, op)) { - ggml_hexagon_dump_op_supp(sess->name, op, false); + return true; +} + +static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, const struct ggml_tensor * op) { + auto dev_ctx = static_cast(dev->context); + auto sess = dev_ctx->session(); + + // reject ops that match the filter + if (opt_opfilter && std::regex_match(ggml_op_desc(op), *opt_opfilter)) { return false; } @@ -4102,11 +7234,13 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons case GGML_OP_RMS_NORM: case GGML_OP_SCALE: case GGML_OP_CLAMP: + case GGML_OP_LEAKY_RELU: supp = ggml_hexagon_supported_unary(sess, op); break; case GGML_OP_SQR: case GGML_OP_SQRT: + case GGML_OP_LOG: supp = ggml_hexagon_supported_unary(sess, op); break; @@ -4125,12 +7259,15 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons case GGML_UNARY_OP_SIGMOID: case GGML_UNARY_OP_SOFTPLUS: case GGML_UNARY_OP_TANH: + case GGML_UNARY_OP_ABS: case GGML_UNARY_OP_SILU: case GGML_UNARY_OP_GELU: case GGML_UNARY_OP_GELU_QUICK: + case GGML_UNARY_OP_RELU: supp = ggml_hexagon_supported_unary(sess, op); break; default: + supp = false; break; } break; @@ -4139,10 +7276,13 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons switch (ggml_get_glu_op(op)) { case GGML_GLU_OP_SWIGLU: case GGML_GLU_OP_SWIGLU_OAI: + case GGML_GLU_OP_SWIGLU_CLAMP: case GGML_GLU_OP_GEGLU: + case GGML_GLU_OP_GEGLU_QUICK: supp = ggml_hexagon_supported_activations(sess, op); break; default: + supp = false; break; } break; @@ -4179,6 +7319,10 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons supp = ggml_hexagon_supported_argsort(sess, op); break; + case GGML_OP_TOP_K: + supp = ggml_hexagon_supported_top_k(sess, op); + break; + case GGML_OP_SSM_CONV: supp = ggml_hexagon_supported_ssm_conv(sess, op); break; @@ -4219,6 +7363,10 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons supp = ggml_hexagon_supported_pad(sess, op); break; + case GGML_OP_ROLL: + supp = ggml_hexagon_supported_roll(sess, op); + break; + default: break; } @@ -4228,31 +7376,14 @@ static bool ggml_backend_hexagon_device_supports_op(ggml_backend_dev_t dev, cons } static bool ggml_backend_hexagon_device_supports_buft(ggml_backend_dev_t dev, ggml_backend_buffer_type_t buft) { - if (buft->iface.get_alignment != ggml_backend_hexagon_buffer_type_get_alignment) { - return false; - } - - auto s0 = static_cast(dev->context); - auto s1 = static_cast(buft->context)->sess; - - // Need session/domain-id for buffers to be compatible - bool supp = (s0->session_id == s1->session_id); + auto dev_ctx = static_cast(dev->context); - HEX_VERBOSE("ggml-hex: %s device-supports-buft %s (%d)\n", s0->name.c_str(), s1->name.c_str(), (int) supp); + bool supp = (buft == &dev_ctx->host_buffer_type) || (buft == &dev_ctx->buffer_type); + HEX_VERBOSE("ggml-hex: %s device-supports-buft %s %s\n", dev_ctx->c_name(), ggml_backend_buft_name(buft), supp ? "yes" : "no"); return supp; } -static ggml_backend_buffer_type_t * ggml_backend_hexagon_device_get_extra_buffers_type(ggml_backend_dev_t dev) { - auto s0 = static_cast(dev->context); - HEX_VERBOSE("ggml-hex: device-get-extra-buft : %s \n", s0->name.c_str()); - - static ggml_backend_buffer_type_t bufts[2]; - bufts[0] = ggml_backend_hexagon_device_get_repack_buffer_type(dev); - bufts[1] = NULL; - return bufts; -} - static const struct ggml_backend_device_i ggml_backend_hexagon_device_i = { /* .get_name = */ ggml_backend_hexagon_device_get_name, /* .get_description = */ ggml_backend_hexagon_device_get_description, @@ -4261,52 +7392,52 @@ static const struct ggml_backend_device_i ggml_backend_hexagon_device_i = { /* .get_props = */ ggml_backend_hexagon_device_get_props, /* .init_backend = */ ggml_backend_hexagon_device_init, /* .get_buffer_type = */ ggml_backend_hexagon_device_get_buffer_type, - /* .get_host_buffer_type = */ NULL, // ggml_backend_hexagon_device_get_host_buffer_type, + /* .get_host_buffer_type = */ ggml_backend_hexagon_device_get_host_buffer_type, /* .buffer_from_host_ptr = */ NULL, // ggml_backend_hexagon_device_buffer_from_ptr, /* .supports_op = */ ggml_backend_hexagon_device_supports_op, /* .supports_buft = */ ggml_backend_hexagon_device_supports_buft, /* .offload_op = */ NULL, // ggml_backend_hexagon_device_offload_op, - /* .event_new = */ NULL, - /* .event_free = */ NULL, - /* .event_synchronize = */ NULL, + /* .event_new = */ ggml_backend_hexagon_device_event_new, + /* .event_free = */ ggml_backend_hexagon_device_event_free, + /* .event_synchronize = */ ggml_backend_hexagon_device_event_synchronize, }; //** backend registry -#define GGML_HEXAGON_MAX_SESSIONS 16 - -struct ggml_hexagon_registry { - ggml_hexagon_registry(ggml_backend_reg_t reg); - ~ggml_hexagon_registry(); - - ggml_backend_device devices[GGML_HEXAGON_MAX_SESSIONS]; -}; - ggml_hexagon_registry::ggml_hexagon_registry(ggml_backend_reg_t reg) { GGML_LOG_INFO("ggml-hex: Hexagon backend (experimental) : allocating new registry : ndev %zu\n", opt_ndev); - GGML_LOG_INFO("ggml-hex: Hexagon Arch version v%d\n", opt_arch); + GGML_LOG_INFO("ggml-hex: Hexagon Arch version v%d, DMA64 %s\n", opt_arch, opt_dma64 ? "enabled" : "disabled"); - // Create devices / sessions + // Create devices for (size_t i = 0; i < opt_ndev; i++) { - devices[i].iface = ggml_backend_hexagon_device_i; - devices[i].reg = reg; - try { - devices[i].context = new ggml_hexagon_session(i, &devices[i]); - } catch (const std::exception & exc) { - GGML_LOG_ERROR("ggml-hex: failed to create device/session %zu\n", i); - devices[i].context = nullptr; + const auto & cfg = opt_device_configs[i]; + if (cfg.mdev_group.empty()) { + GGML_LOG_INFO("ggml-hex: device %zu: %s (phys=%d, virt=%d, domain=%s:%d)\n", + i, cfg.name.c_str(), cfg.physical_idx, cfg.virtual_idx, cfg.domain_name.c_str(), cfg.domain_id); + } else { + std::string peers_str; + for (const auto & p : cfg.mdev_group) { + if (!peers_str.empty()) peers_str += ", "; + peers_str += p.name + " (phys=" + std::to_string(p.physical_idx) + ")"; + } + GGML_LOG_INFO("ggml-hex: device %zu: %s (phys=%d, virt=%d, domain=%s:%d) [mdev peers: %s]\n", + i, cfg.name.c_str(), cfg.physical_idx, cfg.virtual_idx, cfg.domain_name.c_str(), cfg.domain_id, peers_str.c_str()); } + devices[i].iface = ggml_backend_hexagon_device_i; + devices[i].reg = reg; + devices[i].context = new ggml_backend_hexagon_device_context(i, opt_device_configs[i], &devices[i]); } + } ggml_hexagon_registry::~ggml_hexagon_registry() { GGML_LOG_INFO("ggml-hex: releasing registry\n"); - // Release devices / sessions + // Release devices for (size_t i = 0; i < opt_ndev; i++) { - auto sess = static_cast(devices[i].context); - delete sess; + auto dev_ctx = static_cast(devices[i].context); + delete dev_ctx; } } @@ -4330,14 +7461,174 @@ static ggml_backend_dev_t ggml_backend_hexagon_reg_get_device(ggml_backend_reg_t return &hreg->devices[index]; } -static void * ggml_backend_hexagon_get_proc_address(ggml_backend_reg_t reg, const char * name) { - if (strcmp(name, "ggml_backend_dev_get_extra_bufts") == 0 && opt_hostbuf) { - ggml_backend_dev_get_extra_bufts_t fct = ggml_backend_hexagon_device_get_extra_buffers_type; - return (void *) fct; +// ** communication context for tensor-split allreduce + +static void * ggml_backend_hexagon_comm_init(ggml_backend_t * backends, size_t n_backends) { + if (n_backends < 2 || n_backends > 4) { + return nullptr; } - return NULL; + for (size_t i = 0; i < n_backends; ++i) { + if (!ggml_backend_is_hexagon(backends[i])) { + return nullptr; + } + } + + for (size_t i = 0; i < n_backends; i++) { + auto sess_i = static_cast(backends[i]->context); + for (size_t j = i + 1; j < n_backends; j++) { + auto sess_j = static_cast(backends[j]->context); + if (sess_i->phys_idx == sess_j->phys_idx) { + return nullptr; + } + } + } + + auto * ctx = new ggml_backend_hexagon_comm_context(); + ctx->backends.assign(backends, backends + n_backends); + ctx->n_backends = n_backends; + + static ggml_hexagon_tensor_extra fence_extra { {}, 0, GGML_HEXAGON_TENSOR_FENCE }; + for (size_t i = 0; i < n_backends; i++) { + auto sess_i = static_cast(backends[i]->context); + ctx->fence_slots[i] = (volatile uint32_t *) sess_i->alloc_fence(1); + ctx->fence_tensors[i] = {}; + ctx->fence_tensors[i].buffer = &sess_i->fence_buf->backend_buffer; + ctx->fence_tensors[i].extra = &fence_extra; + ctx->fence_tensors[i].data = (void *) ctx->fence_slots[i]; + ctx->fence_tensors[i].type = GGML_TYPE_I32; + ctx->fence_tensors[i].ne[0] = 4; + ctx->fence_tensors[i].ne[1] = 1; + ctx->fence_tensors[i].ne[2] = 1; + ctx->fence_tensors[i].ne[3] = 1; + ctx->fence_tensors[i].nb[0] = sizeof(int32_t); + ctx->fence_tensors[i].nb[1] = sizeof(int32_t); + ctx->fence_tensors[i].nb[2] = sizeof(int32_t); + ctx->fence_tensors[i].nb[3] = sizeof(int32_t); + ctx->fence_tensors[i].op = GGML_OP_NONE; + } + + return ctx; +} + +static void ggml_backend_hexagon_comm_free(void * comm_ctx_v) { + if (!comm_ctx_v) return; + auto * ctx = static_cast(comm_ctx_v); + for (size_t i = 0; i < ctx->n_backends; i++) { + auto sess_i = static_cast(ctx->backends[i]->context); + sess_i->free_fence((void *) ctx->fence_slots[i], 1); + } + delete ctx; +} + +static bool ggml_backend_hexagon_comm_allreduce_tensor(void * comm_ctx_v, struct ggml_tensor ** tensors) { + if (opt_ar_select == 0 || !comm_ctx_v) return false; + auto * comm_ctx = static_cast(comm_ctx_v); + const size_t n_backends = comm_ctx->n_backends; + + if (n_backends < 2 || n_backends > 4) return false; + + for (size_t i = 0; i < n_backends; i++) { + auto sess_i = static_cast(comm_ctx->backends[i]->context); + for (size_t j = i + 1; j < n_backends; j++) { + auto sess_j = static_cast(comm_ctx->backends[j]->context); + if (sess_i->phys_idx == sess_j->phys_idx) { + return false; + } + } + } + + for (size_t i = 0; i < n_backends; i++) { + if (!tensors[i] || !tensors[i]->buffer || !ggml_backend_buffer_is_hexagon(tensors[i]->buffer)) { + return false; + } + if (tensors[i]->type != tensors[0]->type) { + return false; + } + if (!ggml_is_contiguous(tensors[i])) { + return false; + } + if (ggml_nelements(tensors[i]) != ggml_nelements(tensors[0])) { + return false; + } + } + + if (tensors[0]->type != GGML_TYPE_F16 && tensors[0]->type != GGML_TYPE_F32) { + return false; + } + + for (size_t r = 0; r < n_backends; r++) { + auto sess = static_cast(comm_ctx->backends[r]->context); + struct htp_allreduce_kernel_params kparams; + if (!ggml_hexagon_precompute_allreduce_params(sess, tensors[r], (uint32_t) r, (uint32_t) n_backends, false, false, &kparams)) { + return false; + } + } + + uint32_t max_seq = static_cast(comm_ctx->backends[0]->context)->fence_seq; + for (size_t i = 1; i < n_backends; i++) { + auto sess_i = static_cast(comm_ctx->backends[i]->context); + if ((int32_t)(sess_i->fence_seq - max_seq) > 0) { + max_seq = sess_i->fence_seq; + } + } + if (++max_seq == 0) max_seq = 1; + uint32_t fence_seq_entry = max_seq; + if (++max_seq == 0) max_seq = 1; + uint32_t fence_seq_exit = max_seq; + + for (size_t i = 0; i < n_backends; i++) { + auto sess_i = static_cast(comm_ctx->backends[i]->context); + sess_i->fence_seq = max_seq; + } + + std::vector data_tensors(n_backends); + std::vector sync_tensors(n_backends); + for (size_t i = 0; i < n_backends; i++) { + data_tensors[i] = tensors[i]; + sync_tensors[i] = &comm_ctx->fence_tensors[i]; + } + + for (size_t r = 0; r < n_backends; r++) { + auto sess = static_cast(comm_ctx->backends[r]->context); + sess->enqueue_allreduce(tensors[r], data_tensors, sync_tensors, (uint32_t) r, (uint32_t) n_backends, fence_seq_entry, fence_seq_exit); + for (size_t j = 0; j < n_backends; j++) { + if (r != j) { + sess->add_peer(static_cast(comm_ctx->backends[j]->context)); + } + } + } + + return true; +} + +static ggml_backend_buffer_type_t ggml_backend_hexagon_split_buffer_type(int main_device, const float * tensor_split) { + GGML_UNUSED(tensor_split); + auto reg = ggml_backend_hexagon_reg(); + auto dev = ggml_backend_reg_dev_get(reg, main_device); + if (!dev) { + dev = ggml_backend_reg_dev_get(reg, 0); + } + if (!dev) return nullptr; + auto dev_ctx = static_cast(dev->context); + return &dev_ctx->buffer_type; +} + +static void * ggml_backend_hexagon_get_proc_address(ggml_backend_reg_t reg, const char * name) { GGML_UNUSED(reg); + if (strcmp(name, "ggml_backend_split_buffer_type") == 0) { + return (void *) ggml_backend_hexagon_split_buffer_type; + } + if (strcmp(name, "ggml_backend_comm_init") == 0) { + return (void *) ggml_backend_hexagon_comm_init; + } + if (strcmp(name, "ggml_backend_comm_free") == 0) { + return (void *) ggml_backend_hexagon_comm_free; + } + if (strcmp(name, "ggml_backend_comm_allreduce_tensor") == 0) { + return (void *) ggml_backend_hexagon_comm_allreduce_tensor; + } + return NULL; } template std::vector str_to_vec(const char* str) { @@ -4360,6 +7651,91 @@ template std::string vec_to_str(std::vector v) { return str; } +static void ggml_hexagon_resolve_device_domain(ggml_hexagon_device_config & cfg, bool discovery_supported, const std::unordered_map & cdsp_map) { + if (discovery_supported) { + auto it = cdsp_map.find(cfg.physical_idx); + if (it != cdsp_map.end()) { + cfg.domain_id = it->second.id; + cfg.domain_name = it->second.name; + } else { + GGML_LOG_ERROR("ggml-hex: physical CDSP core %d not found on device (%zu CDSP core(s) available)\n", + cfg.physical_idx, cdsp_map.size()); + cfg.domain_id = -1; + cfg.domain_name = ""; + } + } else { + switch (cfg.physical_idx) { + case 0: + cfg.domain_id = 3; + cfg.domain_name = CDSP_DOMAIN_NAME; + break; + case 1: + cfg.domain_id = 4; + cfg.domain_name = "cdsp1"; + break; + default: + GGML_LOG_ERROR("ggml-hex: physical CDSP core %d not supported without dynamic discovery\n", + cfg.physical_idx); + cfg.domain_id = -1; + cfg.domain_name = ""; + break; + } + } + for (auto & sub_cfg : cfg.mdev_group) { + ggml_hexagon_resolve_device_domain(sub_cfg, discovery_supported, cdsp_map); + } +} + +// Enumerate NPU (aka CDSP) domains via FASTRPC_GET_DOMAINS if supported, +// and populate domain_id and domain_name for all configured devices. +static void ggml_hexagon_discover_devices() { + std::unordered_map cdsp_map; + bool discovery_supported = false; + + system_req_payload domain_info = {}; + domain_info.id = FASTRPC_GET_DOMAINS; + domain_info.sys.domains = nullptr; + domain_info.sys.max_domains = 0; + domain_info.sys.flags = DOMAINS_LIST_FLAGS_SET_TYPE(0, FASTRPC_NSP); + + int err = remote_system_request(&domain_info); + if (err == AEE_SUCCESS && domain_info.sys.num_domains > 0) { + std::vector domains(domain_info.sys.num_domains); + domain_info.sys.domains = domains.data(); + domain_info.sys.max_domains = (int) domains.size(); + + err = remote_system_request(&domain_info); + if (err == AEE_SUCCESS) { + discovery_supported = true; + const int n_domains = std::min(domain_info.sys.num_domains, (int) domains.size()); + for (int i = 0; i < n_domains; i++) { + GGML_LOG_INFO("ggml-hex: FASTRPC_GET_DOMAINS[%d]: type %d id %d name '%s' status %d instance-id %d\n", + i, (int) domains[i].type, domains[i].id, domains[i].name, domains[i].status, domains[i].instance_id); + if (domains[i].type != FASTRPC_NSP) { + GGML_LOG_DEBUG("ggml-hex: skipping non-CDSP domain (type=%d)\n", (int) domains[i].type); + continue; + } + if (!domains[i].status) { + GGML_LOG_WARN("ggml-hex: skipping CDSP domain id=%d (status=down)\n", domains[i].id); + continue; + } + cdsp_map[domains[i].instance_id] = domains[i]; + GGML_LOG_INFO("ggml-hex: using CDSP domain: instance-id %d id %d name '%s'\n", + domains[i].instance_id, domains[i].id, domains[i].name); + } + } else { + GGML_LOG_WARN("ggml-hex: FASTRPC_GET_DOMAINS fetch failed (0x%x), using static CDSP domains\n", (unsigned) err); + } + } else if (err != AEE_SUCCESS) { + GGML_LOG_DEBUG("ggml-hex: FASTRPC_GET_DOMAINS query failed (0x%x), using static CDSP domains\n", (unsigned) err); + } + + // Populate domain IDs and names for all configured devices + for (size_t i = 0; i < opt_ndev; i++) { + ggml_hexagon_resolve_device_domain(opt_device_configs[i], discovery_supported, cdsp_map); + } +} + static void ggml_hexagon_init(ggml_backend_reg * reg) { // Basic sanity checks to make sure definitions match static_assert((unsigned int) HTP_TYPE_Q4_0 == (unsigned int) GGML_TYPE_Q4_0, @@ -4372,10 +7748,12 @@ static void ggml_hexagon_init(ggml_backend_reg * reg) { "please update hexagon_type to match ggml_type"); static_assert((unsigned int) HTP_TYPE_IQ4_NL == (unsigned int) GGML_TYPE_IQ4_NL, "please update hexagon_type to match ggml_type"); + static_assert((unsigned int) HTP_TYPE_Q4_K == (unsigned int) GGML_TYPE_Q4_K, + "please update hexagon_type to match ggml_type"); + static_assert((unsigned int) HTP_TYPE_Q6_K == (unsigned int) GGML_TYPE_Q6_K, + "please update hexagon_type to match ggml_type"); const char * str_verbose = getenv("GGML_HEXAGON_VERBOSE"); - const char * str_hostbuf = getenv("GGML_HEXAGON_HOSTBUF"); - const char * str_opstage = getenv("GGML_HEXAGON_OPSTAGE"); const char * str_opbatch = getenv("GGML_HEXAGON_OPBATCH"); const char * str_opqueue = getenv("GGML_HEXAGON_OPQUEUE"); const char * str_oppoll = getenv("GGML_HEXAGON_OPPOLL"); @@ -4384,15 +7762,18 @@ static void ggml_hexagon_init(ggml_backend_reg * reg) { const char * str_profile = getenv("GGML_HEXAGON_PROFILE"); const char * str_etm = getenv("GGML_HEXAGON_ETM"); const char * str_nhvx = getenv("GGML_HEXAGON_NHVX"); - const char * str_use_hmx = getenv("GGML_HEXAGON_USE_HMX"); const char * str_nhmx = getenv("GGML_HEXAGON_NHMX"); const char * str_mm_select = getenv("GGML_HEXAGON_MM_SELECT"); const char * str_fa_select = getenv("GGML_HEXAGON_FA_SELECT"); + const char * str_gdn_select = getenv("GGML_HEXAGON_GDN_SELECT"); + const char * str_ar_select = getenv("GGML_HEXAGON_AR_SELECT"); const char * str_ndev = getenv("GGML_HEXAGON_NDEV"); const char * str_arch = getenv("GGML_HEXAGON_ARCH"); const char * str_vmem = getenv("GGML_HEXAGON_VMEM"); const char * str_mbuf = getenv("GGML_HEXAGON_MBUF"); const char * str_optrace = getenv("GGML_HEXAGON_OPTRACE"); + const char * str_hostbuf = getenv("GGML_HEXAGON_HOSTBUF"); + const char * str_dma64 = getenv("GGML_HEXAGON_DMA64"); // Init Arch first since it affects other defaults if (!str_arch) { @@ -4420,13 +7801,12 @@ static void ggml_hexagon_init(ggml_backend_reg * reg) { // Update vmem default opt_vmem = opt_arch >= 75 ? HTP_OP_MAX_VMEM_DEFAULT : 3000 * MiB; + opt_dma64 = opt_arch > 79 && (!str_dma64 || atoi(str_dma64) != 0); auto RE_ICASE = std::regex_constants::icase; opt_opfilter = str_opfilter ? new std::regex(str_opfilter, RE_ICASE) : NULL; opt_verbose = str_verbose ? atoi(str_verbose) : 0; - opt_hostbuf = str_hostbuf ? atoi(str_hostbuf) : opt_hostbuf; - opt_opstage = str_opstage ? strtoul(str_opstage, NULL, 0) : opt_opstage; opt_opbatch = str_opbatch ? strtoul(str_opbatch, NULL, 0) : opt_opbatch; opt_opqueue = str_opqueue ? strtoul(str_opqueue, NULL, 0) : opt_opqueue; opt_optrace = str_optrace ? strtoul(str_optrace, NULL, 0) : (opt_opbatch * 256); @@ -4435,16 +7815,198 @@ static void ggml_hexagon_init(ggml_backend_reg * reg) { opt_profile = str_profile ? atoi(str_profile) : 0; opt_etm = str_etm ? atoi(str_etm) : 0; opt_nhvx = str_nhvx ? strtoul(str_nhvx, NULL, 0) : opt_nhvx; - opt_nhmx = str_nhmx ? atoi(str_nhmx) : (str_use_hmx ? atoi(str_use_hmx) : opt_nhmx); + opt_nhmx = str_nhmx ? atoi(str_nhmx) : opt_nhmx; opt_mm_select = str_mm_select ? atoi(str_mm_select) : opt_mm_select; opt_fa_select = str_fa_select ? atoi(str_fa_select) : opt_fa_select; - opt_ndev = str_ndev ? strtoul(str_ndev, NULL, 0) : opt_ndev; - opt_hostbuf = str_hostbuf ? atoi(str_hostbuf) : opt_hostbuf; + opt_gdn_select = str_gdn_select ? atoi(str_gdn_select) : opt_gdn_select; + opt_ar_select = str_ar_select ? atoi(str_ar_select) : opt_ar_select; opt_mbuf = str_mbuf ? strtoul(str_mbuf, NULL, 0) * MiB : opt_mbuf; opt_vmem = str_vmem ? strtoul(str_vmem, NULL, 0) * MiB : opt_vmem; + opt_hostbuf = str_hostbuf ? atoi(str_hostbuf) != 0 : opt_hostbuf; + + // Parse device configuration + const char * str_devices = getenv("GGML_HEXAGON_DEVICES"); + if (!str_devices && str_ndev && str_ndev[0] != '\0') { + GGML_LOG_WARN("DEPRECATED: GGML_HEXAGON_NDEV is deprecated. use GGML_HEXAGON_DEVICES instead\n"); + str_devices = str_ndev; + } + + if (str_devices && str_devices[0] != '\0') { + bool is_single_number = true; + for (int i = 0; str_devices[i] != '\0'; i++) { + if (!isdigit((unsigned char)str_devices[i])) { + is_single_number = false; + break; + } + } + if (is_single_number) { + int n = atoi(str_devices); + if (n < 1) n = 1; + if (n > GGML_HEXAGON_MAX_SESSIONS) n = GGML_HEXAGON_MAX_SESSIONS; + opt_ndev = n; + for (size_t i = 0; i < opt_ndev; i++) { + opt_device_configs[i].physical_idx = 0; + opt_device_configs[i].virtual_idx = (int)i; + opt_device_configs[i].name = "HTP" + std::to_string(i); + opt_device_configs[i].mdev_group.clear(); + } + } else { + std::string s_devices(str_devices); + std::vector items; + std::string curr_item; + int bracket_depth = 0; + for (char ch : s_devices) { + if (ch == '[') { + bracket_depth++; + curr_item += ch; + } else if (ch == ']') { + if (bracket_depth > 0) bracket_depth--; + curr_item += ch; + } else if (ch == ',' && bracket_depth == 0) { + size_t s = curr_item.find_first_not_of(" \t\r\n"); + size_t e = curr_item.find_last_not_of(" \t\r\n"); + if (s != std::string::npos) { + items.push_back(curr_item.substr(s, e - s + 1)); + } + curr_item.clear(); + } else { + curr_item += ch; + } + } + size_t s = curr_item.find_first_not_of(" \t\r\n"); + size_t e = curr_item.find_last_not_of(" \t\r\n"); + if (s != std::string::npos) { + items.push_back(curr_item.substr(s, e - s + 1)); + } + + opt_ndev = 0; + for (const auto & item : items) { + size_t b_open = item.find('['); + size_t b_close = item.rfind(']'); + + if (b_open != std::string::npos && b_close != std::string::npos && b_close > b_open) { + // Grouped / composite syntax: Name[phys_spec:virt] or Name[phys_spec] + std::string dev_name = item.substr(0, b_open); + std::string content = item.substr(b_open + 1, b_close - b_open - 1); + + int virt = 0; + std::string phys_spec = content; + size_t colon_pos = content.find(':'); + if (colon_pos != std::string::npos) { + phys_spec = content.substr(0, colon_pos); + try { + virt = std::stoi(content.substr(colon_pos + 1)); + } catch (...) { + virt = 0; + } + } else { + size_t dev_colon = dev_name.find(':'); + if (dev_colon != std::string::npos) { + try { + virt = std::stoi(dev_name.substr(dev_colon + 1)); + } catch (...) { + virt = 0; + } + } + } + + // Parse physical indices from phys_spec (e.g. 0-1, 0,1, 0-3, etc.) + std::vector phys_list; + std::stringstream pss(phys_spec); + std::string p_part; + while (std::getline(pss, p_part, ',')) { + size_t ps = p_part.find_first_not_of(" \t\r\n"); + size_t pe = p_part.find_last_not_of(" \t\r\n"); + if (ps == std::string::npos) continue; + p_part = p_part.substr(ps, pe - ps + 1); + + size_t dash_pos = p_part.find('-'); + if (dash_pos != std::string::npos) { + try { + int p_start = std::stoi(p_part.substr(0, dash_pos)); + int p_end = std::stoi(p_part.substr(dash_pos + 1)); + for (int p = p_start; p <= p_end; p++) { + if (std::find(phys_list.begin(), phys_list.end(), p) == phys_list.end()) { + phys_list.push_back(p); + } + } + } catch (...) { + GGML_LOG_WARN("ggml-hex: failed to parse physical range in '%s'\n", p_part.c_str()); + } + } else { + try { + int p = std::stoi(p_part); + if (std::find(phys_list.begin(), phys_list.end(), p) == phys_list.end()) { + phys_list.push_back(p); + } + } catch (...) { + GGML_LOG_WARN("ggml-hex: failed to parse physical index in '%s'\n", p_part.c_str()); + } + } + } - if (opt_ndev > GGML_HEXAGON_MAX_SESSIONS) { - opt_ndev = GGML_HEXAGON_MAX_SESSIONS; + if (phys_list.empty()) { + phys_list.push_back(0); + } + + if (opt_ndev < GGML_HEXAGON_MAX_SESSIONS) { + auto & cfg = opt_device_configs[opt_ndev]; + cfg.name = dev_name; + cfg.physical_idx = phys_list[0]; + cfg.virtual_idx = virt; + cfg.mdev_group.clear(); + + for (size_t k = 1; k < phys_list.size(); k++) { + ggml_hexagon_device_config sub_cfg; + sub_cfg.physical_idx = phys_list[k]; + sub_cfg.virtual_idx = virt; + sub_cfg.name = "HTP" + std::to_string(phys_list[k]) + ":" + std::to_string(virt); + cfg.mdev_group.push_back(sub_cfg); + } + opt_ndev++; + } else { + GGML_LOG_WARN("ggml-hex: max sessions limit reached (%d), ignoring device %s\n", GGML_HEXAGON_MAX_SESSIONS, item.c_str()); + } + } else if (item.rfind("HTP", 0) == 0) { + std::string rest = item.substr(3); + size_t colon_pos = rest.find(':'); + int phys = 0; + int virt = 0; + try { + if (colon_pos == std::string::npos) { + phys = std::stoi(rest); + virt = 0; + } else { + phys = std::stoi(rest.substr(0, colon_pos)); + virt = std::stoi(rest.substr(colon_pos + 1)); + } + } catch (...) { + GGML_LOG_WARN("ggml-hex: failed to parse device index in '%s'\n", item.c_str()); + continue; + } + + if (opt_ndev < GGML_HEXAGON_MAX_SESSIONS) { + opt_device_configs[opt_ndev].physical_idx = phys; + opt_device_configs[opt_ndev].virtual_idx = virt; + opt_device_configs[opt_ndev].name = colon_pos == std::string::npos + ? "HTP" + std::to_string(phys) + : "HTP" + std::to_string(phys) + ":" + std::to_string(virt); + opt_device_configs[opt_ndev].mdev_group.clear(); + opt_ndev++; + } else { + GGML_LOG_WARN("ggml-hex: max sessions limit reached (%d), ignoring device %s\n", GGML_HEXAGON_MAX_SESSIONS, item.c_str()); + } + } else { + GGML_LOG_WARN("ggml-hex: invalid device name format '%s', must start with HTP\n", item.c_str()); + } + } + } + } else { + opt_ndev = 1; + opt_device_configs[0].physical_idx = 0; + opt_device_configs[0].virtual_idx = 0; + opt_device_configs[0].name = "HTP0"; + opt_device_configs[0].mdev_group.clear(); } #if defined(__ANDROID__) @@ -4454,6 +8016,9 @@ static void ggml_hexagon_init(ggml_backend_reg * reg) { } #endif + // Resolve domain info for all configured devices + ggml_hexagon_discover_devices(); + if (str_profile) { opt_pmu_evt = [&]() -> std::vector { auto v = str_to_vec(str_profile); diff --git a/ggml/src/ggml-hexagon/htp-drv.cpp b/ggml/src/ggml-hexagon/htp-drv.cpp index 4f079080..437e367c 100644 --- a/ggml/src/ggml-hexagon/htp-drv.cpp +++ b/ggml/src/ggml-hexagon/htp-drv.cpp @@ -73,6 +73,7 @@ typedef int (*remote_handle64_close_pfn_t)(remote_handle h); typedef int (*remote_handle_control_pfn_t)(uint32_t req, void* data, uint32_t datalen); typedef int (*remote_handle64_control_pfn_t)(remote_handle64 h, uint32_t req, void* data, uint32_t datalen); typedef int (*remote_session_control_pfn_t)(uint32_t req, void *data, uint32_t datalen); +typedef int (*remote_system_request_pfn_t)(system_req_payload * req); // // Driver API pfns @@ -99,6 +100,7 @@ remote_handle64_close_pfn_t remote_handle64_close_pfn = nullptr; remote_handle_control_pfn_t remote_handle_control_pfn = nullptr; remote_handle64_control_pfn_t remote_handle64_control_pfn = nullptr; remote_session_control_pfn_t remote_session_control_pfn = nullptr; +remote_system_request_pfn_t remote_system_request_pfn = nullptr; // // Driver API @@ -206,6 +208,13 @@ HTPDRV_API int remote_session_control(uint32_t req, void * data, uint32_t datale return remote_session_control_pfn(req, data, datalen); } +HTPDRV_API int remote_system_request(system_req_payload * req) { + if (!remote_system_request_pfn) { + return AEE_EUNSUPPORTEDAPI; + } + return remote_system_request_pfn(req); +} + #ifdef _WIN32 static std::string wstr_to_str(std::wstring_view wstr) { @@ -367,6 +376,7 @@ int htpdrv_init() { dlsym(handle.get(), remote_handle64_control_pfn_t, remote_handle64_control_pfn, remote_handle64_control, false); dlsym(handle.get(), remote_session_control_pfn_t, remote_session_control_pfn, remote_session_control, false); dlsym(handle.get(), remote_handle64_close_pfn_t, remote_handle64_close_pfn, remote_handle64_close, false); + dlsym(handle.get(), remote_system_request_pfn_t, remote_system_request_pfn, remote_system_request, true); lib_cdsp_rpc_handle = std::move(handle); initialized = true; diff --git a/ggml/src/ggml-hexagon/htp-drv.h b/ggml/src/ggml-hexagon/htp-drv.h index f3cc0da7..8232780e 100644 --- a/ggml/src/ggml-hexagon/htp-drv.h +++ b/ggml/src/ggml-hexagon/htp-drv.h @@ -116,6 +116,8 @@ HTPDRV_API domain * htpdrv_get_domain(int domain_id); */ HTPDRV_API int htpdrv_get_arch(int domain, int * arch); +HTPDRV_API int remote_system_request(system_req_payload * req); + #ifdef __cplusplus } #endif diff --git a/ggml/src/ggml-hexagon/htp-opnode.h b/ggml/src/ggml-hexagon/htp-opnode.h index b0c859da..803aa3f5 100644 --- a/ggml/src/ggml-hexagon/htp-opnode.h +++ b/ggml/src/ggml-hexagon/htp-opnode.h @@ -8,60 +8,111 @@ #include #include #include +#include #include #include "htp-ops.h" #include "htp/matmul-ops.h" #include "htp/flash-attn-ops.h" #include "htp/unary-ops.h" +#include "htp/binary-ops.h" +#include "htp/allreduce-ops.h" +#include "htp/ssm-conv.h" +#include "htp/gated-delta-net-ops.h" +#include "htp/softmax-ops.h" struct htp_opnode { - ggml_tensor * node = nullptr; - - std::vector fused; - - htp_op_code opcode = HTP_OP_INVALID; + ggml_tensor * node { nullptr }; + htp_op_code opcode { HTP_OP_INVALID }; + int32_t kernel_params[HTP_OP_MAX_KERN_PARAMS] {0}; + + std::vector fused; + std::vector> dummy; + + std::vector inputs; + std::vector outputs; + std::string name; + + int n_active_src(const ggml_tensor * t) const { + if (!t) return 0; + for (int i = GGML_MAX_SRC - 1; i >= 0; i--) { + if (t->src[i]) { + return i + 1; + } + } + return 0; + } - std::vector extra_dsts; + void init(ggml_tensor * node) { + this->node = node; + if (this->node) { + this->name = ggml_op_desc(this->node); - int32_t kernel_params[HTP_OP_MAX_KERN_PARAMS] = {0}; + // Build inputs (preserving optional nullptrs) + int n_inputs = n_active_src(this->node); + this->inputs.resize(n_inputs, nullptr); + for (int i = 0; i < n_inputs; i++) { + this->inputs[i] = this->node->src[i]; + } - htp_opnode(ggml_tensor * node = nullptr, std::vector fused = {}, htp_op_code opcode = HTP_OP_INVALID, std::vector extra_dsts = {}) - : node(node), fused(std::move(fused)), opcode(opcode), extra_dsts(std::move(extra_dsts)) {} + // Build outputs + this->outputs.push_back(this->dst()); + } + } - ggml_op op() const { - return node->op; + htp_opnode(htp_op_code opcode = HTP_OP_INVALID, ggml_tensor * node = nullptr) : opcode(opcode) { + init(node); } - const ggml_tensor * dst() const { - return fused.empty() ? node : fused.back(); + ggml_op op() const { return node->op; } + const ggml_tensor * src0() const { return node->src[0]; } + const ggml_tensor * src1() const { return node->src[1]; } + const ggml_tensor * dst() const { return outputs.empty() ? node : outputs.back(); } + + ggml_tensor * add_dummy(const ggml_tensor & t) { + dummy.push_back(std::make_shared(t)); + return dummy.back().get(); } void add_fused(ggml_tensor * t, bool extra_dst = false) { fused.push_back(t); + + name += "+"; + name += ggml_op_desc(t); + if (extra_dst) { - extra_dsts.push_back(t); + outputs.push_back(t); + } else { + outputs.clear(); + outputs.push_back(t); } - } - std::vector get_outputs() const { - std::vector res; - if (extra_dsts.empty()) { - res.push_back(dst()); - } else { - res.push_back(node); - for (const auto * x : extra_dsts) { - res.push_back(x); + // Remove the newly fused intermediate output tensor t from inputs (if it was there) + inputs.erase(std::remove(inputs.begin(), inputs.end(), t), inputs.end()); + + // Append new inputs from t, preserving middle nullptrs + int n_inputs = n_active_src(t); + for (int i = 0; i < n_inputs; i++) { + const auto * src = t->src[i]; + if (!src) { + inputs.push_back(nullptr); + } else if (src != node && + std::find(fused.begin(), fused.end(), src) == fused.end() && + std::find(inputs.begin(), inputs.end(), src) == inputs.end()) { + inputs.push_back(src); } } - return res; } - const ggml_tensor * src0() const { - return node->src[0]; + const std::vector & get_inputs() const { + return inputs; } - const ggml_tensor * src1() const { - return node->src[1]; + const std::vector & get_outputs() const { + return outputs; + } + + std::string op_name() const { + return name; } bool is_empty() const { @@ -81,75 +132,6 @@ struct htp_opnode { bool same_input(const htp_opnode& n) const { return n.src1() == this->src1(); } - - std::vector get_inputs() const { - if (fused.empty()) { - int last_non_null = -1; - for (int i = 0; i < GGML_MAX_SRC; i++) { - if (node->src[i]) { - last_non_null = i; - } - } - std::vector inputs(last_non_null + 1, nullptr); - for (int i = 0; i <= last_non_null; i++) { - inputs[i] = node->src[i]; - } - return inputs; - } - - std::vector inputs(GGML_MAX_SRC, nullptr); - std::vector outputs; - outputs.push_back(node); - for (const auto * f : fused) { - outputs.push_back(f); - } - - auto contains = [&](const std::vector & vec, const ggml_tensor * t) { - for (const auto * x : vec) { - if (x == t) return true; - } - return false; - }; - - int count = 0; - auto add_input = [&](const ggml_tensor * t) { - if (t && !contains(outputs, t) && !contains(inputs, t)) { - if (count < (int)inputs.size()) { - inputs[count++] = t; - } else { - inputs.push_back(t); - } - } - }; - - for (int i = 0; i < GGML_MAX_SRC; i++) { - if (node->src[i]) { - add_input(node->src[i]); - } - } - for (const auto * f : fused) { - for (int i = 0; i < GGML_MAX_SRC; i++) { - if (f->src[i]) { - add_input(f->src[i]); - } - } - } - - inputs.resize(count); - return inputs; - } - - std::string op_name() const { - if (fused.empty()) { - return ggml_op_desc(node); - } - std::string name = ggml_op_desc(node); - for (const auto * f : fused) { - name += "+"; - name += ggml_op_desc(f); - } - return name; - } }; struct htp_opformat { @@ -337,7 +319,7 @@ struct htp_opformat { } void format_kernel_params(char * str, size_t max_size, const htp_opnode & node) { if (node.opcode == HTP_OP_MUL_MAT || node.opcode == HTP_OP_MUL_MAT_ID || - node.opcode == HTP_OP_MUL_MAT_QKV || node.opcode == HTP_OP_MUL_MAT_FFN || + node.opcode == HTP_OP_MUL_MAT_NX || node.opcode == HTP_OP_MUL_MAT_ID_NX || node.opcode == HTP_OP_MUL_MAT_ADD) { const auto * kparams = (const struct htp_mm_kernel_params *) node.kernel_params; const char * path = "unknown"; @@ -347,10 +329,6 @@ struct htp_opformat { } else if (type == HTP_MM_KERNEL_HVX_F16_F16_VTCM || type == HTP_MM_KERNEL_HVX_F32_F32_VTCM || type == HTP_MM_KERNEL_HVX_QUANT_ROW || type == HTP_MM_KERNEL_HVX_QUANT_BLOCK) { path = "hvx-tiled"; - } else if (type == HTP_MM_KERNEL_HVX_F16_F16_DDR || type == HTP_MM_KERNEL_HVX_F16_F32_DDR || - type == HTP_MM_KERNEL_HVX_F32_F32_DDR || type == HTP_MM_KERNEL_HVX_F32_F16_DDR || - type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - path = "hvx-flat"; } snprintf(str, max_size, "%s vtcm %d", path, (int) kparams->vtcm_size); } else if (node.opcode == HTP_OP_FLASH_ATTN_EXT) { @@ -366,6 +344,29 @@ struct htp_opformat { } else if (htp_op_is_unary(node.opcode)) { const auto * kparams = (const struct htp_unary_kernel_params *) node.kernel_params; snprintf(str, max_size, "%s vtcm %d", kparams->col_tile ? "wide-row" : "row-block", (int) kparams->vtcm_size); + } else if (node.opcode == HTP_OP_MDEV_GROUP && node.node) { + snprintf(str, max_size, "idx %d count %d", (int) node.node->op_params[0], (int) node.dst()->ne[1]); + } else if ((node.opcode == HTP_OP_FENCE || node.opcode == HTP_OP_CPY_FENCE) && node.node) { + snprintf(str, max_size, "seq 0x%x", (uint32_t) node.node->op_params[0]); + } else if (node.opcode == HTP_OP_ALLREDUCE && node.node) { + snprintf(str, max_size, "seq 0x%x -> 0x%x", (uint32_t) node.node->op_params[0], (uint32_t) node.node->op_params[1]); + } else if (node.opcode == HTP_OP_SSM_CONV) { + const auto * kparams = (const struct htp_ssm_conv_kernel_params *) node.kernel_params; + snprintf(str, max_size, "%s vtcm %d", kparams->n_t == 1 ? "decode" : "prefill", (int) kparams->vtcm_size); + } else if (node.opcode == HTP_OP_SOFTMAX) { + const auto * kparams = (const struct htp_softmax_kernel_params *) node.kernel_params; + snprintf(str, max_size, "k%d nth %d vtcm %d", (int) kparams->kernel_id, (int) kparams->n_threads, (int) kparams->vtcm_size); + } else if (node.opcode == HTP_OP_GATED_DELTA_NET) { + const auto * kparams = (const struct htp_gdn_kernel_params *) node.kernel_params; + const char * path = (kparams->kernel_type == HTP_GDN_KERNEL_HMX_CHUNKED) ? "hmx-chunked" : "hvx-recurrent"; + snprintf(str, max_size, "%s-%s vtcm %u", + path, + kparams->kda ? "kda" : "scalar", + (unsigned int) (kparams->vtcm_size ? kparams->vtcm_size : kparams->vtcm_per_thread * kparams->n_threads)); + } else if (node.opcode == HTP_OP_MUL || node.opcode == HTP_OP_ADD || node.opcode == HTP_OP_ADD_ID || + node.opcode == HTP_OP_SUB || node.opcode == HTP_OP_DIV) { + const auto * kparams = (const struct htp_binary_kernel_params *) node.kernel_params; + snprintf(str, max_size, "vtcm %u", (unsigned int) kparams->vtcm_size); } else { snprintf(str, max_size, "----"); } diff --git a/ggml/src/ggml-hexagon/htp/CMakeLists.txt b/ggml/src/ggml-hexagon/htp/CMakeLists.txt index b00aa2bc..821f08c0 100644 --- a/ggml/src/ggml-hexagon/htp/CMakeLists.txt +++ b/ggml/src/ggml-hexagon/htp/CMakeLists.txt @@ -43,6 +43,8 @@ add_library(${HTP_LIB} SHARED pad-ops.c argsort-ops.c im2col-ops.c + roll-ops.c + allreduce-ops.c ) target_compile_definitions(${HTP_LIB} PRIVATE diff --git a/ggml/src/ggml-hexagon/htp/act-ops.c b/ggml/src/ggml-hexagon/htp/act-ops.c index 9973c088..d59ac077 100644 --- a/ggml/src/ggml-hexagon/htp/act-ops.c +++ b/ggml/src/ggml-hexagon/htp/act-ops.c @@ -3,19 +3,18 @@ #pragma clang diagnostic ignored "-Wunused-but-set-variable" #include -#include #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "hex-common.h" #include "htp-tensor.h" #include "htp-vtcm.h" @@ -54,13 +53,24 @@ const uint32_t nb2 = dst->nb[2]; \ const uint32_t nb3 = dst->nb[3]; +struct htp_act_context; + +typedef void (*glu_compute_fn_t)(const float * restrict src0, + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx); + struct htp_act_context { struct htp_ops_context * octx; + glu_compute_fn_t compute; + const char * op_str; + // Precomputed values - const uint8_t * data_src0; - const uint8_t * data_src1; - uint8_t * data_dst; + dma_addr_t data_src0; + dma_addr_t data_src1; + dma_addr_t data_dst; size_t src0_row_size; size_t src1_row_size; @@ -80,6 +90,7 @@ struct htp_act_context { uint32_t block; uint32_t src0_nrows; uint32_t src0_nrows_per_thread; + uint32_t row_start; int nc; uint8_t * vtcm_src0; @@ -134,10 +145,10 @@ static inline void htp_act_vtcm_layout_build(struct htp_act_vtcm_layout * L, // swiglu(x) = x1 * sigmoid(x0) static void swiglu_f32(const float * restrict src0, - const float * restrict src1, - float * restrict dst, - const uint32_t num_rows, - const struct htp_act_context * actx) { + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx) { htp_glu_op_preamble; for (uint32_t ib = 0; ib < num_rows; ib++) { @@ -152,10 +163,10 @@ static void swiglu_f32(const float * restrict src0, // out = x * sigmoid(alpha * x) * (clamp(y, -limit, limit) + 1.f) static void swiglu_oai_f32(const float * restrict src0, - const float * restrict src1, - float * restrict dst, - const uint32_t num_rows, - const struct htp_act_context * actx) { + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx) { htp_glu_op_preamble; const float alpha = ((const float *) (actx->octx->op_params))[2]; const float limit = ((const float *) (actx->octx->op_params))[3]; @@ -180,9 +191,76 @@ static void swiglu_oai_f32(const float * restrict src0, } } +static void swiglu_clamp_f32(const float * restrict src0, + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx) { + htp_glu_op_preamble; + const float limit = ((const float *) (actx->octx->op_params))[3]; + + for (uint32_t ib = 0; ib < num_rows; ib++) { + const uint8_t * restrict src0_ptr = (const uint8_t *) src0 + (ib * src0_row_size_aligned); + const uint8_t * restrict src1_ptr = (const uint8_t *) src1 + (ib * src1_row_size_aligned); + uint8_t * restrict dst_ptr = (uint8_t *) dst + (ib * dst_row_size_aligned); + + hvx_min_scalar_f32((uint8_t *) src0_ptr, src0_ptr, limit, nc); + hvx_clamp_scalar_f32((uint8_t *) src1_ptr, src1_ptr, -limit, limit, nc); + hvx_sigmoid_f32_aa(dst_ptr, src0_ptr, nc); + hvx_mul_mul_f32_aa(dst_ptr, src0_ptr, dst_ptr, src1_ptr, nc); + } +} + static const float GELU_COEF_A = 0.044715f; static const float SQRT_2_OVER_PI = 0.79788456080286535587989211986876f; +static inline HVX_Vector hvx_vec_fast_sigmoid_f32_2it(HVX_Vector v) { + v = Q6_Vqf32_vmpy_VsfVsf(v, Q6_V_vsplat_R(FAST_SIGMOID_LOG2F)); + v = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(v), Q6_V_vsplat_R(FAST_SIGMOID_C3)); + + HVX_Vector in_int = hvx_vec_truncate_f32(Q6_Vsf_equals_Vqf32(v)); + HVX_Vector x = Q6_Vqf32_vsub_Vqf32Vsf(v, Q6_Vsf_equals_Vw(in_int)); + HVX_Vector xx = Q6_Vqf32_vmpy_Vqf32Vqf32(x, x); + + HVX_Vector v1 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(xx), Q6_V_vsplat_R(FAST_SIGMOID_C2)); + v1 = Q6_Vqf32_vadd_Vqf32Vsf(v1, Q6_V_vsplat_R(FAST_SIGMOID_LOG2F)); + + HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(x), Q6_V_vsplat_R(FAST_SIGMOID_C1)); + v2 = Q6_Vqf32_vmpy_Vqf32Vqf32(v2, xx); + v2 = Q6_Vqf32_vadd_Vqf32Vqf32(v2, x); + + HVX_Vector v3 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vqf32(v2, v1)); + v3 = Q6_Vw_vaslacc_VwVwR(v3, in_int, 24); + + HVX_Vector v4 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_Vqf32Vqf32(v2, v1)); + HVX_Vector v5 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(v3, v4)); + + // Newton-Raphson with 2 iterations + HVX_Vector two_sf = hvx_vec_splat_f32(2.0f); + HVX_Vector i_sf = Q6_Vw_vsub_VwVw(Q6_V_vsplat_R(0x7EEEEBB3), v5); + HVX_Vector r_qf = Q6_Vqf32_vmpy_VsfVsf( + i_sf, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(two_sf, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(i_sf, v5))))); + r_qf = Q6_Vqf32_vmpy_Vqf32Vqf32( + r_qf, Q6_Vqf32_vsub_VsfVsf(two_sf, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(r_qf), v5)))); + HVX_Vector res = Q6_Vsf_equals_Vqf32(r_qf); + + res = Q6_Vqf32_vmpy_VsfVsf(v3, res); + + return Q6_Vsf_equals_Vqf32(res); +} + +static inline HVX_Vector hvx_vec_fast_sigmoid_f32_guard_2it(HVX_Vector v, + HVX_Vector one, + HVX_Vector max_exp, + HVX_Vector min_exp) { + const HVX_VectorPred pred_max = Q6_Q_vcmp_gt_VsfVsf(max_exp, v); + const HVX_VectorPred pred_min = Q6_Q_vcmp_gt_VsfVsf(v, min_exp); + + HVX_Vector out = hvx_vec_fast_sigmoid_f32_2it(v); + out = Q6_V_vmux_QVV(pred_max, out, one); + return Q6_V_vmux_QVV(pred_min, out, Q6_V_vzero()); +} + static inline void hvx_geglu_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { assert((unsigned long) dst % 128 == 0); assert((unsigned long) src0 % 128 == 0); @@ -200,20 +278,13 @@ static inline void hvx_geglu_f32_aa(uint8_t * restrict dst, const uint8_t * rest const HVX_Vector v_coef_a_times_sqrt = hvx_vec_splat_f32(GELU_COEF_A_TIMES_SQRT); const HVX_Vector v_sqrt_2_pi = hvx_vec_splat_f32(SQRT_2_OVER_PI); - const HVX_Vector v_half = hvx_vec_splat_f32(0.5f); const HVX_Vector v_one = hvx_vec_splat_f32(1.0f); - const HVX_Vector v_two = hvx_vec_splat_f32(2.0f); - - // Hoisted fast sigmoid / inverse constants to avoid loop-internal overhead - const HVX_Vector v_log2f = Q6_V_vsplat_R(FAST_SIGMOID_LOG2F); - const HVX_Vector v_c1 = Q6_V_vsplat_R(FAST_SIGMOID_C1); - const HVX_Vector v_c2 = Q6_V_vsplat_R(FAST_SIGMOID_C2); - const HVX_Vector v_inv_aprox = Q6_V_vsplat_R(0x7EEEEBB3); const HVX_Vector v_max_exp = hvx_vec_splat_f32(87.0f); const HVX_Vector v_min_exp = hvx_vec_splat_f32(-87.0f); uint32_t i = 0; + _Pragma("unroll(4)") for (; i < nvec; i++) { HVX_Vector x = vsrc0[i]; HVX_Vector g = vsrc1[i]; @@ -223,56 +294,13 @@ static inline void hvx_geglu_f32_aa(uint8_t * restrict dst, const uint8_t * rest coef = hvx_vec_add_f32_f32(coef, v_sqrt_2_pi); HVX_Vector inner = hvx_vec_mul_f32_f32(x, coef); - // y2 = 2 * inner - HVX_Vector y2 = hvx_vec_mul_f32_f32(inner, v_two); - - // Sigmoid guard check predicates - HVX_VectorPred pred_max = Q6_Q_vcmp_gt_VsfVsf(v_max_exp, y2); - HVX_VectorPred pred_min = Q6_Q_vcmp_gt_VsfVsf(y2, v_min_exp); - - // Fast sigmoid approximation - HVX_Vector v = Q6_Vqf32_vmpy_VsfVsf(y2, v_log2f); - v = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(v), v_half); - - HVX_Vector in_int = hvx_vec_truncate_f32(Q6_Vsf_equals_Vqf32(v)); - HVX_Vector x_sig = Q6_Vqf32_vsub_Vqf32Vsf(v, Q6_Vsf_equals_Vw(in_int)); - HVX_Vector xx_sig = Q6_Vqf32_vmpy_Vqf32Vqf32(x_sig, x_sig); - - HVX_Vector v1 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(xx_sig), v_c2); - v1 = Q6_Vqf32_vadd_Vqf32Vsf(v1, v_log2f); - - HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(x_sig), v_c1); - v2 = Q6_Vqf32_vmpy_Vqf32Vqf32(v2, xx_sig); - v2 = Q6_Vqf32_vadd_Vqf32Vqf32(v2, x_sig); - - HVX_Vector v3 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vqf32(v2, v1)); - v3 = Q6_Vw_vaslacc_VwVwR(v3, in_int, 24); - - HVX_Vector v4 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_Vqf32Vqf32(v2, v1)); - HVX_Vector v5 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(v3, v4)); - - // Fast division (Newton-Raphson with 2 iterations) - HVX_Vector i_sf = Q6_Vw_vsub_VwVw(v_inv_aprox, v5); - HVX_Vector r_qf = Q6_Vqf32_vmpy_VsfVsf( - i_sf, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(v_two, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(i_sf, v5))))); - r_qf = Q6_Vqf32_vmpy_Vqf32Vqf32( - r_qf, Q6_Vqf32_vsub_VsfVsf(v_two, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(r_qf), v5)))); - HVX_Vector res_inv = Q6_Vsf_equals_Vqf32(r_qf); - - HVX_Vector sig2y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(v3, res_inv)); - - // Sigmoid guards - sig2y = Q6_V_vmux_QVV(pred_max, sig2y, v_one); - sig2y = Q6_V_vmux_QVV(pred_min, sig2y, Q6_V_vzero()); - - // tanh(inner) = 2 * sigmoid(2 * inner) - 1 - HVX_Vector tanh_val = hvx_vec_mul_f32_f32(sig2y, v_two); - tanh_val = hvx_vec_sub_f32_f32(tanh_val, v_one); + // y2 = 2 * inner = inner + inner + HVX_Vector y2 = hvx_vec_add_f32_f32(inner, inner); - HVX_Vector tanh_plus_one = hvx_vec_add_f32_f32(tanh_val, v_one); - HVX_Vector half_x = hvx_vec_mul_f32_f32(x, v_half); - HVX_Vector gelu_x = hvx_vec_mul_f32_f32(half_x, tanh_plus_one); + // Fast sigmoid approximation (2 iterations) + HVX_Vector sig2y = hvx_vec_fast_sigmoid_f32_guard_2it(y2, v_one, v_max_exp, v_min_exp); + HVX_Vector gelu_x = hvx_vec_mul_f32_f32(x, sig2y); vdst[i] = hvx_vec_mul_f32_f32(gelu_x, g); } @@ -285,61 +313,61 @@ static inline void hvx_geglu_f32_aa(uint8_t * restrict dst, const uint8_t * rest coef = hvx_vec_add_f32_f32(coef, v_sqrt_2_pi); HVX_Vector inner = hvx_vec_mul_f32_f32(x, coef); - HVX_Vector y2 = hvx_vec_mul_f32_f32(inner, v_two); + HVX_Vector y2 = hvx_vec_add_f32_f32(inner, inner); - HVX_VectorPred pred_max = Q6_Q_vcmp_gt_VsfVsf(v_max_exp, y2); - HVX_VectorPred pred_min = Q6_Q_vcmp_gt_VsfVsf(y2, v_min_exp); + HVX_Vector sig2y = hvx_vec_fast_sigmoid_f32_guard_2it(y2, v_one, v_max_exp, v_min_exp); - HVX_Vector v = Q6_Vqf32_vmpy_VsfVsf(y2, v_log2f); - v = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(v), v_half); - - HVX_Vector in_int = hvx_vec_truncate_f32(Q6_Vsf_equals_Vqf32(v)); - HVX_Vector x_sig = Q6_Vqf32_vsub_Vqf32Vsf(v, Q6_Vsf_equals_Vw(in_int)); - HVX_Vector xx_sig = Q6_Vqf32_vmpy_Vqf32Vqf32(x_sig, x_sig); - - HVX_Vector v1 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(xx_sig), v_c2); - v1 = Q6_Vqf32_vadd_Vqf32Vsf(v1, v_log2f); - - HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(x_sig), v_c1); - v2 = Q6_Vqf32_vmpy_Vqf32Vqf32(v2, xx_sig); - v2 = Q6_Vqf32_vadd_Vqf32Vqf32(v2, x_sig); - - HVX_Vector v3 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vqf32(v2, v1)); - v3 = Q6_Vw_vaslacc_VwVwR(v3, in_int, 24); + HVX_Vector gelu_x = hvx_vec_mul_f32_f32(x, sig2y); + HVX_Vector res = hvx_vec_mul_f32_f32(gelu_x, g); + hvx_vec_store_a((void *) &vdst[i], nloe * sizeof(float), res); + } +} - HVX_Vector v4 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_Vqf32Vqf32(v2, v1)); - HVX_Vector v5 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(v3, v4)); +static inline void hvx_geglu_quick_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src0 % 128 == 0); + assert((unsigned long) src1 % 128 == 0); - HVX_Vector i_sf = Q6_Vw_vsub_VwVw(v_inv_aprox, v5); - HVX_Vector r_qf = Q6_Vqf32_vmpy_VsfVsf( - i_sf, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(v_two, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(i_sf, v5))))); - r_qf = Q6_Vqf32_vmpy_Vqf32Vqf32( - r_qf, Q6_Vqf32_vsub_VsfVsf(v_two, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(r_qf), v5)))); - HVX_Vector res_inv = Q6_Vsf_equals_Vqf32(r_qf); + HVX_Vector * restrict vdst = (HVX_Vector *) dst; + const HVX_Vector * restrict vsrc0 = (const HVX_Vector *) src0; + const HVX_Vector * restrict vsrc1 = (const HVX_Vector *) src1; - HVX_Vector sig2y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(v3, res_inv)); + const uint32_t epv = 128 / sizeof(float); + const uint32_t nvec = n / epv; + const uint32_t nloe = n % epv; - sig2y = Q6_V_vmux_QVV(pred_max, sig2y, v_one); - sig2y = Q6_V_vmux_QVV(pred_min, sig2y, Q6_V_vzero()); + const HVX_Vector v_scale = hvx_vec_splat_f32(1.702f); + const HVX_Vector v_one = hvx_vec_splat_f32(1.0f); + const HVX_Vector v_max_exp = hvx_vec_splat_f32(87.0f); + const HVX_Vector v_min_exp = hvx_vec_splat_f32(-87.0f); - HVX_Vector tanh_val = hvx_vec_mul_f32_f32(sig2y, v_two); - tanh_val = hvx_vec_sub_f32_f32(tanh_val, v_one); + uint32_t i = 0; - HVX_Vector tanh_plus_one = hvx_vec_add_f32_f32(tanh_val, v_one); - HVX_Vector half_x = hvx_vec_mul_f32_f32(x, v_half); - HVX_Vector gelu_x = hvx_vec_mul_f32_f32(half_x, tanh_plus_one); + _Pragma("unroll(4)") + for (; i < nvec; i++) { + HVX_Vector x = vsrc0[i]; + HVX_Vector g = vsrc1[i]; + HVX_Vector scaled_x = hvx_vec_mul_f32_f32(x, v_scale); + HVX_Vector sigmoid_x = hvx_vec_fast_sigmoid_f32_guard_2it(scaled_x, v_one, v_max_exp, v_min_exp); + vdst[i] = hvx_vec_mul_f32_f32(hvx_vec_mul_f32_f32(x, sigmoid_x), g); + } - HVX_Vector res = hvx_vec_mul_f32_f32(gelu_x, g); - hvx_vec_store_a((void *) &vdst[i], nloe * sizeof(float), res); + if (nloe) { + HVX_Vector x = vsrc0[i]; + HVX_Vector g = vsrc1[i]; + HVX_Vector scaled_x = hvx_vec_mul_f32_f32(x, v_scale); + HVX_Vector sigmoid_x = hvx_vec_fast_sigmoid_f32_guard_2it(scaled_x, v_one, v_max_exp, v_min_exp); + HVX_Vector result = hvx_vec_mul_f32_f32(hvx_vec_mul_f32_f32(x, sigmoid_x), g); + hvx_vec_store_a((void *) &vdst[i], nloe * sizeof(float), result); } } // geglu(x, g) = gelu(x) * g static void geglu_f32(const float * restrict src0, - const float * restrict src1, - float * restrict dst, - const uint32_t num_rows, - const struct htp_act_context * actx) { + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx) { htp_glu_op_preamble; for (uint32_t ib = 0; ib < num_rows; ib++) { @@ -351,109 +379,117 @@ static void geglu_f32(const float * restrict src0, } } -#define DEFINE_GLU_PER_THREAD(NAME, OP_STR, CORE_EXPR) \ - static void glu_##NAME##_f32_per_thread(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_act_context * actx = (struct htp_act_context *) data; \ - htp_act_preamble; \ - \ - struct htp_thread_trace * tr = actx->octx->ctx ? &actx->octx->ctx->trace[ith] : NULL; \ - \ - size_t src0_row_size = actx->src0_row_size; \ - size_t src1_row_size = actx->src1_row_size; \ - size_t dst_row_size = actx->dst_row_size; \ - \ - size_t src0_row_stride = actx->src0_row_stride; \ - size_t src1_row_stride = actx->src1_row_stride; \ - \ - const uint32_t src0_nrows = actx->src0_nrows; \ - const uint32_t src0_nrows_per_thread = actx->src0_nrows_per_thread; \ - \ - const uint32_t src0_start_row = src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \ - \ - /* no work for this thread */ \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - const uint8_t * restrict data_src0 = actx->data_src0; \ - const uint8_t * restrict data_src1 = actx->data_src1; \ - uint8_t * restrict data_dst = actx->data_dst; \ - \ - const size_t src0_row_size_aligned = actx->src0_row_size_aligned; \ - const size_t src1_row_size_aligned = actx->src1_row_size_aligned; \ - const size_t dst_row_size_aligned = actx->dst_row_size_aligned; \ - \ - uint8_t * restrict src0_spad_data = actx->vtcm_src0 + (ith * actx->vtcm_src0_size_per_thread); \ - uint8_t * restrict src1_spad_data = actx->vtcm_src1 + (ith * actx->vtcm_src1_size_per_thread); \ - uint8_t * restrict dst_spad_data = actx->vtcm_dst + (ith * actx->vtcm_dst_size_per_thread); \ - \ - size_t src0_spad_half_size = actx->src0_spad_half_size; \ - size_t src1_spad_half_size = actx->src1_spad_half_size; \ - size_t dst_spad_half_size = actx->dst_spad_half_size; \ - \ - const int BLOCK = actx->block; \ - if (BLOCK == 0) { \ - FARF(ERROR, \ - OP_STR \ - " : current VTCM reservation %zu is too small for even 1 row per thread, needed at least %zu\n", \ - actx->vtcm_src0_size_per_thread, src0_row_size_aligned); \ - return; \ - } \ - \ - dma_queue * dma_queue = actx->octx->ctx->dma[ith]; \ - \ - /* See discussion: https://github.com/ggml-org/llama.cpp/pull/18151#issuecomment-3678235379 */ \ - for (uint32_t ir = src0_start_row, spad_idx = 0; ir < src0_end_row && spad_idx < 2; ir += BLOCK, spad_idx++) { \ - const uint32_t block_size = MIN(BLOCK, src0_end_row - ir); \ - \ - /* Dummy DMA transation for sequencing (interleaving dst,src,dst,...) */ \ - dma_queue_push_vtcm_to_ddr(dma_queue, \ - dma_make_ptr(data_dst, dst_spad_data + (spad_idx * dst_spad_half_size)), \ - dst_row_size, dst_row_size_aligned, 0); \ - \ - dma_queue_push( \ - dma_queue, \ - dma_make_ptr(src0_spad_data + (spad_idx * src0_spad_half_size), data_src0 + (ir * src0_row_stride)), \ - src0_row_size_aligned, src0_row_stride, src0_row_size, block_size); \ - dma_queue_push( \ - dma_queue, \ - dma_make_ptr(src1_spad_data + (spad_idx * src1_spad_half_size), data_src1 + (ir * src1_row_stride)), \ - src1_row_size_aligned, src1_row_stride, src1_row_size, block_size); \ - } \ - \ - for (uint32_t ir = src0_start_row; ir < src0_end_row; ir += BLOCK) { \ - const uint32_t block_size = MIN(BLOCK, src0_end_row - ir); \ - \ - float * dst_spad = (float *) dma_queue_pop(dma_queue).src; \ - float * src0_spad = (float *) dma_queue_pop(dma_queue).dst; \ - float * src1_spad = (float *) dma_queue_pop(dma_queue).dst; \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); \ - CORE_EXPR; \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); \ - \ - dma_queue_push_vtcm_to_ddr(dma_queue, dma_make_ptr(data_dst + (ir * dst_row_size), dst_spad), \ - dst_row_size, dst_row_size_aligned, block_size); \ - \ - /* prefetch N+2 loop iteration if any */ \ - const uint32_t pref_block = (ir + BLOCK * 2); \ - if (pref_block < src0_end_row) { \ - const uint32_t pref_block_size = MIN(BLOCK, src0_end_row - pref_block); \ - dma_queue_push(dma_queue, dma_make_ptr(src0_spad, data_src0 + (pref_block * src0_row_stride)), \ - src0_row_size_aligned, src0_row_stride, src0_row_size, pref_block_size); \ - dma_queue_push(dma_queue, dma_make_ptr(src1_spad, data_src1 + (pref_block * src1_row_stride)), \ - src1_row_size_aligned, src1_row_stride, src1_row_size, pref_block_size); \ - } \ - } \ - \ - dma_queue_flush(dma_queue); \ - \ +// geglu_quick(x, g) = x * sigmoid(1.702 * x) * g +static void geglu_quick_f32(const float * restrict src0, + const float * restrict src1, + float * restrict dst, + const uint32_t num_rows, + const struct htp_act_context * actx) { + htp_glu_op_preamble; + + for (uint32_t ib = 0; ib < num_rows; ib++) { + const uint8_t * restrict src0_ptr = (const uint8_t *) src0 + (ib * src0_row_size_aligned); + const uint8_t * restrict src1_ptr = (const uint8_t *) src1 + (ib * src1_row_size_aligned); + uint8_t * restrict dst_ptr = (uint8_t *) dst + (ib * dst_row_size_aligned); + + hvx_geglu_quick_f32_aa(dst_ptr, src0_ptr, src1_ptr, nc); + } +} + +static void glu_f32_per_thread(unsigned int nth, unsigned int ith, void * data) { + struct htp_act_context * actx = (struct htp_act_context *) data; + htp_act_preamble; + + struct htp_thread_trace * tr = actx->octx->ctx ? &actx->octx->ctx->trace[ith] : NULL; + + size_t src0_row_size = actx->src0_row_size; + size_t src1_row_size = actx->src1_row_size; + size_t dst_row_size = actx->dst_row_size; + + size_t src0_row_stride = actx->src0_row_stride; + size_t src1_row_stride = actx->src1_row_stride; + + const uint32_t src0_nrows = actx->src0_nrows; + const uint32_t src0_nrows_per_thread = actx->src0_nrows_per_thread; + + const uint32_t src0_start_row = actx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, actx->row_start + src0_nrows); + + /* no work for this thread */ + if (src0_start_row >= src0_end_row) { + return; } -DEFINE_GLU_PER_THREAD(swiglu, "swiglu-f32", swiglu_f32(src0_spad, src1_spad, dst_spad, block_size, actx)) -DEFINE_GLU_PER_THREAD(swiglu_oai, "swiglu-oai-f32", swiglu_oai_f32(src0_spad, src1_spad, dst_spad, block_size, actx)) -DEFINE_GLU_PER_THREAD(geglu, "geglu-f32", geglu_f32(src0_spad, src1_spad, dst_spad, block_size, actx)) + const dma_addr_t data_src0 = actx->data_src0; + const dma_addr_t data_src1 = actx->data_src1; + const dma_addr_t data_dst = actx->data_dst; + + const size_t src0_row_size_aligned = actx->src0_row_size_aligned; + const size_t src1_row_size_aligned = actx->src1_row_size_aligned; + const size_t dst_row_size_aligned = actx->dst_row_size_aligned; + + uint8_t * restrict src0_spad_data = actx->vtcm_src0 + (ith * actx->vtcm_src0_size_per_thread); + uint8_t * restrict src1_spad_data = actx->vtcm_src1 + (ith * actx->vtcm_src1_size_per_thread); + uint8_t * restrict dst_spad_data = actx->vtcm_dst + (ith * actx->vtcm_dst_size_per_thread); + + size_t src0_spad_half_size = actx->src0_spad_half_size; + size_t src1_spad_half_size = actx->src1_spad_half_size; + size_t dst_spad_half_size = actx->dst_spad_half_size; + + const int BLOCK = actx->block; + if (BLOCK == 0) { + FARF(ERROR, "%s : VTCM reservation %zu is too small, needed %zu\n", + actx->op_str, actx->vtcm_src0_size_per_thread, src0_row_size_aligned); + return; + } + + dma_queue * dma_q = actx->octx->ctx->dma[ith]; + glu_compute_fn_t compute = actx->compute; + + for (uint32_t ir = src0_start_row, spad_idx = 0; ir < src0_end_row && spad_idx < 2; ir += BLOCK, spad_idx++) { + const uint32_t block_size = MIN(BLOCK, src0_end_row - ir); + + /* Dummy DMA transation for sequencing (interleaving dst,src,dst,...) */ + dma_queue_push(dma_q, + dma_make_data(data_dst, dst_spad_data + (spad_idx * dst_spad_half_size)), + dst_row_size, dst_row_size_aligned, dst_row_size, 0); + + dma_queue_push(dma_q, + dma_make_data(src0_spad_data + (spad_idx * src0_spad_half_size), data_src0 + (ir * src0_row_stride)), + src0_row_size_aligned, src0_row_stride, src0_row_size, block_size); + + dma_queue_push(dma_q, + dma_make_data(src1_spad_data + (spad_idx * src1_spad_half_size), data_src1 + (ir * src1_row_stride)), + src1_row_size_aligned, src1_row_stride, src1_row_size, block_size); + } + + for (uint32_t ir = src0_start_row; ir < src0_end_row; ir += BLOCK) { + const uint32_t block_size = MIN(BLOCK, src0_end_row - ir); + + float * dst_spad = (float *) dma_queue_pop(dma_q).src; + float * src0_spad = (float *) dma_queue_pop(dma_q).dst; + float * src1_spad = (float *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); + compute(src0_spad, src1_spad, dst_spad, block_size, actx); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); + + dma_queue_push(dma_q, dma_make_data(data_dst + (ir * dst_row_size), dst_spad), + dst_row_size, dst_row_size_aligned, dst_row_size, block_size); + + /* prefetch N+2 loop iteration if any */ + const uint32_t pref_block = (ir + BLOCK * 2); + if (pref_block < src0_end_row) { + const uint32_t pref_block_size = MIN(BLOCK, src0_end_row - pref_block); + dma_queue_push(dma_q, dma_make_data(src0_spad, data_src0 + (pref_block * src0_row_stride)), + src0_row_size_aligned, src0_row_stride, src0_row_size, pref_block_size); + dma_queue_push(dma_q, dma_make_data(src1_spad, data_src1 + (pref_block * src1_row_stride)), + src1_row_size_aligned, src1_row_stride, src1_row_size, pref_block_size); + } + } + + dma_queue_flush(dma_q); +} static int execute_op_activations_f32(struct htp_ops_context * octx) { const struct htp_tensor * src0 = octx->src[0]; @@ -465,23 +501,33 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - worker_callback_t act_op_func; - const char * op_type = NULL; + glu_compute_fn_t compute_fn = NULL; + const char * op_type = NULL; switch (octx->op) { case HTP_OP_GLU_SWIGLU: - act_op_func = (worker_callback_t)glu_swiglu_f32_per_thread; - op_type = "swiglu-f32"; + compute_fn = swiglu_f32; + op_type = "swiglu-f32"; break; case HTP_OP_GLU_SWIGLU_OAI: - act_op_func = (worker_callback_t)glu_swiglu_oai_f32_per_thread; - op_type = "swiglu-oai-f32"; + compute_fn = swiglu_oai_f32; + op_type = "swiglu-oai-f32"; + break; + + case HTP_OP_GLU_SWIGLU_CLAMP: + compute_fn = swiglu_clamp_f32; + op_type = "swiglu-clamp-f32"; break; case HTP_OP_GLU_GEGLU: - act_op_func = (worker_callback_t)glu_geglu_f32_per_thread; - op_type = "geglu-f32"; + compute_fn = geglu_f32; + op_type = "geglu-f32"; + break; + + case HTP_OP_GLU_GEGLU_QUICK: + compute_fn = geglu_quick_f32; + op_type = "geglu-quick-f32"; break; default: FARF(ERROR, "Unsupported activations Op %u\n", octx->op); @@ -489,14 +535,30 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) { } const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, src0_nrows); + const size_t dst_row_size = dst->ne[0] * SIZEOF_FP32; + + uint32_t row_start = 0; + uint32_t nrows = src0_nrows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(float), (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } + + const uint32_t n_threads = octx->n_threads; // row_size = bytes of useful data per row (what the kernel touches / what DMA copies). // row_stride = bytes between successive rows in DDR (may exceed row_size for non-contig src). - const size_t nc_bytes = dst->ne[0] * SIZEOF_FP32; - const size_t src0_row_size = nc_bytes; - const size_t src1_row_size = nc_bytes; - const size_t dst_row_size = nc_bytes; + const size_t nc_bytes = dst_row_size; + const size_t src0_row_size = nc_bytes; + const size_t src1_row_size = nc_bytes; const size_t src0_row_stride = src0->nb[1]; const size_t src1_row_stride = src1 ? src1->nb[1] : src0->nb[1]; @@ -526,15 +588,13 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) { L.src0_bytes_per_thread * n_threads, L.src1_bytes_per_thread * n_threads, L.dst_bytes_per_thread * n_threads); } - if ((octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - return HTP_STATUS_OK; - } - // Prepare context struct htp_act_context actx; - actx.octx = octx; + actx.octx = octx; + actx.compute = compute_fn; + actx.op_str = op_type; - actx.src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; + actx.src0_nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); actx.src0_row_size = src0_row_size; actx.src1_row_size = src1_row_size; @@ -561,15 +621,20 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) { actx.dst_spad_half_size = L.dst_bytes_per_thread / 2; actx.block = actx.src0_spad_half_size / actx.src0_row_size_aligned; - actx.src0_nrows = src0_nrows; + actx.src0_nrows = nrows; + actx.row_start = row_start; actx.nc = dst->ne[0]; - // Pointers and GLU logic - const uint8_t * data_src0 = (const uint8_t *) src0->data; - const uint8_t * data_src1 = src1 ? (const uint8_t *) src1->data : NULL; + // Addresses and GLU logic + dma_addr_t data_src0 = src0->data; + dma_addr_t data_src1 = src1 ? src1->data : 0; - if (!src1 && (octx->op == HTP_OP_GLU_SWIGLU || octx->op == HTP_OP_GLU_SWIGLU_OAI || octx->op == HTP_OP_GLU_GEGLU)) { + if (!src1 && (octx->op == HTP_OP_GLU_SWIGLU || + octx->op == HTP_OP_GLU_SWIGLU_OAI || + octx->op == HTP_OP_GLU_SWIGLU_CLAMP || + octx->op == HTP_OP_GLU_GEGLU || + octx->op == HTP_OP_GLU_GEGLU_QUICK)) { const int32_t swapped = octx->op_params[1]; data_src1 = data_src0; actx.src1_row_size = actx.src0_row_size; @@ -584,9 +649,9 @@ static int execute_op_activations_f32(struct htp_ops_context * octx) { actx.data_src0 = data_src0; actx.data_src1 = data_src1; - actx.data_dst = (uint8_t *) dst->data; + actx.data_dst = dst->data; - worker_pool_run_func(octx->ctx->worker_pool, act_op_func, &actx, n_threads); + work_queue_run(octx->ctx->work_queue, (worker_callback_t)glu_f32_per_thread, &actx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/allreduce-ops.c b/ggml/src/ggml-hexagon/htp/allreduce-ops.c new file mode 100644 index 00000000..7b577bef --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/allreduce-ops.c @@ -0,0 +1,457 @@ +#pragma clang diagnostic ignored "-Wunused-variable" +#pragma clang diagnostic ignored "-Wunused-function" +#pragma clang diagnostic ignored "-Wunused-but-set-variable" + +#include +#include +#include +#include +#include + +#define GGML_COMMON_DECL_C +#include "ggml-common.h" +#include "htp-ctx.h" +#include "htp-ops.h" +#include "hvx-utils.h" +#include "htp-tensor.h" +#include "dma-queue.h" +#include "hex-profile.h" +#include "allreduce-ops.h" +#include "htp-fence.h" + +struct htp_allreduce_context { + struct htp_ops_context * octx; + uint32_t n_ranks; + uint32_t n_dsts; + uint32_t nelem; + uint32_t ne0; + uint32_t ne1; + uint32_t row_size_aligned; + uint32_t rank_elem_start; + uint32_t rank_nelem; + uint32_t elems_per_thread; + uint32_t block_elems; + uint32_t vtcm_size_per_thread; + bool is_row_bcast; + uint8_t * src_spad_base[HTP_ALLREDUCE_MAX_RANKS]; + uint8_t * dst_spad_base; + uint8_t * res_spad_base; +}; + +#define DEFINE_ALLREDUCE_THREAD_DMA_1D(SUFFIX, TYPE, HVX_ADD_FN, HAS_ADD) \ +static void allreduce_thread_dma_1d_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_allreduce_context * actx = (struct htp_allreduce_context *) data; \ + struct htp_ops_context * octx = actx->octx; \ + \ + const uint32_t n_ranks = actx->n_ranks; \ + const uint32_t n_dsts = actx->n_dsts; \ + const uint32_t block_elems = actx->block_elems; \ + \ + const uint32_t dr = actx->elems_per_thread; \ + const uint32_t ir0 = actx->rank_elem_start + dr * ith; \ + const uint32_t ir1 = MIN(ir0 + dr, actx->rank_elem_start + actx->rank_nelem); \ + if (ir0 >= ir1) return; \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + \ + uint8_t * src_spad_base[HTP_ALLREDUCE_MAX_RANKS]; \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + src_spad_base[s] = actx->src_spad_base[s] + (ith * actx->vtcm_size_per_thread); \ + } \ + uint8_t * dst_spad_base = actx->dst_spad_base + (ith * actx->vtcm_size_per_thread); \ + uint8_t * res_spad_base = HAS_ADD ? (actx->res_spad_base + (ith * actx->vtcm_size_per_thread)) : NULL; \ + \ + const size_t spad_half = actx->vtcm_size_per_thread / 2; \ + uint32_t ir_prefetch = ir0; \ + int spad_idx = 0; \ + \ + for (int k = 0; k < 2 && ir_prefetch < ir1; k++) { \ + uint32_t cur_elems = MIN(block_elems, ir1 - ir_prefetch); \ + size_t cur_bytes = cur_elems * sizeof(TYPE); \ + uint8_t * d_spad = dst_spad_base + spad_idx * spad_half; \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + dma_addr_t d_ddr = octx->dsts[d]->data + ir_prefetch * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(d_ddr, d_spad), cur_bytes, cur_bytes, cur_bytes, 0); \ + } \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + uint8_t * s_spad = src_spad_base[s] + spad_idx * spad_half; \ + const dma_addr_t s_ddr = octx->src[s]->data + ir_prefetch * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(s_spad, s_ddr), cur_bytes, cur_bytes, cur_bytes, 1); \ + } \ + if (HAS_ADD) { \ + uint8_t * r_spad = res_spad_base + spad_idx * spad_half; \ + const dma_addr_t r_ddr = octx->src[2 * n_ranks]->data + ir_prefetch * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(r_spad, r_ddr), cur_bytes, cur_bytes, cur_bytes, 1); \ + } \ + ir_prefetch += cur_elems; \ + spad_idx ^= 1; \ + } \ + \ + for (uint32_t ir = ir0; ir < ir1; ) { \ + uint32_t cur_elems = MIN(block_elems, ir1 - ir); \ + size_t cur_bytes = cur_elems * sizeof(TYPE); \ + uint8_t * d_spad = NULL; \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + d_spad = (uint8_t *) dma_queue_pop(dma_q).src; \ + } \ + uint8_t * s_spad[HTP_ALLREDUCE_MAX_RANKS]; \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + s_spad[s] = (uint8_t *) dma_queue_pop(dma_q).dst; \ + } \ + uint8_t * r_spad = HAS_ADD ? (uint8_t *) dma_queue_pop(dma_q).dst : NULL; \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); \ + HVX_ADD_FN(d_spad, s_spad[0], s_spad[1], cur_elems); \ + for (uint32_t s = 2; s < n_ranks; s++) { \ + HVX_ADD_FN(d_spad, d_spad, s_spad[s], cur_elems); \ + } \ + if (HAS_ADD) { \ + HVX_ADD_FN(d_spad, d_spad, r_spad, cur_elems); \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + dma_addr_t d_ddr = octx->dsts[d]->data + ir * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(d_ddr, d_spad), cur_bytes, cur_bytes, cur_bytes, 1); \ + } \ + if (ir_prefetch < ir1) { \ + uint32_t next_elems = MIN(block_elems, ir1 - ir_prefetch); \ + size_t next_bytes = next_elems * sizeof(TYPE); \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + const dma_addr_t s_next = octx->src[s]->data + ir_prefetch * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(s_spad[s], s_next), next_bytes, next_bytes, next_bytes, 1); \ + } \ + if (HAS_ADD) { \ + const dma_addr_t r_next = octx->src[2 * n_ranks]->data + ir_prefetch * sizeof(TYPE); \ + dma_queue_push(dma_q, dma_make_data(r_spad, r_next), next_bytes, next_bytes, next_bytes, 1); \ + } \ + ir_prefetch += next_elems; \ + } \ + ir += cur_elems; \ + } \ + dma_queue_flush(dma_q); \ +} + +DEFINE_ALLREDUCE_THREAD_DMA_1D(f16, __fp16, hvx_add_f16_aaa, 0) +DEFINE_ALLREDUCE_THREAD_DMA_1D(f32, float, hvx_add_f32_aaa, 0) +DEFINE_ALLREDUCE_THREAD_DMA_1D(add_f16, __fp16, hvx_add_f16_aaa, 1) +DEFINE_ALLREDUCE_THREAD_DMA_1D(add_f32, float, hvx_add_f32_aaa, 1) + +#define DEFINE_ALLREDUCE_THREAD_DMA_2D(SUFFIX, TYPE, HVX_ADD_FN, HAS_ADD, IS_ROW_BCAST) \ +static void allreduce_thread_dma_2d_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_allreduce_context * actx = (struct htp_allreduce_context *) data; \ + struct htp_ops_context * octx = actx->octx; \ + \ + const uint32_t n_ranks = actx->n_ranks; \ + const uint32_t n_dsts = actx->n_dsts; \ + const uint32_t ne0 = actx->ne0; \ + const uint32_t block_rows = actx->block_elems; \ + const uint32_t row_size_aligned = actx->row_size_aligned; \ + const uint32_t row_bytes = ne0 * sizeof(TYPE); \ + \ + const uint32_t dr = actx->elems_per_thread; \ + const uint32_t r0 = actx->rank_elem_start + dr * ith; \ + const uint32_t r1 = MIN(r0 + dr, actx->rank_elem_start + actx->rank_nelem); \ + if (r0 >= r1) return; \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + \ + uint8_t * src_spad_base[HTP_ALLREDUCE_MAX_RANKS]; \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + src_spad_base[s] = actx->src_spad_base[s] + (ith * actx->vtcm_size_per_thread); \ + } \ + uint8_t * dst_spad_base = actx->dst_spad_base + (ith * actx->vtcm_size_per_thread); \ + uint8_t * res_spad_base = HAS_ADD ? (IS_ROW_BCAST ? actx->res_spad_base : (actx->res_spad_base + (ith * actx->vtcm_size_per_thread))) : NULL; \ + \ + const size_t spad_half = actx->vtcm_size_per_thread / 2; \ + uint32_t r_prefetch = r0; \ + int spad_idx = 0; \ + \ + for (int k = 0; k < 2 && r_prefetch < r1; k++) { \ + uint32_t cur_rows = MIN(block_rows, r1 - r_prefetch); \ + uint8_t * d_spad = dst_spad_base + spad_idx * spad_half; \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + dma_addr_t d_ddr = octx->dsts[d]->data + r_prefetch * octx->dsts[d]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(d_ddr, d_spad), octx->dsts[d]->nb[1], row_size_aligned, row_bytes, 0); \ + } \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + uint8_t * s_spad = src_spad_base[s] + spad_idx * spad_half; \ + const dma_addr_t s_ddr = octx->src[s]->data + r_prefetch * octx->src[s]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(s_spad, s_ddr), row_size_aligned, octx->src[s]->nb[1], row_bytes, cur_rows); \ + } \ + if (HAS_ADD && !IS_ROW_BCAST) { \ + uint8_t * r_spad = res_spad_base + spad_idx * spad_half; \ + const dma_addr_t r_ddr = octx->src[2 * n_ranks]->data + r_prefetch * octx->src[2 * n_ranks]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(r_spad, r_ddr), row_size_aligned, octx->src[2 * n_ranks]->nb[1], row_bytes, cur_rows); \ + } \ + r_prefetch += cur_rows; \ + spad_idx ^= 1; \ + } \ + \ + for (uint32_t r = r0; r < r1; ) { \ + uint32_t cur_rows = MIN(block_rows, r1 - r); \ + uint8_t * d_spad = NULL; \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + d_spad = (uint8_t *) dma_queue_pop(dma_q).src; \ + } \ + uint8_t * s_spad[HTP_ALLREDUCE_MAX_RANKS]; \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + s_spad[s] = (uint8_t *) dma_queue_pop(dma_q).dst; \ + } \ + uint8_t * r_spad = (HAS_ADD && !IS_ROW_BCAST) ? (uint8_t *) dma_queue_pop(dma_q).dst : NULL; \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) r); \ + for (uint32_t row = 0; row < cur_rows; row++) { \ + uint8_t * d_row = d_spad + row * row_size_aligned; \ + const uint8_t * s0_row = s_spad[0] + row * row_size_aligned; \ + const uint8_t * s1_row = s_spad[1] + row * row_size_aligned; \ + HVX_ADD_FN(d_row, s0_row, s1_row, ne0); \ + for (uint32_t s = 2; s < n_ranks; s++) { \ + const uint8_t * ss_row = s_spad[s] + row * row_size_aligned; \ + HVX_ADD_FN(d_row, d_row, ss_row, ne0); \ + } \ + if (HAS_ADD) { \ + const uint8_t * res_row = IS_ROW_BCAST ? res_spad_base : (r_spad + row * row_size_aligned); \ + HVX_ADD_FN(d_row, d_row, res_row, ne0); \ + } \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) r); \ + for (uint32_t d = 0; d < n_dsts; d++) { \ + dma_addr_t d_ddr = octx->dsts[d]->data + r * octx->dsts[d]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(d_ddr, d_spad), octx->dsts[d]->nb[1], row_size_aligned, row_bytes, cur_rows); \ + } \ + if (r_prefetch < r1) { \ + uint32_t next_rows = MIN(block_rows, r1 - r_prefetch); \ + for (uint32_t s = 0; s < n_ranks; s++) { \ + const dma_addr_t s_next = octx->src[s]->data + r_prefetch * octx->src[s]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(s_spad[s], s_next), row_size_aligned, octx->src[s]->nb[1], row_bytes, next_rows); \ + } \ + if (HAS_ADD && !IS_ROW_BCAST) { \ + const dma_addr_t r_next = octx->src[2 * n_ranks]->data + r_prefetch * octx->src[2 * n_ranks]->nb[1]; \ + dma_queue_push(dma_q, dma_make_data(r_spad, r_next), row_size_aligned, octx->src[2 * n_ranks]->nb[1], row_bytes, next_rows); \ + } \ + r_prefetch += next_rows; \ + } \ + r += cur_rows; \ + } \ + dma_queue_flush(dma_q); \ +} + +DEFINE_ALLREDUCE_THREAD_DMA_2D(f16, __fp16, hvx_add_f16_aaa, 0, 0) +DEFINE_ALLREDUCE_THREAD_DMA_2D(f32, float, hvx_add_f32_aaa, 0, 0) +DEFINE_ALLREDUCE_THREAD_DMA_2D(add_f16, __fp16, hvx_add_f16_aaa, 1, 0) +DEFINE_ALLREDUCE_THREAD_DMA_2D(add_f32, float, hvx_add_f32_aaa, 1, 0) +DEFINE_ALLREDUCE_THREAD_DMA_2D(add_bcast_f16, __fp16, hvx_add_f16_aaa, 1, 1) +DEFINE_ALLREDUCE_THREAD_DMA_2D(add_bcast_f32, float, hvx_add_f32_aaa, 1, 1) + +static int validate_allreduce( + struct htp_ops_context * octx, + const struct htp_allreduce_kernel_params * kparams, + uint32_t n_ranks +) { + if (!htp_ops_context_set_n_threads(octx, (uint32_t) kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } + + if (kparams->vtcm_size_per_thread <= 0 || kparams->vtcm_size <= 0) { + return HTP_STATUS_INVAL_PARAMS; + } + + const bool has_add = (octx->op == HTP_OP_ALLREDUCE_ADD); + const size_t n_vtcm_buffers = htp_allreduce_vtcm_buffer_count( + n_ranks, octx->n_threads, has_add, kparams->is_row_bcast != 0); + const size_t vtcm_size = n_vtcm_buffers * (size_t) kparams->vtcm_size_per_thread; + if (vtcm_size != (size_t) kparams->vtcm_size) { + return HTP_STATUS_INVAL_PARAMS; + } + if (vtcm_size > octx->ctx->vtcm_size) { + return HTP_STATUS_VTCM_TOO_SMALL; + } + + if (octx->dst->type != HTP_TYPE_F16 && octx->dst->type != HTP_TYPE_F32) { + return HTP_STATUS_NO_SUPPORT; + } + + return HTP_STATUS_OK; +} + +int op_allreduce(struct htp_ops_context * octx) { + if (octx->ctx->mdev.count > 1 && octx->ctx->mdev.idx > 0) { + return HTP_STATUS_OK; + } + + const struct htp_allreduce_kernel_params * kparams = (const struct htp_allreduce_kernel_params *) octx->kernel_params; + const struct htp_tensor * dst = octx->dst; + + const uint32_t rank = (uint32_t) kparams->rank; + const uint32_t n_ranks = (uint32_t) kparams->n_ranks; + + if (n_ranks < 2 || n_ranks > HTP_ALLREDUCE_MAX_RANKS || rank >= n_ranks) { + return HTP_STATUS_INVAL_PARAMS; + } + + const uint32_t fence_seq_entry = (uint32_t) octx->op_params[0]; + const uint32_t fence_seq_exit = (uint32_t) octx->op_params[1]; + + const struct htp_tensor * my_sync = octx->src[n_ranks + rank]; + atomic_uint * my_fence = (atomic_uint *) (uintptr_t) my_sync->data; + + const int status = validate_allreduce(octx, kparams, n_ranks); + if (status != HTP_STATUS_OK) { + if (status == HTP_STATUS_NO_SUPPORT) { + FARF(ERROR, "ggml-hex: allreduce unsupported type %d : rank %u\n", dst->type, rank); + } + htp_fence_write(my_fence, fence_seq_exit, status); + return status; + } + + const bool has_add = (octx->op == HTP_OP_ALLREDUCE_ADD); + const uint32_t nelem = dst->ne[0] * dst->ne[1] * dst->ne[2] * dst->ne[3]; + + // 1. Entry Barrier: Synchronize all ranks before reading + struct htp_thread_trace * tr0 = &octx->ctx->trace[0]; + htp_trace_event_start(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_entry); + + htp_fence_write(my_fence, fence_seq_entry, octx->status); + + for (uint32_t j = 0; j < n_ranks; j++) { + if (j == rank) continue; + const struct htp_tensor * peer_sync = octx->src[n_ranks + j]; + atomic_uint * peer_fence = (atomic_uint *) (uintptr_t) peer_sync->data; + uint64_t spins = 0; + while (1) { + uint32_t peer_seq; + uint32_t peer_status; + htp_fence_read(peer_fence, &peer_seq, &peer_status); + if ((int32_t)(peer_seq - fence_seq_entry) >= 0) { + if (peer_status > HTP_STATUS_OK) { + FARF(ERROR, "ggml-hex: allreduce entry peer %u failed with status %u\n", j, peer_status); + htp_fence_write(my_fence, fence_seq_exit, peer_status); + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_entry); + return peer_status; + } + break; + } + if (++spins > HTP_FENCE_TIMEOUT) { + FARF(ERROR, "ggml-hex: allreduce entry fence-wait TIMEOUT : rank %u waiting on %u fence %p seq 0x%x peer-seq 0x%x\n", + rank, j, peer_fence, fence_seq_entry, peer_seq); + htp_fence_write(my_fence, fence_seq_exit, HTP_STATUS_INTERNAL_ERR); + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_entry); + return HTP_STATUS_INTERNAL_ERR; + } + hex_pause(); + } + } + asm volatile ("syncht" : : : "memory"); + + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_entry); + + // 2. Multi-threaded Reduction across assigned rank chunk + if (nelem > 0) { + const uint32_t n_threads = (uint32_t) kparams->n_threads; + const uint32_t block_elems = (uint32_t) kparams->block_elems; + const uint32_t elems_per_thread = (uint32_t) kparams->elems_per_thread; + const uint32_t vtcm_size_per_thread = (uint32_t) kparams->vtcm_size_per_thread; + + struct htp_allreduce_context actx; + actx.octx = octx; + actx.n_ranks = n_ranks; + actx.n_dsts = (uint32_t) kparams->n_dsts ? (uint32_t) kparams->n_dsts : n_ranks; + actx.nelem = nelem; + actx.ne0 = (uint32_t) kparams->ne0; + actx.ne1 = (uint32_t) kparams->ne1; + actx.row_size_aligned = (uint32_t) kparams->row_size_aligned; + actx.rank_elem_start = (uint32_t) kparams->rank_elem_start; + actx.rank_nelem = (uint32_t) kparams->rank_nelem; + actx.elems_per_thread = elems_per_thread; + actx.block_elems = block_elems; + actx.vtcm_size_per_thread = vtcm_size_per_thread; + actx.is_row_bcast = (kparams->is_row_bcast != 0); + + work_queue_func_t reduce_fun = NULL; + switch (kparams->kernel_type) { + case HTP_ALLREDUCE_KERNEL_DMA_1D: + if (has_add) { + reduce_fun = (dst->type == HTP_TYPE_F16) ? allreduce_thread_dma_1d_add_f16 : allreduce_thread_dma_1d_add_f32; + } else { + reduce_fun = (dst->type == HTP_TYPE_F16) ? allreduce_thread_dma_1d_f16 : allreduce_thread_dma_1d_f32; + } + break; + case HTP_ALLREDUCE_KERNEL_DMA_2D: + if (has_add) { + if (kparams->is_row_bcast) { + reduce_fun = (dst->type == HTP_TYPE_F16) ? allreduce_thread_dma_2d_add_bcast_f16 : allreduce_thread_dma_2d_add_bcast_f32; + } else { + reduce_fun = (dst->type == HTP_TYPE_F16) ? allreduce_thread_dma_2d_add_f16 : allreduce_thread_dma_2d_add_f32; + } + } else { + reduce_fun = (dst->type == HTP_TYPE_F16) ? allreduce_thread_dma_2d_f16 : allreduce_thread_dma_2d_f32; + } + break; + default: + FARF(ERROR, "ggml-hex: allreduce unsupported kernel %d : rank %u\n", kparams->kernel_type, rank); + htp_fence_write(my_fence, fence_seq_exit, HTP_STATUS_NO_SUPPORT); + return HTP_STATUS_NO_SUPPORT; + } + + uint8_t * vtcm_ptr = (uint8_t *) octx->ctx->vtcm_base; + for (uint32_t s = 0; s < n_ranks; s++) { + actx.src_spad_base[s] = vtcm_ptr; + vtcm_ptr += n_threads * vtcm_size_per_thread; + } + actx.dst_spad_base = vtcm_ptr; + vtcm_ptr += n_threads * vtcm_size_per_thread; + if (has_add) { + actx.res_spad_base = vtcm_ptr; + vtcm_ptr += (actx.is_row_bcast ? 1 : n_threads) * vtcm_size_per_thread; + } + + if (has_add && actx.is_row_bcast) { + const dma_addr_t r_ddr = octx->src[2 * n_ranks]->data; + const uint32_t row_bytes = actx.ne0 * (dst->type == HTP_TYPE_F16 ? sizeof(__fp16) : sizeof(float)); + dma_queue * dma_q = octx->ctx->dma[0]; + dma_queue_push(dma_q, dma_make_data(actx.res_spad_base, r_ddr), actx.row_size_aligned, 0, row_bytes, 1); + dma_queue_pop(dma_q); + } + + work_queue_run(octx->ctx->work_queue, reduce_fun, &actx, n_threads); + } + + // 4. Exit Barrier: Synchronize all ranks after writing + htp_trace_event_start(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_exit); + + htp_fence_write(my_fence, fence_seq_exit, octx->status); + + for (uint32_t j = 0; j < n_ranks; j++) { + if (j == rank) continue; + const struct htp_tensor * peer_sync = octx->src[n_ranks + j]; + atomic_uint * peer_fence = (atomic_uint *) (uintptr_t) peer_sync->data; + uint64_t spins = 0; + while (1) { + uint32_t peer_seq; + uint32_t peer_status; + htp_fence_read(peer_fence, &peer_seq, &peer_status); + if ((int32_t)(peer_seq - fence_seq_exit) >= 0) { + if (peer_status > HTP_STATUS_OK) { + FARF(ERROR, "ggml-hex: allreduce exit peer %u failed with status %u\n", j, peer_status); + htp_fence_write(my_fence, fence_seq_exit, peer_status); + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_exit); + return peer_status; + } + break; + } + if (++spins > HTP_FENCE_TIMEOUT) { + FARF(ERROR, "ggml-hex: allreduce exit fence-wait TIMEOUT : rank %u waiting on %u fence %p seq 0x%x peer-seq 0x%x\n", + rank, j, peer_fence, fence_seq_exit, peer_seq); + htp_fence_write(my_fence, fence_seq_exit, HTP_STATUS_INTERNAL_ERR); + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_exit); + return HTP_STATUS_INTERNAL_ERR; + } + hex_pause(); + } + } + asm volatile ("syncht" : : : "memory"); + + htp_trace_event_stop(tr0, HTP_TRACE_EVT_FENCE, (uint16_t) fence_seq_exit); + + return octx->status; +} diff --git a/ggml/src/ggml-hexagon/htp/allreduce-ops.h b/ggml/src/ggml-hexagon/htp/allreduce-ops.h new file mode 100644 index 00000000..0aed2b8b --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/allreduce-ops.h @@ -0,0 +1,51 @@ +#ifndef ALLREDUCE_OPS_H +#define ALLREDUCE_OPS_H + +#include +#include +#include + +#define HTP_ALLREDUCE_MAX_RANKS 4 + +#ifdef __cplusplus +extern "C" { +#endif + +enum htp_allreduce_kernel_type { + HTP_ALLREDUCE_KERNEL_UNSUPPORTED = 0, + HTP_ALLREDUCE_KERNEL_DMA_1D, + HTP_ALLREDUCE_KERNEL_DMA_2D, +}; + +static inline size_t htp_allreduce_vtcm_buffer_count( + uint32_t n_ranks, + uint32_t n_threads, + bool has_add, + bool is_row_bcast +) { + return (size_t) (n_ranks + 1) * n_threads + (has_add ? (is_row_bcast ? 1 : n_threads) : 0); +} + +struct htp_allreduce_kernel_params { + int32_t rank; + int32_t n_ranks; + int32_t n_threads; + int32_t block_elems; // 1D: block_elems, 2D: block_rows + int32_t elems_per_thread; // 1D: nelem_per_thread, 2D: nrows_per_thread + int32_t vtcm_size_per_thread; + int32_t vtcm_size; + int32_t kernel_type; + int32_t ne0; + int32_t ne1; + int32_t row_size_aligned; + int32_t rank_elem_start; + int32_t rank_nelem; + int32_t n_dsts; + int32_t is_row_bcast; +}; + +#ifdef __cplusplus +} +#endif + +#endif /* ALLREDUCE_OPS_H */ diff --git a/ggml/src/ggml-hexagon/htp/argsort-ops.c b/ggml/src/ggml-hexagon/htp/argsort-ops.c index 774faef5..9e9e7465 100644 --- a/ggml/src/ggml-hexagon/htp/argsort-ops.c +++ b/ggml/src/ggml-hexagon/htp/argsort-ops.c @@ -9,11 +9,12 @@ #include "ggml.h" #include "hvx-utils.h" -#include "hex-dma.h" +#include "dma-queue.h" +#include "hex-common.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "htp-tensor.h" #ifndef MIN #define MIN(a, b) ((a) < (b) ? (a) : (b)) @@ -22,6 +23,9 @@ struct htp_argsort_context { struct htp_ops_context * octx; uint32_t nrows_per_thread; + uint32_t total_rows; + uint32_t row_start; + uint32_t row_end; uint8_t * vtcm_base; size_t vtcm_per_thread; }; @@ -166,6 +170,21 @@ static void quicksort_values_indices_desc(float * values, int32_t * indices, int if (i < right) quicksort_values_indices_desc(values, indices, i, right); } +static uint32_t top_k_max_value_index(const float * values, uint32_t n, float * value) { + uint32_t index = 0; + float max_value = values[0]; + + for (uint32_t i = 1; i < n; i++) { + if (values[i] > max_value) { + max_value = values[i]; + index = i; + } + } + + *value = max_value; + return index; +} + // LUT for ramp initialization of argsort output (first 32 members) int32_t argosrt_ramp_lut[32] __attribute__((aligned(VLEN))) = { 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, @@ -299,6 +318,129 @@ static inline void bitonic_sort_generic_hvx(uint8_t * values, uint8_t * indices, } } + +// Sorts descending; values pre-padded with -INFINITY. init_indices=true resets +// indices to a fresh ramp (normal full-row sort); false leaves caller-supplied +// indices in place and only permutes them (used when merging candidates, to +// preserve their original global index). + +static void bitonic_sort_vtcm_desc(uint8_t * values, uint8_t * indices, uint32_t n_vec, bool init_indices) { + HVX_Vector zero_vec = Q6_V_vzero(); + HVX_Vector idx_vec = *(HVX_Vector *)argosrt_ramp_lut; + + HVX_VectorPred pred_all_1s = Q6_Q_vcmp_eq_VwVw(zero_vec, zero_vec); + HVX_VectorPred pred_all_0s = Q6_Q_not_Q(pred_all_1s); + + if (init_indices) { + // Initialize indices ramp (values are already populated by the caller) + for (uint32_t v = 0; v < n_vec; v++) { + HVX_Vector idx = Q6_Vw_vadd_VwVw(idx_vec, Q6_V_vsplat_R(v * 32)); + *(HVX_Vector *)(indices + v * 128) = idx; + } + } + + int M = 5; + while ((1u << (M - 5)) < n_vec) M++; + + for (int s = 1; s <= M; s++) { + for (int stage_d = s - 1; stage_d >= 0; stage_d--) { + int d = 1 << stage_d; + if (d >= 32) { + uint32_t v_dist = d / 32; + for (uint32_t v1 = 0; v1 < n_vec; v1++) { + if ((v1 & v_dist) == 0) { + uint32_t v2 = v1 + v_dist; + bool asc = (s < M) ? ((((v1 * 32) >> s) % 2) == 0) : false; + + HVX_Vector Vv1 = *(HVX_Vector *)(values + v1 * 128); + HVX_Vector Iv1 = *(HVX_Vector *)(indices + v1 * 128); + HVX_Vector Vv2 = *(HVX_Vector *)(values + v2 * 128); + HVX_Vector Iv2 = *(HVX_Vector *)(indices + v2 * 128); + + vec_cas(&Vv1, &Iv1, &Vv2, &Iv2, asc); + + *(HVX_Vector *)(values + v1 * 128) = Vv1; + *(HVX_Vector *)(indices + v1 * 128) = Iv1; + *(HVX_Vector *)(values + v2 * 128) = Vv2; + *(HVX_Vector *)(indices + v2 * 128) = Iv2; + } + } + } else { + if (s < 5) { + HVX_VectorPred dir_mask = Q6_Q_vcmp_eq_VwVw(Q6_V_vand_VV(idx_vec, Q6_V_vsplat_R(1 << s)), zero_vec); + for (uint32_t v = 0; v < n_vec; v++) { + HVX_Vector Vv = *(HVX_Vector *)(values + v * 128); + HVX_Vector Iv = *(HVX_Vector *)(indices + v * 128); + + bitonic_cas_32(&Vv, &Iv, d, dir_mask, idx_vec, zero_vec); + + *(HVX_Vector *)(values + v * 128) = Vv; + *(HVX_Vector *)(indices + v * 128) = Iv; + } + } else { + for (uint32_t v = 0; v < n_vec; v++) { + bool asc = (s < M) ? ((((v * 32) >> s) % 2) == 0) : false; + HVX_VectorPred dir_mask = asc ? pred_all_1s : pred_all_0s; + + HVX_Vector Vv = *(HVX_Vector *)(values + v * 128); + HVX_Vector Iv = *(HVX_Vector *)(indices + v * 128); + + bitonic_cas_32(&Vv, &Iv, d, dir_mask, idx_vec, zero_vec); + + *(HVX_Vector *)(values + v * 128) = Vv; + *(HVX_Vector *)(indices + v * 128) = Iv; + } + } + } + } + } +} + +static void top_k_select_tiled(const uint8_t * src, uint32_t n, uint32_t k, + float * values_buf, int32_t * indices_buf, + float * top_values, int32_t * top_indices) { + const uint32_t tile_elems = 1024; + uint32_t n_tiles = (n + tile_elems - 1) / tile_elems; + uint32_t candidate_count = n_tiles * k; + uint32_t merge_n_vec = hmx_ceil_div(candidate_count, 32); + uint32_t merge_n_vec_pow2 = 1; + while (merge_n_vec_pow2 < merge_n_vec) merge_n_vec_pow2 <<= 1; + uint32_t merge_elems = merge_n_vec_pow2 * 32; + float * candidate_values = values_buf + tile_elems; + int32_t * candidate_indices = indices_buf + tile_elems; + uint32_t candidate_pos = 0; + + for (uint32_t offset = 0; offset < n; offset += tile_elems) { + uint32_t tile_count = MIN(tile_elems, n - offset); + hvx_copy_f32_au((uint8_t *) values_buf, src + offset * sizeof(float), tile_count); + if (tile_count < tile_elems) { + hvx_splat_f32_u((uint8_t *) (values_buf + tile_count), -INFINITY, tile_elems - tile_count); + } + + bitonic_sort_vtcm_desc((uint8_t *) values_buf, (uint8_t *) indices_buf, tile_elems / 32, true); + uint32_t tile_k = MIN(k, tile_count); + for (uint32_t j = 0; j < tile_k; j++) { + candidate_values[candidate_pos] = values_buf[j]; + candidate_indices[candidate_pos] = indices_buf[j] + (int32_t) offset; + candidate_pos++; + } + } + + if (merge_elems > candidate_pos) { + hvx_splat_f32_u((uint8_t *) (candidate_values + candidate_pos), -INFINITY, merge_elems - candidate_pos); + for (uint32_t j = candidate_pos; j < merge_elems; j++) { + candidate_indices[j] = 0; + } + } + + bitonic_sort_vtcm_desc((uint8_t *) candidate_values, (uint8_t *) candidate_indices, merge_n_vec_pow2, false); + + for (uint32_t j = 0; j < k; j++) { + top_values[j] = candidate_values[j]; + top_indices[j] = candidate_indices[j]; + } +} + __attribute__((always_inline)) static inline void sort32_f32_hvx(uint8_t * values, uint8_t * indices, enum ggml_sort_order order) { bitonic_sort_generic_hvx(values, indices, 1, order == GGML_SORT_ORDER_ASC); @@ -336,10 +478,9 @@ static void htp_argsort_f32_##ne00##_##order_name(unsigned int n, unsigned int i const struct htp_tensor * src0 = octx->src[0]; \ const struct htp_tensor * dst = octx->dst; \ uint8_t * spad = actx->vtcm_base + actx->vtcm_per_thread * i; \ - uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; \ uint32_t rows_per_thread = actx->nrows_per_thread; \ - uint32_t start_row = rows_per_thread * i; \ - uint32_t end_row = MIN(start_row + rows_per_thread, total_rows); \ + uint32_t start_row = actx->row_start + rows_per_thread * i; \ + uint32_t end_row = MIN(start_row + rows_per_thread, actx->row_end); \ size_t values_size = hex_round_up(ne00 * sizeof(float), 128); \ float * values_buf = (float *) spad; \ int32_t * indices_buf = (int32_t *) (spad + values_size); \ @@ -386,9 +527,6 @@ static void htp_argsort_f32_fallback(unsigned int n, unsigned int i, void * data // Dimensions uint32_t ne00 = src0->ne[0]; - uint32_t ne01 = src0->ne[1]; - uint32_t ne02 = src0->ne[2]; - uint32_t ne03 = src0->ne[3]; uint32_t nb01 = src0->nb[1]; @@ -398,10 +536,9 @@ static void htp_argsort_f32_fallback(unsigned int n, unsigned int i, void * data enum ggml_sort_order order = (enum ggml_sort_order) octx->op_params[0]; // Rows to process - uint32_t total_rows = ne01 * ne02 * ne03; uint32_t rows_per_thread = actx->nrows_per_thread; - uint32_t start_row = rows_per_thread * i; - uint32_t end_row = MIN(start_row + rows_per_thread, total_rows); + uint32_t start_row = actx->row_start + rows_per_thread * i; + uint32_t end_row = MIN(start_row + rows_per_thread, actx->row_end); size_t values_size = hex_round_up(ne00 * sizeof(float), 128); uint32_t num_vec_ind_values = hmx_ceil_div(ne00, VLEN/(sizeof(int32_t))); @@ -451,8 +588,32 @@ int op_argsort(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - const uint32_t total_rows = octx->src[0]->ne[1] * octx->src[0]->ne[2] * octx->src[0]->ne[3]; - const uint32_t n_threads = MIN(total_rows, octx->n_threads); + const struct htp_tensor * src0 = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + const uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const size_t dst_row_size = dst->ne[0] * sizeof(int32_t); + + uint32_t row_start = 0; + uint32_t row_end = total_rows; + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(int32_t), (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + row_end = range.start + range.count; + } + + const uint32_t nrows = row_end - row_start; + if (nrows == 0) { + return HTP_STATUS_OK; + } + + const uint32_t n_threads = octx->n_threads; // Allocate scratchpad // We need 1 row of float + 1 row of int32 per thread. @@ -478,7 +639,10 @@ int op_argsort(struct htp_ops_context * octx) { struct htp_argsort_context actx; actx.octx = octx; - actx.nrows_per_thread = (total_rows + n_threads - 1) / n_threads; + actx.nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); + actx.total_rows = nrows; + actx.row_start = row_start; + actx.row_end = row_end; actx.vtcm_base = (uint8_t *) octx->ctx->vtcm_base; actx.vtcm_per_thread = spad_per_thread; @@ -508,7 +672,424 @@ int op_argsort(struct htp_ops_context * octx) { } // Run jobs - worker_pool_run_func(octx->ctx->worker_pool, job_func, &actx, n_threads); + work_queue_run(octx->ctx->work_queue, job_func, &actx, n_threads); + + return HTP_STATUS_OK; +} + +// ggml_compute_forward_top_k +// +// Reuses ARGSORT's sort kernels. Only the first `k` indices are copied +// to dst, and there's no asc/desc param -- always largest-first. + +struct htp_top_k_context { + struct htp_ops_context * octx; + uint32_t nrows_per_thread; + uint32_t row_start; + uint32_t row_end; + uint8_t * vtcm_base; + size_t vtcm_per_thread; + uint32_t k; +}; + +#define HTP_TOP_K_FN(ne00, sort_fn) \ +static void htp_top_k_f32_##ne00(unsigned int n, unsigned int i, void * data) { \ + struct htp_top_k_context * actx = (struct htp_top_k_context *)data; \ + struct htp_ops_context * octx = actx->octx; \ + const struct htp_tensor * src0 = octx->src[0]; \ + const struct htp_tensor * dst = octx->dst; \ + uint8_t * spad = actx->vtcm_base + actx->vtcm_per_thread * i; \ + uint32_t row_start = actx->row_start; \ + uint32_t row_end = actx->row_end; \ + uint32_t rows_per_thread = actx->nrows_per_thread; \ + uint32_t start_row = row_start + rows_per_thread * i; \ + uint32_t end_row = MIN(start_row + rows_per_thread, row_end); \ + size_t values_size = hex_round_up(ne00 * sizeof(float), 128); \ + float * values_buf = (float *) spad; \ + int32_t * indices_buf = (int32_t *) (spad + values_size); \ + uint32_t nb01 = src0->nb[1]; \ + uint32_t nb1 = dst->nb[1]; \ + uint32_t k = actx->k; \ + struct htp_thread_trace * tr = &octx->ctx->trace[i]; \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, start_row); \ + for (uint32_t r = start_row; r < end_row; r++) { \ + uint32_t src_offset = r * nb01; \ + uint32_t dst_offset = r * nb1; \ + uint8_t * src_ptr = (uint8_t *) src0->data + src_offset; \ + uint8_t * dst_ptr = (uint8_t *) dst->data + dst_offset; \ + hex_l2fetch(src_ptr, ne00 * sizeof(float), ne00 * sizeof(float), 1); \ + hvx_copy_f32_au((uint8_t*)values_buf, src_ptr, ne00); \ + sort_fn((uint8_t*)values_buf, (uint8_t*)indices_buf, GGML_SORT_ORDER_DESC); \ + hvx_copy_f32_ua(dst_ptr, (const uint8_t *) indices_buf, k); \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, start_row); \ +} + +HTP_TOP_K_FN(32, sort32_f32_hvx) +HTP_TOP_K_FN(64, sort64_f32_hvx) +HTP_TOP_K_FN(128, sort128_f32_hvx) +HTP_TOP_K_FN(256, sort256_f32_hvx) +HTP_TOP_K_FN(512, sort512_f32_hvx) +HTP_TOP_K_FN(1024, sort1024_f32_hvx) + +static void htp_top_k_f32_fallback(unsigned int n, unsigned int i, void * data) { + struct htp_top_k_context * actx = (struct htp_top_k_context *)data; + struct htp_ops_context * octx = actx->octx; + + // Unpack context + const struct htp_tensor * src0 = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + + // Scratchpad memory + uint8_t * spad = actx->vtcm_base + actx->vtcm_per_thread * i; + + // Dimensions + uint32_t ne00 = src0->ne[0]; + uint32_t nb01 = src0->nb[1]; + + uint32_t nb1 = dst->nb[1]; + + uint32_t k = actx->k; + + // Rows to process + uint32_t row_start = actx->row_start; + uint32_t row_end = actx->row_end; + uint32_t rows_per_thread = actx->nrows_per_thread; + uint32_t start_row = row_start + rows_per_thread * i; + uint32_t end_row = MIN(start_row + rows_per_thread, row_end); + + // Pad ne00 to n_vec*32 (n_vec a power of 2) for the bitonic network; + // pad with -INFINITY so it never lands in the top-k. + uint32_t n_vec = hmx_ceil_div(ne00, 32); + uint32_t n_vec_pow2 = 1; + while (n_vec_pow2 < n_vec) n_vec_pow2 <<= 1; + uint32_t ne00_padded = n_vec_pow2 * 32; + + size_t values_size = hex_round_up(ne00_padded * sizeof(float), 128); + float * values_buf = (float *) spad; + int32_t * indices_buf = (int32_t *) (spad + values_size); + + struct htp_thread_trace * tr = &octx->ctx->trace[i]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, start_row); + + for (uint32_t r = start_row; r < end_row; r++) { + uint32_t src_offset = r * nb01; + uint32_t dst_offset = r * nb1; + + uint8_t * src_ptr = (uint8_t *) src0->data + src_offset; + uint8_t * dst_ptr = (uint8_t *) dst->data + dst_offset; + + hex_l2fetch(src_ptr, ne00 * sizeof(float), ne00 * sizeof(float), 1); + + if (k <= 64 && ne00 > 1024) { + float top_values[64]; + int32_t top_indices[64]; + top_k_select_tiled(src_ptr, ne00, k, values_buf, indices_buf, top_values, top_indices); + memcpy(dst_ptr, top_indices, k * sizeof(int32_t)); + continue; + } + + hvx_copy_f32_au((uint8_t*)values_buf, src_ptr, ne00); + + // Fills the indices ramp itself, so no init needed here. + if (ne00_padded > ne00) { + hvx_splat_f32_u((uint8_t *)(values_buf + ne00), -INFINITY, ne00_padded - ne00); + } + bitonic_sort_vtcm_desc((uint8_t*)values_buf, (uint8_t*)indices_buf, n_vec_pow2, true); + + // Copy top-k indices back to DDR + hvx_copy_f32_ua(dst_ptr, (const uint8_t *) indices_buf, k); + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, start_row); +} + +// Single row (ne01=ne02=ne03=1) + large ne00 would otherwise run on one +// HVX thread while the rest sit idle. Split the row into n_chunks +// power-of-two chunks, sort each in parallel with bitonic_sort_vtcm_desc, +// then merge the n_chunks*local_k winners with one more sort that carries +// the global index through instead of re-deriving it. +struct htp_top_k_chunk_ctx { + struct htp_ops_context * octx; + uint8_t * vtcm_base; + size_t phase1_slot_size; + uint32_t ne00; + uint32_t chunk_elems; + uint32_t local_k; + size_t merge_values_off; + size_t merge_indices_off; +}; + +static void htp_top_k_chunk_job(unsigned int n, unsigned int i, void * data) { + struct htp_top_k_chunk_ctx * cctx = (struct htp_top_k_chunk_ctx *) data; + struct htp_ops_context * octx = cctx->octx; + const struct htp_tensor * src0 = octx->src[0]; + + uint32_t chunk_elems = cctx->chunk_elems; + uint32_t ne00 = cctx->ne00; + uint32_t local_k = cctx->local_k; + uint32_t chunk_base = i * chunk_elems; + + uint8_t * spad = cctx->vtcm_base + cctx->phase1_slot_size * i; + size_t values_size = hex_round_up(chunk_elems * sizeof(float), 128); + float * values_buf = (float *) spad; + int32_t * indices_buf = (int32_t *) (spad + values_size); + + uint32_t real_count = (chunk_base < ne00) ? MIN(chunk_elems, ne00 - chunk_base) : 0; + + if (real_count == 0) { + float * merge_values = (float *) (cctx->vtcm_base + cctx->merge_values_off); + int32_t * merge_indices = (int32_t *) (cctx->vtcm_base + cctx->merge_indices_off); + for (uint32_t j = 0; j < local_k; j++) { + merge_values[i * local_k + j] = -INFINITY; + merge_indices[i * local_k + j] = 0; + } + return; + } + + if (local_k > 1 && local_k <= 64 && chunk_elems > 1024) { + uint8_t * src_ptr = (uint8_t *) src0->data + (size_t) chunk_base * sizeof(float); + float top_values[64]; + int32_t top_indices[64]; + float * merge_values = (float *) (cctx->vtcm_base + cctx->merge_values_off); + int32_t * merge_indices = (int32_t *) (cctx->vtcm_base + cctx->merge_indices_off); + for (uint32_t j = 0; j < local_k; j++) { + top_values[j] = -INFINITY; + top_indices[j] = 0; + } + top_k_select_tiled(src_ptr, real_count, local_k, values_buf, indices_buf, top_values, top_indices); + + for (uint32_t j = 0; j < local_k; j++) { + merge_values[i * local_k + j] = top_values[j]; + merge_indices[i * local_k + j] = top_indices[j] + (int32_t) chunk_base; + } + return; + } + + if (real_count > 0) { + uint8_t * src_ptr = (uint8_t *) src0->data + (size_t) chunk_base * sizeof(float); + hex_l2fetch(src_ptr, real_count * sizeof(float), real_count * sizeof(float), 1); + hvx_copy_f32_au((uint8_t *) values_buf, src_ptr, real_count); + } + if (chunk_elems > real_count) { + hvx_splat_f32_u((uint8_t *) (values_buf + real_count), -INFINITY, chunk_elems - real_count); + } + + if (local_k == 1 && cctx->ne00 >= 128*1024) { + float max_value; + uint32_t max_index = top_k_max_value_index(values_buf, real_count, &max_value); + float * merge_values = (float *) (cctx->vtcm_base + cctx->merge_values_off); + int32_t * merge_indices = (int32_t *) (cctx->vtcm_base + cctx->merge_indices_off); + merge_values[i] = max_value; + merge_indices[i] = (int32_t) (max_index + chunk_base); + return; + } + + // chunk_elems is always a power-of-two multiple of 32 + bitonic_sort_vtcm_desc((uint8_t *) values_buf, (uint8_t *) indices_buf, chunk_elems / 32, true); + + float * merge_values = (float *) (cctx->vtcm_base + cctx->merge_values_off); + int32_t * merge_indices = (int32_t *) (cctx->vtcm_base + cctx->merge_indices_off); + + for (uint32_t j = 0; j < local_k; j++) { + merge_values[i * local_k + j] = values_buf[j]; + merge_indices[i * local_k + j] = indices_buf[j] + (int32_t) chunk_base; + } +} + +struct htp_top_k_merge_ctx { + struct htp_ops_context * octx; + uint8_t * vtcm_base; + size_t merge_values_off; + size_t merge_indices_off; + uint32_t merge_elems; + uint32_t total_candidates; + uint32_t k; +}; + +static void htp_top_k_merge_job(unsigned int n, unsigned int i, void * data) { + struct htp_top_k_merge_ctx * mctx = (struct htp_top_k_merge_ctx *) data; + struct htp_ops_context * octx = mctx->octx; + const struct htp_tensor * dst = octx->dst; + + float * merge_values = (float *) (mctx->vtcm_base + mctx->merge_values_off); + int32_t * merge_indices = (int32_t *) (mctx->vtcm_base + mctx->merge_indices_off); + + if (mctx->merge_elems > mctx->total_candidates) { + uint32_t pad = mctx->merge_elems - mctx->total_candidates; + hvx_splat_f32_u((uint8_t *) (merge_values + mctx->total_candidates), -INFINITY, pad); + for (uint32_t j = mctx->total_candidates; j < mctx->merge_elems; j++) { + merge_indices[j] = 0; + } + } + + // Preserve the global indices computed in phase 1 -- init_indices=false + // so they aren't overwritten with a local ramp. + bitonic_sort_vtcm_desc((uint8_t *) merge_values, (uint8_t *) merge_indices, mctx->merge_elems / 32, false); + + hvx_copy_f32_ua((uint8_t *) dst->data, (const uint8_t *) merge_indices, mctx->k); +} + +static int op_top_k_single_row_threaded(struct htp_ops_context * octx, uint32_t ne00, uint32_t k) { + uint32_t n_threads_avail = octx->n_threads; + + uint32_t n_vec = hmx_ceil_div(ne00, 32); + uint32_t n_vec_pow2 = 1; + while (n_vec_pow2 < n_vec) n_vec_pow2 <<= 1; + + // Largest power-of-two chunk count that both fits the available + // threads and evenly divides n_vec_pow2 + uint32_t n_chunks = 1; + while (n_chunks * 2 <= n_threads_avail && n_chunks * 2 <= n_vec_pow2) { + n_chunks *= 2; + } + + uint32_t chunk_n_vec = n_vec_pow2 / n_chunks; + uint32_t chunk_elems = chunk_n_vec * 32; + uint32_t local_k = MIN(k, chunk_elems); + + uint32_t total_candidates = n_chunks * local_k; + uint32_t merge_n_vec = hmx_ceil_div(total_candidates, 32); + uint32_t merge_n_vec_pow2 = 1; + while (merge_n_vec_pow2 < merge_n_vec) merge_n_vec_pow2 <<= 1; + uint32_t merge_elems = merge_n_vec_pow2 * 32; + + size_t phase1_values_size = hex_round_up(chunk_elems * sizeof(float), 128); + size_t phase1_indices_size = hex_round_up(chunk_elems * sizeof(int32_t), 128); + size_t phase1_slot_size = hex_round_up(phase1_values_size + phase1_indices_size, 256); + size_t phase1_total_size = phase1_slot_size * n_chunks; + + size_t merge_values_size = hex_round_up(merge_elems * sizeof(float), 128); + size_t merge_indices_size = hex_round_up(merge_elems * sizeof(int32_t), 128); + size_t merge_values_off = phase1_total_size; + size_t merge_indices_off = merge_values_off + merge_values_size; + + size_t total_vtcm = phase1_total_size + merge_values_size + merge_indices_size; + if (octx->ctx->vtcm_size < total_vtcm) { + return HTP_STATUS_VTCM_TOO_SMALL; + } + + uint8_t * vtcm_base = (uint8_t *) octx->ctx->vtcm_base; + + struct htp_top_k_chunk_ctx cctx; + cctx.octx = octx; + cctx.vtcm_base = vtcm_base; + cctx.phase1_slot_size = phase1_slot_size; + cctx.ne00 = ne00; + cctx.chunk_elems = chunk_elems; + cctx.local_k = local_k; + cctx.merge_values_off = merge_values_off; + cctx.merge_indices_off = merge_indices_off; + + work_queue_run(octx->ctx->work_queue, htp_top_k_chunk_job, &cctx, n_chunks); + + struct htp_top_k_merge_ctx mctx; + mctx.octx = octx; + mctx.vtcm_base = vtcm_base; + mctx.merge_values_off = merge_values_off; + mctx.merge_indices_off = merge_indices_off; + mctx.merge_elems = merge_elems; + mctx.total_candidates = total_candidates; + mctx.k = k; + + work_queue_run(octx->ctx->work_queue, htp_top_k_merge_job, &mctx, 1); + + return HTP_STATUS_OK; +} + +int op_top_k(struct htp_ops_context * octx) { + // Check supported types + if (octx->src[0]->type != HTP_TYPE_F32) { + return HTP_STATUS_NO_SUPPORT; + } + + const struct htp_tensor * src0 = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + + const uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const size_t dst_row_size = dst->ne[0] * sizeof(int32_t); + + uint32_t row_start = 0; + uint32_t row_end = total_rows; + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(int32_t), (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, rows_per_chunk, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + row_end = range.start + range.count; + } + + const uint32_t nrows = row_end - row_start; + if (nrows == 0) { + return HTP_STATUS_OK; + } + + uint32_t ne00 = src0->ne[0]; + uint32_t k = dst->ne[0]; + + // Single row + large ne00: the per-row dispatch below would run on one + // HVX thread while the rest sit idle. Split the row across threads. + if (total_rows == 1 && ne00 > 1024) { + int status = op_top_k_single_row_threaded(octx, ne00, k); + if (status != HTP_STATUS_VTCM_TOO_SMALL) { + return status; + } + // else: fall through to the single-thread path below. + } + + const uint32_t n_threads = MIN(nrows, octx->n_threads); + + // Scratchpad layout: values + indices + // For bitonic: need padding to power-of-2 size + // Allocate for worst case (bitonic with padding) + uint32_t n_vec = hmx_ceil_div(ne00, 32); + uint32_t n_vec_pow2 = 1; + while (n_vec_pow2 < n_vec) n_vec_pow2 <<= 1; + uint32_t ne00_padded = n_vec_pow2 * 32; + + size_t values_size = hex_round_up(ne00_padded * sizeof(float), 128); + size_t indices_size = hex_round_up(ne00_padded * sizeof(int32_t), 128); + size_t spad_per_thread = values_size + indices_size; + + // Make sure we round up to 256 for alignment requirements + spad_per_thread = hex_round_up(spad_per_thread, 256); + + size_t total_spad_size = spad_per_thread * n_threads; + + if (octx->ctx->vtcm_size < total_spad_size) { + FARF(ERROR, "top_k: VTCM size too small. Needed %zu, have %zu", total_spad_size, octx->ctx->vtcm_size); + return HTP_STATUS_VTCM_TOO_SMALL; + } + + FARF(HIGH, "top_k: %ux%ux%ux%u -> %ux%ux%ux%u (0x%x, 0x%x)", + octx->src[0]->ne[0], octx->src[0]->ne[1], octx->src[0]->ne[2], octx->src[0]->ne[3], + octx->dst->ne[0], octx->dst->ne[1], octx->dst->ne[2], octx->dst->ne[3], + octx->src[0]->data, octx->dst->data); + + struct htp_top_k_context actx; + const struct fastdiv_values n_threads_div = init_fastdiv_values(n_threads); + actx.octx = octx; + actx.nrows_per_thread = fastdiv(nrows + n_threads - 1, &n_threads_div); + actx.row_start = row_start; + actx.row_end = row_end; + actx.vtcm_base = (uint8_t *) octx->ctx->vtcm_base; + actx.vtcm_per_thread = spad_per_thread; + actx.k = k; + + worker_callback_t job_func = htp_top_k_f32_fallback; + switch (ne00) { + case 1024: job_func = htp_top_k_f32_1024; break; + case 512: job_func = htp_top_k_f32_512; break; + case 256: job_func = htp_top_k_f32_256; break; + case 128: job_func = htp_top_k_f32_128; break; + case 64: job_func = htp_top_k_f32_64; break; + case 32: job_func = htp_top_k_f32_32; break; + default: job_func = htp_top_k_f32_fallback; break; + } + + // Run jobs + work_queue_run(octx->ctx->work_queue, job_func, &actx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/binary-ops.c b/ggml/src/ggml-hexagon/htp/binary-ops.c index db617796..155e8523 100644 --- a/ggml/src/ggml-hexagon/htp/binary-ops.c +++ b/ggml/src/ggml-hexagon/htp/binary-ops.c @@ -8,14 +8,16 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "hex-profile.h" +#include "binary-ops.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" #include "htp-tensor.h" #ifndef MIN @@ -25,6 +27,8 @@ // Context for binary operations struct htp_binary_context { struct htp_ops_context * octx; + struct htp_binary_vtcm_layout vtcm_layout; + uint8_t * vtcm_base; struct fastdiv_values src0_dim1_div; // ne01 struct fastdiv_values src0_dim2_div; // ne02 @@ -36,39 +40,44 @@ struct htp_binary_context { uint32_t block_max; uint32_t nrows_per_thread; + uint32_t total_rows; + uint32_t row_start; size_t src0_row_size_aligned; size_t src1_row_size_aligned; size_t dst_row_size_aligned; + size_t row_size_bytes; bool split_at_ne01; bool split_at_ne02; + + void * compute; }; #define htp_binary_preamble \ const struct htp_tensor * src0 = octx->src[0]; \ const struct htp_tensor * src1 = octx->src[1]; \ const struct htp_tensor * dst = octx->dst; \ - \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne02 = src0->ne[2]; \ - const uint32_t ne03 = src0->ne[3]; \ - \ - const uint32_t ne10 = src1->ne[0]; \ - const uint32_t ne11 = src1->ne[1]; \ - const uint32_t ne12 = src1->ne[2]; \ - const uint32_t ne13 = src1->ne[3]; \ - \ - const uint32_t nb01 = src0->nb[1]; \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t nb11 = src1->nb[1]; \ - const uint32_t nb12 = src1->nb[2]; \ - const uint32_t nb13 = src1->nb[3]; \ - \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ + \ + const uint32_t ne00 = src0->ne[0]; \ + const uint32_t ne01 = src0->ne[1]; \ + const uint32_t ne02 = src0->ne[2]; \ + const uint32_t ne03 = src0->ne[3]; \ + \ + const uint32_t ne10 = src1->ne[0]; \ + const uint32_t ne11 = src1->ne[1]; \ + const uint32_t ne12 = src1->ne[2]; \ + const uint32_t ne13 = src1->ne[3]; \ + \ + const uint32_t nb01 = src0->nb[1]; \ + const uint32_t nb02 = src0->nb[2]; \ + const uint32_t nb03 = src0->nb[3]; \ + \ + const uint32_t nb11 = src1->nb[1]; \ + const uint32_t nb12 = src1->nb[2]; \ + const uint32_t nb13 = src1->nb[3]; \ + \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ const uint32_t nb3 = dst->nb[3]; static inline uint32_t calc_block_size(struct htp_binary_context * bctx, uint32_t ir, uint32_t end_row, uint32_t ne01, uint32_t ne02) { @@ -92,115 +101,302 @@ static inline uint32_t calc_block_size(struct htp_binary_context * bctx, uint32_ return MIN(bctx->block_max, block_limit); } -// Macro for scalar op switch -#define COMPUTE_SCALAR_OP(DST, SRC, VAL, TYPE, N) \ - if(TYPE == HTP_TYPE_F32) { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_scalar_f32_aa(DST, SRC, *(float *)VAL, N); break; \ - case HTP_OP_SUB: hvx_sub_scalar_f32_aa(DST, SRC, *(float *)VAL, N); break; \ - case HTP_OP_MUL: hvx_mul_scalar_f32_aa(DST, SRC, *(float *)VAL, N); break; \ - case HTP_OP_DIV: hvx_mul_scalar_f32_aa(DST, SRC, 1.0f / (*(float *)VAL), N); break; \ - default: break; \ - } \ - } \ - else { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_scalar_f16_aa(DST, SRC, *(_Float16 *)VAL, N); break; \ - case HTP_OP_SUB: hvx_sub_scalar_f16_aa(DST, SRC, *(_Float16 *)VAL, N); break; \ - case HTP_OP_MUL: hvx_mul_scalar_f16_aa(DST, SRC, *(_Float16 *)VAL, N); break; \ - case HTP_OP_DIV: hvx_div_scalar_f16_aa(DST, SRC, *(_Float16 *)VAL, N); break; \ - default: break; \ - } \ - } +// Out-of-line compute micro-kernels + +typedef void (*compute_scalar_dma_t)( + uint8_t * dst, const uint8_t * src0, const void * s1_table, + uint32_t cur_i11, uint32_t ne11, uint32_t n_rows, + size_t dst_stride, size_t src0_stride, uint32_t ne00); + +#define DEFINE_COMPUTE_SCALAR_DMA(NAME, TYPE, HVX_STMT) \ +static void compute_scalar_dma_##NAME( \ + uint8_t * dst, const uint8_t * src0, const void * s1_table, \ + uint32_t cur_i11, uint32_t ne11, uint32_t n_rows, \ + size_t dst_stride, size_t src0_stride, uint32_t ne00) { \ + const TYPE * table = (const TYPE *) s1_table; \ + for (uint32_t r = 0; r < n_rows; r++) { \ + uint8_t * r_dst = dst + r * dst_stride; \ + const uint8_t * r_src0 = src0 + r * src0_stride; \ + TYPE val = table[cur_i11]; \ + HVX_STMT; \ + if (ne11 > 1 && ++cur_i11 == ne11) { \ + cur_i11 = 0; \ + } \ + } \ +} + +DEFINE_COMPUTE_SCALAR_DMA(add_f32, float, hvx_add_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(add_f16, _Float16, hvx_add_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(sub_f32, float, hvx_sub_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(sub_f16, _Float16, hvx_sub_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(mul_f32, float, hvx_mul_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(mul_f16, _Float16, hvx_mul_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR_DMA(div_f32, float, hvx_mul_scalar_f32_aa(r_dst, r_src0, 1.0f / (val), ne00)) +DEFINE_COMPUTE_SCALAR_DMA(div_f16, _Float16, hvx_div_scalar_f16_aa(r_dst, r_src0, val, ne00)) + +typedef void (*compute_scalar_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_ptr, uint32_t s1_stride, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00); + +#define DEFINE_COMPUTE_SCALAR(NAME, TYPE, HVX_STMT) \ +static void compute_scalar_##NAME( \ + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_ptr, uint32_t s1_stride, \ + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00) { \ + for (uint32_t r = 0; r < n_rows; r++) { \ + uint8_t * r_dst = dst + r * dst_stride; \ + const uint8_t * r_src0 = src0 + r * src0_stride; \ + TYPE val = *(const TYPE *)(src1_ptr + r * s1_stride); \ + HVX_STMT; \ + } \ +} + +DEFINE_COMPUTE_SCALAR(add_f32, float, hvx_add_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(add_f16, _Float16, hvx_add_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(sub_f32, float, hvx_sub_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(sub_f16, _Float16, hvx_sub_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(mul_f32, float, hvx_mul_scalar_f32_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(mul_f16, _Float16, hvx_mul_scalar_f16_aa(r_dst, r_src0, val, ne00)) +DEFINE_COMPUTE_SCALAR(div_f32, float, hvx_mul_scalar_f32_aa(r_dst, r_src0, 1.0f / (val), ne00)) +DEFINE_COMPUTE_SCALAR(div_f16, _Float16, hvx_div_scalar_f16_aa(r_dst, r_src0, val, ne00)) + +typedef void (*compute_same_shape_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, size_t src1_stride, uint32_t ne00); + +#define DEFINE_COMPUTE_SAME_SHAPE(NAME, HVX_FN) \ +static void compute_same_shape_##NAME( \ + uint8_t * dst, const uint8_t * src0, const uint8_t * src1, \ + uint32_t n_rows, size_t dst_stride, size_t src0_stride, size_t src1_stride, uint32_t ne00) { \ + for (uint32_t r = 0; r < n_rows; r++) { \ + HVX_FN(dst + r * dst_stride, src0 + r * src0_stride, src1 + r * src1_stride, ne00); \ + } \ +} + +DEFINE_COMPUTE_SAME_SHAPE(add_f32, hvx_add_f32_aaa) +DEFINE_COMPUTE_SAME_SHAPE(add_f16, hvx_add_f16_aaa) +DEFINE_COMPUTE_SAME_SHAPE(sub_f32, hvx_sub_f32_aaa) +DEFINE_COMPUTE_SAME_SHAPE(sub_f16, hvx_sub_f16_aaa) +DEFINE_COMPUTE_SAME_SHAPE(mul_f32, hvx_mul_f32_aaa) +DEFINE_COMPUTE_SAME_SHAPE(mul_f16, hvx_mul_f16_aaa) +DEFINE_COMPUTE_SAME_SHAPE(div_f32, hvx_div_f32_aaa) +DEFINE_COMPUTE_SAME_SHAPE(div_f16, hvx_div_f16_aaa) + +typedef void (*compute_row_bcast_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00); + +#define DEFINE_COMPUTE_ROW_BCAST(NAME, HVX_FN) \ +static void compute_row_bcast_##NAME( \ + uint8_t * dst, const uint8_t * src0, const uint8_t * src1, \ + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00) { \ + for (uint32_t r = 0; r < n_rows; r++) { \ + HVX_FN(dst + r * dst_stride, src0 + r * src0_stride, src1, ne00); \ + } \ +} + +DEFINE_COMPUTE_ROW_BCAST(add_f32, hvx_add_f32_aaa) +DEFINE_COMPUTE_ROW_BCAST(add_f16, hvx_add_f16_aaa) +DEFINE_COMPUTE_ROW_BCAST(sub_f32, hvx_sub_f32_aaa) +DEFINE_COMPUTE_ROW_BCAST(sub_f16, hvx_sub_f16_aaa) +DEFINE_COMPUTE_ROW_BCAST(mul_f32, hvx_mul_f32_aaa) +DEFINE_COMPUTE_ROW_BCAST(mul_f16, hvx_mul_f16_aaa) +DEFINE_COMPUTE_ROW_BCAST(div_f32, hvx_div_f32_aaa) +DEFINE_COMPUTE_ROW_BCAST(div_f16, hvx_div_f16_aaa) + +typedef void (*compute_complex_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_plane, + uint32_t i01, uint32_t ne11, const struct fastdiv_values * div11, uint32_t nb11, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00); + +#define DEFINE_COMPUTE_COMPLEX(NAME, HVX_FN) \ +static void compute_complex_##NAME( \ + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_plane, \ + uint32_t i01, uint32_t ne11, const struct fastdiv_values * div11, uint32_t nb11, \ + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00) { \ + for (uint32_t r = 0; r < n_rows; r++) { \ + uint32_t i11 = fastmodulo(i01 + r, ne11, div11); \ + const uint8_t * r_src1 = src1_plane + i11 * nb11; \ + HVX_FN(dst + r * dst_stride, src0 + r * src0_stride, r_src1, ne00); \ + } \ +} + +DEFINE_COMPUTE_COMPLEX(add_f32, hvx_add_f32_aau) +DEFINE_COMPUTE_COMPLEX(add_f16, hvx_add_f16_aau) +DEFINE_COMPUTE_COMPLEX(sub_f32, hvx_sub_f32_aau) +DEFINE_COMPUTE_COMPLEX(sub_f16, hvx_sub_f16_aau) +DEFINE_COMPUTE_COMPLEX(mul_f32, hvx_mul_f32_aau) +DEFINE_COMPUTE_COMPLEX(mul_f16, hvx_mul_f16_aau) +DEFINE_COMPUTE_COMPLEX(div_f32, hvx_div_f32_aau) +DEFINE_COMPUTE_COMPLEX(div_f16, hvx_div_f16_aau) + +typedef void (*compute_repeat_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_plane, + uint32_t i01, uint32_t ne11, const struct fastdiv_values * div11, uint32_t nb11, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00, uint32_t ne10); + +#define DEFINE_COMPUTE_REPEAT(NAME, TYPE, HVX_FN) \ +static void compute_repeat_##NAME( \ + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_plane, \ + uint32_t i01, uint32_t ne11, const struct fastdiv_values * div11, uint32_t nb11, \ + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00, uint32_t ne10) { \ + for (uint32_t r = 0; r < n_rows; r++) { \ + uint32_t i11 = fastmodulo(i01 + r, ne11, div11); \ + const uint8_t * r_src1_row = src1_plane + i11 * nb11; \ + uint8_t * r_dst = dst + r * dst_stride; \ + const uint8_t * r_src0 = src0 + r * src0_stride; \ + for (uint32_t c = 0; c < ne00; c += ne10) { \ + uint32_t len = MIN(ne10, ne00 - c); \ + HVX_FN(r_dst + c * sizeof(TYPE), r_src0 + c * sizeof(TYPE), r_src1_row, len); \ + } \ + } \ +} -// Macro for vector op switch (All Aligned) -#define COMPUTE_VECTOR_OP_AAA(DST, SRC0, SRC1, TYPE, N) \ - if(TYPE == HTP_TYPE_F32) { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f32_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f32_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f32_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f32_aaa(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ - } \ - else { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f16_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f16_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f16_aaa(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f16_aaa(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ +DEFINE_COMPUTE_REPEAT(add_f32, float, hvx_add_f32_uuu) +DEFINE_COMPUTE_REPEAT(add_f16, _Float16, hvx_add_f16_uuu) +DEFINE_COMPUTE_REPEAT(sub_f32, float, hvx_sub_f32_uuu) +DEFINE_COMPUTE_REPEAT(sub_f16, _Float16, hvx_sub_f16_uuu) +DEFINE_COMPUTE_REPEAT(mul_f32, float, hvx_mul_f32_uuu) +DEFINE_COMPUTE_REPEAT(mul_f16, _Float16, hvx_mul_f16_uuu) +DEFINE_COMPUTE_REPEAT(div_f32, float, hvx_div_f32_uuu) +DEFINE_COMPUTE_REPEAT(div_f16, _Float16, hvx_div_f16_uuu) + +typedef void (*compute_add_id_t)( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_data, const char * src2_data, + uint32_t i01, uint32_t i02, uint32_t nb20, uint32_t nb21, uint32_t src1_stride, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00); + +static void compute_add_id_f32( + uint8_t * dst, const uint8_t * src0, const uint8_t * src1_data, const char * src2_data, + uint32_t i01, uint32_t i02, uint32_t nb20, uint32_t nb21, uint32_t src1_stride, + uint32_t n_rows, size_t dst_stride, size_t src0_stride, uint32_t ne00) { + for (uint32_t r = 0; r < n_rows; r++) { + uint32_t r_i01 = i01 + r; + const int32_t idx = *(const int32_t *)(src2_data + r_i01 * nb20 + i02 * nb21); + if (idx < 0) { + memcpy(dst + r * dst_stride, src0 + r * src0_stride, ne00 * sizeof(float)); + continue; + } + const uint8_t * r_src1 = src1_data + idx * src1_stride; + const uint8_t * r_src0 = src0 + r * src0_stride; + uint8_t * r_dst = dst + r * dst_stride; + hvx_add_f32_aaa(r_dst, r_src0, r_src1, ne00); } +} + +// 1a. Scalar src1 in VTCM via DMA (ne10 == 1, ne12 == 1, ne13 == 1) +static void binary_thread_scalar_dma(unsigned int nth, unsigned int ith, void * data) { + struct htp_binary_context * bctx = (struct htp_binary_context *) data; + struct htp_ops_context * octx = bctx->octx; + htp_binary_preamble; + + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); + if (start_row >= end_row) return; + + FARF(HIGH, "binary-scalar-dma: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; + const void * s1_table = VTCM_LAYOUT_PTR(const void, bctx->vtcm_base, layout->off_src1); + + dma_queue * dma_q = octx->ctx->dma[ith]; + uint32_t ir_prefetch = start_row; + int spad_idx = 0; + + for (int k = 0; k < 2 && ir_prefetch < end_row; k++) { + uint32_t current_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t i03, i02, i01, rem; + i03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + rem = ir_prefetch - i03 * (ne02 * ne01); + i02 = fastdiv(rem, &bctx->src0_dim1_div); + i01 = rem - i02 * ne01; + + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; -// Macro for vector op switch (Dst Aligned, Src0 Aligned, Src1 Unaligned) -#define COMPUTE_VECTOR_OP_AAU(DST, SRC0, SRC1, TYPE, N) \ - if(TYPE == HTP_TYPE_F32) { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f32_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f32_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f32_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f32_aau(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ - } \ - else { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f16_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f16_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f16_aau(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f16_aau(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ + uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; + uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; + + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, + current_block_size); + ir_prefetch += current_block_size; + spad_idx ^= 1; } -// Macro for vector op switch (All Unaligned - generic loop used in element repeat) -#define COMPUTE_VECTOR_OP_UUU(DST, SRC0, SRC1, TYPE, N) \ - if(TYPE == HTP_TYPE_F32) { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f32_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f32_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f32_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f32_uuu(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ - } \ - else { \ - switch (octx->op) { \ - case HTP_OP_ADD: hvx_add_f16_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_SUB: hvx_sub_f16_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_MUL: hvx_mul_f16_uuu(DST, SRC0, SRC1, N); break; \ - case HTP_OP_DIV: hvx_div_f16_uuu(DST, SRC0, SRC1, N); break; \ - default: break; \ - } \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_scalar_dma_t compute = (compute_scalar_dma_t) bctx->compute; + + for (uint32_t ir = start_row; ir < end_row; ) { + uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); + + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; + + uint32_t i03, i02, i01, rem; + i03 = fastdiv(ir, &bctx->src0_dim12_div); + rem = ir - i03 * (ne02 * ne01); + i02 = fastdiv(rem, &bctx->src0_dim1_div); + i01 = rem - i02 * ne01; + + uint32_t cur_i11 = fastmodulo(i01, ne11, &bctx->src1_dim1_div); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, s1_table, cur_i11, ne11, current_block_size, + bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, + current_block_size); + + if (ir_prefetch < end_row) { + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, + next_block_size); + ir_prefetch += next_block_size; + } + ir += current_block_size; } -// 1. Scalar src1 (ne10 == 1) -static void binary_job_scalar(unsigned int nth, unsigned int ith, void * data) { + dma_queue_flush(dma_q); +} + +// 1b. Scalar src1 dynamic / pointer (ne10 == 1) +static void binary_thread_scalar(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; htp_binary_preamble; - const uint32_t src0_type = octx->src[0]->type; - const uint32_t row_size_bytes = (src0_type == HTP_TYPE_F32) ? ne00 * sizeof(float) : ne00 * sizeof(_Float16); - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - FARF(HIGH, "binary-scalar: %d/%d (%u:%u) row-size %u (%u)", ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + FARF(HIGH, "binary-scalar: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; - // Preamble for (int k = 0; k < 2 && ir_prefetch < end_row; k++) { uint32_t current_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); uint32_t i03, i02, i01, rem; @@ -209,24 +405,26 @@ static void binary_job_scalar(unsigned int nth, unsigned int ith, void * data) { i02 = fastdiv(rem, &bctx->src0_dim1_div); i01 = rem - i02 * ne01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } - // Main loop + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_scalar_t compute = (compute_scalar_t) bctx->compute; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i03, i02, i01, rem; i03 = fastdiv(ir, &bctx->src0_dim12_div); @@ -234,65 +432,62 @@ static void binary_job_scalar(unsigned int nth, unsigned int ith, void * data) { i02 = fastdiv(rem, &bctx->src0_dim1_div); i01 = rem - i02 * ne01; - // src1 indices (broadcast/repeat) uint32_t i13 = fastmodulo(i03, ne13, &bctx->src1_dim3_div); uint32_t i12 = fastmodulo(i02, ne12, &bctx->src1_dim2_div); uint32_t i11 = fastmodulo(i01, ne11, &bctx->src1_dim1_div); - uint8_t * src1_ptr = (uint8_t *)src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; + const uint8_t * src1_ptr = (const uint8_t *)(uintptr_t) src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; uint32_t s1_stride = (ne11 == 1) ? 0 : nb11; - for (uint32_t r = 0; r < current_block_size; r++) { - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; - COMPUTE_SCALAR_OP(r_dst, r_src0, src1_ptr, src0_type, ne00); - src1_ptr += s1_stride; - } + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, src1_ptr, s1_stride, current_block_size, + bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { - uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03, p02, p01, prem; - p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - prem = ir_prefetch - p03 * (ne02 * ne01); - p02 = fastdiv(prem, &bctx->src0_dim1_div); - p01 = prem - p02 * ne01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); - ir_prefetch += next_block_size; + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); + ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } // 2. Vector Same Shape (ne1x == ne0x) or Simple Broadcast -static void binary_job_vector_same_shape(unsigned int nth, unsigned int ith, void * data) { +static void binary_thread_vector_same_shape(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; htp_binary_preamble; - const uint32_t src0_type = octx->src[0]->type; - const uint32_t row_size_bytes = (src0_type == HTP_TYPE_F32) ? ne00 * sizeof(float) : ne00 * sizeof(_Float16); - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - FARF(HIGH, "binary-same-shape: %d/%d (%u:%u) row-size %u (%u)", ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + FARF(HIGH, "binary-same-shape: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * src1_spad_base = octx->src1_spad.data + (ith * octx->src1_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * src1_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src1) + (ith * layout->src1_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t src1_spad_half = octx->src1_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + size_t src0_spad_half = layout->src0_spad_half_size; + size_t src1_spad_half = layout->src1_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; @@ -308,94 +503,95 @@ static void binary_job_vector_same_shape(unsigned int nth, unsigned int ith, voi uint32_t i12 = (ne12 == 1) ? 0 : i02; uint32_t i11 = (ne11 == 1) ? 0 : i01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * src1_curr = (uint8_t *)src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t src1_curr = src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * s1_spad = src1_spad_base + spad_idx * src1_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); - dma_queue_push(q, dma_make_ptr(s1_spad, src1_curr), bctx->src1_row_size_aligned, nb11, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(s1_spad, src1_curr), bctx->src1_row_size_aligned, nb11, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_same_shape_t compute = (compute_same_shape_t) bctx->compute; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; - uint8_t * s1_spad = (uint8_t *) dma_queue_pop(q).dst; - - for (uint32_t r = 0; r < current_block_size; r++) { - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_src1 = s1_spad + r * bctx->src1_row_size_aligned; - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; - COMPUTE_VECTOR_OP_AAA(r_dst, r_src0, r_src1, src0_type, ne00); - } + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; + uint8_t * s1_spad = (uint8_t *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, s1_spad, current_block_size, + bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, bctx->src1_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); uint32_t i03, i02, i01, rem; i03 = fastdiv(ir, &bctx->src0_dim12_div); rem = ir - i03 * (ne02 * ne01); i02 = fastdiv(rem, &bctx->src0_dim1_div); i01 = rem - i02 * ne01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { - uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03, p02, p01, prem; - p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - prem = ir_prefetch - p03 * (ne02 * ne01); - p02 = fastdiv(prem, &bctx->src0_dim1_div); - p01 = prem - p02 * ne01; + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; - uint32_t p13 = (ne13 == 1) ? 0 : p03; - uint32_t p12 = (ne12 == 1) ? 0 : p02; - uint32_t p11 = (ne11 == 1) ? 0 : p01; + uint32_t p13 = (ne13 == 1) ? 0 : p03; + uint32_t p12 = (ne12 == 1) ? 0 : p02; + uint32_t p11 = (ne11 == 1) ? 0 : p01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - uint8_t * s1_next = (uint8_t *)src1->data + p13 * nb13 + p12 * nb12 + p11 * nb11; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_addr_t s1_next = src1->data + p13 * nb13 + p12 * nb12 + p11 * nb11; - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); - dma_queue_push(q, dma_make_ptr(s1_spad, s1_next), bctx->src1_row_size_aligned, nb11, row_size_bytes, next_block_size); + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); + dma_queue_push(dma_q, dma_make_data(s1_spad, s1_next), bctx->src1_row_size_aligned, nb11, row_size_bytes, next_block_size); - ir_prefetch += next_block_size; + ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } // 3. Row Broadcast (ne11 == 1, ne12 == 1, single row src1) -static void binary_job_vector_row_broadcast(unsigned int nth, unsigned int ith, void * data) { +static void binary_thread_vector_row_broadcast(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; htp_binary_preamble; - const uint32_t src0_type = octx->src[0]->type; - const uint32_t row_size_bytes = (src0_type == HTP_TYPE_F32) ? ne00 * sizeof(float) : ne00 * sizeof(_Float16); - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - FARF(HIGH, "binary-row-bcast: %d/%d (%u:%u) row-size %u (%u)", ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + FARF(HIGH, "binary-row-bcast: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * src1_spad_base = octx->src1_spad.data + (ith * octx->src1_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; - void * s1_ptr = (void *) src1_spad_base; + void * s1_ptr = VTCM_LAYOUT_PTR(void, bctx->vtcm_base, layout->off_src1); for (int k = 0; k < 2 && ir_prefetch < end_row; k++) { uint32_t current_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); @@ -404,73 +600,76 @@ static void binary_job_vector_row_broadcast(unsigned int nth, unsigned int ith, uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_row_bcast_t compute = (compute_row_bcast_t) bctx->compute; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; - - for (uint32_t r = 0; r < current_block_size; r++) { - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_src1 = (uint8_t *)s1_ptr; // Constant - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; - COMPUTE_VECTOR_OP_AAA(r_dst, r_src0, r_src1, src0_type, ne00); - } + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, (const uint8_t *)s1_ptr, current_block_size, + bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); uint32_t i03 = fastdiv(ir, &bctx->src0_dim12_div); uint32_t rem = ir - i03 * (ne02 * ne01); uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { - uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - uint32_t prem = ir_prefetch - p03 * (ne02 * ne01); - uint32_t p02 = fastdiv(prem, &bctx->src0_dim1_div); - uint32_t p01 = prem - p02 * ne01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); - ir_prefetch += next_block_size; + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); + ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } // 4. Vector Complex (ne10 == ne00, complex broadcast) -static void binary_job_vector_complex(unsigned int nth, unsigned int ith, void * data) { +static void binary_thread_vector_complex(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; htp_binary_preamble; - const uint32_t src0_type = octx->src[0]->type; - const uint32_t row_size_bytes = (src0_type == HTP_TYPE_F32) ? ne00 * sizeof(float) : ne00 * sizeof(_Float16); - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - FARF(HIGH, "binary-complex: %d/%d (%u:%u) row-size %u (%u)", ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + FARF(HIGH, "binary-complex: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; @@ -481,82 +680,81 @@ static void binary_job_vector_complex(unsigned int nth, unsigned int ith, void * uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_complex_t compute = (compute_complex_t) bctx->compute; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i03 = fastdiv(ir, &bctx->src0_dim12_div); uint32_t rem = ir - i03 * (ne02 * ne01); uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - for (uint32_t r = 0; r < current_block_size; r++) { - uint32_t r_i01 = i01 + r; - uint32_t i13 = fastmodulo(i03, ne13, &bctx->src1_dim3_div); - uint32_t i12 = fastmodulo(i02, ne12, &bctx->src1_dim2_div); - uint32_t i11 = fastmodulo(r_i01, ne11, &bctx->src1_dim1_div); + uint32_t i13 = fastmodulo(i03, ne13, &bctx->src1_dim3_div); + uint32_t i12 = fastmodulo(i02, ne12, &bctx->src1_dim2_div); + const uint8_t * src1_plane = (const uint8_t *)(uintptr_t) src1->data + i13 * nb13 + i12 * nb12; - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_src1 = (uint8_t *)src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, src1_plane, i01, ne11, &bctx->src1_dim1_div, nb11, + current_block_size, bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - // Read src1 from DDR (unaligned) - COMPUTE_VECTOR_OP_AAU(r_dst, r_src0, r_src1, src0_type, ne00); - } - - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { - uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - uint32_t prem = ir_prefetch - p03 * (ne02 * ne01); - uint32_t p02 = fastdiv(prem, &bctx->src0_dim1_div); - uint32_t p01 = prem - p02 * ne01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); - ir_prefetch += next_block_size; + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); + ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } // 5. Element Repeat (ne10 != ne00) -static void binary_job_element_repeat(unsigned int nth, unsigned int ith, void * data) { +static void binary_thread_element_repeat(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; htp_binary_preamble; - const uint32_t src0_type = octx->src[0]->type; - const uint32_t elem_size_bytes = (src0_type == HTP_TYPE_F32) ? sizeof(float) : sizeof(_Float16); - const uint32_t row_size_bytes = ne00 * elem_size_bytes;; - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - FARF(HIGH, "binary-repeat: %d/%d (%u:%u) row-size %u (%u)", ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); + FARF(HIGH, "binary-repeat: %d/%d (%u:%u) row-size %u (%u)", + ith, nth, start_row, end_row, nb01, bctx->dst_row_size_aligned); - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; @@ -567,66 +765,62 @@ static void binary_job_element_repeat(unsigned int nth, unsigned int ith, void * uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + compute_repeat_t compute = (compute_repeat_t) bctx->compute; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i03 = fastdiv(ir, &bctx->src0_dim12_div); uint32_t rem = ir - i03 * (ne02 * ne01); uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - for (uint32_t r = 0; r < current_block_size; r++) { - uint32_t r_i01 = i01 + r; - uint32_t i13 = fastmodulo(i03, ne13, &bctx->src1_dim3_div); - uint32_t i12 = fastmodulo(i02, ne12, &bctx->src1_dim2_div); - uint32_t i11 = fastmodulo(r_i01, ne11, &bctx->src1_dim1_div); - - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_src1_row = (uint8_t *)src1->data + i13 * nb13 + i12 * nb12 + i11 * nb11; - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; - - // Repeat src1 row - for (uint32_t c = 0; c < ne00; c += ne10) { - uint32_t len = MIN(ne10, ne00 - c); - // Use UUU for speed and simplicity - COMPUTE_VECTOR_OP_UUU(r_dst + c * elem_size_bytes, r_src0 + c * elem_size_bytes, r_src1_row, src0_type, len); - } - } + uint32_t i13 = fastmodulo(i03, ne13, &bctx->src1_dim3_div); + uint32_t i12 = fastmodulo(i02, ne12, &bctx->src1_dim2_div); + const uint8_t * src1_plane = (const uint8_t *)(uintptr_t) src1->data + i13 * nb13 + i12 * nb12; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute(d_spad, s0_spad, src1_plane, i01, ne11, &bctx->src1_dim1_div, nb11, + current_block_size, bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00, ne10); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { - uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - uint32_t prem = ir_prefetch - p03 * (ne02 * ne01); - uint32_t p02 = fastdiv(prem, &bctx->src0_dim1_div); - uint32_t p01 = prem - p02 * ne01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); - ir_prefetch += next_block_size; + uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); + ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } // 6. ADD_ID (src1 gathered via src2 indices) -static void binary_job_add_id(unsigned int nth, unsigned int ith, void * data) { +static void binary_thread_add_id_f32(unsigned int nth, unsigned int ith, void * data) { struct htp_binary_context * bctx = (struct htp_binary_context *) data; struct htp_ops_context * octx = bctx->octx; @@ -639,28 +833,29 @@ static void binary_job_add_id(unsigned int nth, unsigned int ith, void * data) { const uint32_t ne01 = src0->ne[1]; const uint32_t ne02 = src0->ne[2]; const uint32_t ne03 = src0->ne[3]; - const uint32_t ne11 = src1->ne[1]; // for bounds check const uint32_t nb01 = src0->nb[1]; const uint32_t nb02 = src0->nb[2]; const uint32_t nb03 = src0->nb[3]; - const uint32_t nb11 = src1->nb[1]; // src1 row stride + const uint32_t src1_stride = bctx->src1_row_size_aligned; const uint32_t nb1 = dst->nb[1]; const uint32_t nb2 = dst->nb[2]; const uint32_t nb3 = dst->nb[3]; - const uint32_t total_rows = ne01 * ne02 * ne03; - const uint32_t start_row = bctx->nrows_per_thread * ith; - const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, total_rows); + const uint32_t row_size_bytes = bctx->row_size_bytes; + const uint32_t start_row = bctx->row_start + bctx->nrows_per_thread * ith; + const uint32_t end_row = MIN(start_row + bctx->nrows_per_thread, bctx->row_start + bctx->total_rows); if (start_row >= end_row) return; - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); - size_t src0_spad_half = octx->src0_spad.size_per_thread / 2; - size_t dst_spad_half = octx->dst_spad.size_per_thread / 2; + const struct htp_binary_vtcm_layout * layout = &bctx->vtcm_layout; + uint8_t * src0_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_src0) + (ith * layout->src0_bytes_per_thread); + uint8_t * dst_spad_base = VTCM_LAYOUT_PTR(uint8_t, bctx->vtcm_base, layout->off_dst) + (ith * layout->dst_bytes_per_thread); + const uint8_t * vtcm_src1 = VTCM_LAYOUT_PTR(const uint8_t, bctx->vtcm_base, layout->off_src1); + size_t src0_spad_half = layout->src0_spad_half_size; + size_t dst_spad_half = layout->dst_spad_half_size; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint32_t ir_prefetch = start_row; int spad_idx = 0; @@ -671,155 +866,139 @@ static void binary_job_add_id(unsigned int nth, unsigned int ith, void * data) { uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - uint8_t * src0_curr = (uint8_t *)src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_addr_t src0_curr = src0->data + i03 * nb03 + i02 * nb02 + i01 * nb01; + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; uint8_t * s0_spad = src0_spad_base + spad_idx * src0_spad_half; uint8_t * d_spad = dst_spad_base + spad_idx * dst_spad_half; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, ne00 * sizeof(float), 0); - dma_queue_push(q, dma_make_ptr(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, ne00 * sizeof(float), current_block_size); + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, 0); + dma_queue_push(dma_q, dma_make_data(s0_spad, src0_curr), bctx->src0_row_size_aligned, nb01, row_size_bytes, current_block_size); ir_prefetch += current_block_size; spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t ir = start_row; ir < end_row; ) { uint32_t current_block_size = calc_block_size(bctx, ir, end_row, ne01, ne02); - uint8_t * d_spad = (uint8_t *) dma_queue_pop(q).src; - uint8_t * s0_spad = (uint8_t *) dma_queue_pop(q).dst; + uint8_t * d_spad = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * s0_spad = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i03 = fastdiv(ir, &bctx->src0_dim12_div); uint32_t rem = ir - i03 * (ne02 * ne01); uint32_t i02 = fastdiv(rem, &bctx->src0_dim1_div); uint32_t i01 = rem - i02 * ne01; - for (uint32_t r = 0; r < current_block_size; r++) { - uint32_t r_i01 = i01 + r; // linear within block since we split at ne01 - - const int32_t idx = *(int32_t *)((char *)src2->data + r_i01 * src2->nb[0] + i02 * src2->nb[1]); - - uint8_t * r_src1 = (uint8_t *)src1->data + idx * nb11; - uint8_t * r_src0 = s0_spad + r * bctx->src0_row_size_aligned; - uint8_t * r_dst = d_spad + r * bctx->dst_row_size_aligned; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + compute_add_id_t compute = (compute_add_id_t) bctx->compute; + compute(d_spad, s0_spad, vtcm_src1, (const char *)(uintptr_t)src2->data, + i01, i02, src2->nb[0], src2->nb[1], src1_stride, + current_block_size, bctx->dst_row_size_aligned, bctx->src0_row_size_aligned, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - hvx_add_f32_aau(r_dst, r_src0, r_src1, ne00); - } - - uint8_t * dst_curr = (uint8_t *)dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; - dma_queue_push(q, dma_make_ptr(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, ne00 * sizeof(float), current_block_size); + dma_addr_t dst_curr = dst->data + i03 * nb3 + i02 * nb2 + i01 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_curr, d_spad), nb1, bctx->dst_row_size_aligned, row_size_bytes, current_block_size); if (ir_prefetch < end_row) { uint32_t next_block_size = calc_block_size(bctx, ir_prefetch, end_row, ne01, ne02); - uint32_t p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); - uint32_t prem = ir_prefetch - p03 * (ne02 * ne01); - uint32_t p02 = fastdiv(prem, &bctx->src0_dim1_div); - uint32_t p01 = prem - p02 * ne01; - uint8_t * s0_next = (uint8_t *)src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; - dma_queue_push(q, dma_make_ptr(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, ne00 * sizeof(float), next_block_size); + uint32_t p03, p02, p01, prem; + p03 = fastdiv(ir_prefetch, &bctx->src0_dim12_div); + prem = ir_prefetch - p03 * (ne02 * ne01); + p02 = fastdiv(prem, &bctx->src0_dim1_div); + p01 = prem - p02 * ne01; + dma_addr_t s0_next = src0->data + p03 * nb03 + p02 * nb02 + p01 * nb01; + dma_queue_push(dma_q, dma_make_data(s0_spad, s0_next), bctx->src0_row_size_aligned, nb01, row_size_bytes, next_block_size); ir_prefetch += next_block_size; } ir += current_block_size; } - dma_queue_flush(q); + + dma_queue_flush(dma_q); } static int execute_op_binary(struct htp_ops_context * octx) { const struct htp_tensor * src0 = octx->src[0]; const struct htp_tensor * src1 = octx->src[1]; const struct htp_tensor * dst = octx->dst; + const struct htp_binary_kernel_params * kparams = (const struct htp_binary_kernel_params *) octx->kernel_params; const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, src0_nrows); - // Use packed row sizes for VTCM allocation + // Use packed row sizes for VTCM allocation and alignment const uint32_t src0_type = octx->src[0]->type; const size_t elem_size = (src0_type == HTP_TYPE_F32) ? sizeof(float) : sizeof(_Float16); const size_t src0_row_size = src0->ne[0] * elem_size; const size_t src1_row_size = src1->ne[0] * elem_size; const size_t dst_row_size = dst->ne[0] * elem_size; - size_t src0_row_size_aligned = hex_round_up(src0_row_size, VLEN); - size_t src1_row_size_aligned = hex_round_up(src1_row_size, VLEN); - size_t dst_row_size_aligned = hex_round_up(dst_row_size, VLEN); - - bool is_add_id = (octx->op == HTP_OP_ADD_ID); - bool is_scalar = !is_add_id && (src1->ne[0] == 1); - - bool is_transposed = (src0->nb[1] < src0_row_size || src1->nb[1] < src1_row_size || dst->nb[1] < dst_row_size); - - bool is_same_shape = !is_add_id && !is_scalar && !is_transposed && - (src1->ne[0] == src0->ne[0] && src0->ne[0] % VLEN == 0) && - (src1->ne[1] == src0->ne[1] || src1->ne[1] == 1) && - (src1->ne[2] == src0->ne[2] || src1->ne[2] == 1) && - (src1->ne[3] == src0->ne[3] || src1->ne[3] == 1); - - bool is_row_bcast = is_same_shape && (src1->ne[1] == 1 && src1->ne[2] == 1 && src1->ne[3] == 1); - bool is_complex = !is_add_id && !is_scalar && !is_same_shape && (src1->ne[0] == src0->ne[0]); - bool is_repeat = !is_add_id && !is_scalar && !is_same_shape && (src1->ne[0] != src0->ne[0]); + uint32_t row_start = 0; + uint32_t nrows = src0_nrows; - size_t spad_row_total; - if (is_same_shape) { - spad_row_total = 2 * (src0_row_size_aligned + src1_row_size_aligned + dst_row_size_aligned); - } else { - spad_row_total = 2 * (src0_row_size_aligned + dst_row_size_aligned); + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, (uint32_t) elem_size, (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, rows_per_chunk, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; } - size_t rows_per_buffer = octx->ctx->vtcm_size / (n_threads * spad_row_total); - - // Adjust for static src1 in row_bcast case - if (is_row_bcast) { - size_t needed_static = src1_row_size_aligned; - if (octx->ctx->vtcm_size < needed_static) return HTP_STATUS_VTCM_TOO_SMALL; - size_t avail = octx->ctx->vtcm_size - needed_static; - rows_per_buffer = avail / (n_threads * spad_row_total); + if (nrows == 0) { + return HTP_STATUS_OK; } - if (rows_per_buffer < 1) { - FARF(ERROR, "binary: VTCM too small\n"); - return HTP_STATUS_VTCM_TOO_SMALL; + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; } - octx->src0_spad.size_per_thread = rows_per_buffer * 2 * src0_row_size_aligned; - octx->dst_spad.size_per_thread = rows_per_buffer * 2 * dst_row_size_aligned; - - if (is_add_id || is_scalar || is_complex || is_repeat || is_row_bcast) { - octx->src1_spad.size_per_thread = 0; - } else { - octx->src1_spad.size_per_thread = rows_per_buffer * 2 * src1_row_size_aligned; - } - - octx->dst_spad.size = n_threads * octx->dst_spad.size_per_thread; - octx->src0_spad.size = n_threads * octx->src0_spad.size_per_thread; - if (is_row_bcast) { - octx->src1_spad.size = src1_row_size_aligned; - } else { - octx->src1_spad.size = n_threads * octx->src1_spad.size_per_thread; + const uint32_t n_threads = octx->n_threads; + const size_t src0_row_size_aligned = kparams->src0_row_size_aligned; + const size_t src1_row_size_aligned = kparams->src1_row_size_aligned; + const size_t dst_row_size_aligned = kparams->dst_row_size_aligned; + + if (htp_tensor_is_extended(src1)) { + if (kparams->kernel_type != HTP_BINARY_KERNEL_SAME_SHAPE && + kparams->kernel_type != HTP_BINARY_KERNEL_ROW_BCAST && + kparams->kernel_type != HTP_BINARY_KERNEL_SCALAR_DMA && + kparams->kernel_type != HTP_BINARY_KERNEL_ADD_ID) { + return HTP_STATUS_NO_SUPPORT; + } } - if (octx->ctx->vtcm_size < (octx->src0_spad.size + octx->src1_spad.size + octx->dst_spad.size)) { - return HTP_STATUS_VTCM_TOO_SMALL; + if (octx->op == HTP_OP_ADD_ID && htp_tensor_is_extended(octx->src[2])) { + return HTP_STATUS_NO_SUPPORT; } - octx->src0_spad.data = octx->ctx->vtcm_base; octx->src0_spad.src = NULL; - octx->src1_spad.data = octx->src0_spad.data + octx->src0_spad.size; octx->src1_spad.src = NULL; - octx->dst_spad.data = octx->src1_spad.data + octx->src1_spad.size; octx->dst_spad.src = NULL; + struct htp_binary_context bctx; + bctx.vtcm_base = (uint8_t *) octx->ctx->vtcm_base; + htp_binary_vtcm_layout_build(&bctx.vtcm_layout, kparams, octx->ctx->vtcm_size); - if ((octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - return HTP_STATUS_OK; + if (bctx.vtcm_layout.rows_per_buffer == 0 || bctx.vtcm_layout.total_bytes > octx->ctx->vtcm_size) { + return HTP_STATUS_VTCM_TOO_SMALL; } - dma_queue * q = octx->ctx->dma[0]; - if (is_row_bcast) { - dma_queue_push(q, dma_make_ptr(octx->src1_spad.data, (const void *) src1->data), src1_row_size_aligned, 0, src1->ne[0] * elem_size, 1); + dma_queue * dma_q = octx->ctx->dma[0]; + uint8_t * vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, bctx.vtcm_base, bctx.vtcm_layout.off_src1); + if (kparams->kernel_type == HTP_BINARY_KERNEL_ROW_BCAST) { + dma_queue_push(dma_q, dma_make_data(vtcm_src1, src1->data), bctx.vtcm_layout.src1_size, 0, src1->ne[0] * elem_size, 1); + } else if (kparams->kernel_type == HTP_BINARY_KERNEL_SCALAR_DMA) { + dma_queue_push(dma_q, dma_make_data(vtcm_src1, src1->data), bctx.vtcm_layout.src1_size, 0, src1->ne[1] * elem_size, 1); + } else if (kparams->kernel_type == HTP_BINARY_KERNEL_ADD_ID) { + dma_queue_push(dma_q, dma_make_data(vtcm_src1, src1->data), + kparams->src1_row_size_aligned, src1->nb[1], + src1->ne[0] * elem_size, src1->ne[1]); } - struct htp_binary_context bctx; bctx.octx = octx; - bctx.nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; - bctx.block_max = rows_per_buffer; + bctx.nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); + bctx.total_rows = nrows; + bctx.row_start = row_start; + bctx.block_max = bctx.vtcm_layout.rows_per_buffer; bctx.src0_row_size_aligned = src0_row_size_aligned; bctx.src1_row_size_aligned = src1_row_size_aligned; bctx.dst_row_size_aligned = dst_row_size_aligned; + bctx.row_size_bytes = src0_row_size; bctx.src0_dim1_div = init_fastdiv_values(src0->ne[1]); bctx.src0_dim2_div = init_fastdiv_values(src0->ne[2]); @@ -835,22 +1014,156 @@ static int execute_op_binary(struct htp_ops_context * octx) { bool src0_contig_dim2 = (src0->nb[3] == src0->ne[2] * src0->nb[2]); bool dst_contig_dim2 = (dst->nb[3] == src0->ne[2] * dst->nb[2]); - bctx.split_at_ne01 = (src0->ne[2] > 1) && ((src1->ne[1] > 1) || (src1->ne[2] > 1) || !src0_contig_dim1 || !dst_contig_dim1); + bctx.split_at_ne01 = (octx->op == HTP_OP_ADD_ID) || + ((src0->ne[2] > 1) && ((src1->ne[1] > 1) || (src1->ne[2] > 1) || !src0_contig_dim1 || !dst_contig_dim1)); bctx.split_at_ne02 = (src0->ne[3] > 1) && ((src1->ne[2] > 1) || (src1->ne[3] > 1) || !src0_contig_dim2 || !dst_contig_dim2); - worker_callback_t worker_func; - if (is_add_id) worker_func = binary_job_add_id; - else if (is_scalar) worker_func = binary_job_scalar; - else if (is_row_bcast) worker_func = binary_job_vector_row_broadcast; - else if (is_same_shape) worker_func = binary_job_vector_same_shape; - else if (is_complex) worker_func = binary_job_vector_complex; - else worker_func = binary_job_element_repeat; + worker_callback_t worker_func = NULL; + void * compute_func = NULL; + + switch (kparams->kernel_type) { + case HTP_BINARY_KERNEL_SAME_SHAPE: + worker_func = binary_thread_vector_same_shape; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_same_shape_add_f32; break; + case HTP_OP_SUB: compute_func = compute_same_shape_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_same_shape_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_same_shape_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_same_shape_add_f16; break; + case HTP_OP_SUB: compute_func = compute_same_shape_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_same_shape_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_same_shape_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_ROW_BCAST: + worker_func = binary_thread_vector_row_broadcast; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_row_bcast_add_f32; break; + case HTP_OP_SUB: compute_func = compute_row_bcast_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_row_bcast_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_row_bcast_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_row_bcast_add_f16; break; + case HTP_OP_SUB: compute_func = compute_row_bcast_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_row_bcast_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_row_bcast_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_SCALAR_DMA: + worker_func = binary_thread_scalar_dma; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_scalar_dma_add_f32; break; + case HTP_OP_SUB: compute_func = compute_scalar_dma_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_scalar_dma_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_scalar_dma_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_scalar_dma_add_f16; break; + case HTP_OP_SUB: compute_func = compute_scalar_dma_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_scalar_dma_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_scalar_dma_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_SCALAR: + worker_func = binary_thread_scalar; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_scalar_add_f32; break; + case HTP_OP_SUB: compute_func = compute_scalar_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_scalar_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_scalar_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_scalar_add_f16; break; + case HTP_OP_SUB: compute_func = compute_scalar_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_scalar_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_scalar_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_COMPLEX: + worker_func = binary_thread_vector_complex; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_complex_add_f32; break; + case HTP_OP_SUB: compute_func = compute_complex_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_complex_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_complex_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_complex_add_f16; break; + case HTP_OP_SUB: compute_func = compute_complex_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_complex_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_complex_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_REPEAT: + worker_func = binary_thread_element_repeat; + if (src0_type == HTP_TYPE_F32) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_repeat_add_f32; break; + case HTP_OP_SUB: compute_func = compute_repeat_sub_f32; break; + case HTP_OP_MUL: compute_func = compute_repeat_mul_f32; break; + case HTP_OP_DIV: compute_func = compute_repeat_div_f32; break; + default: break; + } + } else if (src0_type == HTP_TYPE_F16) { + switch (octx->op) { + case HTP_OP_ADD: compute_func = compute_repeat_add_f16; break; + case HTP_OP_SUB: compute_func = compute_repeat_sub_f16; break; + case HTP_OP_MUL: compute_func = compute_repeat_mul_f16; break; + case HTP_OP_DIV: compute_func = compute_repeat_div_f16; break; + default: break; + } + } + break; + case HTP_BINARY_KERNEL_ADD_ID: + if (octx->op == HTP_OP_ADD_ID && src0_type == HTP_TYPE_F32) { + worker_func = binary_thread_add_id_f32; + compute_func = (void *) compute_add_id_f32; + } + break; + default: break; + } + + if (!worker_func || !compute_func) { + return HTP_STATUS_NO_SUPPORT; + } - if (is_row_bcast) { - dma_queue_pop(q); + bctx.compute = compute_func; + + if (kparams->kernel_type == HTP_BINARY_KERNEL_ROW_BCAST || + kparams->kernel_type == HTP_BINARY_KERNEL_SCALAR_DMA || + kparams->kernel_type == HTP_BINARY_KERNEL_ADD_ID) { + dma_queue_pop(dma_q); } - worker_pool_run_func(octx->ctx->worker_pool, worker_func, &bctx, n_threads); + work_queue_run(octx->ctx->work_queue, worker_func, &bctx, n_threads); return HTP_STATUS_OK; } @@ -870,4 +1183,3 @@ int op_binary(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - diff --git a/ggml/src/ggml-hexagon/htp/binary-ops.h b/ggml/src/ggml-hexagon/htp/binary-ops.h new file mode 100644 index 00000000..b99f2ad6 --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/binary-ops.h @@ -0,0 +1,111 @@ +#ifndef HTP_BINARY_OPS_H +#define HTP_BINARY_OPS_H + +#include +#include +#include + +#include "hex-common.h" +#include "htp-ops.h" +#include "htp-vtcm.h" + +enum htp_binary_kernel_type { + HTP_BINARY_KERNEL_SAME_SHAPE = 0, + HTP_BINARY_KERNEL_ROW_BCAST, + HTP_BINARY_KERNEL_SCALAR_DMA, + HTP_BINARY_KERNEL_SCALAR, + HTP_BINARY_KERNEL_ADD_ID, + HTP_BINARY_KERNEL_COMPLEX, + HTP_BINARY_KERNEL_REPEAT, +}; + +struct htp_binary_kernel_params { + uint32_t kernel_type; + uint32_t n_threads; + uint32_t rows_per_buffer; + + uint32_t src0_row_size_aligned; + uint32_t src1_row_size_aligned; + uint32_t dst_row_size_aligned; + + uint32_t src1_size; + uint32_t vtcm_size; +}; + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_binary_kernel_params) <= 128, "htp_binary_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_binary_kernel_params) <= 128, "htp_binary_kernel_params is too large for kernel_params blob"); +#endif + +struct htp_binary_vtcm_layout { + size_t total_bytes; + size_t off_src0; + size_t off_src1; + size_t off_dst; + + size_t src0_bytes_per_thread; + size_t src1_bytes_per_thread; + size_t dst_bytes_per_thread; + + size_t src0_spad_half_size; + size_t src1_spad_half_size; + size_t dst_spad_half_size; + + size_t src1_size; + uint32_t rows_per_buffer; +}; + +static inline void htp_binary_vtcm_layout_build( + struct htp_binary_vtcm_layout * L, + const struct htp_binary_kernel_params * kparams, + size_t vtcm_size +) { + memset(L, 0, sizeof(*L)); + + const uint32_t n_threads = kparams->n_threads; + if (n_threads == 0) { + return; + } + + const size_t spad_row_total = (kparams->kernel_type == HTP_BINARY_KERNEL_SAME_SHAPE) + ? 2 * (kparams->src0_row_size_aligned + kparams->src1_row_size_aligned + kparams->dst_row_size_aligned) + : 2 * (kparams->src0_row_size_aligned + kparams->dst_row_size_aligned); + + if (spad_row_total == 0 || vtcm_size < kparams->src1_size) { + return; + } + + const size_t rows_per_buffer = (vtcm_size - kparams->src1_size) / (n_threads * spad_row_total); + if (rows_per_buffer == 0) { + return; + } + + L->rows_per_buffer = (uint32_t) rows_per_buffer; + L->src1_size = kparams->src1_size; + + L->src0_bytes_per_thread = rows_per_buffer * 2 * kparams->src0_row_size_aligned; + L->dst_bytes_per_thread = rows_per_buffer * 2 * kparams->dst_row_size_aligned; + L->src1_bytes_per_thread = (kparams->kernel_type == HTP_BINARY_KERNEL_SAME_SHAPE) + ? rows_per_buffer * 2 * kparams->src1_row_size_aligned + : 0; + + L->src0_spad_half_size = L->src0_bytes_per_thread / 2; + L->src1_spad_half_size = L->src1_bytes_per_thread / 2; + L->dst_spad_half_size = L->dst_bytes_per_thread / 2; + + const size_t src0_total = n_threads * L->src0_bytes_per_thread; + const size_t src1_total = (kparams->src1_size > 0) + ? kparams->src1_size + : n_threads * L->src1_bytes_per_thread; + const size_t dst_total = n_threads * L->dst_bytes_per_thread; + + size_t off = 0; + VTCM_LAYOUT_ALLOC(off, off_src0, src0_total); + VTCM_LAYOUT_ALLOC(off, off_src1, src1_total); + VTCM_LAYOUT_ALLOC(off, off_dst, dst_total); + + L->total_bytes = off; +} + +#endif diff --git a/ggml/src/ggml-hexagon/htp/concat-ops.c b/ggml/src/ggml-hexagon/htp/concat-ops.c index 51d39e8d..1fa6ec1b 100644 --- a/ggml/src/ggml-hexagon/htp/concat-ops.c +++ b/ggml/src/ggml-hexagon/htp/concat-ops.c @@ -1,9 +1,12 @@ +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" #include "hexagon_types.h" #include "hexagon_protos.h" #include "hvx_hexagon_protos.h" -#include "hex-dma.h" +#include "dma-queue.h" #include "htp-vtcm.h" #include "hvx-utils.h" #include "hex-fastdiv.h" @@ -13,6 +16,10 @@ struct htp_concat_context { struct htp_ops_context * octx; uint32_t dim; uint32_t nrows_per_thread; + uint32_t row_start; + uint32_t nrows; + uint32_t elem_start; + uint32_t nelems; struct fastdiv_values div_ne0; struct fastdiv_values div_ne1; struct fastdiv_values div_ne2; @@ -28,13 +35,13 @@ static void concat_2d_f32_transposed(unsigned int nth, unsigned int ith, void * const uint32_t src0_ne0 = src0->ne[0]; const uint32_t src1_ne0 = src1->ne[0]; - const uint32_t ne1 = dst->ne[1]; - const uint32_t start_i = ith * cctx->nrows_per_thread; - const uint32_t end_i = (start_i + cctx->nrows_per_thread < ne1) ? (start_i + cctx->nrows_per_thread) : ne1; + const uint32_t row_end = cctx->row_start + cctx->nrows; + const uint32_t start_i = cctx->row_start + ith * cctx->nrows_per_thread; + const uint32_t end_i = (start_i + cctx->nrows_per_thread < row_end) ? (start_i + cctx->nrows_per_thread) : row_end; if (start_i >= end_i) return; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint8_t * spad0_base = octx->src0_spad.data + ith * octx->src0_spad.size_per_thread; uint8_t * spad1_base = octx->src1_spad.data + ith * octx->src1_spad.size_per_thread; @@ -51,21 +58,24 @@ static void concat_2d_f32_transposed(unsigned int nth, unsigned int ith, void * const uint32_t spad0_row_bytes = hex_round_up((src0_ne0 + src1_ne0_padded) * sizeof(float), VLEN); uint32_t mu = src1_ne0_padded * spad1_stride; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t i = start_i; i < end_i; i += block_i) { uint32_t current_block_i = (end_i - i < block_i) ? (end_i - i) : block_i; uint32_t src1_width_bytes = current_block_i * sizeof(float); - uint8_t * src1_ptr = (uint8_t *)src1->data + i * src1->nb[1]; - dma_queue_push(q, dma_make_ptr(spad1_base, src1_ptr), spad1_stride, src1->nb[0], src1_width_bytes, src1_ne0); + const dma_addr_t src1_addr = src1->data + i * src1->nb[1]; + dma_queue_push(dma_q, dma_make_data(spad1_base, src1_addr), spad1_stride, src1->nb[0], src1_width_bytes, src1_ne0); uint32_t src0_row_bytes = src0_ne0 * sizeof(float); - uint8_t * src0_ptr = (uint8_t *)src0->data + i * src0->nb[1]; - dma_queue_push(q, dma_make_ptr(spad0_base, src0_ptr), spad0_row_bytes, src0->nb[1], src0_row_bytes, current_block_i); + const dma_addr_t src0_addr = src0->data + i * src0->nb[1]; + dma_queue_push(dma_q, dma_make_data(spad0_base, src0_addr), spad0_row_bytes, src0->nb[1], src0_row_bytes, current_block_i); - dma_queue_pop(q); // src1 + dma_queue_pop(dma_q); // src1 HVX_Vector * vtcm_tmp = (HVX_Vector *)(spad1_base + src1_ne0_padded * spad1_stride); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) i); for (uint32_t j = 0; j < src1_ne0_padded; j += 32) { #pragma unroll(4) for (uint32_t ii = 0; ii < current_block_i; ii++) { @@ -75,13 +85,14 @@ static void concat_2d_f32_transposed(unsigned int nth, unsigned int ith, void * hvx_vmemu(dst_ptr) = vtcm_tmp[ii]; } } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) i); - dma_queue_pop(q); // src0 + dma_queue_pop(dma_q); // src0 - uint8_t * dst_ptr = (uint8_t *)dst->data + i * dst->nb[1]; - dma_queue_push(q, dma_make_ptr(dst_ptr, spad0_base), dst->nb[1], spad0_row_bytes, (src0_ne0 + src1_ne0) * sizeof(float), current_block_i); + const dma_addr_t dst_addr = dst->data + i * dst->nb[1]; + dma_queue_push(dma_q, dma_make_data(dst_addr, spad0_base), dst->nb[1], spad0_row_bytes, (src0_ne0 + src1_ne0) * sizeof(float), current_block_i); - dma_queue_pop(q); + dma_queue_pop(dma_q); } } @@ -95,13 +106,13 @@ static void concat_2d_f16_transposed(unsigned int nth, unsigned int ith, void * const uint32_t src0_ne0 = src0->ne[0]; const uint32_t src1_ne0 = src1->ne[0]; - const uint32_t ne1 = dst->ne[1]; - const uint32_t start_i = ith * cctx->nrows_per_thread; - const uint32_t end_i = (start_i + cctx->nrows_per_thread < ne1) ? (start_i + cctx->nrows_per_thread) : ne1; + const uint32_t row_end = cctx->row_start + cctx->nrows; + const uint32_t start_i = cctx->row_start + ith * cctx->nrows_per_thread; + const uint32_t end_i = (start_i + cctx->nrows_per_thread < row_end) ? (start_i + cctx->nrows_per_thread) : row_end; if (start_i >= end_i) return; - dma_queue * q = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; uint8_t * spad0_base = octx->src0_spad.data + ith * octx->src0_spad.size_per_thread; uint8_t * spad1_base = octx->src1_spad.data + ith * octx->src1_spad.size_per_thread; @@ -118,21 +129,24 @@ static void concat_2d_f16_transposed(unsigned int nth, unsigned int ith, void * const uint32_t spad0_row_bytes = hex_round_up((src0_ne0 + src1_ne0_padded) * sizeof(__fp16), VLEN); uint32_t mu = src1_ne0_padded * spad1_stride; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t i = start_i; i < end_i; i += block_i) { uint32_t current_block_i = (end_i - i < block_i) ? (end_i - i) : block_i; uint32_t src1_width_bytes = current_block_i * sizeof(__fp16); - uint8_t * src1_ptr = (uint8_t *)src1->data + i * src1->nb[1]; - dma_queue_push(q, dma_make_ptr(spad1_base, src1_ptr), spad1_stride, src1->nb[0], src1_width_bytes, src1_ne0); + const dma_addr_t src1_addr = src1->data + i * src1->nb[1]; + dma_queue_push(dma_q, dma_make_data(spad1_base, src1_addr), spad1_stride, src1->nb[0], src1_width_bytes, src1_ne0); uint32_t src0_row_bytes = src0_ne0 * sizeof(__fp16); - uint8_t * src0_ptr = (uint8_t *)src0->data + i * src0->nb[1]; - dma_queue_push(q, dma_make_ptr(spad0_base, src0_ptr), spad0_row_bytes, src0->nb[1], src0_row_bytes, current_block_i); + const dma_addr_t src0_addr = src0->data + i * src0->nb[1]; + dma_queue_push(dma_q, dma_make_data(spad0_base, src0_addr), spad0_row_bytes, src0->nb[1], src0_row_bytes, current_block_i); - dma_queue_pop(q); // src1 + dma_queue_pop(dma_q); // src1 HVX_Vector * vtcm_tmp = (HVX_Vector *)(spad1_base + src1_ne0_padded * spad1_stride); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) i); for (uint32_t j = 0; j < src1_ne0_padded; j += 64) { #pragma unroll(4) for (uint32_t ii = 0; ii < current_block_i; ii++) { @@ -142,13 +156,14 @@ static void concat_2d_f16_transposed(unsigned int nth, unsigned int ith, void * hvx_vmemu(dst_ptr) = vtcm_tmp[ii]; } } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) i); - dma_queue_pop(q); // src0 + dma_queue_pop(dma_q); // src0 - uint8_t * dst_ptr = (uint8_t *)dst->data + i * dst->nb[1]; - dma_queue_push(q, dma_make_ptr(dst_ptr, spad0_base), dst->nb[1], spad0_row_bytes, (src0_ne0 + src1_ne0) * sizeof(__fp16), current_block_i); + const dma_addr_t dst_addr = dst->data + i * dst->nb[1]; + dma_queue_push(dma_q, dma_make_data(dst_addr, spad0_base), dst->nb[1], spad0_row_bytes, (src0_ne0 + src1_ne0) * sizeof(__fp16), current_block_i); - dma_queue_pop(q); + dma_queue_pop(dma_q); } } @@ -164,11 +179,14 @@ static void concat_generic(unsigned int nth, unsigned int ith, void * data) { const uint32_t type_size = (dst->type == HTP_TYPE_F32 || dst->type == HTP_TYPE_I32) ? 4 : 2; const uint32_t ne[4] = {dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]}; - const uint32_t total_elements = ne[0] * ne[1] * ne[2] * ne[3]; - const uint32_t chunk_size = (total_elements + nth - 1) / nth; - const uint32_t start_idx = MIN(ith * chunk_size, total_elements); - const uint32_t end_idx = MIN(start_idx + chunk_size, total_elements); + // Per-device element range aligned to prevent false sharing + const uint32_t elem_start = cctx->elem_start; + const uint32_t nelems = cctx->nelems; + const uint32_t chunk_size = (nelems + nth - 1) / nth; + + const uint32_t start_idx = MIN(elem_start + ith * chunk_size, elem_start + nelems); + const uint32_t end_idx = MIN(start_idx + chunk_size, elem_start + nelems); // Naive scalar element-wise copy for (uint32_t idx = start_idx; idx < end_idx; idx++) { @@ -236,13 +254,28 @@ int op_concat(struct htp_ops_context * octx) { void (*worker_func)(unsigned int, unsigned int, void *) = concat_generic; if (dim == 0 && is_2d && is_src1_transposed && !is_src0_transposed) { - n_threads = MIN(dst->ne[1], n_threads); - if (n_threads < 1) { - n_threads = 1; + const uint32_t total_rows = dst->ne[1]; + const size_t dst_data_row_size = dst->ne[0] * type_size; + uint32_t row_start = 0; + uint32_t nrows = total_rows; + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, type_size, (uint32_t) dst_data_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; } + + if (nrows == 0) { + return HTP_STATUS_OK; + } + + cctx.row_start = row_start; + cctx.nrows = nrows; + uint32_t block_i = (type_size == 4) ? 32 : 64; - cctx.nrows_per_thread = hmx_ceil_div(dst->ne[1], n_threads); + cctx.nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); // Allocate VTCM uint32_t spad1_stride = block_i * type_size; @@ -270,8 +303,30 @@ int op_concat(struct htp_ops_context * octx) { } else { worker_func = concat_2d_f16_transposed; } + } else { + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(src1) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + const uint32_t total_elements = dst->ne[0] * dst->ne[1] * dst->ne[2] * dst->ne[3]; + uint32_t elem_start = 0; + uint32_t nelems = total_elements; + if (octx->ctx->mdev.count > 1) { + const uint32_t elems_per_chunk = HEX_L2_LINE_SIZE / type_size; + const bool can_split = htp_tensor_mdev_data_aligned(dst) && htp_tensor_is_contiguous(dst, type_size) && !htp_tensor_is_permuted(dst); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_elements, can_split ? elems_per_chunk : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + elem_start = range.start; + nelems = range.count; + } + + if (nelems == 0) { + return HTP_STATUS_OK; + } + + cctx.elem_start = elem_start; + cctx.nelems = nelems; } - worker_pool_run_func(octx->ctx->worker_pool, worker_func, &cctx, n_threads); + work_queue_run(octx->ctx->work_queue, worker_func, &cctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/cpy-ops.c b/ggml/src/ggml-hexagon/htp/cpy-ops.c index ae507eff..4453dda3 100644 --- a/ggml/src/ggml-hexagon/htp/cpy-ops.c +++ b/ggml/src/ggml-hexagon/htp/cpy-ops.c @@ -4,6 +4,7 @@ #include #include +#include #include #include @@ -14,6 +15,8 @@ #include "htp-ops.h" #include "htp-ops.h" #include "hvx-utils.h" +#include "htp-tensor.h" +#include "htp-fence.h" struct htp_copy_context { struct htp_ops_context * octx; @@ -27,9 +30,49 @@ struct htp_copy_context { uint32_t src0_blocks_per_row; uint32_t dst_blocks_per_row; + uint32_t elem_start; + uint32_t nelem; + uint32_t elem_per_thread; + uint32_t src0_nrows_per_thread; + uint32_t row_start; + uint32_t nrows; + + struct fastdiv_values div_ne01; + struct fastdiv_values div_ne02_ne01; + + struct fastdiv_values div_ne0; + struct fastdiv_values div_ne1_ne0; + struct fastdiv_values div_ne2_ne1_ne0; + struct fastdiv_values div_ne00; + struct fastdiv_values div_ne01_ne00; + struct fastdiv_values div_ne02_ne01_ne00; }; +static inline void cpy_dma_sametype_reshape_contig( + dma_queue * dma_q, + dma_addr_t dst, + dma_addr_t src0, + uint32_t total_bytes +) { + if (total_bytes == 0) { + return; + } + + const uint32_t max_chunk = DMA_SAFE_CHUNK_SIZE; + while (total_bytes > 0) { + const uint32_t chunk = MIN(total_bytes, max_chunk); + if (!dma_queue_push(dma_q, dma_make_data(dst, src0), chunk, chunk, chunk, /*nrows=*/ 1)) { + dma_queue_flush(dma_q); + dma_queue_push(dma_q, dma_make_data(dst, src0), chunk, chunk, chunk, /*nrows=*/ 1); + } + dst += chunk; + src0 += chunk; + total_bytes -= chunk; + } + dma_queue_flush(dma_q); +} + #define cpy_preamble \ const struct htp_tensor *src0 = octx->src[0]; \ const struct htp_tensor *dst = octx->dst; \ @@ -52,134 +95,136 @@ struct htp_copy_context { const uint32_t nb0 = dst->nb[0]; \ const uint32_t nb1 = dst->nb[1]; \ const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; \ - \ - const uint32_t nr = ne01; - -#define DEFINE_CPY_SAMESHAPE(NAME, ELEM_TYPE, ELEM_SIZE) \ -static void cpy_thread_##NAME##_sameshape(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_copy_context * ct = (struct htp_copy_context *) data; \ - struct htp_ops_context * octx = ct->octx; \ - cpy_preamble; \ - const uint32_t dr = ct->src0_nrows_per_thread; \ - const uint32_t ir0 = dr * ith; \ - const uint32_t ir1 = (ir0 + dr) < nr ? (ir0 + dr) : nr; \ - if (ir0 >= nr) return; \ - for (uint32_t i03 = 0; i03 < ne03; i03++) { \ - for (uint32_t i02 = 0; i02 < ne02; i02++) { \ - _Pragma("unroll(4)") \ - for (uint32_t i01 = ir0; i01 < ir1; i01++) { \ - uint8_t* dst_ptr = (uint8_t*) dst->data + i01*nb1 + i02*nb2 + i03*nb3; \ - uint8_t* src0_ptr = (uint8_t*) src0->data + i01*nb01 + i02*nb02 + i03*nb03; \ - hex_l2fetch(src0_ptr, ne00 * ELEM_SIZE, nb01, 2); \ - hvx_copy_uu(dst_ptr, src0_ptr, ne00, ELEM_SIZE); \ - } \ - } \ - } \ + const uint32_t nb3 = dst->nb[3]; + +#define DEFINE_CPY_SAMESHAPE(NAME, ELEM_TYPE, ELEM_SIZE) \ +static void cpy_thread_##NAME##_sameshape(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_copy_context * ct = (struct htp_copy_context *) data; \ + struct htp_ops_context * octx = ct->octx; \ + cpy_preamble; \ + const uint32_t dr = ct->src0_nrows_per_thread; \ + const uint32_t ir0 = ct->row_start + dr * ith; \ + const uint32_t ir1 = MIN(ir0 + dr, ct->row_start + ct->nrows); \ + if (ir0 >= ir1) return; \ + const bool contiguous = htp_tensor_is_contiguous(src0, ELEM_SIZE) && htp_tensor_is_contiguous(dst, ELEM_SIZE); \ + if (contiguous) { \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + dma_addr_t dst_addr = dst->data + ir0 * ne00 * ELEM_SIZE; \ + dma_addr_t src0_addr = src0->data + ir0 * ne00 * ELEM_SIZE; \ + cpy_dma_sametype_reshape_contig(dma_q, dst_addr, src0_addr, (ir1 - ir0) * ne00 * ELEM_SIZE); \ + return; \ + } \ + const uint32_t ne02_ne01 = ne02 * ne01; \ + uint32_t i03 = fastdiv(ir0, &ct->div_ne02_ne01); \ + uint32_t rem = ir0 - i03 * ne02_ne01; \ + uint32_t i02 = fastdiv(rem, &ct->div_ne01); \ + uint32_t i01 = rem - i02 * ne01; \ + uint8_t * dst_ptr = (uint8_t *) dst->data + i01*nb1 + i02*nb2 + i03*nb3; \ + uint8_t * src0_ptr = (uint8_t *) src0->data + i01*nb01 + i02*nb02 + i03*nb03; \ + for (uint32_t r = ir0; r < ir1; r++) { \ + hex_l2fetch(src0_ptr, ne00 * ELEM_SIZE, nb01, 2); \ + hvx_copy_uu(dst_ptr, src0_ptr, ne00, ELEM_SIZE); \ + dst_ptr += nb1; \ + src0_ptr += nb01; \ + if (++i01 == ne01) { \ + i01 = 0; \ + if (++i02 == ne02) { \ + i02 = 0; \ + i03++; \ + } \ + dst_ptr = (uint8_t *) dst->data + i02*nb2 + i03*nb3; \ + src0_ptr = (uint8_t *) src0->data + i02*nb02 + i03*nb03; \ + } \ + } \ } -DEFINE_CPY_SAMESHAPE(f32, float, 4) +DEFINE_CPY_SAMESHAPE(f32, float, 4) DEFINE_CPY_SAMESHAPE(f16, __fp16, 2) -#define DEFINE_CPY_RESHAPE(NAME, ELEM_TYPE, ELEM_SIZE) \ -static void cpy_thread_##NAME##_reshape(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_copy_context * ct = (struct htp_copy_context *) data; \ - struct htp_ops_context * octx = ct->octx; \ - cpy_preamble; \ - const uint32_t dr = ct->src0_nrows_per_thread; \ - const uint32_t ir0 = dr * ith; \ - const uint32_t ir1 = (ir0 + dr) < nr ? (ir0 + dr) : nr; \ - if (ir0 >= nr) return; \ - const bool src0_contig = (nb00 == ELEM_SIZE) && \ - (nb01 == ne00 * nb00) && \ - (nb02 == ne01 * nb01) && \ - (nb03 == ne02 * nb02); \ - const bool dst_contig = (nb0 == ELEM_SIZE) && \ - (nb1 == ne0 * nb0) && \ - (nb2 == ne1 * nb1) && \ - (nb3 == ne2 * nb2); \ - if (src0_contig && dst_contig) { \ - for (int64_t i03 = 0; i03 < ne03; i03++) { \ - for (int64_t i02 = 0; i02 < ne02; i02++) { \ - uint8_t * src_ptr = (uint8_t *) src0->data + i03*nb03 + i02*nb02 + ir0*nb01; \ - uint32_t flat = ((i03*ne02 + i02)*ne01 + ir0) * ne00; \ - uint8_t * dst_ptr = (uint8_t *) dst->data + flat * ELEM_SIZE; \ - hvx_copy_uu(dst_ptr, src_ptr, (ir1 - ir0) * ne00, ELEM_SIZE); \ - } \ - } \ - return; \ - } \ - const bool reshape_flat_fast = (ne03 == 1 && ne2 == 1 && ne3 == 1) && \ - (ne0 == ne00 * ne01) && (ne1 == ne02) && \ - (nb00 == ELEM_SIZE) && (nb0 == ELEM_SIZE); \ - if (reshape_flat_fast) { \ - for (uint32_t i02 = 0; i02 < ne02; i02++) { \ - for (uint32_t i01 = ir0; i01 < ir1; i01++) { \ - uint8_t * src0_ptr = (uint8_t *) src0->data + i01 * nb01 + i02 * nb02; \ - uint8_t * dst_ptr = (uint8_t *) dst->data + i01 * ne00 * ELEM_SIZE + i02 * nb1; \ - hvx_copy_uu(dst_ptr, src0_ptr, ne00, ELEM_SIZE); \ - } \ - } \ - return; \ - } \ - int64_t k10 = 0; \ - int64_t i11 = 0; \ - int64_t i12 = 0; \ - int64_t i13 = 0; \ - const int64_t nk00 = ct->src0_blocks_per_row; \ - const int64_t nk0 = ct->dst_blocks_per_row; \ - for (int64_t i03 = 0; i03 < ne03; i03++) { \ - for (int64_t i02 = 0; i02 < ne02; i02++) { \ - k10 += nk00 * ir0; \ - while (k10 >= nk0) { \ - k10 -= nk0; \ - if (++i11 == ne1) { \ - i11 = 0; \ - if (++i12 == ne2) { \ - i12 = 0; \ - if (++i13 == ne3) { \ - i13 = 0; \ - } \ - } \ - } \ - } \ - for (int64_t i01 = ir0; i01 < ir1; i01++) { \ - for (int64_t k00 = 0; k00 < nk00; k00++) { \ - const char * src0_ptr = ((char *) src0->data + k00*nb00 + i01*nb01 + i02*nb02 + i03*nb03); \ - char * dst_ptr = ((char *) dst->data + k10*nb0 + i11*nb1 + i12*nb2 + i13*nb3); \ - memcpy(dst_ptr, src0_ptr, ELEM_SIZE); \ - if (++k10 == nk0) { \ - k10 = 0; \ - if (++i11 == ne1) { \ - i11 = 0; \ - if (++i12 == ne2) { \ - i12 = 0; \ - if (++i13 == ne3) { \ - i13 = 0; \ - } \ - } \ - } \ - } \ - } \ - } \ - k10 += nk00 * (ne01 - ir1); \ - while (k10 >= nk0) { \ - k10 -= nk0; \ - if (++i11 == ne1) { \ - i11 = 0; \ - if (++i12 == ne2) { \ - i12 = 0; \ - if (++i13 == ne3) { \ - i13 = 0; \ - } \ - } \ - } \ - } \ - } \ - } \ +#define DEFINE_CPY_RESHAPE(NAME, ELEM_TYPE, ELEM_SIZE) \ +static void cpy_thread_##NAME##_reshape(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_copy_context * ct = (struct htp_copy_context *) data; \ + struct htp_ops_context * octx = ct->octx; \ + cpy_preamble; \ + const uint32_t th_nelem = ct->elem_per_thread; \ + const uint32_t th_start = ct->elem_start + ith * th_nelem; \ + const uint32_t th_end = MIN(th_start + th_nelem, ct->elem_start + ct->nelem); \ + if (th_start >= th_end) return; \ + \ + if (htp_tensor_is_contiguous(src0, ELEM_SIZE) && htp_tensor_is_contiguous(dst, ELEM_SIZE)) { \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + dma_addr_t dst_addr = dst->data + th_start * ELEM_SIZE; \ + dma_addr_t src0_addr = src0->data + th_start * ELEM_SIZE; \ + cpy_dma_sametype_reshape_contig(dma_q, dst_addr, src0_addr, (th_end - th_start) * ELEM_SIZE); \ + return; \ + } \ + \ + const uint32_t ne01_ne00 = ne01 * ne00; \ + const uint32_t ne02_ne01_ne00 = ne02 * ne01_ne00; \ + const uint32_t ne1_ne0 = ne1 * ne0; \ + const uint32_t ne2_ne1_ne0 = ne2 * ne1_ne0; \ + \ + uint32_t e = th_start; \ + uint32_t i13 = fastdiv(e, &ct->div_ne2_ne1_ne0); \ + uint32_t rem = e - i13 * ne2_ne1_ne0; \ + uint32_t i12 = fastdiv(rem, &ct->div_ne1_ne0); \ + uint32_t rem2 = rem - i12 * ne1_ne0; \ + uint32_t i11 = fastdiv(rem2, &ct->div_ne0); \ + uint32_t i10 = rem2 - i11 * ne0; \ + \ + uint32_t i03 = fastdiv(e, &ct->div_ne02_ne01_ne00); \ + uint32_t rem_s = e - i03 * ne02_ne01_ne00; \ + uint32_t i02 = fastdiv(rem_s, &ct->div_ne01_ne00); \ + uint32_t rem2_s = rem_s - i02 * ne01_ne00; \ + uint32_t i01 = fastdiv(rem2_s, &ct->div_ne00); \ + uint32_t i00 = rem2_s - i01 * ne00; \ + \ + char * dst_ptr = (char *) dst->data + i10*nb0 + i11*nb1 + i12*nb2 + i13*nb3; \ + const char * src0_ptr = (const char *) src0->data + i00*nb00 + i01*nb01 + i02*nb02 + i03*nb03; \ + \ + const bool rows_contig = (nb00 == ELEM_SIZE) && (nb0 == ELEM_SIZE); \ + \ + while (e < th_end) { \ + uint32_t run = 1; \ + if (rows_contig) { \ + run = MIN(MIN(ne00 - i00, ne0 - i10), th_end - e); \ + hvx_copy_uu((uint8_t *) dst_ptr, (const uint8_t *) src0_ptr, run, ELEM_SIZE); \ + } else { \ + *((ELEM_TYPE *) dst_ptr) = *((const ELEM_TYPE *) src0_ptr); \ + } \ + e += run; \ + \ + dst_ptr += run * nb0; \ + i10 += run; \ + if (i10 == ne0) { \ + i10 = 0; \ + if (++i11 == ne1) { \ + i11 = 0; \ + if (++i12 == ne2) { \ + i12 = 0; \ + i13++; \ + } \ + } \ + dst_ptr = (char *) dst->data + i11*nb1 + i12*nb2 + i13*nb3; \ + } \ + \ + src0_ptr += run * nb00; \ + i00 += run; \ + if (i00 == ne00) { \ + i00 = 0; \ + if (++i01 == ne01) { \ + i01 = 0; \ + if (++i02 == ne02) { \ + i02 = 0; \ + i03++; \ + } \ + } \ + src0_ptr = (const char *) src0->data + i01*nb01 + i02*nb02 + i03*nb03; \ + } \ + } \ } -DEFINE_CPY_RESHAPE(f32, float, 4) +DEFINE_CPY_RESHAPE(f32, float, 4) DEFINE_CPY_RESHAPE(f16, __fp16, 2) static void cpy_thread_f16_f32_sameshape(unsigned int nth, unsigned int ith, void * data) { @@ -187,22 +232,33 @@ static void cpy_thread_f16_f32_sameshape(unsigned int nth, unsigned int ith, voi struct htp_ops_context * octx = ct->octx; cpy_preamble; - // parallelize by src0 rows const uint32_t dr = ct->src0_nrows_per_thread; - const uint32_t ir0 = dr * ith; - const uint32_t ir1 = (ir0 + dr) < nr ? (ir0 + dr) : nr; - if (ir0 >= nr) return; + const uint32_t ir0 = ct->row_start + dr * ith; + const uint32_t ir1 = MIN(ir0 + dr, ct->row_start + ct->nrows); + if (ir0 >= ir1) return; - // copy by rows - for (uint32_t i03 = 0; i03 < ne03; i03++) { - for (uint32_t i02 = 0; i02 < ne02; i02++) { - #pragma unroll(2) - for (uint32_t i01 = ir0; i01 < ir1; i01++) { - uint8_t* dst_ptr = (uint8_t*) dst->data + i01*nb1 + i02*nb2 + i03*nb3; - uint8_t* src0_ptr = (uint8_t*) src0->data + i01*nb01 + i02*nb02 + i03*nb03; - hex_l2fetch(src0_ptr, ne00 * sizeof(float), nb01, 2); - hvx_copy_f16_f32_uu(dst_ptr, src0_ptr, ne00); + const uint32_t ne02_ne01 = ne02 * ne01; + uint32_t i03 = fastdiv(ir0, &ct->div_ne02_ne01); + uint32_t rem = ir0 - i03 * ne02_ne01; + uint32_t i02 = fastdiv(rem, &ct->div_ne01); + uint32_t i01 = rem - i02 * ne01; + + uint8_t* dst_ptr = (uint8_t*) dst->data + i01*nb1 + i02*nb2 + i03*nb3; + uint8_t* src0_ptr = (uint8_t*) src0->data + i01*nb01 + i02*nb02 + i03*nb03; + + for (uint32_t r = ir0; r < ir1; r++) { + hex_l2fetch(src0_ptr, ne00 * sizeof(float), nb01, 2); + hvx_copy_f16_f32_uu(dst_ptr, src0_ptr, ne00); + dst_ptr += nb1; + src0_ptr += nb01; + if (++i01 == ne01) { + i01 = 0; + if (++i02 == ne02) { + i02 = 0; + i03++; } + dst_ptr = (uint8_t*) dst->data + i02*nb2 + i03*nb3; + src0_ptr = (uint8_t*) src0->data + i02*nb02 + i03*nb03; } } } @@ -212,30 +268,101 @@ static void cpy_thread_f32_f16_sameshape(unsigned int nth, unsigned int ith, voi struct htp_ops_context * octx = ct->octx; cpy_preamble; - // parallelize by src0 rows const uint32_t dr = ct->src0_nrows_per_thread; - const uint32_t ir0 = dr * ith; - const uint32_t ir1 = (ir0 + dr) < nr ? (ir0 + dr) : nr; - if (ir0 >= nr) return; + const uint32_t ir0 = ct->row_start + dr * ith; + const uint32_t ir1 = MIN(ir0 + dr, ct->row_start + ct->nrows); + if (ir0 >= ir1) return; + + const uint32_t ne02_ne01 = ne02 * ne01; + uint32_t i03 = fastdiv(ir0, &ct->div_ne02_ne01); + uint32_t rem = ir0 - i03 * ne02_ne01; + uint32_t i02 = fastdiv(rem, &ct->div_ne01); + uint32_t i01 = rem - i02 * ne01; + + uint8_t* dst_ptr = (uint8_t*) dst->data + i01*nb1 + i02*nb2 + i03*nb3; + uint8_t* src0_ptr = (uint8_t*) src0->data + i01*nb01 + i02*nb02 + i03*nb03; + + for (uint32_t r = ir0; r < ir1; r++) { + hex_l2fetch(src0_ptr, ne00 * sizeof(__fp16), nb01, 2); + hvx_copy_f32_f16_uu(dst_ptr, src0_ptr, ne00); + dst_ptr += nb1; + src0_ptr += nb01; + if (++i01 == ne01) { + i01 = 0; + if (++i02 == ne02) { + i02 = 0; + i03++; + } + dst_ptr = (uint8_t*) dst->data + i02*nb2 + i03*nb3; + src0_ptr = (uint8_t*) src0->data + i02*nb02 + i03*nb03; + } + } +} + +static inline void cpy_dma_push_2d_chunked( + dma_queue * dma_q, + dma_addr_t dst, + dma_addr_t src, + size_t dst_stride, + size_t src_stride, + size_t row_size, + uint32_t nrows +) { + while (nrows > 0) { + const uint32_t cur_rows = MIN(nrows, DMA_MAX_NROWS); + if (!dma_queue_push(dma_q, dma_make_data(dst, src), dst_stride, src_stride, row_size, cur_rows)) { + dma_queue_flush(dma_q); + dma_queue_push(dma_q, dma_make_data(dst, src), dst_stride, src_stride, row_size, cur_rows); + } + dst += cur_rows * dst_stride; + src += cur_rows * src_stride; + nrows -= cur_rows; + } +} + +static inline void cpy_dma_sametype_sameshape( + struct htp_ops_context * octx, + const struct htp_tensor * dst, + const struct htp_tensor * src0, + uint32_t elem_size, + uint32_t ne00, uint32_t ne01, uint32_t ne02, uint32_t ne03, + uint32_t nb01, uint32_t nb02, uint32_t nb03, + uint32_t nb1, uint32_t nb2, uint32_t nb3 +) { + const bool contiguous = htp_tensor_is_contiguous(src0, elem_size) && htp_tensor_is_contiguous(dst, elem_size); + + dma_queue * dma_q = octx->ctx->dma[0]; + + if (contiguous) { + cpy_dma_sametype_reshape_contig(dma_q, dst->data, src0->data, ne00 * elem_size * ne01 * ne02 * ne03); + return; + } + + const bool contiguous_outer = + (ne02 == 1 || (nb02 == ne01 * nb01 && nb2 == ne01 * nb1)) && + (ne03 == 1 || (nb03 == ne02 * nb02 && nb3 == ne02 * nb2)); + + if (contiguous_outer) { + uint32_t total_rows = ne01 * ne02 * ne03; + cpy_dma_push_2d_chunked(dma_q, dst->data, src0->data, nb1, nb01, ne00 * elem_size, total_rows); + dma_queue_flush(dma_q); + return; + } - // copy by rows for (uint32_t i03 = 0; i03 < ne03; i03++) { for (uint32_t i02 = 0; i02 < ne02; i02++) { - #pragma unroll(2) - for (uint32_t i01 = ir0; i01 < ir1; i01++) { - uint8_t* dst_ptr = (uint8_t*) dst->data + i01*nb1 + i02*nb2 + i03*nb3; - uint8_t* src0_ptr = (uint8_t*) src0->data + i01*nb01 + i02*nb02 + i03*nb03; - hex_l2fetch(src0_ptr, ne00 * sizeof(__fp16), nb01, 2); - hvx_copy_f32_f16_uu(dst_ptr, src0_ptr, ne00); - } + dma_addr_t dst_data = dst->data + i02 * nb2 + i03 * nb3; + dma_addr_t src0_data = src0->data + i02 * nb02 + i03 * nb03; + cpy_dma_push_2d_chunked(dma_q, dst_data, src0_data, nb1, nb01, ne00 * elem_size, ne01); } } + + dma_queue_flush(dma_q); } -int op_cpy(struct htp_ops_context * octx) { +static int exec_cpy(struct htp_ops_context * octx, bool * use_dma) { cpy_preamble; - - const uint32_t n_threads = MIN(nr, octx->n_threads); + *use_dma = false; struct htp_copy_context ct; ct.octx = octx; @@ -254,42 +381,141 @@ int op_cpy(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { - return HTP_STATUS_OK; - } - const bool sametype = (src0->type == dst->type); - const bool transposed = (nb00 > nb01) || (nb0 > nb1); + const bool transposed = (nb00 > nb01) || (nb0 > nb1) || + (nb00 != ct.src0_type_size) || (nb0 != ct.dst_type_size) || + (nb01 < ne00 * ct.src0_type_size) || (nb1 < ne0 * ct.dst_type_size); const bool sameshape = !transposed && (ne00 == ne0 && ne01 == ne1 && ne02 == ne2 && ne03 == ne3); - ct.src0_nrows_per_thread = (nr + n_threads - 1) / n_threads; + const uint32_t n_threads = octx->n_threads; - worker_callback_t copy_fun; + const bool src_is_contiguous = htp_tensor_is_contiguous(src0, ct.src0_type_size); + const bool dst_is_contiguous = htp_tensor_is_contiguous(dst, ct.dst_type_size); - if (sametype && sameshape) { - if (src0->type == HTP_TYPE_F32) { - copy_fun = cpy_thread_f32_sameshape; - } else { - copy_fun = cpy_thread_f16_sameshape; + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + if (!sametype) { + return HTP_STATUS_NO_SUPPORT; } - } else if (sameshape) { - /**/ if (dst->type == HTP_TYPE_F16 && src0->type == HTP_TYPE_F32) - copy_fun = cpy_thread_f16_f32_sameshape; - else if (dst->type == HTP_TYPE_F32 && src0->type == HTP_TYPE_F16) - copy_fun = cpy_thread_f32_f16_sameshape; - else + if (!sameshape && !(src_is_contiguous && dst_is_contiguous && octx->ctx->mdev.count <= 1)) { return HTP_STATUS_NO_SUPPORT; - } else if (sametype) { - if (src0->type == HTP_TYPE_F32) { - copy_fun = cpy_thread_f32_reshape; + } + } + + if (sameshape) { + const uint32_t total_rows = ne01 * ne02 * ne03; + const uint32_t row_size = ne00 * ct.dst_type_size; + + ct.div_ne01 = init_fastdiv_values(ne01); + ct.div_ne02_ne01 = init_fastdiv_values(ne02 * ne01); + + uint32_t row_start = 0; + uint32_t nrows = total_rows; + + if (octx->ctx->mdev.count > 1) { + const uint32_t rows_per_chunk = (row_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(row_size, HEX_L2_LINE_SIZE)) : 1; + const bool can_split = htp_tensor_mdev_data_aligned(dst) && dst_is_contiguous; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, can_split ? rows_per_chunk : 0, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } + + ct.row_start = row_start; + ct.nrows = nrows; + ct.src0_nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); + + if (sametype && (octx->ctx->mdev.count <= 1 || htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst))) { + if (octx->ctx->mdev.idx == 0) { + *use_dma = true; + cpy_dma_sametype_sameshape(octx, dst, src0, ct.src0_type_size, ne00, ne01, ne02, ne03, nb01, nb02, nb03, nb1, nb2, nb3); + } } else { - copy_fun = cpy_thread_f16_reshape; + work_queue_func_t copy_fun = NULL; + if (sametype) { + copy_fun = (src0->type == HTP_TYPE_F32) ? cpy_thread_f32_sameshape : cpy_thread_f16_sameshape; + } else if (dst->type == HTP_TYPE_F16 && src0->type == HTP_TYPE_F32) { + copy_fun = cpy_thread_f16_f32_sameshape; + } else if (dst->type == HTP_TYPE_F32 && src0->type == HTP_TYPE_F16) { + copy_fun = cpy_thread_f32_f16_sameshape; + } else { + return HTP_STATUS_NO_SUPPORT; + } + work_queue_run(octx->ctx->work_queue, copy_fun, &ct, n_threads); + } + } else if (sametype) { + const uint32_t total_elems = ne0 * ne1 * ne2 * ne3; + const uint32_t elems_per_line = (ct.dst_type_size == 4) ? 32 : 64; + + if (octx->ctx->mdev.count <= 1 && dst_is_contiguous && src_is_contiguous) { + *use_dma = true; + cpy_dma_sametype_reshape_contig(octx->ctx->dma[0], dst->data, src0->data, total_elems * ct.dst_type_size); + return HTP_STATUS_OK; + } + + ct.div_ne0 = init_fastdiv_values(ne0); + ct.div_ne1_ne0 = init_fastdiv_values(ne1 * ne0); + ct.div_ne2_ne1_ne0 = init_fastdiv_values(ne2 * ne1 * ne0); + ct.div_ne00 = init_fastdiv_values(ne00); + ct.div_ne01_ne00 = init_fastdiv_values(ne01 * ne00); + ct.div_ne02_ne01_ne00 = init_fastdiv_values(ne02 * ne01 * ne00); + + uint32_t elem_start = 0; + uint32_t nelem = total_elems; + + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && dst_is_contiguous; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_elems, can_split ? elems_per_line : 0, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + elem_start = range.start; + nelem = range.count; } + + if (nelem == 0) { + return HTP_STATUS_OK; + } + + ct.elem_start = elem_start; + ct.nelem = nelem; + ct.elem_per_thread = fastdiv(nelem + n_threads - 1, &octx->n_threads_div); + + work_queue_func_t copy_fun = (src0->type == HTP_TYPE_F32) ? cpy_thread_f32_reshape : cpy_thread_f16_reshape; + work_queue_run(octx->ctx->work_queue, copy_fun, &ct, n_threads); } else { return HTP_STATUS_NO_SUPPORT; } - worker_pool_run_func(octx->ctx->worker_pool, copy_fun, &ct, n_threads); - return HTP_STATUS_OK; } + +int op_cpy(struct htp_ops_context * octx) { + bool use_dma = false; + int status = exec_cpy(octx, &use_dma); + + htp_ops_context_set_status(octx, status); + + if (octx->op == HTP_OP_CPY_FENCE) { + if (!use_dma) { + htp_flush_dirty_ranges(octx->ctx); + } + + htp_mdev_group_barrier(octx); + + if (octx->ctx->mdev.idx == 0) { + const struct htp_tensor * sync = octx->src[1]; + if (htp_tensor_is_extended(sync)) { + return HTP_STATUS_NO_SUPPORT; + } + const uint32_t seq = (uint32_t) octx->op_params[0]; + atomic_uint * sync_fence = (atomic_uint *) (uintptr_t) sync->data; + htp_fence_write(sync_fence, seq, octx->status); + + FARF(HIGH, "ggml-hex: sync-release : fence %p seq 0x%x status %d\n", sync_fence, seq, octx->status); + } + } + + return octx->status; +} diff --git a/ggml/src/ggml-hexagon/htp/cumsum-ops.c b/ggml/src/ggml-hexagon/htp/cumsum-ops.c index 2d45c39f..eaab7d7e 100644 --- a/ggml/src/ggml-hexagon/htp/cumsum-ops.c +++ b/ggml/src/ggml-hexagon/htp/cumsum-ops.c @@ -7,35 +7,37 @@ #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ctx.h" #include "htp-ops.h" #include "htp-tensor.h" #include "hvx-types.h" #include "hvx-utils.h" -#include "hex-dma.h" +#include "dma-queue.h" #define htp_cumsum_tensors_preamble \ const struct htp_tensor * restrict src0 = octx->src[0]; \ const struct htp_tensor * restrict dst = octx->dst; \ - \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne02 = src0->ne[2]; \ - const uint32_t ne03 = src0->ne[3]; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb00 = src0->nb[0]; \ - const uint32_t nb01 = src0->nb[1]; \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t nb0 = dst->nb[0]; \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ + \ + const uint32_t ne00 = src0->ne[0]; \ + const uint32_t ne01 = src0->ne[1]; \ + const uint32_t ne02 = src0->ne[2]; \ + const uint32_t ne03 = src0->ne[3]; \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + const uint32_t ne2 = dst->ne[2]; \ + const uint32_t ne3 = dst->ne[3]; \ + \ + const uint32_t nb00 = src0->nb[0]; \ + const uint32_t nb01 = src0->nb[1]; \ + const uint32_t nb02 = src0->nb[2]; \ + const uint32_t nb03 = src0->nb[3]; \ + \ + const uint32_t nb0 = dst->nb[0]; \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ const uint32_t nb3 = dst->nb[3]; struct htp_cumsum_context { @@ -46,13 +48,14 @@ struct htp_cumsum_context { size_t dst_row_size_aligned; uint32_t rows_per_thread; uint32_t total_rows; + uint32_t row_start; }; #define htp_cumsum_preamble \ struct htp_cumsum_context * cctx = (struct htp_cumsum_context *) data; \ struct htp_ops_context * octx = cctx->octx; \ htp_cumsum_tensors_preamble; \ - dma_queue * dma_queue = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; // --------------------------------------------------------------------------- // HVX prefix scan helpers @@ -116,11 +119,8 @@ static inline void hvx_cumsum_row_f32(const float * restrict src, float * restri static void cumsum_thread_f32_dma(unsigned int nth, unsigned int ith, void * data) { htp_cumsum_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - - const uint32_t ir0 = cctx->rows_per_thread * ith; - const uint32_t ir1 = MIN(ir0 + cctx->rows_per_thread, cctx->total_rows); + const uint32_t ir0 = cctx->row_start + cctx->rows_per_thread * ith; + const uint32_t ir1 = MIN(ir0 + cctx->rows_per_thread, cctx->row_start + cctx->total_rows); if (ir0 >= ir1) { return; @@ -131,49 +131,51 @@ static void cumsum_thread_f32_dma(unsigned int nth, unsigned int ith, void * dat const size_t src_row_size_aligned = cctx->src_row_size_aligned; const size_t dst_row_size_aligned = cctx->dst_row_size_aligned; - const uint8_t * src_data = (const uint8_t *) src0->data; - uint8_t * dst_data = (uint8_t *) dst->data; + const dma_addr_t src_data = src0->data; + const dma_addr_t dst_data = dst->data; uint8_t * src_spad = octx->src0_spad.data + (ith * src_row_size_aligned * 2); uint8_t * dst_spad = octx->dst_spad.data + (ith * dst_row_size_aligned * 2); for (uint32_t ir = ir0, spad_idx = 0; ir < ir1 && spad_idx < 2; ir++, spad_idx++) { // Dummy dst writeback to establish queue ordering - dma_queue_push_vtcm_to_ddr(dma_queue, - dma_make_ptr(dst_data, dst_spad + (spad_idx * dst_row_size_aligned)), - dst_row_size, dst_row_size_aligned, 0); - - dma_queue_push_ddr_to_vtcm(dma_queue, - dma_make_ptr(src_spad + (spad_idx * src_row_size_aligned), - src_data + (ir * src_row_size)), - src_row_size_aligned, src_row_size, 1); + dma_queue_push(dma_q, + dma_make_data(dst_data, dst_spad + (spad_idx * dst_row_size_aligned)), + dst_row_size, dst_row_size_aligned, dst_row_size, 0); + + dma_queue_push(dma_q, + dma_make_data(src_spad + (spad_idx * src_row_size_aligned), + src_data + (ir * src_row_size)), + src_row_size_aligned, src_row_size, src_row_size, 1); } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t ir = ir0; ir < ir1; ir++) { - float * dst_spad_row = (float *) dma_queue_pop(dma_queue).src; - float * src_spad_row = (float *) dma_queue_pop(dma_queue).dst; + float * dst_spad_row = (float *) dma_queue_pop(dma_q).src; + float * src_spad_row = (float *) dma_queue_pop(dma_q).dst; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); hvx_cumsum_row_f32(src_spad_row, dst_spad_row, ne00); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - dma_queue_push_vtcm_to_ddr(dma_queue, - dma_make_ptr(dst_data + (ir * dst_row_size), (uint8_t *) dst_spad_row), - dst_row_size, dst_row_size_aligned, 1); + dma_queue_push(dma_q, + dma_make_data(dst_data + (ir * dst_row_size), dst_spad_row), + dst_row_size, dst_row_size_aligned, dst_row_size, 1); const uint32_t next_row = ir + 2; if (next_row < ir1) { - dma_queue_push_ddr_to_vtcm(dma_queue, - dma_make_ptr((uint8_t *) src_spad_row, src_data + (next_row * src_row_size)), - src_row_size_aligned, src_row_size, 1); + dma_queue_push(dma_q, + dma_make_data(src_spad_row, src_data + (next_row * src_row_size)), + src_row_size_aligned, src_row_size, src_row_size, 1); } } - dma_queue_flush(dma_queue); - t2 = HAP_perf_get_qtimer_count(); + dma_queue_flush(dma_q); - FARF(HIGH, "cumsum-f32-dma %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u usec %u\n", + FARF(HIGH, "cumsum-f32-dma %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u\n", ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ir0, ir1, - dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]); } // --------------------------------------------------------------------------- @@ -183,14 +185,14 @@ static void cumsum_thread_f32_dma(unsigned int nth, unsigned int ith, void * dat static void cumsum_thread_f32(unsigned int nth, unsigned int ith, void * data) { htp_cumsum_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - const uint8_t * src_data = (const uint8_t *) src0->data; uint8_t * dst_data = (uint8_t *) dst->data; - const uint32_t ir0 = cctx->rows_per_thread * ith; - const uint32_t ir1 = MIN(ir0 + cctx->rows_per_thread, cctx->total_rows); + const uint32_t ir0 = cctx->row_start + cctx->rows_per_thread * ith; + const uint32_t ir1 = MIN(ir0 + cctx->rows_per_thread, cctx->row_start + cctx->total_rows); + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); for (uint32_t ir = ir0; ir < ir1; ir++) { const float * restrict src_row = (const float *) (src_data + ir * cctx->src_row_size); @@ -198,24 +200,36 @@ static void cumsum_thread_f32(unsigned int nth, unsigned int ith, void * data) { hvx_cumsum_row_f32(src_row, dst_row, ne00); } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); - FARF(HIGH, "cumsum-f32 %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u usec %u\n", + FARF(HIGH, "cumsum-f32 %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u\n", ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ir0, ir1, - dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]); } int op_cumsum_f32(struct htp_ops_context * octx) { const struct htp_tensor * src0 = octx->src[0]; const struct htp_tensor * dst = octx->dst; - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + const uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const size_t dst_data_row_size = dst->ne[0] * sizeof(float); + + uint32_t row_start = 0; + uint32_t nrows = total_rows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(float), (uint32_t) dst_data_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { return HTP_STATUS_OK; } - const uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, total_rows); + const uint32_t n_threads = octx->n_threads; const size_t src_row_size = src0->nb[1]; const size_t dst_row_size = dst->nb[1]; @@ -240,14 +254,18 @@ int op_cumsum_f32(struct htp_ops_context * octx) { .dst_row_size = dst_row_size, .src_row_size_aligned = src_row_size_aligned, .dst_row_size_aligned = dst_row_size_aligned, - .rows_per_thread = (total_rows + n_threads - 1) / n_threads, - .total_rows = total_rows, + .rows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .total_rows = nrows, + .row_start = row_start, }; if (octx->ctx->vtcm_size < spad_per_thread * n_threads) { - worker_pool_run_func(octx->ctx->worker_pool, cumsum_thread_f32, &cctx, n_threads); + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + work_queue_run(octx->ctx->work_queue, cumsum_thread_f32, &cctx, n_threads); } else { - worker_pool_run_func(octx->ctx->worker_pool, cumsum_thread_f32_dma, &cctx, n_threads); + work_queue_run(octx->ctx->work_queue, cumsum_thread_f32_dma, &cctx, n_threads); } return HTP_STATUS_OK; diff --git a/ggml/src/ggml-hexagon/htp/diag-ops.c b/ggml/src/ggml-hexagon/htp/diag-ops.c index 9b3194d9..162214d3 100644 --- a/ggml/src/ggml-hexagon/htp/diag-ops.c +++ b/ggml/src/ggml-hexagon/htp/diag-ops.c @@ -5,27 +5,30 @@ #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" #include "hvx-types.h" #include "hex-utils.h" #include "hvx-copy.h" -#include "hex-dma.h" +#include "dma-queue.h" #define htp_diag_tensors_preamble \ const struct htp_tensor * restrict src0 = octx->src[0]; \ const struct htp_tensor * restrict dst = octx->dst; \ - \ - const uint32_t ne02 = src0->ne[2]; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ + \ + const uint32_t ne02 = src0->ne[2]; \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + \ + const uint32_t nb02 = src0->nb[2]; \ + const uint32_t nb03 = src0->nb[3]; \ + \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ const uint32_t nb3 = dst->nb[3]; struct htp_diag_context { @@ -36,6 +39,7 @@ struct htp_diag_context { size_t dst_row_size_aligned; uint32_t batches_per_thread; uint32_t total_batches; + uint32_t batch_start; }; #define htp_diag_preamble \ @@ -55,13 +59,10 @@ static inline void hvx_diag_row_f32(const float * restrict src, float * restrict static void diag_thread_f32_dma(unsigned int nth, unsigned int ith, void * data) { htp_diag_preamble; - dma_queue * dma_queue = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - - const uint32_t ib0 = dctx->batches_per_thread * ith; - const uint32_t ib1 = MIN(ib0 + dctx->batches_per_thread, dctx->total_batches); + const uint32_t ib0 = dctx->batch_start + dctx->batches_per_thread * ith; + const uint32_t ib1 = MIN(ib0 + dctx->batches_per_thread, dctx->batch_start + dctx->total_batches); if (ib0 >= ib1) { return; @@ -72,47 +73,48 @@ static void diag_thread_f32_dma(unsigned int nth, unsigned int ith, void * data) const size_t src_batch_size_aligned = dctx->src_batch_size_aligned; const size_t dst_row_size_aligned = dctx->dst_row_size_aligned; - const uint8_t * src_data = (const uint8_t *) src0->data; - uint8_t * dst_data = (uint8_t *) dst->data; + const dma_addr_t src_data = src0->data; + const dma_addr_t dst_data = dst->data; // 1 src buffer + 1 dst row buffer per thread in VTCM uint8_t * src_spad = octx->src0_spad.data + (ith * src_batch_size_aligned); uint8_t * dst_spad = octx->dst_spad.data + (ith * dst_row_size_aligned); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t ib = ib0; ib < ib1; ib++) { const uint32_t i3 = ib / ne02; const uint32_t i2 = ib % ne02; - const uint8_t * src_batch = src_data + i3 * nb03 + i2 * nb02; + const dma_addr_t src_batch = src_data + i3 * nb03 + i2 * nb02; // Fetch source vector into VTCM - dma_queue_push_ddr_to_vtcm(dma_queue, - dma_make_ptr(src_spad, src_batch), - src_batch_size_aligned, src_batch_size, 1); - dma_queue_flush(dma_queue); + dma_queue_push(dma_q, + dma_make_data(src_spad, src_batch), + src_batch_size_aligned, src_batch_size, src_batch_size, 1); + dma_queue_flush(dma_q); const float * src_spad_f32 = (const float *) src_spad; float * dst_spad_f32 = (float *) dst_spad; for (uint32_t i1 = 0; i1 < ne1; i1++) { // Compute row in VTCM + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) (ib * ne1 + i1)); hvx_diag_row_f32(src_spad_f32, dst_spad_f32, i1, ne0); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) (ib * ne1 + i1)); // Write completed row back to DDR - uint8_t * dst_row = dst_data + i3 * nb3 + i2 * nb2 + i1 * nb1; - dma_queue_push_vtcm_to_ddr(dma_queue, - dma_make_ptr(dst_row, dst_spad), - dst_row_size, dst_row_size_aligned, 1); - dma_queue_flush(dma_queue); + const dma_addr_t dst_row = dst_data + i3 * nb3 + i2 * nb2 + i1 * nb1; + dma_queue_push(dma_q, + dma_make_data(dst_row, dst_spad), + dst_row_size, dst_row_size_aligned, dst_row_size, 1); + dma_queue_flush(dma_q); } } - t2 = HAP_perf_get_qtimer_count(); - - FARF(HIGH, "diag-f32-dma %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u usec %u\n", + FARF(HIGH, "diag-f32-dma %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u\n", ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ib0, ib1, - dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]); } // --------------------------------------------------------------------------- @@ -122,14 +124,14 @@ static void diag_thread_f32_dma(unsigned int nth, unsigned int ith, void * data) static void diag_thread_f32(unsigned int nth, unsigned int ith, void * data) { htp_diag_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - const uint8_t * src_data = (const uint8_t *) src0->data; uint8_t * dst_data = (uint8_t *) dst->data; - const uint32_t ib0 = dctx->batches_per_thread * ith; - const uint32_t ib1 = MIN(ib0 + dctx->batches_per_thread, dctx->total_batches); + const uint32_t ib0 = dctx->batch_start + dctx->batches_per_thread * ith; + const uint32_t ib1 = MIN(ib0 + dctx->batches_per_thread, dctx->batch_start + dctx->total_batches); + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ib0); for (uint32_t ib = ib0; ib < ib1; ib++) { const uint32_t i3 = ib / ne02; @@ -143,24 +145,48 @@ static void diag_thread_f32(unsigned int nth, unsigned int ith, void * data) { } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ib0); - FARF(HIGH, "diag-f32 %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u usec %u\n", + FARF(HIGH, "diag-f32 %d/%d: %ux%ux%ux%u (%u:%u) -> %ux%ux%ux%u\n", ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ib0, ib1, - dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]); } int op_diag_f32(struct htp_ops_context * octx) { const struct htp_tensor * src0 = octx->src[0]; const struct htp_tensor * dst = octx->dst; - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + const uint32_t total_batches = src0->ne[2] * src0->ne[3]; + const size_t dst_batch_size = dst->ne[1] * dst->nb[1]; + + uint32_t batch_start = 0; + uint32_t nbatches = total_batches; + + if (octx->ctx->mdev.count > 1) { + bool can_split = htp_tensor_mdev_data_aligned(dst) && (dst->ne[0] == 1 || dst->nb[0] == sizeof(float)) && !htp_tensor_is_permuted(dst); + uint32_t batches_per_chunk = 1; + if (can_split) { + if (dst->ne[2] > 1 && (dst->nb[2] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0 && + (dst->ne[3] <= 1 || (dst->nb[3] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0)) { + batches_per_chunk = 1; + } else if (dst->nb[2] == dst_batch_size && + (dst->ne[3] <= 1 || dst->nb[3] == dst->nb[2] * dst->ne[2])) { + batches_per_chunk = (dst_batch_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(dst_batch_size, HEX_L2_LINE_SIZE)) : 1; + } else { + can_split = false; + } + } + + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_batches, can_split ? batches_per_chunk : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + batch_start = range.start; + nbatches = range.count; + } + + if (nbatches == 0) { return HTP_STATUS_OK; } - const uint32_t total_batches = src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, total_batches); + const uint32_t n_threads = octx->n_threads; const size_t src_batch_size = src0->ne[0] * sizeof(float); const size_t dst_row_size = dst->ne[0] * sizeof(float); @@ -185,14 +211,18 @@ int op_diag_f32(struct htp_ops_context * octx) { .dst_row_size = dst_row_size, .src_batch_size_aligned = src_batch_size_aligned, .dst_row_size_aligned = dst_row_size_aligned, - .batches_per_thread = (total_batches + n_threads - 1) / n_threads, - .total_batches = total_batches, + .batches_per_thread = fastdiv(nbatches + n_threads - 1, &octx->n_threads_div), + .total_batches = nbatches, + .batch_start = batch_start, }; if (octx->ctx->vtcm_size < spad_per_thread * n_threads) { - worker_pool_run_func(octx->ctx->worker_pool, diag_thread_f32, &dctx, n_threads); + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + work_queue_run(octx->ctx->work_queue, diag_thread_f32, &dctx, n_threads); } else { - worker_pool_run_func(octx->ctx->worker_pool, diag_thread_f32_dma, &dctx, n_threads); + work_queue_run(octx->ctx->work_queue, diag_thread_f32_dma, &dctx, n_threads); } return HTP_STATUS_OK; diff --git a/ggml/src/ggml-hexagon/htp/dma-queue.c b/ggml/src/ggml-hexagon/htp/dma-queue.c index 4beded1d..464e4b84 100644 --- a/ggml/src/ggml-hexagon/htp/dma-queue.c +++ b/ggml/src/ggml-hexagon/htp/dma-queue.c @@ -22,58 +22,69 @@ static inline uintptr_t align_up(uintptr_t addr, size_t align) { return (addr + align - 1) & ~(align - 1); } -size_t dma_queue_sizeof(size_t capacity) { +static inline size_t dma_ring_sizeof(size_t capacity) { capacity = pow2_ceil(capacity); - size_t size_q = sizeof(dma_queue); - size_t offset_r = align_up(size_q, HEX_L2_LINE_SIZE); size_t size_r = sizeof(dma_ring); - size_t offset_desc = align_up(offset_r + size_r, HEX_L2_LINE_SIZE); + size_t offset_desc = align_up(size_r, HEX_L2_LINE_SIZE); size_t size_desc = capacity * sizeof(dma_descriptor_2d); - size_t offset_dptr = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE); - size_t size_dptr = capacity * sizeof(dma_ptr); + size_t offset_data = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE); + size_t size_data = capacity * sizeof(dma_data); - return offset_dptr + size_dptr; + return offset_data + size_data; } -size_t dma_queue_alignof(void) { - return HEX_L2_LINE_SIZE; -} - -dma_queue_t dma_queue_init(void * ptr, size_t capacity, uintptr_t vtcm_base, size_t vtcm_size, struct htp_thread_trace * trace) { +static inline dma_ring * dma_ring_init(void * ptr, size_t capacity, struct htp_thread_trace * trace) { capacity = pow2_ceil(capacity); - size_t size_q = sizeof(dma_queue); - size_t offset_r = align_up(size_q, HEX_L2_LINE_SIZE); size_t size_r = sizeof(dma_ring); - size_t offset_desc = align_up(offset_r + size_r, HEX_L2_LINE_SIZE); + size_t offset_desc = align_up(size_r, HEX_L2_LINE_SIZE); size_t size_desc = capacity * sizeof(dma_descriptor_2d); - size_t offset_dptr = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE); - size_t size_dptr = capacity * sizeof(dma_ptr); - - size_t total_size = offset_dptr + size_dptr; - memset(ptr, 0, total_size); - - dma_queue * q = (dma_queue *) ptr; - dma_ring * r = (dma_ring *) ((uintptr_t) ptr + offset_r); - - q->ring = r; - q->nocache = 0; - q->alias = false; + size_t offset_data = align_up(offset_desc + size_desc, HEX_L2_LINE_SIZE); + dma_ring * r = (dma_ring *) ptr; r->trace = trace; - r->vtcm_base = vtcm_base; - r->vtcm_end = vtcm_base + vtcm_size; r->capacity = capacity; r->idx_mask = capacity - 1; r->push_idx = 0; r->pop_idx = 0; + r->desc = (dma_descriptor_2d *) ((uintptr_t) ptr + offset_desc); + r->data = (dma_data *) ((uintptr_t) ptr + offset_data); + r->tail = &r->desc[capacity - 1]; + + return r; +} + +size_t dma_queue_sizeof(size_t capacity) { + size_t size_q = sizeof(dma_queue); + size_t offset_r0 = align_up(size_q, HEX_L2_LINE_SIZE); + size_t size_r0 = dma_ring_sizeof(capacity); + size_t offset_r1 = align_up(offset_r0 + size_r0, HEX_L2_LINE_SIZE); + size_t size_r1 = dma_ring_sizeof(DMA_FALLBACK_CAPACITY); + + return offset_r1 + size_r1; +} + +size_t dma_queue_alignof(void) { + return HEX_L2_LINE_SIZE; +} + +dma_queue_t dma_queue_init(void * ptr, size_t capacity, struct htp_thread_trace * trace) { + size_t total_size = dma_queue_sizeof(capacity); + memset(ptr, 0, total_size); + + dma_queue * q = (dma_queue *) ptr; + + size_t size_q = sizeof(dma_queue); + size_t offset_r0 = align_up(size_q, HEX_L2_LINE_SIZE); + size_t size_r0 = dma_ring_sizeof(capacity); + size_t offset_r1 = align_up(offset_r0 + size_r0, HEX_L2_LINE_SIZE); - r->desc = (dma_descriptor_2d *) ((uintptr_t) ptr + offset_desc); - r->dptr = (dma_ptr *) ((uintptr_t) ptr + offset_dptr); - r->tail = &r->desc[capacity - 1]; + q->ring0 = dma_ring_init((void *) ((uintptr_t) ptr + offset_r0), capacity, trace); + q->ring1 = dma_ring_init((void *) ((uintptr_t) ptr + offset_r1), DMA_FALLBACK_CAPACITY, trace); + q->alias = false; - FARF(HIGH, "dma-queue: capacity %u, unified memory size %zu\n", capacity, total_size); + FARF(HIGH, "dma-queue: capacity %u, unified memory size %zu\n", (unsigned) capacity, total_size); return q; } @@ -86,13 +97,13 @@ size_t dma_queue_alias_sizeof(void) { return sizeof(dma_queue); } -dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q, uint8_t nocache) { +dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q) { dma_queue * q = (dma_queue *) ptr; memset(q, 0, sizeof(dma_queue)); - q->ring = main_q->ring; - q->nocache = nocache; - q->alias = true; + q->ring0 = main_q->ring0; + q->ring1 = main_q->ring1; + q->alias = true; return q; } @@ -101,4 +112,101 @@ void dma_queue_alias_free(dma_queue_t q) { (void) q; } +bool dma_queue_push_fallback_2d(dma_queue * q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { + dma_ring * r0 = q->ring0; + dma_ring * r1 = q->ring1; + + if (((r0->push_idx + 1) & r0->idx_mask) == r0->pop_idx) { + return false; + } + + r1->tail = r0->tail; + + size_t rem_rows = nrows; + dma_addr_t cur_dst = ddata.dst; + dma_addr_t cur_src = ddata.src; + + while (rem_rows > 0) { + const uint32_t cur_rows = MIN(rem_rows, DMA_MAX_NROWS); + dma_data cur_data = dma_make_data(cur_dst, cur_src); + if (!dma_ring_push_single_2d(r1, cur_data, dst_stride, src_stride, row_size, cur_rows)) { + dma_ring_flush(r1); + dma_ring_push_single_2d(r1, cur_data, dst_stride, src_stride, row_size, cur_rows); + } + cur_dst += cur_rows * dst_stride; + cur_src += cur_rows * src_stride; + rem_rows -= cur_rows; + } + + dma_ring_flush(r1); + r0->tail = r1->tail; + + return dma_ring_push_single_2d(r0, ddata, 0, 0, 0, /*nrows=*/ 0); +} + +bool dma_queue_push_fallback_contig(dma_queue * q, dma_data ddata, size_t total) { + dma_ring * r0 = q->ring0; + dma_ring * r1 = q->ring1; + + if (((r0->push_idx + 1) & r0->idx_mask) == r0->pop_idx) { + return false; + } + + r1->tail = r0->tail; + + size_t rem_bytes = total; + dma_addr_t cur_dst = ddata.dst; + dma_addr_t cur_src = ddata.src; + + while (rem_bytes > 0) { + const uint32_t cur_bytes = MIN(rem_bytes, DMA_SAFE_CHUNK_SIZE); + dma_data cur_data = dma_make_data(cur_dst, cur_src); + if (!dma_ring_push_single_1d(r1, cur_data, cur_bytes)) { + dma_ring_flush(r1); + dma_ring_push_single_1d(r1, cur_data, cur_bytes); + } + cur_dst += cur_bytes; + cur_src += cur_bytes; + rem_bytes -= cur_bytes; + } + + dma_ring_flush(r1); + r0->tail = r1->tail; + + return dma_ring_push_single_1d(r0, ddata, /*size=*/ 0); +} + +#if __HVX_ARCH__ < 75 + +bool dma_queue_push_fallback_1d(dma_queue * q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { + dma_ring * r0 = q->ring0; + dma_ring * r1 = q->ring1; + + if (((r0->push_idx + 1) & r0->idx_mask) == r0->pop_idx) { + return false; + } + + r1->tail = r0->tail; + + size_t rem_rows = nrows; + dma_addr_t cur_dst = ddata.dst; + dma_addr_t cur_src = ddata.src; + + while (rem_rows > 0) { + dma_data cur_data = dma_make_data(cur_dst, cur_src); + if (!dma_ring_push_single_1d(r1, cur_data, row_size)) { + dma_ring_flush(r1); + dma_ring_push_single_1d(r1, cur_data, row_size); + } + cur_dst += dst_stride; + cur_src += src_stride; + rem_rows -= 1; + } + + dma_ring_flush(r1); + r0->tail = r1->tail; + + return dma_ring_push_single_1d(r0, ddata, /*size=*/ 0); +} +#endif diff --git a/ggml/src/ggml-hexagon/htp/dma-queue.h b/ggml/src/ggml-hexagon/htp/dma-queue.h index 264284bd..c77359f8 100644 --- a/ggml/src/ggml-hexagon/htp/dma-queue.h +++ b/ggml/src/ggml-hexagon/htp/dma-queue.h @@ -3,8 +3,10 @@ #include #include +#include #include #include +#include #include "hex-utils.h" #include "hex-profile.h" @@ -24,8 +26,8 @@ typedef struct dma_descriptor_1d_s { uint32_t src_bypass:1; uint32_t order:1; uint32_t done:1; - void * src; - void * dst; + uint32_t src; + uint32_t dst; } dma_descriptor_1d; #if __HVX_ARCH__ < 75 @@ -40,8 +42,8 @@ typedef struct dma_descriptor_2d_s { uint32_t src_bypass:1; uint32_t order:1; uint32_t done:1; - void * src; - void * dst; + uint32_t src; + uint32_t dst; uint32_t desc_type:8; uint32_t reserved1:24; uint32_t row_size:16; @@ -64,10 +66,18 @@ typedef struct dma_descriptor_2d_s { uint32_t src_bypass:1; uint32_t order:1; uint32_t done:1; - void * src; - void * dst; + uint32_t src; + uint32_t dst; uint32_t desc_type:8; +#if __HVX_ARCH__ > 79 + uint32_t src_upper:8; + uint32_t dst_upper:8; + uint32_t allocation:2; + uint32_t reserved0:2; + uint32_t transform:4; +#else uint32_t reserved0:24; +#endif uint32_t row_size:24; uint32_t nrows_lo:8; uint32_t nrows_hi:8; @@ -78,45 +88,63 @@ typedef struct dma_descriptor_2d_s { #endif +#if __HVX_ARCH__ > 79 +typedef uint64_t dma_addr_t; +#else +typedef uint32_t dma_addr_t; +#endif + typedef struct { - void *dst; - const void *src; -} dma_ptr; + dma_addr_t dst; + dma_addr_t src; +} dma_data; + +// Hardware descriptor field limits +#define DMA_MAX_NROWS 0xFFFFu // 16-bit HW descriptor limit (65535) +#define DMA_MAX_SIZE_16B 0xFFFFu // 16-bit HW descriptor limit for row_size (65535) +#define DMA_MAX_STRIDE_16B 0xFFFFu // 16-bit HW descriptor limit for strides (65535) +#define DMA_MAX_SIZE_24B 0x00FFFFFFu // 24-bit HW descriptor limit for row_size / 1D size (16MB - 1) +#define DMA_MAX_STRIDE_24B 0x00FFFFFFu // 24-bit HW descriptor limit for strides (16MB - 1) +#define DMA_SAFE_CHUNK_SIZE 0x00F00000u // ~15MB safe contiguous chunk size + +#define DMA_FALLBACK_CAPACITY 16u // descriptors in secondary fallback ring typedef struct dma_ring_s dma_ring; struct dma_ring_s { dma_descriptor_2d * desc; // descriptor pointers dma_descriptor_2d * tail; // tail pointer - dma_ptr * dptr; // dst/src pointers + dma_data * data; // dst/src data uint32_t push_idx; uint32_t pop_idx; uint32_t capacity; uint32_t idx_mask; struct htp_thread_trace * trace; - uintptr_t vtcm_base; - uintptr_t vtcm_end; }; typedef struct dma_queue_s dma_queue; typedef dma_queue * dma_queue_t; struct dma_queue_s { - dma_ring * ring; // Points to the descriptor ring state - uint8_t nocache; // Queue-specific bypass flag + dma_ring * ring0; // Main descriptor ring state + dma_ring * ring1; // Secondary fallback descriptor ring state bool alias; // When set, dma_queue_delete will not free the ring }; - - size_t dma_queue_sizeof(size_t capacity); size_t dma_queue_alignof(void); -dma_queue_t dma_queue_init(void * ptr, size_t capacity, uintptr_t vtcm_base, size_t vtcm_size, struct htp_thread_trace * trace); +dma_queue_t dma_queue_init(void * ptr, size_t capacity, struct htp_thread_trace * trace); void dma_queue_free(dma_queue_t q); size_t dma_queue_alias_sizeof(void); -dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q, uint8_t nocache); +dma_queue_t dma_queue_alias_init(void * ptr, dma_queue_t main_q); void dma_queue_alias_free(dma_queue_t q); +bool dma_queue_push_fallback_2d(dma_queue * q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows); +bool dma_queue_push_fallback_contig(dma_queue * q, dma_data ddata, size_t total); +#if __HVX_ARCH__ < 75 +bool dma_queue_push_fallback_1d(dma_queue * q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows); +#endif + // TODO: technically we don't need these and could use Q6_dmstart/wait/etc instead // but those do not seem to always compiler properly. static inline void dmstart(void * next) { @@ -141,36 +169,37 @@ static inline unsigned int dmwait(void) { return ret; } -static inline dma_ptr dma_make_ptr(void *dst, const void *src) +static inline dma_data dma_make_data_impl(dma_addr_t dst, dma_addr_t src) { - dma_ptr p = { dst, src }; - return p; + dma_data d = { dst, src }; + return d; } -static inline bool dma_is_vtcm(const dma_queue * q, const void * ptr) { - return (uintptr_t) ptr >= q->ring->vtcm_base && (uintptr_t) ptr < q->ring->vtcm_end; -} +#define dma_make_data(dst, src) dma_make_data_impl((dma_addr_t) (dst), (dma_addr_t) (src)) + +static inline bool dma_ring_push_single_1d(dma_ring * r, dma_data ddata, size_t size) { +#if __HVX_ARCH__ > 79 + assert(!((ddata.src | ddata.dst) >> 32) || size == 0); +#endif -static inline bool dma_queue_push_single_1d(dma_queue * q, dma_ptr dptr, size_t size) { - dma_ring * r = q->ring; if (((r->push_idx + 1) & r->idx_mask) == r->pop_idx) { return false; } dma_descriptor_1d * desc = (dma_descriptor_1d *) &r->desc[r->push_idx]; - desc->src = (void *) dptr.src; - desc->dst = (void *) dptr.dst; + desc->src = (uint32_t) ddata.src; + desc->dst = (uint32_t) ddata.dst; desc->size = size; - r->dptr[r->push_idx] = dptr; + r->data[r->push_idx] = ddata; htp_trace_event_start(r->trace, HTP_TRACE_EVT_DMA, r->push_idx); if (size) { desc->next = NULL; desc->desc_size = 0; // 1D mode - desc->src_bypass = dma_is_vtcm(q, dptr.src) ? 1 : q->nocache; - desc->dst_bypass = dma_is_vtcm(q, dptr.dst) ? 1 : q->nocache; + desc->src_bypass = 1; + desc->dst_bypass = 1; desc->order = 0; desc->done = 0; @@ -185,8 +214,17 @@ static inline bool dma_queue_push_single_1d(dma_queue * q, dma_ptr dptr, size_t return true; } -static inline bool dma_queue_push_single_2d(dma_queue * q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { - dma_ring * r = q->ring; +static inline bool dma_ring_push_single_2d(dma_ring * r, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { +#if __HVX_ARCH__ > 79 + const uint32_t src_hi = (uint32_t) (ddata.src >> 32); + const uint32_t dst_hi = (uint32_t) (ddata.dst >> 32); + const bool is_ext = (src_hi | dst_hi) != 0; + + if (is_ext && ((ddata.src >> 40) || (ddata.dst >> 40))) { + return false; + } +#endif + if (((r->push_idx + 1) & r->idx_mask) == r->pop_idx) { return false; } @@ -194,34 +232,44 @@ static inline bool dma_queue_push_single_2d(dma_queue * q, dma_ptr dptr, size_t dma_descriptor_2d * desc = &r->desc[r->push_idx]; desc->next = NULL; - desc->reserved0 = 0; desc->reserved1 = 0; desc->desc_size = 1; // 2d mode - desc->src_bypass = dma_is_vtcm(q, dptr.src) ? 1 : q->nocache; - desc->dst_bypass = dma_is_vtcm(q, dptr.dst) ? 1 : q->nocache; + desc->src_bypass = 1; + desc->dst_bypass = 1; desc->src_comp = 0; desc->dst_comp = 0; desc->order = 0; desc->done = 0; desc->src_stride = src_stride; desc->dst_stride = dst_stride; - desc->src = (void *) dptr.src; - desc->dst = (void *) dptr.dst; + desc->src = (uint32_t) ddata.src; + desc->dst = (uint32_t) ddata.dst; desc->row_size = row_size; #if __HVX_ARCH__ < 75 + desc->reserved0 = 0; desc->desc_type = 0; // 2d (16-bit) mode desc->nrows = nrows; desc->src_offset = 0; desc->dst_offset = 0; #else +#if __HVX_ARCH__ > 79 + desc->src_upper = src_hi; + desc->dst_upper = dst_hi; + desc->allocation = 0; + desc->reserved0 = 0; + desc->transform = 0; + desc->desc_type = is_ext ? 10 : 9; // 2d 40-bit or 24-bit mode +#else + desc->reserved0 = 0; desc->desc_type = 9; // 2d (24-bit) mode +#endif desc->nrows_lo = (nrows & 0xff); desc->nrows_hi = (nrows >> 8); desc->offset = 0; #endif - r->dptr[r->push_idx] = dptr; + r->data[r->push_idx] = ddata; htp_trace_event_start(r->trace, HTP_TRACE_EVT_DMA, r->push_idx); @@ -236,134 +284,160 @@ static inline bool dma_queue_push_single_2d(dma_queue * q, dma_ptr dptr, size_t return true; } -static inline dma_ptr dma_queue_pop(dma_queue * q) { - dma_ring * r = q->ring; - dma_ptr dptr = { NULL }; +static inline dma_data dma_ring_pop(dma_ring * r) { + dma_data ddata = { 0 }; if (r->push_idx == r->pop_idx) { - return dptr; + return ddata; } - dma_descriptor_2d * desc = &r->desc[r->pop_idx]; + ddata = r->data[r->pop_idx]; + + volatile dma_descriptor_2d * desc = &r->desc[r->pop_idx]; // Wait for desc to complete if (!desc->done) { + // FARF(ALWAYS, "dma-poll: idx %u dst %p src %p", r->pop_idx, ddata.dst, ddata.src); while (!desc->done) { dmpoll(); } } - dptr = r->dptr[r->pop_idx]; - htp_trace_event_stop(r->trace, HTP_TRACE_EVT_DMA, r->pop_idx); r->pop_idx = (r->pop_idx + 1) & r->idx_mask; - return dptr; + return ddata; } -static inline dma_ptr dma_queue_pop_nowait(dma_queue * q) { - dma_ring * r = q->ring; - dma_ptr dptr = { NULL }; +static inline dma_data dma_ring_pop_nowait(dma_ring * r) { + dma_data ddata = { 0 }; if (r->push_idx == r->pop_idx) { - return dptr; + return ddata; } - dptr = r->dptr[r->pop_idx]; + ddata = r->data[r->pop_idx]; htp_trace_event_stop(r->trace, HTP_TRACE_EVT_DMA, r->pop_idx); r->pop_idx = (r->pop_idx + 1) & r->idx_mask; - return dptr; + return ddata; +} + +static inline bool dma_ring_empty(dma_ring * r) { + return r->push_idx == r->pop_idx; +} + +static inline void dma_ring_flush(dma_ring * r) { + while (!dma_ring_empty(r)) { + dma_ring_pop(r); + } +} + +static inline uint32_t dma_ring_depth(dma_ring * r) { + return (r->push_idx - r->pop_idx) & r->idx_mask; +} + +static inline uint32_t dma_ring_capacity(dma_ring * r) { + return r->capacity; +} + +static inline bool dma_queue_push_single_1d(dma_queue * q, dma_data ddata, size_t size) { + return dma_ring_push_single_1d(q->ring0, ddata, size); +} + +static inline bool dma_queue_push_single_2d(dma_queue * q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { + return dma_ring_push_single_2d(q->ring0, ddata, dst_stride, src_stride, row_size, nrows); +} + +static inline dma_data dma_queue_pop(dma_queue * q) { + return dma_ring_pop(q->ring0); +} + +static inline dma_data dma_queue_pop_nowait(dma_queue * q) { + return dma_ring_pop_nowait(q->ring0); } static inline bool dma_queue_empty(dma_queue * q) { - return q->ring->push_idx == q->ring->pop_idx; + return dma_ring_empty(q->ring0); } static inline void dma_queue_flush(dma_queue * q) { - while (dma_queue_pop(q).dst != NULL) ; + dma_ring_flush(q->ring0); } static inline uint32_t dma_queue_depth(dma_queue * q) { - return (q->ring->push_idx - q->ring->pop_idx) & q->ring->idx_mask; + return dma_ring_depth(q->ring0); } static inline uint32_t dma_queue_capacity(dma_queue * q) { - return q->ring->capacity; + return dma_ring_capacity(q->ring0); } #if __HVX_ARCH__ < 75 -// Overflow-safe DMA push: all 2d descriptor fields (row_size, nrows, src_stride, dst_stride) are 16-bit, max 65535. -// This version transparently handles values that exceed the 16-bit limit and submits chained DMA transtions. - -#define DMA_MAX_FIELD_VAL 65535u - -static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { +static inline bool dma_queue_push(dma_queue *q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { // Fast path: everything fits in 16 bits if (nrows == 0 || __builtin_expect( - row_size <= DMA_MAX_FIELD_VAL && - nrows <= DMA_MAX_FIELD_VAL && - src_stride <= DMA_MAX_FIELD_VAL && - dst_stride <= DMA_MAX_FIELD_VAL, 1)) { - return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows); + nrows <= DMA_MAX_NROWS && + row_size <= DMA_MAX_SIZE_16B && + src_stride <= DMA_MAX_STRIDE_16B && + dst_stride <= DMA_MAX_STRIDE_16B, 1)) { + return dma_ring_push_single_2d(q->ring0, ddata, dst_stride, src_stride, row_size, nrows); } - // Contiguous block - // Use 1d DMA mode which supports sizes up to 24-bits (16MB) + // Contiguous block: 1D DMA mode supports up to 24-bit size (16MB) if (nrows == 1 || (row_size == src_stride && row_size == dst_stride)) { size_t total = row_size * nrows; - return dma_queue_push_single_1d(q, dptr, total); + if (total <= DMA_MAX_SIZE_24B) { + return dma_ring_push_single_1d(q->ring0, ddata, total); + } + return dma_queue_push_fallback_contig(q, ddata, total); } - // Stride overflow - fall back to row-by-row. - { - const uint8_t *src = (const uint8_t *) dptr.src; - uint8_t *dst = (uint8_t *) dptr.dst; - size_t r = 0; - while (r + 1 < nrows) { - dma_ptr p = dma_make_ptr(dst + r * dst_stride, src + r * src_stride); - if (!dma_queue_push_single_1d(q, p, row_size)) { - dma_queue_flush(q); - } else { - r++; - } - } - dma_queue_flush(q); - dma_ptr p = dma_make_ptr(dst + r * dst_stride, src + r * src_stride); - return dma_queue_push_single_1d(q, p, row_size); + // Row count overflow with 16-bit strides: chunk 2D descriptors via fallback ring + if (row_size <= DMA_MAX_SIZE_16B && src_stride <= DMA_MAX_STRIDE_16B && dst_stride <= DMA_MAX_STRIDE_16B) { + return dma_queue_push_fallback_2d(q, ddata, dst_stride, src_stride, row_size, nrows); } + + // Stride or row_size overflow: row-by-row 1D via fallback ring + return dma_queue_push_fallback_1d(q, ddata, dst_stride, src_stride, row_size, nrows); } #else // HVX_ARCH >= 75 -static inline bool dma_queue_push(dma_queue *q, dma_ptr dptr, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { - // On v75 and up we always use 2d 24-bit mode - return dma_queue_push_single_2d(q, dptr, dst_stride, src_stride, row_size, nrows); -} +static inline bool dma_queue_push(dma_queue *q, dma_data ddata, size_t dst_stride, size_t src_stride, size_t row_size, size_t nrows) { + if (nrows == 0 || __builtin_expect( + nrows <= DMA_MAX_NROWS && + row_size <= DMA_MAX_SIZE_24B && + src_stride <= DMA_MAX_STRIDE_24B && + dst_stride <= DMA_MAX_STRIDE_24B, 1)) { + return dma_ring_push_single_2d(q->ring0, ddata, dst_stride, src_stride, row_size, nrows); + } -#endif + // Contiguous block exceeding 24 bits + if (nrows == 1 || (row_size == src_stride && row_size == dst_stride)) { + size_t total = row_size * nrows; + return dma_queue_push_fallback_contig(q, ddata, total); + } -static inline bool dma_queue_push_ddr_to_vtcm(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) { - return dma_queue_push(q, dptr, dst_row_size, src_row_size, src_row_size, nrows); + return dma_queue_push_fallback_2d(q, ddata, dst_stride, src_stride, row_size, nrows); } -static inline bool dma_queue_push_vtcm_to_ddr(dma_queue * q, dma_ptr dptr, size_t dst_row_size, size_t src_row_size, size_t nrows) { - return dma_queue_push(q, dptr, dst_row_size, src_row_size, dst_row_size, nrows); -} +#endif #define DMA_CACHE_MAX_SIZE 256U +// Fully assoc LRU cache typedef struct { uint8_t *base; uint32_t line_size; uint32_t capacity; - uint32_t src[DMA_CACHE_MAX_SIZE]; + dma_addr_t src[DMA_CACHE_MAX_SIZE]; uint16_t age[DMA_CACHE_MAX_SIZE]; -} dma_cache; +} dma_cache_fa; -static inline void dma_cache_init(dma_cache *c, uint8_t *base, uint32_t line_size, uint32_t capacity) +static inline void dma_cache_fa_init(dma_cache_fa *c, uint8_t *base, uint32_t line_size, uint32_t capacity) { c->capacity = (capacity > DMA_CACHE_MAX_SIZE) ? DMA_CACHE_MAX_SIZE : capacity; c->base = base; @@ -375,14 +449,14 @@ static inline void dma_cache_init(dma_cache *c, uint8_t *base, uint32_t line_siz } } -static inline bool dma_cache_push(dma_queue *q, dma_cache *c, const uint8_t * src, uint32_t dst_stride, uint32_t src_stride, uint32_t row_size, uint32_t nrows) +static inline bool dma_cache_fa_push(dma_queue *q, dma_cache_fa *c, dma_addr_t src_addr, uint32_t dst_stride, uint32_t src_stride, uint32_t row_size, uint32_t nrows) { uint32_t o_idx = 0; uint16_t o_age = 0; uint8_t * dst = 0; for (unsigned i=0; i < c->capacity; i++) { - if (c->src[i] == (uint32_t) src) { + if (c->src[i] == src_addr) { c->age[i] = 0; dst = c->base + (i * c->line_size); nrows = 0; // dummy dma } else { @@ -392,12 +466,46 @@ static inline bool dma_cache_push(dma_queue *q, dma_cache *c, const uint8_t * sr } if (!dst) { c->age[o_idx] = 0; - c->src[o_idx] = (uint32_t) src; + c->src[o_idx] = src_addr; dst = c->base + o_idx * c->line_size; // normal nrows dma - return dma_queue_push(q, dma_make_ptr(dst, src), dst_stride, src_stride, row_size, nrows); + return dma_queue_push(q, dma_make_data(dst, src_addr), dst_stride, src_stride, row_size, nrows); + } + + return dma_queue_push_single_1d(q, dma_make_data(dst, src_addr), 0); +} + +// Direct mapped cache +typedef struct { + uint8_t *base; + uint32_t line_size; + uint32_t capacity; + uint32_t idx_mask; + dma_addr_t src[DMA_CACHE_MAX_SIZE]; +} dma_cache_dm; + +static inline void dma_cache_dm_init(dma_cache_dm *c, uint8_t *base, uint32_t line_size, uint32_t capacity) +{ + c->capacity = (capacity > DMA_CACHE_MAX_SIZE) ? DMA_CACHE_MAX_SIZE : capacity; + c->idx_mask = c->capacity - 1; + c->base = base; + c->line_size = line_size; + + for (unsigned i=0; i < c->capacity; i++) { + c->src[i] = 0; + } +} + +static inline bool dma_cache_dm_push(dma_queue *q, dma_cache_dm *c, uint32_t slot, dma_addr_t src_addr, uint32_t dst_stride, uint32_t src_stride, uint32_t row_size, uint32_t nrows) +{ + const uint32_t i = slot & c->idx_mask; + uint8_t * dst = c->base + (i * c->line_size); + + if (c->src[i] == src_addr) { + return dma_queue_push_single_1d(q, dma_make_data(dst, src_addr), 0); // dummy dma } - return dma_queue_push_single_1d(q, dma_make_ptr(dst, src), 0); + c->src[i] = src_addr; + return dma_queue_push(q, dma_make_data(dst, src_addr), dst_stride, src_stride, row_size, nrows); } #ifdef __cplusplus diff --git a/ggml/src/ggml-hexagon/htp/fill-ops.c b/ggml/src/ggml-hexagon/htp/fill-ops.c index 3ccfbe74..212104a2 100644 --- a/ggml/src/ggml-hexagon/htp/fill-ops.c +++ b/ggml/src/ggml-hexagon/htp/fill-ops.c @@ -3,10 +3,11 @@ #pragma clang diagnostic ignored "-Wunused-but-set-variable" #include -#include - #include +#include "hex-common.h" +#include "hex-profile.h" + #include "hvx-copy.h" #include "hvx-utils.h" @@ -14,28 +15,30 @@ #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" // ggml op_params layout for FILL: // op_params[0] (as float) - the scalar fill value -#define fill_preamble \ +#define fill_preamble \ const struct htp_tensor * dst = octx->dst; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; \ - \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + const uint32_t ne2 = dst->ne[2]; \ + const uint32_t ne3 = dst->ne[3]; \ + \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ + const uint32_t nb3 = dst->nb[3]; \ + \ const uint32_t nr = ne1 * ne2 * ne3; struct htp_fill_context { struct htp_ops_context * octx; uint32_t nrows_per_thread; uint32_t total_rows; // ne1 * ne2 * ne3 + uint32_t row_start; bool opt_path; HVX_Vector splat_vec; uint32_t elem_size; @@ -47,10 +50,15 @@ static void fill_thread(unsigned int nth, unsigned int ith, void * data) { fill_preamble; // Parallelise over the flat row index spanning ne1*ne2*ne3 - const uint32_t ir0 = fctx->nrows_per_thread * ith; - const uint32_t ir1 = MIN(ir0 + fctx->nrows_per_thread, fctx->total_rows); + const uint32_t ir0 = fctx->row_start + fctx->nrows_per_thread * ith; + const uint32_t ir1 = MIN(ir0 + fctx->nrows_per_thread, fctx->row_start + fctx->total_rows); - uint64_t t1 = HAP_perf_get_qtimer_count(); + if (ir0 >= ir1) { + return; + } + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); if (fctx->opt_path) { // Opt path: tensor is fully contiguous, treat as flat array @@ -69,9 +77,8 @@ static void fill_thread(unsigned int nth, unsigned int ith, void * data) { } } - uint64_t t2 = HAP_perf_get_qtimer_count(); - FARF(HIGH, "fill %u/%u: rows %u:%u usec %u\n", - ith, nth, ir0, ir1, (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir1); + FARF(HIGH, "fill %u/%u: rows %u:%u\n", ith, nth, ir0, ir1); } int op_fill(struct htp_ops_context * octx) { @@ -81,12 +88,27 @@ int op_fill(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + if (htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + uint32_t row_start = 0; + uint32_t nrows = nr; + + if (octx->ctx->mdev.count > 1) { + const uint32_t row_size = nb1; + const uint32_t rows_per_chunk = (row_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(row_size, HEX_L2_LINE_SIZE)) : 1; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(nr, htp_tensor_mdev_data_aligned(dst) ? rows_per_chunk : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { return HTP_STATUS_OK; } // nr = ne1*ne2*ne3 (flat row count across all outer dims); parallelise over it. - const uint32_t n_threads = MIN(nr, octx->n_threads); + const uint32_t n_threads = octx->n_threads; // Optimize if fully contiguous: skip stride arithmetic, treat as flat array const bool opt_path = (nb2 == nb1 * ne1) && (nb3 == nb2 * ne2); @@ -99,8 +121,9 @@ int op_fill(struct htp_ops_context * octx) { struct htp_fill_context fctx = { .octx = octx, - .nrows_per_thread = (nr + n_threads - 1) / n_threads, - .total_rows = nr, + .nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .total_rows = nrows, + .row_start = row_start, .opt_path = opt_path, }; @@ -117,7 +140,7 @@ int op_fill(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - worker_pool_run_func(octx->ctx->worker_pool, fill_thread, &fctx, n_threads); + work_queue_run(octx->ctx->work_queue, fill_thread, &fctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/flash-attn-ops.c b/ggml/src/ggml-hexagon/htp/flash-attn-ops.c index fe78718c..f079f738 100644 --- a/ggml/src/ggml-hexagon/htp/flash-attn-ops.c +++ b/ggml/src/ggml-hexagon/htp/flash-attn-ops.c @@ -5,7 +5,6 @@ #include #include #include -#include #include #include #include @@ -13,7 +12,7 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hex-fastdiv.h" #include "hex-profile.h" #include "hmx-queue.h" @@ -30,6 +29,8 @@ #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" +#include "hvx-quant.h" #include "flash-attn-ops.h" #include "hvx-fa-kernels.h" @@ -54,6 +55,7 @@ struct htp_fa_context { float scale; float max_bias; + bool has_softcap; __fp16 logit_softcap; uint32_t n_head_log2; @@ -73,6 +75,7 @@ struct htp_fa_context { uint32_t qrows; uint32_t qrows_per_thread; + uint32_t qrow_start; bool is_q_fp32; @@ -84,25 +87,31 @@ struct htp_fa_context { uint8_t * spad_v; uint8_t * spad_m; uint8_t * spad_a; + float * spad_sinks; - uint64_t t_start; + const struct htp_tensor * k; + const struct htp_tensor * v; }; struct hmx_fa_context { const struct htp_ops_context * octx; const struct htp_tensor * sinks; // attention sinks (src[4]), NULL if absent + const struct htp_tensor * k; + const struct htp_tensor * v; bool pipeline; // true when n_kv_blocks >= FA_MIN_KV_BLOCKS && n_threads >= 2 uint32_t n_threads; // Op parameters __fp16 scale; float max_bias; + bool has_softcap; __fp16 logit_softcap; uint32_t n_head_log2; float m0, m1; // Dimensions uint32_t DK, DV; + uint32_t DK_pad, DV_pad; // head_dim rounded up to 64 for HMX tiling uint32_t n_kv; // kv_len uint32_t n_kv_heads; // number of KV heads uint32_t n_heads; // number of Q heads @@ -132,8 +141,8 @@ struct hmx_fa_context { __fp16 * vtcm_v_tiles[2]; // V tiles (column-major, double-buffered) __fp16 * vtcm_s_tiles[2]; // S = QK^T [g_br, Bc] (double-buffered) __fp16 * vtcm_p_tiles[2]; // P = softmax(S) [g_br, Bc] - __fp16 * vtcm_d_tiles; // Diagonal rescale [g_br, g_br] - __fp16 * vtcm_d_inv_l; // Diagonal rescale (1/l) [g_br, g_br] + __fp16 * vtcm_d_tiles[2]; // Diagonal rescale, g_br/32 packed diagonal tiles (double-buffered) + __fp16 * vtcm_d_inv_l; // Diagonal rescale (1/l), same packed layout HVX_Vector * vtcm_m_vec; // Row max [g_br] HVX_Vector * vtcm_l_vec; // Row sum [g_br] HVX_Vector * vtcm_s_rowmax; // Softmax intermediate [g_br] @@ -143,6 +152,7 @@ struct hmx_fa_context { uint8_t * vtcm_hmx_scales_qk; // HMX output scales (qk_scale) __fp16 * vtcm_mask_buf; // VTCM mask buffer [Br * m_line], DMA'd per KV block __fp16 * vtcm_slopes; // ALiBi slopes [g_br] + float * vtcm_sinks; // Attention sinks size_t row_buf_stride; // HVX vectors per row buffer (Bc/64) size_t mask_buf_row_stride; // elements (__fp16) per row in mask buffer size_t q_tile_bytes; @@ -150,7 +160,7 @@ struct hmx_fa_context { size_t col_vec_bytes; size_t d_tile_bytes; bool mask_broadcast; // true when mask->ne[2] == 1 (head-independent, single 2D DMA) - dma_cache m_cache; + dma_cache_fa m_cache; }; static void flash_attn_ext_f16_thread(unsigned int nth, unsigned int ith, void * data) { @@ -199,23 +209,22 @@ static void flash_attn_ext_f16_thread(unsigned int nth, unsigned int ith, void * const uint32_t nb3 = dst->nb[3]; // total rows in q - const uint32_t nr = factx->qrows; - const uint32_t dr = factx->qrows_per_thread; - const uint32_t ir0 = dr * ith; - const uint32_t ir1 = MIN(ir0 + dr, nr); + const uint32_t dr = factx->qrows_per_thread; + const uint32_t ir0 = factx->qrow_start + dr * ith; + const uint32_t ir1 = MIN(ir0 + dr, factx->qrow_start + factx->qrows); if (ir0 >= ir1) return; struct htp_thread_trace * tr = &octx->ctx->trace[ith]; - dma_queue * dma = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; const uint32_t DK = nek0; const uint32_t DV = nev0; const size_t size_q_row = DK * ((q->type == HTP_TYPE_F32) ? 4 : 2); - const size_t size_k_row = DK * sizeof(__fp16); - const size_t size_v_row = DV * sizeof(__fp16); + const size_t size_k_row = htp_tensor_get_row_size(k->type, DK); + const size_t size_v_row = htp_tensor_get_row_size(v->type, DV); // Scratchpad buffers for Q, K, V, Mask, and VKQ32 accumulator uint8_t * spad_q = factx->spad_q + factx->size_q_block * ith; @@ -224,10 +233,13 @@ static void flash_attn_ext_f16_thread(unsigned int nth, unsigned int ith, void * uint8_t * spad_m = factx->spad_m + (mask ? factx->size_m_block * HVX_FA_DMA_CACHE_SIZE : 0) * ith; uint8_t * spad_a = factx->spad_a + factx->size_vkq_acc * ith; - dma_cache m_cache; - dma_cache_init(&m_cache, spad_m, factx->size_m_block, HVX_FA_DMA_CACHE_SIZE); + dma_cache_dm m_cache; + dma_cache_dm_init(&m_cache, spad_m, factx->size_m_block, HVX_FA_DMA_CACHE_SIZE); - for (uint32_t ir = ir0; ir < ir1; ++ir) { + const size_t size_vkq_acc_single = hex_round_up(DV * sizeof(float), 128); + + uint32_t ir = ir0; + while (ir < ir1) { const uint32_t iq3 = fastdiv(ir, &factx->src0_div21); const uint32_t iq2 = fastdiv(ir - iq3*neq2*neq1, &factx->src0_div1); const uint32_t iq1 = (ir - iq3*neq2*neq1 - iq2 * neq1); @@ -238,361 +250,346 @@ static void flash_attn_ext_f16_thread(unsigned int nth, unsigned int ith, void * const uint32_t iv3 = fastdiv(iq3, &factx->broadcast_rv3); const uint32_t iv2 = fastdiv(iq2, &factx->broadcast_rv2); - const __fp16 * mp_base = NULL; - if (mask) { - const uint32_t im2 = fastmodulo(iq2, mask->ne[2], &factx->src3_div2); - const uint32_t im3 = fastmodulo(iq3, mask->ne[3], &factx->src3_div3); - mp_base = (const __fp16 *) ((const uint8_t *) mask->data + iq1*mask->nb[1] + im2*mask->nb[2] + im3*mask->nb[3]); - } + uint32_t G_local = 1; + if (neq1 == 1 && (mask == NULL || mask->ne[2] == 1)) { + while (ir + G_local < ir1 && G_local < FA_HVX_G_MAX) { + const uint32_t next_ir = ir + G_local; + const uint32_t next_iq3 = fastdiv(next_ir, &factx->src0_div21); + const uint32_t next_iq2 = fastdiv(next_ir - next_iq3*neq2*neq1, &factx->src0_div1); + const uint32_t next_iq1 = (next_ir - next_iq3*neq2*neq1 - next_iq2 * neq1); - // Precalculate next row variables if there is a next row - bool has_next_ir = (ir + 1 < ir1); - uint32_t next_ik2 = 0, next_ik3 = 0, next_iv2 = 0, next_iv3 = 0; - const uint8_t * next_q_row_ptr = NULL; - const __fp16 * next_mp_base = NULL; + const uint32_t next_ik3 = fastdiv(next_iq3, &factx->broadcast_rk3); + const uint32_t next_ik2 = fastdiv(next_iq2, &factx->broadcast_rk2); - const uint8_t * next_k_src0 = NULL; - const uint8_t * next_v_src0 = NULL; - const uint8_t * next_m_src0 = NULL; - uint32_t next_block_size0 = 0; + const uint32_t next_iv3 = fastdiv(next_iq3, &factx->broadcast_rv3); + const uint32_t next_iv2 = fastdiv(next_iq2, &factx->broadcast_rv2); - const uint8_t * next_k_src1 = NULL; - const uint8_t * next_v_src1 = NULL; - const uint8_t * next_m_src1 = NULL; - uint32_t next_block_size1 = 0; + if (next_ik2 != ik2 || next_ik3 != ik3 || next_iv2 != iv2 || next_iv3 != iv3 || next_iq1 != iq1 || next_iq3 != iq3) { + break; + } + G_local++; + } + } - if (has_next_ir) { - const uint32_t next_ir = ir + 1; - const uint32_t next_iq3 = fastdiv(next_ir, &factx->src0_div21); - const uint32_t next_iq2 = fastdiv(next_ir - next_iq3*neq2*neq1, &factx->src0_div1); - const uint32_t next_iq1 = (next_ir - next_iq3*neq2*neq1 - next_iq2 * neq1); + uint32_t heads[FA_HVX_G_MAX]; + HVX_Vector slope_vecs[FA_HVX_G_MAX] __attribute__((aligned(128))); + HVX_Vector S_vec[FA_HVX_G_MAX] __attribute__((aligned(128))); + HVX_Vector M_vec[FA_HVX_G_MAX] __attribute__((aligned(128))); + uint8_t * q_ptrs[FA_HVX_G_MAX]; + float * vkq_ptrs[FA_HVX_G_MAX]; - next_ik3 = fastdiv(next_iq3, &factx->broadcast_rk3); - next_ik2 = fastdiv(next_iq2, &factx->broadcast_rk2); + for (uint32_t g = 0; g < G_local; ++g) { + const uint32_t r = ir + g; + const uint32_t r_iq3 = fastdiv(r, &factx->src0_div21); + const uint32_t r_iq2 = fastdiv(r - r_iq3*neq2*neq1, &factx->src0_div1); + const uint32_t r_iq1 = (r - r_iq3*neq2*neq1 - r_iq2 * neq1); - next_iv3 = fastdiv(next_iq3, &factx->broadcast_rv3); - next_iv2 = fastdiv(next_iq2, &factx->broadcast_rv2); + heads[g] = r_iq2; + const __fp16 slope = factx->slopes[r_iq2]; + slope_vecs[g] = hvx_vec_splat_f16(slope); - next_q_row_ptr = (const uint8_t *) q->data + (next_iq1*nbq1 + next_iq2*nbq2 + next_iq3*nbq3); + S_vec[g] = hvx_vec_splat_f32(0.0f); + M_vec[g] = hvx_vec_splat_f32(HTP_FA_M_INITIAL_VAL); - if (mask) { - const uint32_t next_im2 = fastmodulo(next_iq2, mask->ne[2], &factx->src3_div2); - const uint32_t next_im3 = fastmodulo(next_iq3, mask->ne[3], &factx->src3_div3); - next_mp_base = (const __fp16 *) ((const uint8_t *) mask->data + next_iq1*mask->nb[1] + next_im2*mask->nb[2] + next_im3*mask->nb[3]); - } + uint8_t * q_dst = spad_q + g * factx->size_q_row_padded; + q_ptrs[g] = q_dst; - // Precalculate next K/V block 0 source pointers - { - const uint32_t ic_start = 0; - next_block_size0 = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - ic_start); - next_k_src0 = (const uint8_t *) k->data + (ic_start*nbk1 + next_ik2*nbk2 + next_ik3*nbk3); - next_v_src0 = (const uint8_t *) v->data + (ic_start*nbv1 + next_iv2*nbv2 + next_iv3*nbv3); - if (mask) { - next_m_src0 = (const uint8_t *) (next_mp_base + ic_start); - } - } + float * vkq_dst = (float *)(spad_a + g * size_vkq_acc_single); + vkq_ptrs[g] = vkq_dst; + hvx_splat_f32_a((uint8_t *) vkq_dst, 0, DV); - // Precalculate next K/V block 1 source pointers (if n_blocks > 1) - if (factx->n_blocks > 1) { - const uint32_t ic_start = 1 * FLASH_ATTN_BLOCK_SIZE; - next_block_size1 = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - ic_start); - next_k_src1 = (const uint8_t *) k->data + (ic_start*nbk1 + next_ik2*nbk2 + next_ik3*nbk3); - next_v_src1 = (const uint8_t *) v->data + (ic_start*nbv1 + next_iv2*nbv2 + next_iv3*nbv3); - if (mask) { - next_m_src1 = (const uint8_t *) (next_mp_base + ic_start); - } - } + // Fetch Q row g + const dma_addr_t q_row_ptr = q->data + r_iq1*nbq1 + r_iq2*nbq2 + r_iq3*nbq3; + dma_queue_push(dma_q, dma_make_data(q_dst, q_row_ptr), factx->size_q_row_padded, nbq1, size_q_row, 1); } - if (ir == ir0) { - // Fetch Q row - const uint8_t * q_row_ptr = (const uint8_t *) q->data + (iq1*nbq1 + iq2*nbq2 + iq3*nbq3); - dma_queue_push(dma, dma_make_ptr(spad_q, q_row_ptr), factx->size_q_row_padded, nbq1, size_q_row, 1); + dma_addr_t mp_base = 0; + if (mask) { + const uint32_t im2 = fastmodulo(iq2, mask->ne[2], &factx->src3_div2); + const uint32_t im3 = fastmodulo(iq3, mask->ne[3], &factx->src3_div3); + mp_base = mask->data + iq1*mask->nb[1] + im2*mask->nb[2] + im3*mask->nb[3]; + } - // Prefetch first two blocks - for (uint32_t ib = 0; ib < MIN(factx->n_blocks, 2); ++ib) { - const uint32_t ic_start = ib * FLASH_ATTN_BLOCK_SIZE; - const uint32_t current_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - ic_start); + // Prefetch first two blocks + for (uint32_t ib = 0; ib < MIN(factx->n_blocks, 2); ++ib) { + const uint32_t ic_start = ib * FLASH_ATTN_BLOCK_SIZE; + const uint32_t current_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - ic_start); - // K - const uint8_t * k_src = (const uint8_t *) k->data + (ic_start*nbk1 + ik2*nbk2 + ik3*nbk3); - uint8_t * k_dst = spad_k + (ib % 2) * factx->size_k_block; - dma_queue_push(dma, dma_make_ptr(k_dst, k_src), factx->size_k_row_padded, nbk1, size_k_row, current_block_size); + // K + const dma_addr_t k_src = k->data + ic_start*nbk1 + ik2*nbk2 + ik3*nbk3; + uint8_t * k_dst = spad_k + (ib % 2) * factx->size_k_block; + dma_queue_push(dma_q, dma_make_data(k_dst, k_src), factx->size_k_row_padded, nbk1, size_k_row, current_block_size); - // V - const uint8_t * v_src = (const uint8_t *) v->data + (ic_start*nbv1 + iv2*nbv2 + iv3*nbv3); - uint8_t * v_dst = spad_v + (ib % 2) * factx->size_v_block; - dma_queue_push(dma, dma_make_ptr(v_dst, v_src), factx->size_v_row_padded, nbv1, size_v_row, current_block_size); + // V + const dma_addr_t v_src = v->data + ic_start*nbv1 + iv2*nbv2 + iv3*nbv3; + uint8_t * v_dst = spad_v + (ib % 2) * factx->size_v_block; + dma_queue_push(dma_q, dma_make_data(v_dst, v_src), factx->size_v_row_padded, nbv1, size_v_row, current_block_size); - // Mask - if (mask) { - const uint8_t * m_src = (const uint8_t *) (mp_base + ic_start); - // Mask is 1D contiguous for this row - dma_cache_push(dma, &m_cache, m_src, current_block_size * 2, current_block_size * 2, current_block_size * 2, 1); - } + // Mask + if (mask) { + const dma_addr_t m_src = mp_base + ic_start * sizeof(__fp16); + dma_cache_dm_push(dma_q, &m_cache, ib, m_src, current_block_size * 2, current_block_size * 2, current_block_size * 2, 1); } } - const uint32_t h = iq2; // head index - const __fp16 slope = factx->slopes[h]; - - HVX_Vector S_vec = hvx_vec_splat_f32(0.0f); - HVX_Vector M_vec = hvx_vec_splat_f32(HTP_FA_M_INITIAL_VAL); - - // Clear accumulator - hvx_splat_f32_a(spad_a, 0, DV); - float * VKQ32 = (float *) (spad_a + 0); - - uint8_t * q_ptr_vtcm = dma_queue_pop(dma).dst; - if (factx->is_q_fp32) { - hvx_copy_f16_f32_aa(q_ptr_vtcm, q_ptr_vtcm, DK); // inplace convert f32 to f16 + // Pop all Q rows + for (uint32_t g = 0; g < G_local; ++g) { + uint8_t * q_ptr_vtcm = (void *) dma_queue_pop(dma_q).dst; + if (factx->is_q_fp32) { + hvx_copy_f16_f32_aa(q_ptr_vtcm, q_ptr_vtcm, DK); + } } - const HVX_Vector slope_vec = hvx_vec_splat_f16(slope); const HVX_Vector v_neg_inf = Q6_Vh_vsplat_R(0xfbff); - const HVX_Vector v_cap = (factx->logit_softcap != 0.0f) ? hvx_vec_splat_f16(factx->logit_softcap) : Q6_V_vzero(); + const bool has_softcap = factx->has_softcap; + const HVX_Vector v_cap = has_softcap ? hvx_vec_splat_f16(factx->logit_softcap) : Q6_V_vzero(); const HVX_Vector vinf = Q6_Vh_vsplat_R(0xFC00); const HVX_Vector vmin = Q6_Vh_vsplat_R(0xFBFF); const HVX_Vector v_log2e = hvx_vec_splat_f16(EXP_LOG2E_F); const uint32_t stride_v2 = factx->size_v_row_padded * 2; + for (uint32_t ib = 0; ib < factx->n_blocks; ++ib) { const uint32_t ic_start = ib * FLASH_ATTN_BLOCK_SIZE; const uint32_t current_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - ic_start); // Wait for DMA - uint8_t * k_base = dma_queue_pop(dma).dst; // K - uint8_t * v_base = dma_queue_pop(dma).dst; // V - __fp16 * m_base = mask ? dma_queue_pop(dma).dst : NULL; // M - - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_QK, ir); - - // Inner loop processing the block from VTCM - // 1. Compute scores (64 elements FP16) - HVX_Vector scores_f16 = Q6_V_vzero(); - if (current_block_size > 0) { - HVX_Vector scores0 = hvx_dot_f16_f16_aa_rx32(q_ptr_vtcm, k_base, factx->size_k_row_padded, DK, factx->scale); - HVX_Vector scores1 = (current_block_size > 32) ? hvx_dot_f16_f16_aa_rx32(q_ptr_vtcm, k_base + 32 * factx->size_k_row_padded, factx->size_k_row_padded, DK, factx->scale) : Q6_V_vzero(); - scores_f16 = hvx_vec_f32_to_f16(scores0, scores1); + uint8_t * k_base = (void *) dma_queue_pop(dma_q).dst; // K + uint8_t * v_base = (void *) dma_queue_pop(dma_q).dst; // V + __fp16 * m_base = mask ? (__fp16 *) dma_queue_pop(dma_q).dst : NULL; // M + + if (factx->k->type == HTP_TYPE_Q8_0) { + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_K_PREP, ir); + for (uint32_t r = 0; r < current_block_size; ++r) { + __fp16 * row_k = (__fp16 *)(k_base + r * factx->size_k_row_padded); + hvx_dequantize_row_q8_0_f16(row_k, row_k, DK); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_K_PREP, ir); } - - // 2. Softcap (in FP16) - if (factx->logit_softcap != 0.0f) { - scores_f16 = hvx_vec_tanh_f16(scores_f16); - scores_f16 = hvx_vec_mul_f16_f16(scores_f16, v_cap); + if (factx->v->type == HTP_TYPE_Q8_0) { + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_V_PREP, ir); + for (uint32_t r = 0; r < current_block_size; ++r) { + __fp16 * row_v = (__fp16 *)(v_base + r * factx->size_v_row_padded); + hvx_dequantize_row_q8_0_f16(row_v, row_v, DV); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_V_PREP, ir); } - HVX_VectorPred q_tail_keep = Q6_Q_vsetq2_R(current_block_size * sizeof(__fp16)); + for (uint32_t g = 0; g < G_local; ++g) { + const uint32_t head_ir = ir + g; + uint8_t * q_ptr_vtcm = q_ptrs[g]; + float * VKQ32 = vkq_ptrs[g]; + const HVX_Vector slope_vec = slope_vecs[g]; - // 3. Mask (in FP16) - if (mask) { - HVX_Vector m_vals_f16 = *(const HVX_UVector *) m_base; - HVX_VectorPred is_inf = Q6_Q_vcmp_eq_VhVh(m_vals_f16, vinf); - m_vals_f16 = Q6_V_vmux_QVV(is_inf, vmin, m_vals_f16); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_QK, head_ir); - HVX_Vector m_scaled = hvx_vec_mul_f16_f16(m_vals_f16, slope_vec); - scores_f16 = Q6_V_vmux_QVV(q_tail_keep, hvx_vec_add_f16_f16(scores_f16, m_scaled), v_neg_inf); - } else { - scores_f16 = Q6_V_vmux_QVV(q_tail_keep, scores_f16, v_neg_inf); - } + HVX_Vector scores_f16 = Q6_V_vzero(); + if (current_block_size > 0) { + HVX_Vector scores0 = hvx_dot_f16_f16_aa_rx32(q_ptr_vtcm, k_base, factx->size_k_row_padded, DK, factx->scale); + HVX_Vector scores1 = (current_block_size > 32) ? hvx_dot_f16_f16_aa_rx32(q_ptr_vtcm, k_base + 32 * factx->size_k_row_padded, factx->size_k_row_padded, DK, factx->scale) : Q6_V_vzero(); + scores_f16 = hvx_vec_f32_to_f16(scores0, scores1); + } - // Compute block max in FP16 - HVX_Vector v_max_f16 = hvx_vec_reduce_max_f16(scores_f16); - HVX_Vector v_max = Q6_V_lo_W(hvx_vec_f16_to_f32(v_max_f16)); // splat block max in FP32 - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_QK, ir); + if (has_softcap) { + scores_f16 = hvx_vec_tanh_f16(scores_f16); + scores_f16 = hvx_vec_mul_f16_f16(scores_f16, v_cap); + } - if (ib + 1 == factx->n_blocks && has_next_ir) { - // Queue next row's Q row! - dma_queue_push(dma, dma_make_ptr(spad_q, next_q_row_ptr), factx->size_q_row_padded, nbq1, size_q_row, 1); + HVX_VectorPred q_tail_keep = Q6_Q_vsetq2_R(current_block_size * sizeof(__fp16)); - if (factx->n_blocks % 2 == 0) { - // Queue next row's block 0 (into buffer slot 0) - uint8_t * k_dst = spad_k + 0 * factx->size_k_block; - uint8_t * v_dst = spad_v + 0 * factx->size_v_block; + if (mask) { + HVX_Vector m_vals_f16 = *(const HVX_UVector *) m_base; + HVX_VectorPred is_inf = Q6_Q_vcmp_eq_VhVh(m_vals_f16, vinf); + m_vals_f16 = Q6_V_vmux_QVV(is_inf, vmin, m_vals_f16); - // K (block 0 of next row) - dma_queue_push(dma, dma_make_ptr(k_dst, next_k_src0), factx->size_k_row_padded, nbk1, size_k_row, next_block_size0); + HVX_Vector m_scaled = hvx_vec_mul_f16_f16(m_vals_f16, slope_vec); + scores_f16 = Q6_V_vmux_QVV(q_tail_keep, hvx_vec_add_f16_f16(scores_f16, m_scaled), v_neg_inf); + } else { + scores_f16 = Q6_V_vmux_QVV(q_tail_keep, scores_f16, v_neg_inf); + } - // V (block 0 of next row) - dma_queue_push(dma, dma_make_ptr(v_dst, next_v_src0), factx->size_v_row_padded, nbv1, size_v_row, next_block_size0); + HVX_Vector v_max_f16 = hvx_vec_reduce_max_f16(scores_f16); + HVX_Vector v_max = Q6_V_lo_W(hvx_vec_f16_to_f32(v_max_f16)); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_QK, head_ir); - // Mask (block 0 of next row) - if (mask) { - dma_cache_push(dma, &m_cache, next_m_src0, next_block_size0 * 2, next_block_size0 * 2, next_block_size0 * 2, 1); - } + // prefetch K for block ib + 2 after last head finished QK + if (g + 1 == G_local && ib + 2 < factx->n_blocks) { + const uint32_t next_ib = ib + 2; + const uint32_t next_ic_start = next_ib * FLASH_ATTN_BLOCK_SIZE; + const uint32_t next_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - next_ic_start); + + const dma_addr_t k_src = k->data + next_ic_start*nbk1 + ik2*nbk2 + ik3*nbk3; + dma_queue_push(dma_q, dma_make_data(k_base, k_src), factx->size_k_row_padded, nbk1, size_k_row, next_block_size); } - } - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_SFM, ir); - { - // 4. Online Softmax Update - HVX_Vector M_new_vec = Q6_Vsf_vmax_VsfVsf(v_max, M_vec); - HVX_Vector diff_vec = HVX_OP_SUB_F32(M_vec, M_new_vec); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_SFM, head_ir); + { + HVX_Vector M_new_vec = Q6_Vsf_vmax_VsfVsf(v_max, M_vec[g]); + HVX_Vector diff_vec = HVX_OP_SUB_F32(M_vec[g], M_new_vec); - HVX_Vector diff_f16 = hvx_vec_f32_to_f16(diff_vec, diff_vec); - HVX_Vector diff_base2 = hvx_vec_mul_f16_f16(diff_f16, v_log2e); - HVX_Vector ms_f16 = hvx_vec_exp2_f16(diff_base2); - HVX_Vector ms_vec = Q6_V_lo_W(hvx_vec_f16_to_f32(ms_f16)); + HVX_Vector diff_f16 = hvx_vec_f32_to_f16(diff_vec, diff_vec); + HVX_Vector diff_base2 = hvx_vec_mul_f16_f16(diff_f16, v_log2e); + HVX_Vector ms_f16 = hvx_vec_exp2_f16(diff_base2); + HVX_Vector ms_vec = Q6_V_lo_W(hvx_vec_f16_to_f32(ms_f16)); - M_vec = M_new_vec; + M_vec[g] = M_new_vec; - hvx_scale_vec_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, ms_vec); + HVX_Vector v_m_vec_f16 = hvx_vec_f32_to_f16(M_vec[g], M_vec[g]); + HVX_Vector v_s_minus_m = Q6_Vqf16_vsub_VhfVhf(scores_f16, v_m_vec_f16); + HVX_Vector v_s_minus_m_base2 = hvx_vec_mul_f16_f16(Q6_Vhf_equals_Vqf16(v_s_minus_m), v_log2e); - // Compute P = exp2((S - M) * log2(e)) in FP16 - HVX_Vector v_m_vec_f16 = hvx_vec_f32_to_f16(M_vec, M_vec); - HVX_Vector v_s_minus_m = Q6_Vqf16_vsub_VhfVhf(scores_f16, v_m_vec_f16); + HVX_Vector P = hvx_vec_exp2_f16(v_s_minus_m_base2); + P = Q6_V_vmux_QVV(q_tail_keep, P, Q6_V_vzero()); - HVX_Vector v_s_minus_m_base2 = hvx_vec_mul_f16_f16(Q6_Vhf_equals_Vqf16(v_s_minus_m), v_log2e); + HVX_VectorPair P_pair = hvx_vec_f16_to_f32(P); + HVX_Vector P0 = Q6_V_lo_W(P_pair); + HVX_Vector P1 = Q6_V_hi_W(P_pair); + HVX_Vector p_sum_vec = hvx_vec_reduce_sum_f32(HVX_OP_ADD_F32(P0, P1)); - HVX_Vector P = hvx_vec_exp2_f16(v_s_minus_m_base2); - P = Q6_V_vmux_QVV(q_tail_keep, P, Q6_V_vzero()); + S_vec[g] = HVX_OP_ADD_F32(HVX_OP_MUL_F32(S_vec[g], ms_vec), p_sum_vec); - // Convert P to FP32 to update the running sum S_vec - HVX_VectorPair P_pair = hvx_vec_f16_to_f32(P); - HVX_Vector P0 = Q6_V_lo_W(P_pair); - HVX_Vector P1 = Q6_V_hi_W(P_pair); - HVX_Vector p_sum_vec = hvx_vec_reduce_sum_f32(HVX_OP_ADD_F32(P0, P1)); + const uint8_t * v_ptr = v_base; - S_vec = HVX_OP_ADD_F32(HVX_OP_MUL_F32(S_vec, ms_vec), p_sum_vec); + if (DV == 64) { + HVX_VectorPair vkq0 = *((const HVX_VectorPair *) VKQ32); + vkq0 = Q6_W_vcombine_VV( + HVX_OP_MUL_F32(Q6_V_hi_W(vkq0), ms_vec), + HVX_OP_MUL_F32(Q6_V_lo_W(vkq0), ms_vec) + ); - // 5. Accumulate V (F16 * F16 -> F32 accumulator) - const uint8_t * v_ptr = v_base; + for (uint32_t j = 0; j < current_block_size; j += 2) { + HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); + const HVX_Vector * vx0 = (const HVX_Vector *) v_ptr; + if (j + 1 == current_block_size) { + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx0[0]), S0); + break; + } - for (uint32_t j = 0; j < current_block_size; j += 2) { - if (j + 1 == current_block_size) { - HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); - hvx_mad_f32_f16_aa_vec(VKQ32, v_ptr, S0, DV); - break; - } + HVX_Vector S1 = hvx_vec_repl_f16(Q6_V_vror_VR(P, (j + 1) * 2)); + const HVX_Vector * vx1 = (const HVX_Vector *) (v_ptr + factx->size_v_row_padded); + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx0[0]), S0); + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx1[0]), S1); + v_ptr += stride_v2; + } - HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); - HVX_Vector S1 = hvx_vec_repl_f16(Q6_V_vror_VR(P, (j + 1) * 2)); + *((HVX_VectorPair *) VKQ32) = vkq0; + } else if (DV == 128) { + HVX_VectorPair vkq0 = ((const HVX_VectorPair *) VKQ32)[0]; + HVX_VectorPair vkq1 = ((const HVX_VectorPair *) VKQ32)[1]; + vkq0 = Q6_W_vcombine_VV( + HVX_OP_MUL_F32(Q6_V_hi_W(vkq0), ms_vec), + HVX_OP_MUL_F32(Q6_V_lo_W(vkq0), ms_vec) + ); + vkq1 = Q6_W_vcombine_VV( + HVX_OP_MUL_F32(Q6_V_hi_W(vkq1), ms_vec), + HVX_OP_MUL_F32(Q6_V_lo_W(vkq1), ms_vec) + ); + + for (uint32_t j = 0; j < current_block_size; j += 2) { + HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); + const HVX_Vector * vx0 = (const HVX_Vector *) v_ptr; + if (j + 1 == current_block_size) { + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx0[0]), S0); + vkq1 = hvx_vec_mpyacc_f32_f16(vkq1, Q6_Vh_vshuff_Vh(vx0[1]), S0); + break; + } - hvx_mad_f32_f16_aa_rx2_vec(VKQ32, v_ptr, v_ptr + factx->size_v_row_padded, S0, S1, DV); - v_ptr += stride_v2; - } - } - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_SFM, ir); + HVX_Vector S1 = hvx_vec_repl_f16(Q6_V_vror_VR(P, (j + 1) * 2)); + const HVX_Vector * vx1 = (const HVX_Vector *) (v_ptr + factx->size_v_row_padded); + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx0[0]), S0); + vkq0 = hvx_vec_mpyacc_f32_f16(vkq0, Q6_Vh_vshuff_Vh(vx1[0]), S1); + vkq1 = hvx_vec_mpyacc_f32_f16(vkq1, Q6_Vh_vshuff_Vh(vx0[1]), S0); + vkq1 = hvx_vec_mpyacc_f32_f16(vkq1, Q6_Vh_vshuff_Vh(vx1[1]), S1); + v_ptr += stride_v2; + } - // Issue DMA for next+1 block (if exists) - if (ib + 2 < factx->n_blocks) { - const uint32_t next_ib = ib + 2; - const uint32_t next_ic_start = next_ib * FLASH_ATTN_BLOCK_SIZE; - const uint32_t next_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - next_ic_start); + ((HVX_VectorPair *) VKQ32)[0] = vkq0; + ((HVX_VectorPair *) VKQ32)[1] = vkq1; + } else { + hvx_scale_vec_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, ms_vec); - // K - const uint8_t * k_src = (const uint8_t *) k->data + (next_ic_start*nbk1 + ik2*nbk2 + ik3*nbk3); - dma_queue_push(dma, dma_make_ptr(k_base, k_src), factx->size_k_row_padded, nbk1, size_k_row, next_block_size); + for (uint32_t j = 0; j < current_block_size; j += 2) { + if (j + 1 == current_block_size) { + HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); + hvx_mad_f32_f16_aa_vec(VKQ32, v_ptr, S0, DV); + break; + } - // V - const uint8_t * v_src = (const uint8_t *) v->data + (next_ic_start*nbv1 + iv2*nbv2 + iv3*nbv3); - dma_queue_push(dma, dma_make_ptr(v_base, v_src), factx->size_v_row_padded, nbv1, size_v_row, next_block_size); + HVX_Vector S0 = hvx_vec_repl_f16(Q6_V_vror_VR(P, j * 2)); + HVX_Vector S1 = hvx_vec_repl_f16(Q6_V_vror_VR(P, (j + 1) * 2)); - // Mask - if (mask) { - const uint8_t * m_src = (const uint8_t *) (mp_base + next_ic_start); - dma_cache_push(dma, &m_cache, m_src, next_block_size * 2, next_block_size * 2, next_block_size * 2, 1); + hvx_mad_f32_f16_aa_rx2_vec(VKQ32, v_ptr, v_ptr + factx->size_v_row_padded, S0, S1, DV); + v_ptr += stride_v2; + } + } } - } - } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_SFM, head_ir); - if (has_next_ir) { - if (factx->n_blocks % 2 == 0) { - // Queue next row's block 1 (into buffer slot 1, if n_blocks > 1) - if (factx->n_blocks > 1) { - uint8_t * k_dst = spad_k + 1 * factx->size_k_block; - uint8_t * v_dst = spad_v + 1 * factx->size_v_block; + // prefetch V and mask for block ib + 2 after last head finished V accumulation + if (g + 1 == G_local && ib + 2 < factx->n_blocks) { + const uint32_t next_ib = ib + 2; + const uint32_t next_ic_start = next_ib * FLASH_ATTN_BLOCK_SIZE; + const uint32_t next_block_size = MIN(FLASH_ATTN_BLOCK_SIZE, nek1 - next_ic_start); - // K (block 1 of next row) - dma_queue_push(dma, dma_make_ptr(k_dst, next_k_src1), factx->size_k_row_padded, nbk1, size_k_row, next_block_size1); + // V + const dma_addr_t v_src = v->data + next_ic_start*nbv1 + iv2*nbv2 + iv3*nbv3; + dma_queue_push(dma_q, dma_make_data(v_base, v_src), factx->size_v_row_padded, nbv1, size_v_row, next_block_size); - // V (block 1 of next row) - dma_queue_push(dma, dma_make_ptr(v_dst, next_v_src1), factx->size_v_row_padded, nbv1, size_v_row, next_block_size1); - - // Mask (block 1 of next row) + // Mask if (mask) { - dma_cache_push(dma, &m_cache, next_m_src1, next_block_size1 * 2, next_block_size1 * 2, next_block_size1 * 2, 1); + const dma_addr_t m_src = mp_base + next_ic_start * sizeof(__fp16); + dma_cache_dm_push(dma_q, &m_cache, next_ib, m_src, next_block_size * 2, next_block_size * 2, next_block_size * 2, 1); } } - } else { - // Queue next row's block 0 (into buffer slot 0) - { - uint8_t * k_dst = spad_k + 0 * factx->size_k_block; - uint8_t * v_dst = spad_v + 0 * factx->size_v_block; + } // end for g + } // end for ib - // K (block 0 of next row) - dma_queue_push(dma, dma_make_ptr(k_dst, next_k_src0), factx->size_k_row_padded, nbk1, size_k_row, next_block_size0); + for (uint32_t g = 0; g < G_local; ++g) { + const uint32_t head_ir = ir + g; + const uint32_t h = heads[g]; + float * VKQ32 = vkq_ptrs[g]; - // V (block 0 of next row) - dma_queue_push(dma, dma_make_ptr(v_dst, next_v_src0), factx->size_v_row_padded, nbv1, size_v_row, next_block_size0); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, head_ir); - // Mask (block 0 of next row) - if (mask) { - dma_cache_push(dma, &m_cache, next_m_src0, next_block_size0 * 2, next_block_size0 * 2, next_block_size0 * 2, 1); - } - } + float M = hvx_vec_get_f32(M_vec[g]); + float S = hvx_vec_get_f32(S_vec[g]); - // Queue next row's block 1 (into buffer slot 1, if n_blocks > 1) - if (factx->n_blocks > 1) { - uint8_t * k_dst = spad_k + 1 * factx->size_k_block; - uint8_t * v_dst = spad_v + 1 * factx->size_v_block; + if (sinks) { + const float s = factx->spad_sinks[h]; - // K (block 1 of next row) - dma_queue_push(dma, dma_make_ptr(k_dst, next_k_src1), factx->size_k_row_padded, nbk1, size_k_row, next_block_size1); + float vs = 1.0f; - // V (block 1 of next row) - dma_queue_push(dma, dma_make_ptr(v_dst, next_v_src1), factx->size_v_row_padded, nbv1, size_v_row, next_block_size1); + if (s > M) { + HVX_Vector diff_vec = hvx_vec_splat_f32(M - s); + HVX_Vector ms_vec = hvx_vec_exp_f32(diff_vec); + hvx_scale_vec_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, ms_vec); - // Mask (block 1 of next row) - if (mask) { - dma_cache_push(dma, &m_cache, next_m_src1, next_block_size1 * 2, next_block_size1 * 2, next_block_size1 * 2, 1); - } + float ms = hvx_vec_get_f32(ms_vec); + S = S * ms + vs; + } else { + HVX_Vector diff_vec = hvx_vec_splat_f32(s - M); + vs = hvx_vec_get_f32(hvx_vec_exp_f32(diff_vec)); + S += vs; } } - } - - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, ir); - // sinks - float M = hvx_vec_get_f32(M_vec); - float S = hvx_vec_get_f32(S_vec); - if (sinks) { - const float s = ((float *)((char *) sinks->data))[h]; + const float S_inv = S == 0.0f ? 0.0f : 1.0f/S; + hvx_scale_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, S_inv); - float vs = 1.0f; + const uint32_t r_iq3 = fastdiv(head_ir, &factx->src0_div21); + const uint32_t r_iq2 = fastdiv(head_ir - r_iq3*neq2*neq1, &factx->src0_div1); + const uint32_t r_iq1 = (head_ir - r_iq3*neq2*neq1 - r_iq2 * neq1); - if (s > M) { - HVX_Vector diff_vec = hvx_vec_splat_f32(M - s); - HVX_Vector ms_vec = hvx_vec_exp_f32(diff_vec); - hvx_scale_vec_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, ms_vec); + uint8_t * dst_ptr = (uint8_t *) dst->data + r_iq2 * dst->nb[1] + r_iq1 * dst->nb[2] + r_iq3 * dst->nb[3]; - float ms = hvx_vec_get_f32(ms_vec); - S = S * ms + vs; - } else { - HVX_Vector diff_vec = hvx_vec_splat_f32(s - M); - vs = hvx_vec_get_f32(hvx_vec_exp_f32(diff_vec)); - S += vs; + if (dst->type == HTP_TYPE_F32) { + hvx_copy_f32_ua(dst_ptr, (uint8_t *) VKQ32, DV); + } else if (dst->type == HTP_TYPE_F16) { + hvx_copy_f16_f32_ua(dst_ptr, (uint8_t *) VKQ32, DV); } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_O_PROC, head_ir); } - const float S_inv = S == 0.0f ? 0.0f : 1.0f/S; - hvx_scale_f32_aa((uint8_t *) VKQ32, (const uint8_t *) VKQ32, DV, S_inv); - - // Store result - // dst indices - const uint32_t i1 = iq1; - const uint32_t i2 = iq2; - const uint32_t i3 = iq3; - - // dst is permuted: [DV, n_heads, n_tokens, n_seq] - // head stride is nb[1], token stride is nb[2], batch stride is nb[3] - uint8_t * dst_ptr = (uint8_t *) dst->data + i2 * dst->nb[1] + i1 * dst->nb[2] + i3 * dst->nb[3]; - - if (dst->type == HTP_TYPE_F32) { - hvx_copy_f32_ua(dst_ptr, (uint8_t *) VKQ32, DV); - } else if (dst->type == HTP_TYPE_F16) { - hvx_copy_f16_f32_ua(dst_ptr, (uint8_t *) VKQ32, DV); - } - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_O_PROC, ir); + ir += G_local; } } @@ -625,7 +622,13 @@ static void fa_k_interleave_thread(unsigned int n, unsigned int i, void * data) struct htp_thread_trace * tr = &factx->octx->ctx->trace[i]; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_K_PREP, (uint16_t) (args->kv_start + start)); - hmx_interleave_rows_to_tiles(factx->vtcm_k_tiles[args->buf_idx], (const __fp16 *) args->curr_k, total_rows, factx->DK, + if (factx->k->type == HTP_TYPE_Q8_0) { + for (uint32_t r = start; r < end; ++r) { + __fp16 * row_k = (__fp16 *)((char *)args->curr_k + r * args->src_stride * sizeof(__fp16)); + hvx_dequantize_row_q8_0_f16(row_k, row_k, factx->DK); + } + } + hmx_interleave_rows_to_tiles(factx->vtcm_k_tiles[args->buf_idx], (const __fp16 *) args->curr_k, total_rows, factx->DK_pad, args->src_stride, start, end); htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_K_PREP, (uint16_t) (args->kv_start + start)); } @@ -673,7 +676,13 @@ static void fa_v_interleave_thread(unsigned int n, unsigned int i, void * data) struct htp_thread_trace * tr = &factx->octx->ctx->trace[i]; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_FA_V_PREP, (uint16_t) (args->kv_start + start)); - hmx_interleave_cols_to_tiles(v_tiles_dst, (const __fp16 *) args->v_src, total_rows, factx->DV, + if (factx->v->type == HTP_TYPE_Q8_0) { + for (uint32_t r = start; r < end; ++r) { + __fp16 * row_v = (__fp16 *)((char *)args->v_src + r * args->src_stride * sizeof(__fp16)); + hvx_dequantize_row_q8_0_f16(row_v, row_v, factx->DV); + } + } + hmx_interleave_cols_to_tiles(v_tiles_dst, (const __fp16 *) args->v_src, total_rows, factx->DV_pad, args->src_stride, (uint32_t) args->n_col_tiles, start, end); htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_FA_V_PREP, (uint16_t) (args->kv_start + start)); } @@ -747,7 +756,7 @@ static void fa_q_load_thread(unsigned int n, unsigned int i, void * data) { const size_t m_end = hex_smin(m_start + m_bytes_per_t, col_vec_bytes); if (factx->sinks) { - const float * sinks_data = (const float *) (uintptr_t) factx->sinks->data; + const float * sinks_data = factx->vtcm_sinks; float * m_vec = (float *) factx->vtcm_m_vec; const size_t r_start = l_start / sizeof(float); const size_t r_end = l_end / sizeof(float); @@ -782,13 +791,14 @@ static void fa_q_load_thread(unsigned int n, unsigned int i, void * data) { } } - // Initialize vtcm_d_tiles and vtcm_d_inv_l to 0 + // Zero the whole rescale region: vtcm_d_tiles[0], the optional vtcm_d_tiles[1] + // and vtcm_d_inv_l are equal-sized and allocated back to back, so one run covers + // them all. The scatter only ever writes the diagonal, ignore the rest. const size_t d_bytes_per_t = hex_align_up(d_tile_bytes / n, 128); const size_t d_start = i * d_bytes_per_t; const size_t d_end = hex_smin(d_start + d_bytes_per_t, d_tile_bytes); if (d_start < d_tile_bytes) { - hvx_splat_u8_a((char *) factx->vtcm_d_tiles + d_start, 0, d_end - d_start); - hvx_splat_u8_a((char *) factx->vtcm_d_inv_l + d_start, 0, d_end - d_start); + hvx_splat_u8_a((char *) factx->vtcm_d_tiles[0] + d_start, 0, d_end - d_start); } } @@ -798,17 +808,22 @@ static void fa_q_load_thread(unsigned int n, unsigned int i, void * data) { const uint32_t kv_head = args->kv_head; const uint32_t ib3 = args->ib3; - assert(factx->DK == factx->DV); - const bool use_q_dma = (factx->vtcm_q_dma != NULL); __fp16 * q_tiles = factx->vtcm_q_tiles; + const size_t DK_pad = factx->DK_pad; if (use_q_dma) { const size_t g_rows_end = hex_smin(end, n_rows_g); const uint32_t d_limit = factx->is_q_fp32 ? DK / 32 : DK / 64; uint8_t * q_flat = (uint8_t *) factx->vtcm_q_dma; - if (factx->is_q_fp32) { + if (DK_pad != DK) { + if (factx->is_q_fp32) { + hmx_fa_q_prep_fp32_pad(q_tiles, q_flat, start, end, g_rows_end, DK, DK_pad, G, args->n_rows_q, &factx->div_G, args->q_transposed); + } else { + hmx_fa_q_prep_fp16_pad(q_tiles, q_flat, start, end, g_rows_end, DK, DK_pad, G, args->n_rows_q, &factx->div_G, args->q_transposed); + } + } else if (factx->is_q_fp32) { switch (d_limit) { case 2: hmx_fa_q_prep_fp32_d2(q_tiles, q_flat, start, end, g_rows_end, DK, G, args->n_rows_q, &factx->div_G, args->q_transposed); break; case 4: hmx_fa_q_prep_fp32_d4(q_tiles, q_flat, start, end, g_rows_end, DK, G, args->n_rows_q, &factx->div_G, args->q_transposed); break; @@ -824,7 +839,7 @@ static void fa_q_load_thread(unsigned int n, unsigned int i, void * data) { } else { // Fallback: direct-from-DDR/L2 path hmx_fa_q_prep_fallback(q_tiles, q->data, q->nb[1], q->nb[2], q->nb[3], - q_start, kv_head, ib3, start, end, n_rows_g, G, DK, factx->is_q_fp32, &factx->div_G); + q_start, kv_head, ib3, start, end, n_rows_g, G, DK, DK_pad, factx->is_q_fp32, &factx->div_G); } } @@ -918,6 +933,8 @@ static void fa_o_store_thread_f32(unsigned int n, unsigned int i, void * data) { const uint32_t kv_head = args->kv_head; const uint32_t ib3 = args->ib3; + const size_t DV_pad = factx->DV_pad; + size_t q_idx = fastdiv(start, &factx->div_G); size_t h_idx = fastmodulo(start, G, &factx->div_G); @@ -927,7 +944,7 @@ static void fa_o_store_thread_f32(unsigned int n, unsigned int i, void * data) { size_t r0 = r / HMX_FP16_TILE_N_ROWS; size_t r1 = r % HMX_FP16_TILE_N_ROWS; - const __fp16 * tile_row_base = o_tile_src + r0 * HMX_FP16_TILE_N_ROWS * DV; + const __fp16 * tile_row_base = o_tile_src + r0 * HMX_FP16_TILE_N_ROWS * DV_pad; for (uint32_t d = 0; d < DV / 32; ++d) { const HVX_Vector * in_tile = (const HVX_Vector *) (tile_row_base + d * HMX_FP16_TILE_N_ELMS); @@ -938,6 +955,16 @@ static void fa_o_store_thread_f32(unsigned int n, unsigned int i, void * data) { *(HVX_UVector *) (out + d * 32) = Q6_V_hi_W(vp); } } + // Ragged tail: DV not a multiple of 32 (e.g. 72 -> last 8 lanes). Partial vector-write + // for the remaining (DV % 32) floats. + const uint32_t d_tail = DV / 32; + const uint32_t rem = DV - d_tail * 32; + if (rem) { + const HVX_Vector * in_tile = (const HVX_Vector *) (tile_row_base + d_tail * HMX_FP16_TILE_N_ELMS); + HVX_VectorPair vp = hvx_vec_f16_to_f32_shuff(in_tile[r1 / 2]); + HVX_Vector vd = (r1 % 2 == 0) ? Q6_V_lo_W(vp) : Q6_V_hi_W(vp); + hvx_vec_store_u((void *) (out + d_tail * 32), rem * sizeof(float), vd); + } h_idx++; if (h_idx == G) { @@ -972,6 +999,9 @@ static void fa_o_store_thread_f16(unsigned int n, unsigned int i, void * data) { const uint32_t kv_head = args->kv_head; const uint32_t ib3 = args->ib3; + // O-tiles use the padded head dim (DV_pad); dst holds the real DV lanes. + const size_t DV_pad = factx->DV_pad; + size_t q_idx = fastdiv(start, &factx->div_G); size_t h_idx = fastmodulo(start, G, &factx->div_G); @@ -981,7 +1011,7 @@ static void fa_o_store_thread_f16(unsigned int n, unsigned int i, void * data) { size_t r0 = r / HMX_FP16_TILE_N_ROWS; size_t r1 = r % HMX_FP16_TILE_N_ROWS; - const __fp16 * tile_row_base = o_tile_src + r0 * HMX_FP16_TILE_N_ROWS * DV; + const __fp16 * tile_row_base = o_tile_src + r0 * HMX_FP16_TILE_N_ROWS * DV_pad; for (uint32_t d = 0; d < DV / 64; ++d) { const __fp16 * in_dtile = tile_row_base + d * HMX_FP16_TILE_N_ELMS * 2; @@ -994,6 +1024,17 @@ static void fa_o_store_thread_f16(unsigned int n, unsigned int i, void * data) { *(HVX_UVector *) (out + d * 64) = Q6_V_hi_W(vp); } } + // Ragged tail when DV is not a multiple of 64. + const uint32_t d_tail = DV / 64; + const uint32_t rem = DV - d_tail * 64; + if (rem) { + const __fp16 * in_dtile = tile_row_base + d_tail * HMX_FP16_TILE_N_ELMS * 2; + const HVX_Vector * pv_in0 = ((const HVX_Vector *) in_dtile) + r1 / 2; + const HVX_Vector * pv_in1 = pv_in0 + 16; + HVX_VectorPair vp = Q6_W_vdeal_VVR(*pv_in1, *pv_in0, -2); + HVX_Vector vd = (r1 % 2 == 0) ? Q6_V_lo_W(vp) : Q6_V_hi_W(vp); + hvx_vec_store_u((void *) (out + d_tail * 64), rem * sizeof(__fp16), vd); + } h_idx++; if (h_idx == G) { @@ -1432,17 +1473,19 @@ static inline void fa_softmax_impl( const HVX_VectorPred q_32_mask = Q6_Q_vsetq_R(32 * sizeof(__fp16)); HVX_Vector v_exp_m_diff = exp_m_diff_f16; + __fp16 * const d_tiles_out = factx->vtcm_d_tiles[args->buf_idx]; + size_t t0 = r_vec_idx * 2; if (t0 < args->n_row_tiles) { const HVX_Vector v_content = v_exp_m_diff; - __fp16 * out_base = factx->vtcm_d_tiles + t0 * (args->n_row_tiles_g_br + 1) * HMX_FP16_TILE_N_ELMS; + __fp16 * out_base = d_tiles_out + t0 * HMX_FP16_TILE_N_ELMS; Q6_vscatter_QRMVhV(q_32_mask, (size_t) out_base, HMX_FP16_TILE_SIZE - 1, v_offsets, v_content); } size_t t1 = r_vec_idx * 2 + 1; if (t1 < args->n_row_tiles) { const HVX_Vector v_content = Q6_V_vror_VR(v_exp_m_diff, 64); - __fp16 * out_base = factx->vtcm_d_tiles + t1 * (args->n_row_tiles_g_br + 1) * HMX_FP16_TILE_N_ELMS; + __fp16 * out_base = d_tiles_out + t1 * HMX_FP16_TILE_N_ELMS; Q6_vscatter_QRMVhV(q_32_mask, (size_t) out_base, HMX_FP16_TILE_SIZE - 1, v_offsets, v_content); } } @@ -1484,7 +1527,7 @@ static void fa_softmax_thread(unsigned int n, unsigned int i, void * data) { const bool mask_broadcast = factx->mask_broadcast; const bool is_g1 = (args->G == 1); const bool has_alibi = args->has_alibi; - const bool has_softcap = (factx->logit_softcap != 0.0f); + const bool has_softcap = factx->has_softcap; fa_softmax_impl(n, i, data, has_mask, mask_broadcast, is_g1, has_alibi, has_softcap); } @@ -1506,7 +1549,7 @@ static __attribute__((noinline)) void fa_build_d_diag_inv_l(struct hmx_fa_contex v_content = Q6_V_vror_VR(v_content, 64); } - __fp16 * out_base = factx->vtcm_d_inv_l + i * (n_row_tiles_g_br + 1) * HMX_FP16_TILE_N_ELMS; + __fp16 * out_base = factx->vtcm_d_inv_l + i * HMX_FP16_TILE_N_ELMS; Q6_vscatter_QRMVhV(q_32_mask, (size_t) out_base, HMX_FP16_TILE_SIZE - 1, v_offsets, v_content); } } @@ -1519,9 +1562,9 @@ static void fa_phase_softmax_and_build_d(struct hmx_fa_context * factx, const size_t n_row_vec_cnt = hmx_ceil_div(sargs->n_rows_g, 64); worker_callback_t softmax_fn = fa_softmax_thread; - if (sargs->mask == NULL && factx->logit_softcap == 0.0f && !sargs->has_alibi) { + if (sargs->mask == NULL && !factx->has_softcap && !sargs->has_alibi) { softmax_fn = fa_softmax_thread_nomask; - } else if (sargs->mask != NULL && factx->mask_broadcast && factx->logit_softcap == 0.0f && !sargs->has_alibi) { + } else if (sargs->mask != NULL && factx->mask_broadcast && !factx->has_softcap && !sargs->has_alibi) { if (sargs->G == 1) { softmax_fn = fa_softmax_thread_mask_broadcast_g1; } else { @@ -1615,7 +1658,7 @@ static void hmx_fa_o_update_worker(void * data) { const size_t o_stride = n_row_tiles_g_br * HMX_FP16_TILE_N_ELMS; const size_t v_stride = n_tiles_per_bc * HMX_FP16_TILE_N_ELMS; for (size_t r = 0; r < n_row_tiles; ++r) { - const __fp16 * d_diag = d_tiles + r * (n_row_tiles_g_br + 1) * HMX_FP16_TILE_N_ELMS; + const __fp16 * d_diag = d_tiles + r * HMX_FP16_TILE_N_ELMS; const __fp16 * p_tile_in = p_tiles + (r * n_tiles_per_bc) * HMX_FP16_TILE_N_ELMS; const __fp16 * o_rc = o_prev + r * HMX_FP16_TILE_N_ELMS; const __fp16 * v_tile_in = v_tiles; @@ -1654,7 +1697,7 @@ static void hmx_fa_o_norm_worker(void * data) { asm volatile(HMX_SET_BIAS("%0") :: "r"((unsigned int)job->hmx_scales)); const size_t o_stride = n_row_tiles_g_br * HMX_FP16_TILE_N_ELMS; for (size_t r = 0; r < n_row_tiles; ++r) { - const __fp16 * d_diag = d_tiles + r * (n_row_tiles_g_br + 1) * HMX_FP16_TILE_N_ELMS; + const __fp16 * d_diag = d_tiles + r * HMX_FP16_TILE_N_ELMS; const __fp16 * o_rc = o_prev + r * HMX_FP16_TILE_N_ELMS; __fp16 * o_out = o_curr + r * DV_tiles * HMX_FP16_TILE_N_ELMS; @@ -1711,7 +1754,7 @@ static __attribute__((noinline)) void fa_compute_slopes( } static void fa_push_mask_dma_gqa( - dma_queue * dma, + dma_queue * dma_q, const struct htp_tensor * mask, uint32_t q_start, uint32_t im3, @@ -1726,36 +1769,36 @@ static void fa_push_mask_dma_gqa( for (uint32_t g = 0; g < G; ++g) { const uint32_t h_idx = kv_head * G + g; const uint32_t im2 = fastmodulo(h_idx, mask->ne[2], &factx->src3_div2); - const uint8_t * ms_src = (const uint8_t *) mask->data + q_start * mask->nb[1] + - im2 * mask->nb[2] + im3 * mask->nb[3] + kv_start * sizeof(__fp16); + const dma_addr_t ms_src = mask->data + q_start * mask->nb[1] + + im2 * mask->nb[2] + im3 * mask->nb[3] + kv_start * sizeof(__fp16); uint8_t * ms_dst = (uint8_t *) factx->vtcm_mask_buf + g * m_line_bytes; - dma_queue_push(dma, dma_make_ptr(ms_dst, ms_src), G * m_line_bytes, mask->nb[1], kv_rows * sizeof(__fp16), n_rows_q); + dma_queue_push(dma_q, dma_make_data(ms_dst, ms_src), G * m_line_bytes, mask->nb[1], kv_rows * sizeof(__fp16), n_rows_q); } } -static void fa_pop_mask_dma_gqa(dma_queue * dma, uint32_t G) { +static void fa_pop_mask_dma_gqa(dma_queue * dma_q, uint32_t G) { for (uint32_t g = 0; g < G; ++g) { - dma_queue_pop(dma); + dma_queue_pop(dma_q); } } -static inline void fa_prefetch_block(dma_queue * dma, const struct htp_tensor * k, const struct htp_tensor * v, const struct htp_tensor * mask, +static inline void fa_prefetch_block(dma_queue * dma_q, const struct htp_tensor * k, const struct htp_tensor * v, const struct htp_tensor * mask, uint32_t b, size_t Bc, size_t size_k_row_padded, size_t size_k_row, size_t size_v_row_padded, size_t size_v_row, uint32_t ik2, uint32_t ik3, uint32_t iv2, uint32_t iv3, uint32_t q_start, uint32_t im3, uint32_t kv_head, uint32_t G, size_t m_line_bytes, size_t n_rows_q, size_t nek1, size_t prefetch_buf, struct hmx_fa_context * factx) { const uint32_t prefetch_start = b * Bc; const uint32_t prefetch_rows = hex_smin(Bc, nek1 - prefetch_start); - const uint8_t * k_prefetch_src = (const uint8_t *) k->data + prefetch_start * k->nb[1] + ik2 * k->nb[2] + ik3 * k->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx->vtcm_k_fp16[prefetch_buf], k_prefetch_src), size_k_row_padded, k->nb[1], size_k_row, prefetch_rows); - const uint8_t * v_prefetch_src = (const uint8_t *) v->data + prefetch_start * v->nb[1] + iv2 * v->nb[2] + iv3 * v->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx->vtcm_v_fp16[prefetch_buf], v_prefetch_src), size_v_row_padded, v->nb[1], size_v_row, prefetch_rows); + const dma_addr_t k_prefetch_src = k->data + prefetch_start * k->nb[1] + ik2 * k->nb[2] + ik3 * k->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx->vtcm_k_fp16[prefetch_buf], k_prefetch_src), size_k_row_padded, k->nb[1], size_k_row, prefetch_rows); + const dma_addr_t v_prefetch_src = v->data + prefetch_start * v->nb[1] + iv2 * v->nb[2] + iv3 * v->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx->vtcm_v_fp16[prefetch_buf], v_prefetch_src), size_v_row_padded, v->nb[1], size_v_row, prefetch_rows); if (mask) { if (__builtin_expect(factx->mask_broadcast, true)) { - const uint8_t * ms_src = (const uint8_t *) mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + prefetch_start * sizeof(__fp16); - dma_cache_push(dma, &factx->m_cache, ms_src, m_line_bytes, mask->nb[1], prefetch_rows * sizeof(__fp16), n_rows_q); + const dma_addr_t ms_src = mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + prefetch_start * sizeof(__fp16); + dma_cache_fa_push(dma_q, &factx->m_cache, ms_src, m_line_bytes, mask->nb[1], prefetch_rows * sizeof(__fp16), n_rows_q); } else { - fa_push_mask_dma_gqa(dma, mask, q_start, im3, prefetch_start, kv_head, G, m_line_bytes, prefetch_rows, n_rows_q, factx); + fa_push_mask_dma_gqa(dma_q, mask, q_start, im3, prefetch_start, kv_head, G, m_line_bytes, prefetch_rows, n_rows_q, factx); } } } @@ -1793,8 +1836,11 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { const uint32_t DK = neq0; const uint32_t DV = nev0; - // HMX requires head_dim to be multiple of 32 - if (DK % 32 != 0 || DV % 32 != 0) { + // HMX tiles head_dim in units of 64. head_dim need not be 64- (or 32-) aligned: + // we can operate on DK/DV rounded up to 64 with tail lanes [D, D_pad) zero-filled. + const uint32_t DK_pad = hex_round_up(DK, 64); + const uint32_t DV_pad = hex_round_up(DV, 64); + if (DK == 0 || DV == 0) { return HTP_STATUS_NO_SUPPORT; } @@ -1806,9 +1852,13 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { memset(&factx, 0, sizeof(factx)); factx.octx = octx; factx.sinks = octx->src[4]; // NULL if this op has no attention sinks + factx.k = k; + factx.v = v; factx.n_threads = kparams->n_threads; factx.DK = DK; factx.DV = DV; + factx.DK_pad = DK_pad; + factx.DV_pad = DV_pad; factx.n_kv = nek1; factx.n_kv_heads = n_kv_heads; factx.n_heads = neq2; @@ -1828,13 +1878,14 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { factx.src3_div3 = kparams->src3_div3; } - if (kparams->logit_softcap == 0.0f) { + factx.has_softcap = (kparams->logit_softcap != 0.0f); + if (!factx.has_softcap) { factx.scale = (__fp16) (kparams->scale * EXP_LOG2E_F); // log2(e) } else { factx.scale = (__fp16) kparams->scale; } factx.max_bias = kparams->max_bias; - factx.logit_softcap = (__fp16) (kparams->logit_softcap * EXP_LOG2E_F); + factx.logit_softcap = factx.has_softcap ? (__fp16) (kparams->logit_softcap * EXP_LOG2E_F) : 0; factx.n_head_log2 = kparams->n_head_log2; factx.m0 = kparams->m0; @@ -1847,18 +1898,38 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { const uint32_t n_threads = factx.n_threads; const uint32_t G = factx.G; + // Multi-device: split Q blocks across devices + const uint32_t n_q_blocks = (neq1 + Br - 1) / Br; + uint32_t q_start_min = 0; + uint32_t q_start_max = neq1; + + if (octx->ctx->mdev.count > 1) { + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(n_q_blocks, htp_tensor_mdev_data_aligned(dst) ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + const uint32_t block_start = range.start; + const uint32_t block_end = range.start + range.count; + + if (block_start >= block_end) { + return HTP_STATUS_OK; + } + + q_start_min = block_start * Br; + q_start_max = MIN(block_end * Br, neq1); + } + // ======== VTCM allocation (GQA-aware) ======== // K/V row sizes drive the DMA descriptors (not the VTCM layout) and are used - // throughout the KV loop below. - const size_t size_k_row = DK * sizeof(__fp16); - const size_t size_v_row = DV * sizeof(__fp16); - const size_t size_k_row_padded = hex_round_up(size_k_row, 128); - const size_t size_v_row_padded = hex_round_up(size_v_row, 128); + // throughout the KV loop below. The DMA copies only the real DK/DV columns; the + // staging rows are padded to hold DK_pad/DV_pad columns (tail zero-filled below) + // so the HMX interleave/tile logic can operate on 64-aligned head dims. + const size_t size_k_row = htp_tensor_get_row_size(k->type, DK); + const size_t size_v_row = htp_tensor_get_row_size(v->type, DV); + const size_t size_k_row_padded = hex_round_up(DK_pad * sizeof(__fp16), 128); + const size_t size_v_row_padded = hex_round_up(DV_pad * sizeof(__fp16), 128); // Build the VTCM layout once (shared with the host estimator) and place every - // scratch buffer at its computed offset. + // scratch buffer at its computed offset. Padded head dims size the HMX tiles. struct hmx_fa_vtcm_layout L; - hmx_fa_vtcm_layout_build(&L, G, DK, DV, Br, Bc, n_threads, pipeline, factx.is_q_fp32); + hmx_fa_vtcm_layout_build(&L, G, DK_pad, DV_pad, Br, Bc, n_threads, pipeline, factx.is_q_fp32, factx.sinks != NULL, factx.n_heads); if (L.total_bytes > ctx->vtcm_size) { return HTP_STATUS_VTCM_TOO_SMALL; @@ -1882,7 +1953,8 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { factx.vtcm_s_tiles[1] = VTCM_LAYOUT_PTR_OPTIONAL(__fp16, base, L.off_s_tiles[1], pipeline); factx.vtcm_p_tiles[0] = VTCM_LAYOUT_PTR(__fp16, base, L.off_p_tiles[0]); factx.vtcm_p_tiles[1] = VTCM_LAYOUT_PTR_OPTIONAL(__fp16, base, L.off_p_tiles[1], pipeline); - factx.vtcm_d_tiles = VTCM_LAYOUT_PTR(__fp16, base, L.off_d_tiles); + factx.vtcm_d_tiles[0] = VTCM_LAYOUT_PTR(__fp16, base, L.off_d_tiles[0]); + factx.vtcm_d_tiles[1] = VTCM_LAYOUT_PTR_OPTIONAL(__fp16, base, L.off_d_tiles[1], pipeline); factx.vtcm_d_inv_l = VTCM_LAYOUT_PTR(__fp16, base, L.off_d_inv_l); factx.vtcm_m_vec = VTCM_LAYOUT_PTR(HVX_Vector, base, L.off_m_vec); factx.vtcm_l_vec = VTCM_LAYOUT_PTR(HVX_Vector, base, L.off_l_vec); @@ -1899,22 +1971,41 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { factx.col_vec_bytes = L.col_vec_bytes; factx.d_tile_bytes = L.d_tile_bytes; factx.vtcm_slopes = VTCM_LAYOUT_PTR(__fp16, base, L.off_slopes); + factx.vtcm_sinks = VTCM_LAYOUT_PTR_OPTIONAL(float, base, L.off_sinks, factx.sinks != NULL); const size_t m_line_bytes = L.m_line_bytes; // used by the mask DMAs in the KV loop - dma_cache_init(&factx.m_cache, (uint8_t *) factx.vtcm_mask_buf, L.m_buf_slot_bytes, HMX_FA_DMA_CACHE_SIZE); + dma_cache_fa_init(&factx.m_cache, (uint8_t *) factx.vtcm_mask_buf, L.m_buf_slot_bytes, HMX_FA_DMA_CACHE_SIZE); + + // Head-dim padding: the K/V DMA staging buffers and the flat-Q buffer are laid out + // with padded row strides (size_{k,v,q}_row_padded, covering D_pad columns) but the + // DMA only writes the real D columns per row. Zero the whole staging buffers once up + // front so tail lanes [D, D_pad) stay zero for all KV blocks. No-op when already aligned. + if (DK_pad != DK || DV_pad != DV) { + const size_t k_buf_bytes = (size_t) factx.Bc * size_k_row_padded; + const size_t v_buf_bytes = (size_t) factx.Bc * size_v_row_padded; + hvx_splat_u8_a((char *) factx.vtcm_k_fp16[0], 0, k_buf_bytes); + hvx_splat_u8_a((char *) factx.vtcm_k_fp16[1], 0, k_buf_bytes); + hvx_splat_u8_a((char *) factx.vtcm_v_fp16[0], 0, v_buf_bytes); + hvx_splat_u8_a((char *) factx.vtcm_v_fp16[1], 0, v_buf_bytes); + // Flat-Q DMA scratch + if (factx.vtcm_q_dma) { + const size_t q_dma_bytes = hex_align_up(factx.g_br * DK * (factx.is_q_fp32 ? sizeof(float) : sizeof(__fp16)), 128); + hvx_splat_u8_a((char *) factx.vtcm_q_dma, 0, q_dma_bytes); + } + } // ======== Initialize HMX output scales ======== hmx_init_column_scales(factx.vtcm_hmx_scales_id, Q6_V_vsplat_R(0x3c00)); // 1.0 hmx_init_column_scales(factx.vtcm_hmx_scales_qk, hvx_vec_splat_f16(factx.scale)); - // ======== Skip compute if profiling ======== - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { - return HTP_STATUS_OK; - } - // ======== DMA setup ======== - dma_queue * const dma = ctx->dma[0]; + dma_queue * const dma_q = ctx->dma[0]; + + if (factx.sinks) { + dma_queue_push(dma_q, dma_make_data(factx.vtcm_sinks, factx.sinks->data), L.sinks_bytes, 0, factx.sinks->size, 1); + dma_queue_pop(dma_q); + } const size_t n_row_tiles_g_br = g_br / HMX_FP16_TILE_N_ROWS; const size_t n_tiles_per_bc = Bc / HMX_FP16_TILE_N_COLS; @@ -1935,7 +2026,7 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { // ======== Main loop ======== for (uint32_t ib3 = 0; ib3 < neq3; ++ib3) { const uint32_t im3 = mask ? fastmodulo(ib3, mask->ne[3], &factx.src3_div3) : 0; - for (uint32_t q_start = 0; q_start < neq1; q_start += Br) { + for (uint32_t q_start = q_start_min; q_start < q_start_max; q_start += Br) { const uint32_t n_rows_q = hex_smin(Br, neq1 - q_start); const size_t n_rows_g = n_rows_q * G; const size_t g_br_actual = hex_align_up(n_rows_g, HMX_FP16_TILE_N_ROWS); @@ -1949,32 +2040,33 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { // 1. Push Q and KV DMAs for the very first iteration. // Subsequent iterations are enqueued early at the end of the previous iteration. - if (ib3 == 0 && q_start == 0 && kv_head == 0) { - const uint8_t * q_ptr = (const uint8_t *) q->data; + if (ib3 == 0 && q_start == q_start_min && kv_head == 0) { + const dma_addr_t q_ptr = q->data + q_start * q->nb[1] + + (kv_head * factx.G) * q->nb[2] + ib3 * q->nb[3]; const size_t q_row_bytes = q_transposed ? n_rows_q * q_row_bytes_trans_factor : q_row_bytes_untransposed; const size_t n_rows = q_transposed ? factx.G : n_rows_q; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_q_dma, q_ptr), q_row_bytes, hex_smax(q_src_stride, q_row_bytes), q_row_bytes, n_rows); + dma_queue_push(dma_q, dma_make_data(factx.vtcm_q_dma, q_ptr), q_row_bytes, hex_smax(q_src_stride, q_row_bytes), q_row_bytes, n_rows); if (factx.n_kv_blocks > 0) { - const uint8_t * k_src = (const uint8_t *) k->data + ik2 * k->nb[2] + ik3 * k->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_k_fp16[0], k_src), size_k_row_padded, k->nb[1], size_k_row, kv_rows0); + const dma_addr_t k_src = k->data + ik2 * k->nb[2] + ik3 * k->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_k_fp16[0], k_src), size_k_row_padded, k->nb[1], size_k_row, kv_rows0); - const uint8_t * v_src = (const uint8_t *) v->data + iv2 * v->nb[2] + iv3 * v->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_v_fp16[0], v_src), size_v_row_padded, v->nb[1], size_v_row, kv_rows0); + const dma_addr_t v_src = v->data + iv2 * v->nb[2] + iv3 * v->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_v_fp16[0], v_src), size_v_row_padded, v->nb[1], size_v_row, kv_rows0); if (factx.pipeline && mask) { if (__builtin_expect(factx.mask_broadcast, true)) { - const uint8_t * ms_src = (const uint8_t *) mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + 0; - dma_cache_push(dma, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows0 * sizeof(__fp16), n_rows_q); + const dma_addr_t ms_src = mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + 0; + dma_cache_fa_push(dma_q, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows0 * sizeof(__fp16), n_rows_q); } else { - fa_push_mask_dma_gqa(dma, mask, q_start, im3, 0, kv_head, G, m_line_bytes, kv_rows0, n_rows_q, &factx); + fa_push_mask_dma_gqa(dma_q, mask, q_start, im3, 0, kv_head, G, m_line_bytes, kv_rows0, n_rows_q, &factx); } } } } // 2. Pop Q DMA (blocks until Q is loaded) - dma_queue_pop(dma); + dma_queue_pop(dma_q); // ---- Load Q block & Initialize per-block state ---- fa_phase_q_load(&factx, q, q_start, kv_head, ib3, n_rows_g); @@ -2001,12 +2093,12 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { // Prefetch block 1 early if there are multiple blocks if (factx.n_kv_blocks > 1) { - fa_prefetch_block(dma, k, v, mask, 1, Bc, size_k_row_padded, size_k_row, size_v_row_padded, size_v_row, + fa_prefetch_block(dma_q, k, v, mask, 1, Bc, size_k_row_padded, size_k_row, size_v_row_padded, size_v_row, ik2, ik3, iv2, iv3, q_start, im3, kv_head, G, m_line_bytes, n_rows_q, nek1, 1, &factx); } // Prep and start QK-dot(0) - void * curr_k0 = dma_queue_pop(dma).dst; + void * curr_k0 = (void *) dma_queue_pop(dma_q).dst; fa_phase_k_interleave(&factx, kv_rows0, k_src_stride, curr_k0, 0, 0); qk_job[0].q_tiles = factx.vtcm_q_tiles; @@ -2014,7 +2106,7 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { qk_job[0].s_tiles = factx.vtcm_s_tiles[0]; qk_job[0].n_row_tiles = n_row_tiles; qk_job[0].n_col_tiles = hmx_ceil_div(kv_rows0, HMX_FP16_TILE_N_COLS); - qk_job[0].n_dot_tiles = DK / 32; + qk_job[0].n_dot_tiles = DK_pad / 32; qk_job[0].n_tiles_per_bc = n_tiles_per_bc; qk_job[0].hmx_scales = factx.vtcm_hmx_scales_qk; hmx_queue_push(hmx_q, hmx_queue_make_desc(hmx_fa_qk_dot_worker, &qk_job[0])); @@ -2025,27 +2117,50 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { const size_t n_col_tiles = hmx_ceil_div(kv_rows, HMX_FP16_TILE_N_COLS); // ---- 1. Pop and run V-prep for current block ---- - void * curr_v = dma_queue_pop(dma).dst; + void * curr_v = (void *) dma_queue_pop(dma_q).dst; fa_phase_v_interleave(&factx, kv_rows, v_src_stride, curr_v, factx.vtcm_v_tiles[buf_idx], n_tiles_per_bc, kv_start); // ---- 2. Pop and run mask-prep for current block ---- __fp16 * current_mask_vtcm = NULL; if (mask) { if (__builtin_expect(factx.mask_broadcast, true)) { - current_mask_vtcm = (__fp16 *) dma_queue_pop(dma).dst; + current_mask_vtcm = (__fp16 *) dma_queue_pop(dma_q).dst; } else { - fa_pop_mask_dma_gqa(dma, G); + fa_pop_mask_dma_gqa(dma_q, G); current_mask_vtcm = factx.vtcm_mask_buf; } } - // ---- 3. Pop and run K-prep for next block & push next QK-dot ---- + // ---- 3. Start HMX O update for block kv_blk - 1 (reads P[1 - buf_idx], V[1 - buf_idx], D) ---- + // O update relys on the previous block's P and V tiles. + // O update MUST be pushed before the next block's QK-dot: hmx_queue_pop() retires the + // oldest descriptor, so push order alone decides which pop waits for which job. + // If OU went in after QK(i+1), the pop below would retire QK(i+1) and leave + // OU(i-1) in flight into the next iteration, where V-prep overwrites V[prev_buf]. + if (kv_blk > 0) { + const size_t prev_buf = 1 - buf_idx; + ou_job[prev_buf].o_curr = o_tile_curr; + ou_job[prev_buf].o_prev = o_tile_prev; + ou_job[prev_buf].p_tiles = factx.vtcm_p_tiles[prev_buf]; + ou_job[prev_buf].v_tiles = factx.vtcm_v_tiles[prev_buf]; + ou_job[prev_buf].d_tiles = factx.vtcm_d_tiles[prev_buf]; + ou_job[prev_buf].hmx_scales = factx.vtcm_hmx_scales_id; + ou_job[prev_buf].n_row_tiles = n_row_tiles; + ou_job[prev_buf].n_col_tiles = + hmx_ceil_div(hex_smin(Bc, nek1 - (kv_blk - 1) * Bc), HMX_FP16_TILE_N_COLS); + ou_job[prev_buf].n_row_tiles_g_br = n_row_tiles_g_br; + ou_job[prev_buf].n_tiles_per_bc = n_tiles_per_bc; + ou_job[prev_buf].DV = DV_pad; + hmx_queue_push(hmx_q, hmx_queue_make_desc(hmx_fa_o_update_worker, &ou_job[prev_buf])); + } + + // ---- 4. Pop and run K-prep for next block & push next QK-dot ---- if (kv_blk + 1 < factx.n_kv_blocks) { const uint32_t next_start = (kv_blk + 1) * Bc; const uint32_t next_rows = hex_smin(Bc, nek1 - next_start); const size_t next_buf = 1 - buf_idx; - void * next_k = dma_queue_pop(dma).dst; + void * next_k = (void *) dma_queue_pop(dma_q).dst; fa_phase_k_interleave(&factx, next_rows, k_src_stride, next_k, next_start, next_buf); qk_job[next_buf].q_tiles = factx.vtcm_q_tiles; @@ -2053,16 +2168,16 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { qk_job[next_buf].s_tiles = factx.vtcm_s_tiles[next_buf]; qk_job[next_buf].n_row_tiles = n_row_tiles; qk_job[next_buf].n_col_tiles = hmx_ceil_div(next_rows, HMX_FP16_TILE_N_COLS); - qk_job[next_buf].n_dot_tiles = DK / 32; + qk_job[next_buf].n_dot_tiles = DK_pad / 32; qk_job[next_buf].n_tiles_per_bc = n_tiles_per_bc; qk_job[next_buf].hmx_scales = factx.vtcm_hmx_scales_qk; hmx_queue_push(hmx_q, hmx_queue_make_desc(hmx_fa_qk_dot_worker, &qk_job[next_buf])); } - // ---- 4. Wait for current block's QK-dot to finish ---- + // ---- 5. Wait for current block's QK-dot to finish ---- hmx_queue_pop(hmx_q); - // ---- 5. Phase 2: softmax + build_D ---- + // ---- 6. Phase 2: softmax + build_D ---- fa_softmax_args_t sargs; memset(&sargs, 0, sizeof(sargs)); sargs.factx = &factx; @@ -2085,23 +2200,6 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { sargs.mask_vtcm_row_stride = factx.mask_buf_row_stride; sargs.slopes = factx.vtcm_slopes; - // Start HMX O update for block kv_blk - 1 (reads P[1 - buf_idx], V[1 - buf_idx]) - if (kv_blk > 0) { - const size_t prev_buf = 1 - buf_idx; - ou_job[prev_buf].o_curr = o_tile_curr; - ou_job[prev_buf].o_prev = o_tile_prev; - ou_job[prev_buf].p_tiles = factx.vtcm_p_tiles[prev_buf]; - ou_job[prev_buf].v_tiles = factx.vtcm_v_tiles[prev_buf]; - ou_job[prev_buf].d_tiles = factx.vtcm_d_tiles; - ou_job[prev_buf].hmx_scales = factx.vtcm_hmx_scales_id; - ou_job[prev_buf].n_row_tiles = n_row_tiles; - ou_job[prev_buf].n_col_tiles = hmx_ceil_div(hex_smin(Bc, nek1 - (kv_blk - 1) * Bc), HMX_FP16_TILE_N_COLS); - ou_job[prev_buf].n_row_tiles_g_br = n_row_tiles_g_br; - ou_job[prev_buf].n_tiles_per_bc = n_tiles_per_bc; - ou_job[prev_buf].DV = DV; - hmx_queue_push(hmx_q, hmx_queue_make_desc(hmx_fa_o_update_worker, &ou_job[prev_buf])); - } - // Run Softmax on HVX (blocking call) fa_phase_softmax_and_build_d(&factx, &sargs, n_row_tiles, n_row_tiles_g_br); @@ -2113,7 +2211,7 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { // Prefetch block kv_blk + 2 if (kv_blk + 2 < factx.n_kv_blocks) { - fa_prefetch_block(dma, k, v, mask, kv_blk + 2, Bc, size_k_row_padded, size_k_row, size_v_row_padded, size_v_row, + fa_prefetch_block(dma_q, k, v, mask, kv_blk + 2, Bc, size_k_row_padded, size_k_row, size_v_row_padded, size_v_row, ik2, ik3, iv2, iv3, q_start, im3, kv_head, G, m_line_bytes, n_rows_q, nek1, buf_idx, &factx); } @@ -2128,13 +2226,13 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { ou_job[0].o_prev = o_tile_prev; ou_job[0].p_tiles = factx.vtcm_p_tiles[1 - buf_idx]; ou_job[0].v_tiles = factx.vtcm_v_tiles[1 - buf_idx]; - ou_job[0].d_tiles = factx.vtcm_d_tiles; + ou_job[0].d_tiles = factx.vtcm_d_tiles[1 - buf_idx]; ou_job[0].hmx_scales = factx.vtcm_hmx_scales_id; ou_job[0].n_row_tiles = n_row_tiles; ou_job[0].n_col_tiles = last_cols; ou_job[0].n_row_tiles_g_br = n_row_tiles_g_br; ou_job[0].n_tiles_per_bc = n_tiles_per_bc; - ou_job[0].DV = DV; + ou_job[0].DV = DV_pad; hmx_queue_push(hmx_q, hmx_queue_make_desc(hmx_fa_o_update_worker, &ou_job[0])); // Overlapped: run HVX build diag inv L while HMX is busy executing the update @@ -2155,10 +2253,10 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { if (mask) { if (__builtin_expect(factx.mask_broadcast, true)) { - const uint8_t * ms_src = (const uint8_t *) mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + kv_start * sizeof(__fp16); - dma_cache_push(dma, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows * sizeof(__fp16), n_rows_q); + const dma_addr_t ms_src = mask->data + q_start * mask->nb[1] + im3 * mask->nb[3] + kv_start * sizeof(__fp16); + dma_cache_fa_push(dma_q, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows * sizeof(__fp16), n_rows_q); } else { - fa_push_mask_dma_gqa(dma, mask, q_start, im3, kv_start, kv_head, G, m_line_bytes, kv_rows, n_rows_q, &factx); + fa_push_mask_dma_gqa(dma_q, mask, q_start, im3, kv_start, kv_head, G, m_line_bytes, kv_rows, n_rows_q, &factx); } } @@ -2166,14 +2264,14 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { const uint32_t prefetch_start = (kv_blk + 1) * Bc; const uint32_t prefetch_rows = hex_smin(Bc, nek1 - prefetch_start); const size_t prefetch_buf = 1 - buf_idx; - const uint8_t * k_prefetch_src = (const uint8_t *) k->data + prefetch_start * k->nb[1] + ik2 * k->nb[2] + ik3 * k->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_k_fp16[prefetch_buf], k_prefetch_src), size_k_row_padded, k->nb[1], size_k_row, prefetch_rows); - const uint8_t * v_prefetch_src = (const uint8_t *) v->data + prefetch_start * v->nb[1] + iv2 * v->nb[2] + iv3 * v->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_v_fp16[prefetch_buf], v_prefetch_src), size_v_row_padded, v->nb[1], size_v_row, prefetch_rows); + const dma_addr_t k_prefetch_src = k->data + prefetch_start * k->nb[1] + ik2 * k->nb[2] + ik3 * k->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_k_fp16[prefetch_buf], k_prefetch_src), size_k_row_padded, k->nb[1], size_k_row, prefetch_rows); + const dma_addr_t v_prefetch_src = v->data + prefetch_start * v->nb[1] + iv2 * v->nb[2] + iv3 * v->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_v_fp16[prefetch_buf], v_prefetch_src), size_v_row_padded, v->nb[1], size_v_row, prefetch_rows); } // Wait for current K DMA and interleave - void * curr_k = dma_queue_pop(dma).dst; + void * curr_k = (void *) dma_queue_pop(dma_q).dst; fa_phase_k_interleave(&factx, kv_rows, k_src_stride, curr_k, kv_start, 0); { @@ -2182,7 +2280,7 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { qk_job.s_tiles = factx.vtcm_s_tiles[0]; qk_job.n_row_tiles = n_row_tiles; qk_job.n_col_tiles = n_col_tiles; - qk_job.n_dot_tiles = (size_t) (DK / 32); + qk_job.n_dot_tiles = (size_t) (DK_pad / 32); qk_job.n_tiles_per_bc = n_tiles_per_bc; qk_job.hmx_scales = factx.vtcm_hmx_scales_qk; @@ -2191,16 +2289,16 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { } // Wait for current V DMA and interleave - void * curr_v = dma_queue_pop(dma).dst; + void * curr_v = (void *) dma_queue_pop(dma_q).dst; fa_phase_v_interleave(&factx, kv_rows, v_src_stride, curr_v, factx.vtcm_v_tiles[0], n_tiles_per_bc, kv_start); // ---- Phase 3: softmax + build_D ---- __fp16 * current_mask_vtcm = NULL; if (mask) { if (__builtin_expect(factx.mask_broadcast, true)) { - current_mask_vtcm = (__fp16 *) dma_queue_pop(dma).dst; + current_mask_vtcm = (__fp16 *) dma_queue_pop(dma_q).dst; } else { - fa_pop_mask_dma_gqa(dma, G); + fa_pop_mask_dma_gqa(dma_q, G); current_mask_vtcm = factx.vtcm_mask_buf; } } @@ -2232,13 +2330,13 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { ou_job.o_prev = o_tile_prev; ou_job.p_tiles = factx.vtcm_p_tiles[0]; ou_job.v_tiles = factx.vtcm_v_tiles[0]; - ou_job.d_tiles = factx.vtcm_d_tiles; + ou_job.d_tiles = factx.vtcm_d_tiles[0]; ou_job.hmx_scales = factx.vtcm_hmx_scales_id; ou_job.n_row_tiles = n_row_tiles; ou_job.n_col_tiles = n_col_tiles; ou_job.n_row_tiles_g_br = n_row_tiles_g_br; ou_job.n_tiles_per_bc = n_tiles_per_bc; - ou_job.DV = DV; + ou_job.DV = DV_pad; hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_fa_o_update_worker, &ou_job)); if (kv_blk + 1 == factx.n_kv_blocks) { @@ -2263,8 +2361,8 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { if (next_kv_head >= n_kv_heads) { next_kv_head = 0; next_q_start = q_start + Br; - if (next_q_start >= neq1) { - next_q_start = 0; + if (next_q_start >= q_start_max) { + next_q_start = q_start_min; next_ib3 = ib3 + 1; } } @@ -2272,10 +2370,10 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { if (has_next) { const uint32_t next_n_rows_q = hex_smin(Br, neq1 - next_q_start); - const uint8_t * next_q_ptr = (const uint8_t *) q->data + next_q_start * q->nb[1] + (next_kv_head * factx.G) * q->nb[2] + next_ib3 * q->nb[3]; + const dma_addr_t next_q_ptr = q->data + next_q_start * q->nb[1] + (next_kv_head * factx.G) * q->nb[2] + next_ib3 * q->nb[3]; const size_t next_q_row_bytes = q_transposed ? next_n_rows_q * q_row_bytes_trans_factor : q_row_bytes_untransposed; const size_t next_n_rows = q_transposed ? factx.G : next_n_rows_q; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_q_dma, next_q_ptr), next_q_row_bytes, hex_smax(q_src_stride, next_q_row_bytes), next_q_row_bytes, next_n_rows); + dma_queue_push(dma_q, dma_make_data(factx.vtcm_q_dma, next_q_ptr), next_q_row_bytes, hex_smax(q_src_stride, next_q_row_bytes), next_q_row_bytes, next_n_rows); if (factx.n_kv_blocks > 0) { const uint32_t next_ik2 = next_kv_head; @@ -2287,11 +2385,11 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { next_iv3 = fastdiv(next_ib3, &kparams->broadcast_rv3); } - const uint8_t * next_k_src = (const uint8_t *) k->data + next_ik2 * k->nb[2] + next_ik3 * k->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_k_fp16[0], next_k_src), size_k_row_padded, k->nb[1], size_k_row, kv_rows0); + const dma_addr_t next_k_src = k->data + next_ik2 * k->nb[2] + next_ik3 * k->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_k_fp16[0], next_k_src), size_k_row_padded, k->nb[1], size_k_row, kv_rows0); - const uint8_t * next_v_src = (const uint8_t *) v->data + next_iv2 * v->nb[2] + next_iv3 * v->nb[3]; - dma_queue_push(dma, dma_make_ptr(factx.vtcm_v_fp16[0], next_v_src), size_v_row_padded, v->nb[1], size_v_row, kv_rows0); + const dma_addr_t next_v_src = v->data + next_iv2 * v->nb[2] + next_iv3 * v->nb[3]; + dma_queue_push(dma_q, dma_make_data(factx.vtcm_v_fp16[0], next_v_src), size_v_row_padded, v->nb[1], size_v_row, kv_rows0); if (factx.pipeline && mask) { uint32_t next_im3 = im3; @@ -2299,10 +2397,10 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { next_im3 = fastmodulo(next_ib3, mask->ne[3], &factx.src3_div3); } if (__builtin_expect(factx.mask_broadcast, true)) { - const uint8_t * ms_src = (const uint8_t *) mask->data + next_q_start * mask->nb[1] + next_im3 * mask->nb[3] + 0; - dma_cache_push(dma, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows0 * sizeof(__fp16), next_n_rows_q); + const dma_addr_t ms_src = mask->data + next_q_start * mask->nb[1] + next_im3 * mask->nb[3] + 0; + dma_cache_fa_push(dma_q, &factx.m_cache, ms_src, m_line_bytes, mask->nb[1], kv_rows0 * sizeof(__fp16), next_n_rows_q); } else { - fa_push_mask_dma_gqa(dma, mask, next_q_start, next_im3, 0, next_kv_head, G, m_line_bytes, kv_rows0, next_n_rows_q, &factx); + fa_push_mask_dma_gqa(dma_q, mask, next_q_start, next_im3, 0, next_kv_head, G, m_line_bytes, kv_rows0, next_n_rows_q, &factx); } } } @@ -2316,7 +2414,7 @@ int hmx_flash_attn_ext(struct htp_ops_context * octx) { on_job.hmx_scales = factx.vtcm_hmx_scales_id; on_job.n_row_tiles = n_row_tiles; on_job.n_row_tiles_g_br = n_row_tiles_g_br; - on_job.DV = DV; + on_job.DV = DV_pad; hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_fa_o_norm_worker, &on_job)); hmx_queue_pop(ctx->hmx_queue); } @@ -2338,7 +2436,13 @@ int op_flash_attn_ext(struct htp_ops_context * octx) { const struct htp_tensor * dst = octx->dst; // Check support - if ((q->type != HTP_TYPE_F16 && q->type != HTP_TYPE_F32) || k->type != HTP_TYPE_F16 || v->type != HTP_TYPE_F16) { + if ((q->type != HTP_TYPE_F16 && q->type != HTP_TYPE_F32) || + (k->type != HTP_TYPE_F16 && k->type != HTP_TYPE_Q8_0) || + (v->type != HTP_TYPE_F16 && v->type != HTP_TYPE_Q8_0)) { + return HTP_STATUS_NO_SUPPORT; + } + + if (htp_tensor_is_extended(dst)) { return HTP_STATUS_NO_SUPPORT; } @@ -2348,14 +2452,18 @@ int op_flash_attn_ext(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } + if (kparams->kernel_type == HTP_FA_KERNEL_HMX) { return hmx_flash_attn_ext(octx); } struct htp_fa_context factx; factx.octx = octx; - - factx.t_start = HAP_perf_get_qtimer_count(); + factx.k = k; + factx.v = v; factx.src0_div21 = kparams->u.hvx.src0_div21; factx.src0_div1 = kparams->u.hvx.src0_div1; @@ -2375,16 +2483,12 @@ int op_flash_attn_ext(struct htp_ops_context * octx) { factx.size_k_row_padded = kparams->u.hvx.size_k_row_padded; factx.size_v_row_padded = kparams->u.hvx.size_v_row_padded; - size_t size_q_block = factx.size_q_row_padded * 1; // single row for now - factx.size_k_block = factx.size_k_row_padded * FLASH_ATTN_BLOCK_SIZE; - factx.size_v_block = factx.size_v_row_padded * FLASH_ATTN_BLOCK_SIZE; - factx.size_m_block = hex_round_up(FLASH_ATTN_BLOCK_SIZE * sizeof(__fp16), 128); - factx.n_blocks = kparams->n_kv_blocks; factx.scale = kparams->scale; factx.max_bias = kparams->max_bias; - factx.logit_softcap = (__fp16) kparams->logit_softcap; + factx.has_softcap = (kparams->logit_softcap != 0.0f); + factx.logit_softcap = factx.has_softcap ? (__fp16) kparams->logit_softcap : 0; factx.n_head_log2 = kparams->n_head_log2; factx.m0 = kparams->m0; @@ -2399,29 +2503,63 @@ int op_flash_attn_ext(struct htp_ops_context * octx) { } // total rows in q - factx.qrows = kparams->qrows; - factx.qrows_per_thread = kparams->qrows_per_thread; + const uint32_t neq1 = q->ne[1]; + const uint32_t neq2 = q->ne[2]; + const uint32_t neq3 = q->ne[3]; + const uint32_t total_qrows = neq1 * neq2 * neq3; + + uint32_t qrow_start = 0; + uint32_t qrows = total_qrows; + + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && ((dst->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_qrows, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + qrow_start = range.start; + qrows = range.count; + } - size_t size_vkq_acc = hex_round_up(v->ne[0] * sizeof(float), 128); // VKQ32 + if (qrows == 0) { + return HTP_STATUS_OK; + } - factx.size_q_block = size_q_block; - factx.size_vkq_acc = size_vkq_acc; + const uint32_t n_threads = octx->n_threads; - uint8_t * vtcm_cur = octx->ctx->vtcm_base; + factx.qrows = qrows; + factx.qrow_start = qrow_start; + factx.qrows_per_thread = fastdiv(qrows + n_threads - 1, &octx->n_threads_div); - factx.spad_q = vtcm_seq_alloc(&vtcm_cur, size_q_block * octx->n_threads); - factx.spad_k = vtcm_seq_alloc(&vtcm_cur, factx.size_k_block * 2 * octx->n_threads); - factx.spad_v = vtcm_seq_alloc(&vtcm_cur, factx.size_v_block * 2 * octx->n_threads); - factx.spad_m = vtcm_seq_alloc(&vtcm_cur, (mask ? factx.size_m_block * HVX_FA_DMA_CACHE_SIZE : 0) * octx->n_threads); - factx.spad_a = vtcm_seq_alloc(&vtcm_cur, size_vkq_acc * octx->n_threads); + const bool has_mask = (mask != NULL); + const bool has_sinks = (octx->src[4] != NULL); + struct hvx_fa_vtcm_layout L; + hvx_fa_vtcm_layout_build(&L, k->ne[0], v->ne[0], factx.is_q_fp32, has_mask, has_sinks, n_head, n_threads); - if ((size_t) (vtcm_cur - octx->ctx->vtcm_base) > octx->ctx->vtcm_size) { + if (L.total_bytes > octx->ctx->vtcm_size) { return HTP_STATUS_VTCM_TOO_SMALL; } - if (!(octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - work_queue_run(octx->ctx->work_queue, flash_attn_ext_f16_thread, &factx, octx->n_threads); + factx.size_q_block = L.size_q_block; + factx.size_k_block = L.size_k_block; + factx.size_v_block = L.size_v_block; + factx.size_m_block = L.size_m_block; + factx.size_vkq_acc = L.size_vkq_acc; + + uint8_t * const base = octx->ctx->vtcm_base; + + factx.spad_q = VTCM_LAYOUT_PTR(uint8_t, base, L.off_q); + factx.spad_k = VTCM_LAYOUT_PTR(uint8_t, base, L.off_k); + factx.spad_v = VTCM_LAYOUT_PTR(uint8_t, base, L.off_v); + factx.spad_m = VTCM_LAYOUT_PTR_OPTIONAL(uint8_t, base, L.off_m, has_mask); + factx.spad_a = VTCM_LAYOUT_PTR(uint8_t, base, L.off_a); + factx.spad_sinks = VTCM_LAYOUT_PTR_OPTIONAL(float, base, L.off_sinks, has_sinks); + + if (has_sinks) { + const struct htp_tensor * sinks = octx->src[4]; + dma_queue * dma_q = octx->ctx->dma[0]; + dma_queue_push(dma_q, dma_make_data(factx.spad_sinks, sinks->data), L.size_sinks, 0, sinks->size, 1); + dma_queue_pop(dma_q); } + work_queue_run(octx->ctx->work_queue, flash_attn_ext_f16_thread, &factx, n_threads); + return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/flash-attn-ops.h b/ggml/src/ggml-hexagon/htp/flash-attn-ops.h index efe5ce54..22bb8c53 100644 --- a/ggml/src/ggml-hexagon/htp/flash-attn-ops.h +++ b/ggml/src/ggml-hexagon/htp/flash-attn-ops.h @@ -51,6 +51,7 @@ struct htp_fa_kernel_params { uint32_t qrows; uint32_t qrows_per_thread; + uint32_t qrow_start; float m0; float m1; uint32_t n_head_log2; @@ -109,7 +110,7 @@ struct hmx_fa_vtcm_layout { size_t off_v_tiles[2]; size_t off_s_tiles[2]; size_t off_p_tiles[2]; - size_t off_d_tiles; + size_t off_d_tiles[2]; size_t off_d_inv_l; size_t off_m_vec; size_t off_l_vec; @@ -120,15 +121,17 @@ struct hmx_fa_vtcm_layout { size_t off_hmx_scales_qk; size_t off_mask_buf; size_t off_slopes; + size_t off_sinks; // Region byte sizes reused by the device at runtime (not just for allocation). size_t q_tile_bytes; size_t o_tile_bytes; size_t s_tile_bytes; // S and P tiles (same size) - size_t d_tile_bytes; + size_t d_tile_bytes; // d_tiles[0..1] + d_inv_l, allocated back to back size_t m_line_bytes; // one mask row size_t m_buf_slot_bytes; // one dma_cache slot = align_up(Br * m_line_bytes, 4096) size_t col_vec_bytes; + size_t sinks_bytes; // Derived strides. size_t row_buf_stride; // HVX vectors (128B) per row buffer @@ -141,15 +144,22 @@ struct hmx_fa_vtcm_layout { // Build the VTCM layout. static inline void hmx_fa_vtcm_layout_build(struct hmx_fa_vtcm_layout * L, - size_t gqa_factor, size_t DK, size_t DV, - size_t Br, size_t Bc, size_t n_threads, bool pipeline, bool is_q_fp32) { + size_t gqa_factor, size_t DK, size_t DV, + size_t Br, size_t Bc, size_t n_threads, + bool pipeline, bool is_q_fp32, + bool has_sinks, size_t n_heads) { const size_t g_br = hex_align_up(gqa_factor * Br, HMX_FP16_TILE_N_ROWS); const size_t q_tile_size = hex_align_up(g_br * DK * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); const size_t o_tile_size = hex_align_up(g_br * DV * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); const size_t k_tile_size = hex_align_up(Bc * DK * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); const size_t v_tile_size = hex_align_up(Bc * DV * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); const size_t s_tile_size = hex_align_up(g_br * Bc * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); - const size_t d_tile_size = hex_align_up(g_br * g_br * sizeof(__fp16), HTP_FA_HMX_TILE_SIZE); + + // The rescale matrices are diagonal: the HMX kernels only ever load the g_br/32 + // tiles that sit on the diagonal, so store just those, packed back to back with + // a stride of one tile. The old [g_br, g_br] square layout allocated g_br/32 + // times more than it used, which is also why a second D buffer was unaffordable. + const size_t d_tile_size = (g_br / HMX_FP16_TILE_N_ROWS) * HTP_FA_HMX_TILE_SIZE; const size_t q_dma_size = hex_align_up(g_br * DK * (is_q_fp32 ? sizeof(float) : sizeof(__fp16)), 128); const size_t k_dma_size = hex_align_up(Bc * hex_round_up(DK * sizeof(__fp16), 128), 128); @@ -160,6 +170,7 @@ static inline void hmx_fa_vtcm_layout_build(struct hmx_fa_vtcm_layout * L, const size_t m_buf_slot = hex_align_up(Br * m_line_size, 256); const size_t m_buf_size = m_buf_slot * HMX_FA_DMA_CACHE_SIZE; const size_t slopes_size = hex_align_up(g_br * sizeof(__fp16), 128); + const size_t sinks_size = hex_round_up(n_heads * sizeof(float), 128); size_t off = 0; @@ -167,7 +178,8 @@ static inline void hmx_fa_vtcm_layout_build(struct hmx_fa_vtcm_layout * L, VTCM_LAYOUT_ALLOC(off, off_q_tiles, q_tile_size); VTCM_LAYOUT_ALLOC(off, off_o_tiles[0], o_tile_size); VTCM_LAYOUT_ALLOC(off, off_o_tiles[1], o_tile_size); - VTCM_LAYOUT_ALLOC(off, off_d_tiles, d_tile_size); + VTCM_LAYOUT_ALLOC(off, off_d_tiles[0], d_tile_size); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_d_tiles[1], d_tile_size, pipeline); VTCM_LAYOUT_ALLOC(off, off_d_inv_l, d_tile_size); // Group B & C share start offset (Group B tiles must be 2KB aligned) @@ -208,62 +220,108 @@ static inline void hmx_fa_vtcm_layout_build(struct hmx_fa_vtcm_layout * L, VTCM_LAYOUT_ALLOC(off, off_hmx_scales_qk, 256); VTCM_LAYOUT_ALLOC(off, off_mask_buf, m_buf_size); VTCM_LAYOUT_ALLOC(off, off_slopes, slopes_size); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_sinks, sinks_size, has_sinks); L->q_tile_bytes = q_tile_size; L->o_tile_bytes = o_tile_size; L->col_vec_bytes = col_vec_size; L->s_tile_bytes = s_tile_size; - L->d_tile_bytes = d_tile_size; + // Measured from the actual offsets rather than assumed to be N * d_tile_size, so + // that inserting a region between them (or adding padding to VTCM_LAYOUT_ALLOC) + // cannot silently leave the tail of the run unzeroed. + L->d_tile_bytes = (L->off_d_inv_l + d_tile_size) - L->off_d_tiles[0]; L->m_line_bytes = m_line_size; L->m_buf_slot_bytes = m_buf_slot; L->row_buf_stride = row_vec_size / 128; L->mask_buf_row_stride = m_line_size / sizeof(__fp16); + L->sinks_bytes = has_sinks ? sinks_size : 0; L->pipeline = pipeline; L->total_bytes = off; } // Exact VTCM usage for a given (gqa_factor, DK, DV, Br, Bc) configuration. -static inline size_t hmx_fa_compute_vtcm_usage(size_t gqa_factor, size_t DK, size_t DV, size_t Br, size_t Bc, size_t n_threads, bool pipeline, bool is_q_fp32) { +static inline size_t hmx_fa_compute_vtcm_usage(size_t gqa_factor, size_t DK, size_t DV, size_t Br, size_t Bc, size_t n_threads, bool pipeline, bool is_q_fp32, bool has_sinks, size_t n_heads) { struct hmx_fa_vtcm_layout L; - hmx_fa_vtcm_layout_build(&L, gqa_factor, DK, DV, Br, Bc, n_threads, pipeline, is_q_fp32); + hmx_fa_vtcm_layout_build(&L, gqa_factor, DK, DV, Br, Bc, n_threads, pipeline, is_q_fp32, has_sinks, n_heads); return L.total_bytes; } #define FA_HVX_BLOCK_SIZE 64 +#define FA_HVX_G_MAX 8 + +struct hvx_fa_vtcm_layout { + size_t off_q; + size_t off_k; + size_t off_v; + size_t off_m; + size_t off_a; + size_t off_sinks; + + size_t size_q_block; + size_t size_k_block; + size_t size_v_block; + size_t size_m_block; + size_t size_vkq_acc; + size_t size_sinks; + + size_t total_bytes; +}; -static inline size_t hvx_fa_compute_vtcm_usage(size_t DK, size_t DV, bool is_q_fp32, bool has_mask, size_t n_threads) { +static inline void hvx_fa_vtcm_layout_build(struct hvx_fa_vtcm_layout * L, + size_t DK, size_t DV, + bool is_q_fp32, bool has_mask, + bool has_sinks, size_t n_heads, + size_t n_threads) { const size_t size_q_row_padded = hex_round_up(DK * (is_q_fp32 ? 4 : 2), 128); const size_t size_k_row_padded = hex_round_up(DK * sizeof(__fp16), 128); const size_t size_v_row_padded = hex_round_up(DV * sizeof(__fp16), 128); - const size_t size_q_block = size_q_row_padded * 1; + const size_t size_q_block = size_q_row_padded * FA_HVX_G_MAX; const size_t size_k_block = size_k_row_padded * FA_HVX_BLOCK_SIZE; const size_t size_v_block = size_v_row_padded * FA_HVX_BLOCK_SIZE; const size_t size_m_block = hex_round_up(FA_HVX_BLOCK_SIZE * sizeof(__fp16), 128); - const size_t size_vkq_acc = hex_round_up(DV * sizeof(float), 128); + const size_t size_vkq_acc = hex_round_up(DV * sizeof(float), 128) * FA_HVX_G_MAX; + const size_t size_sinks = hex_round_up(n_heads * sizeof(float), 128); - const size_t size_per_thread = size_q_block * 1 - + size_k_block * 2 - + size_v_block * 2 - + (has_mask ? size_m_block * HVX_FA_DMA_CACHE_SIZE : 0) - + size_vkq_acc; + size_t off = 0; + + VTCM_LAYOUT_ALLOC(off, off_q, size_q_block * n_threads); + VTCM_LAYOUT_ALLOC(off, off_k, size_k_block * 2 * n_threads); + VTCM_LAYOUT_ALLOC(off, off_v, size_v_block * 2 * n_threads); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_m, size_m_block * HVX_FA_DMA_CACHE_SIZE * n_threads, has_mask); + VTCM_LAYOUT_ALLOC(off, off_a, size_vkq_acc * n_threads); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_sinks, size_sinks, has_sinks); + + L->size_q_block = size_q_block; + L->size_k_block = size_k_block; + L->size_v_block = size_v_block; + L->size_m_block = size_m_block; + L->size_vkq_acc = size_vkq_acc; + L->size_sinks = has_sinks ? size_sinks : 0; + L->total_bytes = off; +} - return size_per_thread * n_threads; +static inline size_t hvx_fa_compute_vtcm_usage(size_t DK, size_t DV, bool is_q_fp32, bool has_mask, bool has_sinks, size_t n_heads, size_t n_threads) { + struct hvx_fa_vtcm_layout L; + hvx_fa_vtcm_layout_build(&L, DK, DV, is_q_fp32, has_mask, has_sinks, n_heads, n_threads); + return L.total_bytes; } #define FA_MIN_KV_BLOCKS 3 // Cost-based (Br, Bc) search for flash attention with pipeline constraint. static inline int hmx_fa_find_chunk_size(size_t * Br_out, - size_t * Bc_out, - size_t gqa_factor, - size_t DK, - size_t DV, - size_t qo_len, - size_t kv_len, - size_t vtcm_budget, - size_t n_threads, - bool is_q_fp32) { + size_t * Bc_out, + size_t gqa_factor, + size_t DK, + size_t DV, + size_t qo_len, + size_t kv_len, + size_t vtcm_budget, + size_t n_threads, + bool is_q_fp32, + bool has_sinks, + size_t n_heads) { const size_t T = HMX_FP16_TILE_N_ROWS; // 32 const size_t br_unit = hmx_ceil_div(T, gqa_factor); const size_t bc_unit = HMX_FP16_TILE_N_COLS * 2; // 64 @@ -287,7 +345,7 @@ static inline int hmx_fa_find_chunk_size(size_t * Br_out, for (size_t Br = Br_max; Br >= br_unit; Br -= br_unit) { // Try all Bc candidates from Bc_limit down to bc_unit for (size_t Bc = Bc_limit; Bc >= bc_unit; Bc -= bc_unit) { - size_t vtcm_needed = hmx_fa_compute_vtcm_usage(gqa_factor, DK, DV, Br, Bc, n_threads, can_pipeline, is_q_fp32); + size_t vtcm_needed = hmx_fa_compute_vtcm_usage(gqa_factor, DK, DV, Br, Bc, n_threads, can_pipeline, is_q_fp32, has_sinks, n_heads); if (vtcm_needed <= vtcm_budget) { // This Bc fits for this Br! const size_t q_blocks = (qo_len + Br - 1) / Br; diff --git a/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c b/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c index 35518e61..1dd828db 100644 --- a/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c +++ b/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c @@ -1,39 +1,45 @@ -#include #include +#include #include +#include +#include -#include "hvx-utils.h" -#include "hex-fastdiv.h" - -#define GGML_COMMON_DECL_C +#include "hvx-base.h" +#include "hvx-copy.h" +#include "hvx-reduce.h" +#include "hvx-exp.h" +#include "dma-queue.h" #include "ggml-common.h" #include "htp-ctx.h" +#include "htp-tensor.h" +#include "htp-vtcm.h" +#include "hmx-utils.h" +#include "hmx-fa-kernels.h" +#include "hmx-queue.h" +#include "gated-delta-net-ops.h" #ifndef MIN #define MIN(a, b) ((a) < (b) ? (a) : (b)) #endif -#define HTP_GDN_MAX_SV 128 - - struct htp_gdn_context { struct htp_ops_context * octx; - uint32_t rows_per_thread; - size_t state_bytes; + const struct htp_gdn_kernel_params * kparams; + struct htp_gdn_vtcm_layout layout; uint8_t * vtcm_base; - size_t vtcm_per_thread; + uint32_t row_start; + uint32_t nrows; }; -static inline HVX_Vector gdn_mul_dot_f32(float * restrict dst, const float * restrict mul, const float * restrict dot, uint32_t n) { +static inline HVX_Vector gdn_mul_dot_f32(float * restrict dst, const HVX_Vector * restrict mul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc = Q6_V_vzero(); - - const uint32_t epv = 128 / sizeof(float); + const uint32_t epv = 128 / sizeof(float); const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { HVX_Vector vd = hvx_vmemu(dst + i * epv); - HVX_Vector vm = hvx_vmem(mul + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vm = mul[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out = hvx_vec_mul_f32_f32(vd, vm); hvx_vmemu(dst + i * epv) = out; acc = hvx_vec_add_f32_f32(acc, hvx_vec_mul_f32_f32(out, vdot)); @@ -41,29 +47,27 @@ static inline HVX_Vector gdn_mul_dot_f32(float * restrict dst, const float * res if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vd = hvx_vmemu(dst + off); - HVX_Vector vm = hvx_vmem(mul + off); - HVX_Vector vdot = hvx_vmem(dot + off); - HVX_Vector out = hvx_vec_mul_f32_f32(vd, vm); - hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + HVX_Vector vm = mul[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); - HVX_Vector prod = hvx_vec_mul_f32_f32(out, vdot); - acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, prod, Q6_V_vzero())); + HVX_Vector zero = Q6_V_vzero(); + + HVX_Vector out = hvx_vec_mul_f32_f32(hvx_vmemu(dst + off), vm); + hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, hvx_vec_mul_f32_f32(out, vdot), zero)); } return hvx_vec_reduce_sum_f32(acc); } -static inline HVX_Vector gdn_mul_scalar_dot_f32(float * restrict dst, float mul, const float * restrict dot, uint32_t n) { +static inline HVX_Vector gdn_mul_scalar_dot_f32(float * restrict dst, HVX_Vector vmul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc = Q6_V_vzero(); - const HVX_Vector vmul = hvx_vec_splat_f32(mul); - - const uint32_t epv = 128 / sizeof(float); + const uint32_t epv = 128 / sizeof(float); const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { HVX_Vector vd = hvx_vmemu(dst + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vdot = dot[i]; HVX_Vector out = hvx_vec_mul_f32_f32(vd, vmul); hvx_vmemu(dst + i * epv) = out; acc = hvx_vec_add_f32_f32(acc, hvx_vec_mul_f32_f32(out, vdot)); @@ -71,29 +75,28 @@ static inline HVX_Vector gdn_mul_scalar_dot_f32(float * restrict dst, float mul, if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vd = hvx_vmemu(dst + off); - HVX_Vector vdot = hvx_vmem(dot + off); - HVX_Vector out = hvx_vec_mul_f32_f32(vd, vmul); - hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); - HVX_Vector prod = hvx_vec_mul_f32_f32(out, vdot); - acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, prod, Q6_V_vzero())); + HVX_Vector zero = Q6_V_vzero(); + + HVX_Vector out = hvx_vec_mul_f32_f32(hvx_vmemu(dst + off), vmul); + hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, hvx_vec_mul_f32_f32(out, vdot), zero)); } return hvx_vec_reduce_sum_f32(acc); } -static inline HVX_Vector gdn_add_scaled_dot_f32(float * restrict dst, const float * restrict src, - HVX_Vector vscale, const float * restrict dot, uint32_t n) { +static inline HVX_Vector gdn_add_scaled_dot_f32(float * restrict dst, const HVX_Vector * restrict src, + HVX_Vector vscale, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc = Q6_V_vzero(); - - const uint32_t epv = 128 / sizeof(float); + const uint32_t epv = 128 / sizeof(float); const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { HVX_Vector vd = hvx_vmemu(dst + i * epv); - HVX_Vector vs = hvx_vmem(src + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vs = src[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out = hvx_vec_add_f32_f32(vd, hvx_vec_mul_f32_f32(vs, vscale)); hvx_vmemu(dst + i * epv) = out; acc = hvx_vec_add_f32_f32(acc, hvx_vec_mul_f32_f32(out, vdot)); @@ -101,22 +104,22 @@ static inline HVX_Vector gdn_add_scaled_dot_f32(float * restrict dst, const floa if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vd = hvx_vmemu(dst + off); - HVX_Vector vs = hvx_vmem(src + off); - HVX_Vector vdot = hvx_vmem(dot + off); - HVX_Vector out = hvx_vec_add_f32_f32(vd, hvx_vec_mul_f32_f32(vs, vscale)); - hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + HVX_Vector vs = src[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); - HVX_Vector prod = hvx_vec_mul_f32_f32(out, vdot); - acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, prod, Q6_V_vzero())); + HVX_Vector zero = Q6_V_vzero(); + + HVX_Vector out = hvx_vec_add_f32_f32(hvx_vmemu(dst + off), hvx_vec_mul_f32_f32(vs, vscale)); + hvx_vec_store_u(dst + off, nloe * sizeof(float), out); + acc = hvx_vec_add_f32_f32(acc, Q6_V_vmux_QVV(mask, hvx_vec_mul_f32_f32(out, vdot), zero)); } return hvx_vec_reduce_sum_f32(acc); } -static inline void gdn_mul_dot4_f32(float * restrict dst0, float * restrict dst1, - float * restrict dst2, float * restrict dst3, const float * restrict mul, - const float * restrict dot, uint32_t n, float * restrict sums) { +static inline HVX_Vector gdn_mul_dot4_f32(float * restrict dst0, float * restrict dst1, + float * restrict dst2, float * restrict dst3, + const HVX_Vector * restrict mul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); @@ -126,8 +129,8 @@ static inline void gdn_mul_dot4_f32(float * restrict dst0, float * restrict dst1 const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vm = hvx_vmem(mul + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vm = mul[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_mul_f32_f32(hvx_vmemu(dst0 + i * epv), vm); HVX_Vector out1 = hvx_vec_mul_f32_f32(hvx_vmemu(dst1 + i * epv), vm); @@ -147,8 +150,8 @@ static inline void gdn_mul_dot4_f32(float * restrict dst0, float * restrict dst1 if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vm = hvx_vmem(mul + off); - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vm = mul[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -169,23 +172,22 @@ static inline void gdn_mul_dot4_f32(float * restrict dst0, float * restrict dst1 } HVX_Vector_x4 acc = { .v = { acc0, acc1, acc2, acc3 } }; - hvx_vec_store_u(sums, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(acc)); + return hvx_vec_reduce_sum_f32x4(acc); } -static inline void gdn_mul_scalar_dot4_f32(float * restrict dst0, float * restrict dst1, - float * restrict dst2, float * restrict dst3, float mul, - const float * restrict dot, uint32_t n, float * restrict sums) { +static inline HVX_Vector gdn_mul_scalar_dot4_f32(float * restrict dst0, float * restrict dst1, + float * restrict dst2, float * restrict dst3, + HVX_Vector vmul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); HVX_Vector acc3 = Q6_V_vzero(); - const HVX_Vector vmul = hvx_vec_splat_f32(mul); const uint32_t epv = 128 / sizeof(float); const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_mul_f32_f32(hvx_vmemu(dst0 + i * epv), vmul); HVX_Vector out1 = hvx_vec_mul_f32_f32(hvx_vmemu(dst1 + i * epv), vmul); @@ -205,7 +207,7 @@ static inline void gdn_mul_scalar_dot4_f32(float * restrict dst0, float * restri if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -226,13 +228,13 @@ static inline void gdn_mul_scalar_dot4_f32(float * restrict dst0, float * restri } HVX_Vector_x4 acc = { .v = { acc0, acc1, acc2, acc3 } }; - hvx_vec_store_u(sums, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(acc)); + return hvx_vec_reduce_sum_f32x4(acc); } -static inline void gdn_add_scaled_dot4_f32(float * restrict dst0, float * restrict dst1, - float * restrict dst2, float * restrict dst3, const float * restrict src, - const float * restrict scale, const float * restrict dot, uint32_t n, - float * restrict sums) { +static inline HVX_Vector gdn_add_scaled_dot4_f32(float * restrict dst0, float * restrict dst1, + float * restrict dst2, float * restrict dst3, + const HVX_Vector * restrict src, const float * restrict scale, + const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); @@ -246,8 +248,8 @@ static inline void gdn_add_scaled_dot4_f32(float * restrict dst0, float * restri const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vs = hvx_vmem(src + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vs = src[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_add_f32_f32(hvx_vmemu(dst0 + i * epv), hvx_vec_mul_f32_f32(vs, scale0)); HVX_Vector out1 = hvx_vec_add_f32_f32(hvx_vmemu(dst1 + i * epv), hvx_vec_mul_f32_f32(vs, scale1)); @@ -267,8 +269,8 @@ static inline void gdn_add_scaled_dot4_f32(float * restrict dst0, float * restri if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vs = hvx_vmem(src + off); - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vs = src[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -289,14 +291,13 @@ static inline void gdn_add_scaled_dot4_f32(float * restrict dst0, float * restri } HVX_Vector_x4 acc = { .v = { acc0, acc1, acc2, acc3 } }; - hvx_vec_store_u(sums, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(acc)); + return hvx_vec_reduce_sum_f32x4(acc); } -static inline void gdn_mul_dot8_f32(float * restrict dst0, float * restrict dst1, +static inline HVX_Vector gdn_mul_dot8_f32(float * restrict dst0, float * restrict dst1, float * restrict dst2, float * restrict dst3, float * restrict dst4, float * restrict dst5, float * restrict dst6, float * restrict dst7, - const float * restrict mul, const float * restrict dot, uint32_t n, - float * restrict sums) { + const HVX_Vector * restrict mul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); @@ -310,8 +311,8 @@ static inline void gdn_mul_dot8_f32(float * restrict dst0, float * restrict dst1 const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vm = hvx_vmem(mul + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vm = mul[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_mul_f32_f32(hvx_vmemu(dst0 + i * epv), vm); HVX_Vector out1 = hvx_vec_mul_f32_f32(hvx_vmemu(dst1 + i * epv), vm); @@ -343,8 +344,8 @@ static inline void gdn_mul_dot8_f32(float * restrict dst0, float * restrict dst1 if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vm = hvx_vmem(mul + off); - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vm = mul[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -378,14 +379,16 @@ static inline void gdn_mul_dot8_f32(float * restrict dst0, float * restrict dst1 HVX_Vector_x4 accA = { .v = { acc0, acc1, acc2, acc3 } }; HVX_Vector_x4 accB = { .v = { acc4, acc5, acc6, acc7 } }; - hvx_vec_store_u(sums + 0, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accA)); - hvx_vec_store_u(sums + 4, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accB)); + HVX_Vector rA = hvx_vec_reduce_sum_f32x4(accA); + HVX_Vector rB = hvx_vec_reduce_sum_f32x4(accB); + HVX_VectorPred q16 = Q6_Q_vsetq2_R(16); + return Q6_V_vmux_QVV(q16, rA, Q6_V_vror_VR(rB, 128 - 16)); } -static inline void gdn_mul_scalar_dot8_f32(float * restrict dst0, float * restrict dst1, +static inline HVX_Vector gdn_mul_scalar_dot8_f32(float * restrict dst0, float * restrict dst1, float * restrict dst2, float * restrict dst3, float * restrict dst4, float * restrict dst5, float * restrict dst6, float * restrict dst7, - float mul, const float * restrict dot, uint32_t n, float * restrict sums) { + HVX_Vector vmul, const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); @@ -394,13 +397,12 @@ static inline void gdn_mul_scalar_dot8_f32(float * restrict dst0, float * restri HVX_Vector acc5 = Q6_V_vzero(); HVX_Vector acc6 = Q6_V_vzero(); HVX_Vector acc7 = Q6_V_vzero(); - const HVX_Vector vmul = hvx_vec_splat_f32(mul); const uint32_t epv = 128 / sizeof(float); const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_mul_f32_f32(hvx_vmemu(dst0 + i * epv), vmul); HVX_Vector out1 = hvx_vec_mul_f32_f32(hvx_vmemu(dst1 + i * epv), vmul); @@ -432,7 +434,7 @@ static inline void gdn_mul_scalar_dot8_f32(float * restrict dst0, float * restri if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -466,15 +468,17 @@ static inline void gdn_mul_scalar_dot8_f32(float * restrict dst0, float * restri HVX_Vector_x4 accA = { .v = { acc0, acc1, acc2, acc3 } }; HVX_Vector_x4 accB = { .v = { acc4, acc5, acc6, acc7 } }; - hvx_vec_store_u(sums + 0, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accA)); - hvx_vec_store_u(sums + 4, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accB)); + HVX_Vector rA = hvx_vec_reduce_sum_f32x4(accA); + HVX_Vector rB = hvx_vec_reduce_sum_f32x4(accB); + HVX_VectorPred q16 = Q6_Q_vsetq2_R(16); + return Q6_V_vmux_QVV(q16, rA, Q6_V_vror_VR(rB, 128 - 16)); } -static inline void gdn_add_scaled_dot8_f32(float * restrict dst0, float * restrict dst1, +static inline HVX_Vector gdn_add_scaled_dot8_f32(float * restrict dst0, float * restrict dst1, float * restrict dst2, float * restrict dst3, float * restrict dst4, float * restrict dst5, float * restrict dst6, float * restrict dst7, - const float * restrict src, const float * restrict scale, - const float * restrict dot, uint32_t n, float * restrict sums) { + const HVX_Vector * restrict src, const float * restrict scale, + const HVX_Vector * restrict dot, uint32_t n) { HVX_Vector acc0 = Q6_V_vzero(); HVX_Vector acc1 = Q6_V_vzero(); HVX_Vector acc2 = Q6_V_vzero(); @@ -496,8 +500,8 @@ static inline void gdn_add_scaled_dot8_f32(float * restrict dst0, float * restri const uint32_t nvec = n / epv; const uint32_t nloe = n % epv; for (uint32_t i = 0; i < nvec; ++i) { - HVX_Vector vs = hvx_vmem(src + i * epv); - HVX_Vector vdot = hvx_vmem(dot + i * epv); + HVX_Vector vs = src[i]; + HVX_Vector vdot = dot[i]; HVX_Vector out0 = hvx_vec_add_f32_f32(hvx_vmemu(dst0 + i * epv), hvx_vec_mul_f32_f32(vs, scale0)); HVX_Vector out1 = hvx_vec_add_f32_f32(hvx_vmemu(dst1 + i * epv), hvx_vec_mul_f32_f32(vs, scale1)); @@ -529,8 +533,8 @@ static inline void gdn_add_scaled_dot8_f32(float * restrict dst0, float * restri if (nloe) { const uint32_t off = nvec * epv; - HVX_Vector vs = hvx_vmem(src + off); - HVX_Vector vdot = hvx_vmem(dot + off); + HVX_Vector vs = src[nvec]; + HVX_Vector vdot = dot[nvec]; HVX_VectorPred mask = Q6_Q_vsetq2_R(nloe * sizeof(float)); HVX_Vector zero = Q6_V_vzero(); @@ -564,13 +568,190 @@ static inline void gdn_add_scaled_dot8_f32(float * restrict dst0, float * restri HVX_Vector_x4 accA = { .v = { acc0, acc1, acc2, acc3 } }; HVX_Vector_x4 accB = { .v = { acc4, acc5, acc6, acc7 } }; - hvx_vec_store_u(sums + 0, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accA)); - hvx_vec_store_u(sums + 4, 4 * sizeof(float), hvx_vec_reduce_sum_f32x4(accB)); + HVX_Vector rA = hvx_vec_reduce_sum_f32x4(accA); + HVX_Vector rB = hvx_vec_reduce_sum_f32x4(accB); + HVX_VectorPred q16 = Q6_Q_vsetq2_R(16); + return Q6_V_vmux_QVV(q16, rA, Q6_V_vror_VR(rB, 128 - 16)); +} + +static inline void gdn_step_kda_f32( + float * restrict s_work, + float * restrict attn_out, + const float * restrict q_t, + const float * restrict k_t, + const float * restrict v_t, + const float * restrict g_t, + float beta_val, + float scale, + uint32_t S_v +) { + const uint32_t epv = 128 / sizeof(float); + const uint32_t nvec = S_v / epv; + const uint32_t nloe = S_v % epv; + + HVX_Vector vq[4]; + HVX_Vector vk[4]; + HVX_Vector vg[4]; + + for (uint32_t i = 0; i < nvec; ++i) { + vq[i] = hvx_vmemu(q_t + i * epv); + vk[i] = hvx_vmemu(k_t + i * epv); + vg[i] = hvx_vec_exp_f32(hvx_vmemu(g_t + i * epv)); + } + if (nloe) { + vq[nvec] = hvx_vmemu(q_t + nvec * epv); + vk[nvec] = hvx_vmemu(k_t + nvec * epv); + vg[nvec] = hvx_vec_exp_f32(hvx_vmemu(g_t + nvec * epv)); + } + + const HVX_Vector vbeta = hvx_vec_splat_f32(beta_val); + const HVX_Vector vscale = hvx_vec_splat_f32(scale); + + float delta[8] __attribute__((aligned(128))); + + uint32_t j = 0; + for (; j + 8 <= S_v; j += 8) { + float * row0 = s_work + (uint64_t) (j + 0) * S_v; + float * row1 = s_work + (uint64_t) (j + 1) * S_v; + float * row2 = s_work + (uint64_t) (j + 2) * S_v; + float * row3 = s_work + (uint64_t) (j + 3) * S_v; + float * row4 = s_work + (uint64_t) (j + 4) * S_v; + float * row5 = s_work + (uint64_t) (j + 5) * S_v; + float * row6 = s_work + (uint64_t) (j + 6) * S_v; + float * row7 = s_work + (uint64_t) (j + 7) * S_v; + + HVX_Vector vsums = gdn_mul_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, + vg, vk, S_v); + + HVX_Vector vv_t = hvx_vmemu(v_t + j); + HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, vsums); + HVX_Vector vdelta = hvx_vec_mul_f32_f32(diff, vbeta); + hvx_vec_store_u(delta, 8 * sizeof(float), vdelta); + + HVX_Vector vattn = gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, + vk, delta, vq, S_v); + + HVX_Vector res_attn = hvx_vec_mul_f32_f32(vattn, vscale); + hvx_vec_store_u(attn_out + j, 8 * sizeof(float), res_attn); + } + for (; j + 4 <= S_v; j += 4) { + float * row0 = s_work + (uint64_t) (j + 0) * S_v; + float * row1 = s_work + (uint64_t) (j + 1) * S_v; + float * row2 = s_work + (uint64_t) (j + 2) * S_v; + float * row3 = s_work + (uint64_t) (j + 3) * S_v; + + HVX_Vector vsums = gdn_mul_dot4_f32(row0, row1, row2, row3, vg, vk, S_v); + + HVX_Vector vv_t = hvx_vmemu(v_t + j); + HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, vsums); + HVX_Vector vdelta = hvx_vec_mul_f32_f32(diff, vbeta); + hvx_vec_store_u(delta, 4 * sizeof(float), vdelta); + + HVX_Vector vattn = gdn_add_scaled_dot4_f32(row0, row1, row2, row3, vk, delta, vq, S_v); + + HVX_Vector res_attn = hvx_vec_mul_f32_f32(vattn, vscale); + hvx_vec_store_u(attn_out + j, 4 * sizeof(float), res_attn); + } + for (; j < S_v; ++j) { + float * row = s_work + (uint64_t) j * S_v; + HVX_Vector vsum = gdn_mul_dot_f32(row, vg, vk, S_v); + HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); + HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), vbeta); + HVX_Vector vres = gdn_add_scaled_dot_f32(row, vk, vdj, vq, S_v); + attn_out[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale)); + } +} + +static inline void gdn_step_scalar_f32( + float * restrict s_work, + float * restrict attn_out, + const float * restrict q_t, + const float * restrict k_t, + const float * restrict v_t, + const float * restrict g_t, + float beta_val, + float scale, + uint32_t S_v +) { + const uint32_t epv = 128 / sizeof(float); + const uint32_t nvec = S_v / epv; + const uint32_t nloe = S_v % epv; + + HVX_Vector vq[4]; + HVX_Vector vk[4]; + + for (uint32_t i = 0; i < nvec; ++i) { + vq[i] = hvx_vmemu(q_t + i * epv); + vk[i] = hvx_vmemu(k_t + i * epv); + } + if (nloe) { + vq[nvec] = hvx_vmemu(q_t + nvec * epv); + vk[nvec] = hvx_vmemu(k_t + nvec * epv); + } + + const HVX_Vector vgate = hvx_vec_exp_f32(hvx_vec_splat_f32(g_t[0])); + const HVX_Vector vbeta = hvx_vec_splat_f32(beta_val); + const HVX_Vector vscale = hvx_vec_splat_f32(scale); + + float delta[8] __attribute__((aligned(128))); + + uint32_t j = 0; + for (; j + 8 <= S_v; j += 8) { + float * row0 = s_work + (uint64_t) (j + 0) * S_v; + float * row1 = s_work + (uint64_t) (j + 1) * S_v; + float * row2 = s_work + (uint64_t) (j + 2) * S_v; + float * row3 = s_work + (uint64_t) (j + 3) * S_v; + float * row4 = s_work + (uint64_t) (j + 4) * S_v; + float * row5 = s_work + (uint64_t) (j + 5) * S_v; + float * row6 = s_work + (uint64_t) (j + 6) * S_v; + float * row7 = s_work + (uint64_t) (j + 7) * S_v; + + HVX_Vector vsums = gdn_mul_scalar_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, + vgate, vk, S_v); + + HVX_Vector vv_t = hvx_vmemu(v_t + j); + HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, vsums); + HVX_Vector vdelta = hvx_vec_mul_f32_f32(diff, vbeta); + hvx_vec_store_u(delta, 8 * sizeof(float), vdelta); + + HVX_Vector vattn = gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, + vk, delta, vq, S_v); + + HVX_Vector res_attn = hvx_vec_mul_f32_f32(vattn, vscale); + hvx_vec_store_u(attn_out + j, 8 * sizeof(float), res_attn); + } + for (; j + 4 <= S_v; j += 4) { + float * row0 = s_work + (uint64_t) (j + 0) * S_v; + float * row1 = s_work + (uint64_t) (j + 1) * S_v; + float * row2 = s_work + (uint64_t) (j + 2) * S_v; + float * row3 = s_work + (uint64_t) (j + 3) * S_v; + + HVX_Vector vsums = gdn_mul_scalar_dot4_f32(row0, row1, row2, row3, vgate, vk, S_v); + + HVX_Vector vv_t = hvx_vmemu(v_t + j); + HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, vsums); + HVX_Vector vdelta = hvx_vec_mul_f32_f32(diff, vbeta); + hvx_vec_store_u(delta, 4 * sizeof(float), vdelta); + + HVX_Vector vattn = gdn_add_scaled_dot4_f32(row0, row1, row2, row3, vk, delta, vq, S_v); + + HVX_Vector res_attn = hvx_vec_mul_f32_f32(vattn, vscale); + hvx_vec_store_u(attn_out + j, 4 * sizeof(float), res_attn); + } + for (; j < S_v; ++j) { + float * row = s_work + (uint64_t) j * S_v; + HVX_Vector vsum = gdn_mul_scalar_dot_f32(row, vgate, vk, S_v); + HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); + HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), vbeta); + HVX_Vector vres = gdn_add_scaled_dot_f32(row, vk, vdj, vq, S_v); + attn_out[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale)); + } } static void gated_delta_net_f32_pp_thread(unsigned int nth, unsigned int ith, void * data) { struct htp_gdn_context * gctx = (struct htp_gdn_context *) data; struct htp_ops_context * octx = gctx->octx; + const struct htp_gdn_kernel_params * kparams = gctx->kparams; const struct htp_tensor * q = octx->src[0]; const struct htp_tensor * k = octx->src[1]; @@ -580,65 +761,55 @@ static void gated_delta_net_f32_pp_thread(unsigned int nth, unsigned int ith, vo const struct htp_tensor * state = octx->src[5]; const struct htp_tensor * dst = octx->dst; - const uint32_t S_v = v->ne[0]; - const uint32_t H = v->ne[1]; - const uint32_t n_tokens = v->ne[2]; - const uint32_t n_seqs = v->ne[3]; - const uint32_t K = octx->op_params[0]; + const uint32_t S_v = kparams->S_v; + const uint32_t H = kparams->H; + const uint32_t n_tokens = kparams->n_tokens; + const uint32_t n_seqs = kparams->n_seqs; + const uint32_t K = kparams->K; + const uint32_t row_end = gctx->row_start + gctx->nrows; - const uint32_t total_rows = H * n_seqs; - if (ith >= total_rows) { + if (ith >= gctx->nrows) { return; } - const uint32_t rq3 = n_seqs / q->ne[3]; - const uint32_t rk3 = n_seqs / k->ne[3]; - const float scale = 1.0f / sqrtf((float) S_v); - + const struct htp_tensor * dst_cache = octx->dsts[1]; + const float scale = kparams->scale; float * dst_base = (float *) (uintptr_t) dst->data; - float * state_out_base = dst_base + (uint64_t) S_v * H * n_tokens * n_seqs; - const float * state_in_base = (const float *) (uintptr_t) state->data; - - const bool kda = (g->ne[0] == S_v); - float local_gate[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_q[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_k[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_sums[32] __attribute__((aligned(128))); - - dma_queue * dma = octx->ctx->dma[ith]; - size_t state_aligned = (size_t) S_v * S_v * sizeof(float); - state_aligned = (state_aligned + 127) & ~(size_t)127; + float * state_out_base = dst_cache ? (float *) (uintptr_t) dst_cache->data : (dst_base + S_v * H * n_tokens * n_seqs); + + dma_queue * dma_q = octx->ctx->dma[ith]; + const struct htp_gdn_vtcm_layout * layout = &gctx->layout; float * s_work[2]; - s_work[0] = (float *) (gctx->vtcm_base + gctx->vtcm_per_thread * ith); - s_work[1] = s_work[0] + state_aligned / sizeof(float); + s_work[0] = (float *) (gctx->vtcm_base + layout->bytes_per_thread * ith); + s_work[1] = s_work[0] + layout->state_aligned / sizeof(float); - struct fastdiv_values fd_H = init_fastdiv_values(H); - struct fastdiv_values fd_q1 = init_fastdiv_values(q->ne[1]); - struct fastdiv_values fd_k1 = init_fastdiv_values(k->ne[1]); - struct fastdiv_values fd_rq3 = init_fastdiv_values(rq3); - struct fastdiv_values fd_rk3 = init_fastdiv_values(rk3); + const struct fastdiv_values * fd_H = &kparams->div_H; + const struct fastdiv_values * fd_q1 = &kparams->div_q1; + const struct fastdiv_values * fd_k1 = &kparams->div_k1; + const struct fastdiv_values * fd_rq3 = &kparams->div_rq3; + const struct fastdiv_values * fd_rk3 = &kparams->div_rk3; - const uint64_t state_seq_stride = state->nb[3] / sizeof(float); - const uint64_t state_size_per_snap = (uint64_t) S_v * S_v * H * n_seqs; + const uint32_t state_seq_stride = kparams->state_seq_stride; + const uint64_t state_size_per_snap = (uint64_t) kparams->state_size_per_snap; + const dma_addr_t state_out_dma_base = dst_cache ? dst_cache->data : (dst->data + S_v * H * n_tokens * n_seqs * sizeof(float)); - uint32_t ir_prefetch = ith; + uint32_t ir_prefetch = gctx->row_start + ith; int spad_idx = 0; // Prefetch preamble (up to 2 steps) - for (int k = 0; k < 2 && ir_prefetch < total_rows; k++) { - const uint32_t piv1 = fastmodulo(ir_prefetch, H, &fd_H); - const uint32_t piv3 = fastdiv(ir_prefetch, &fd_H); - const float * ps_in = state_in_base + (uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v; - // final state lands in snapshot slot 0 (most-recent-first ordering) - float * ps_out = state_out_base + ((uint64_t) piv3 * H + piv1) * S_v * S_v; + for (int step = 0; step < 2 && ir_prefetch < row_end; step++) { + const uint32_t piv1 = fastmodulo(ir_prefetch, H, fd_H); + const uint32_t piv3 = fastdiv(ir_prefetch, fd_H); + dma_addr_t ps_in = state->data + ((uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v) * sizeof(float); + dma_addr_t ps_out = state_out_dma_base + ((uint64_t) piv3 * H + piv1) * S_v * S_v * sizeof(float); // Push dummy write-back - dma_queue_push(dma, dma_make_ptr(ps_out, s_work[spad_idx]), + dma_queue_push(dma_q, dma_make_data(ps_out, s_work[spad_idx]), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), 0); // Push fetch - dma_queue_push(dma, dma_make_ptr(s_work[spad_idx], ps_in), + dma_queue_push(dma_q, dma_make_data(s_work[spad_idx], ps_in), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); @@ -646,26 +817,27 @@ static void gated_delta_net_f32_pp_thread(unsigned int nth, unsigned int ith, vo spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + int curr_spad_idx = 0; - for (uint32_t ir = ith; ir < total_rows; ir += nth) { - dma_queue_pop(dma); - dma_queue_pop(dma); + for (uint32_t ir = gctx->row_start + ith; ir < row_end; ir += nth) { + dma_queue_pop(dma_q); + dma_queue_pop(dma_q); float * s_work_curr = s_work[curr_spad_idx]; - const uint32_t iv1 = fastmodulo(ir, H, &fd_H); - const uint32_t iv3 = fastdiv(ir, &fd_H); - - const uint32_t iq1 = fastmodulo(iv1, q->ne[1], &fd_q1); - const uint32_t ik1 = fastmodulo(iv1, k->ne[1], &fd_k1); - const uint32_t iq3 = fastdiv(iv3, &fd_rq3); - const uint32_t ik3 = fastdiv(iv3, &fd_rk3); + const uint32_t iv1 = fastmodulo(ir, H, fd_H); + const uint32_t iv3 = fastdiv(ir, fd_H); - // final state lands in snapshot slot 0 (most-recent-first ordering) - float * s_out = state_out_base + ((uint64_t) iv3 * H + iv1) * S_v * S_v; + const uint32_t iq1 = fastmodulo(iv1, q->ne[1], fd_q1); + const uint32_t ik1 = fastmodulo(iv1, k->ne[1], fd_k1); + const uint32_t iq3 = fastdiv(iv3, fd_rq3); + const uint32_t ik3 = fastdiv(iv3, fd_rk3); + dma_addr_t s_out = state_out_dma_base + ((uint64_t) iv3 * H + iv1) * S_v * S_v * sizeof(float); float * attn_data = dst_base + ((uint64_t) iv3 * n_tokens * H + iv1) * S_v; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); for (uint32_t t = 0; t < n_tokens; ++t) { const float * q_t = (const float *) ((const uint8_t *) (uintptr_t) q->data + (uint64_t) iq3 * q->nb[3] + (uint64_t) t * q->nb[2] + (uint64_t) iq1 * q->nb[1]); @@ -678,146 +850,36 @@ static void gated_delta_net_f32_pp_thread(unsigned int nth, unsigned int ith, vo const float beta_val = *(const float *) ((const uint8_t *) (uintptr_t) beta->data + (uint64_t) iv3 * beta->nb[3] + (uint64_t) t * beta->nb[2] + (uint64_t) iv1 * beta->nb[1]); - hvx_copy_f32_au((uint8_t *) local_q, (const uint8_t *) q_t, S_v); - hvx_copy_f32_au((uint8_t *) local_k, (const uint8_t *) k_t, S_v); - - if (kda) { - hvx_exp_f32((uint8_t *) local_gate, (const uint8_t *) g_t, S_v, false); - - uint32_t j = 0; - for (; j + 8 <= S_v; j += 8) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - float * row4 = s_work_curr + (uint64_t) (j + 4) * S_v; - float * row5 = s_work_curr + (uint64_t) (j + 5) * S_v; - float * row6 = s_work_curr + (uint64_t) (j + 6) * S_v; - float * row7 = s_work_curr + (uint64_t) (j + 7) * S_v; - gdn_mul_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 8 * sizeof(float), res_attn); - } - for (; j + 4 <= S_v; j += 4) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - gdn_mul_dot4_f32(row0, row1, row2, row3, local_gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot4_f32(row0, row1, row2, row3, local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 4 * sizeof(float), res_attn); - } - HVX_Vector vscale_splat = hvx_vec_splat_f32(scale); - for (; j < S_v; ++j) { - float * row = s_work_curr + (uint64_t) j * S_v; - HVX_Vector vsum = gdn_mul_dot_f32(row, local_gate, local_k, S_v); - HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); - HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), hvx_vec_splat_f32(beta_val)); - HVX_Vector vres = gdn_add_scaled_dot_f32(row, local_k, vdj, local_q, S_v); - attn_data[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale_splat)); - } + if (kparams->kda) { + gdn_step_kda_f32(s_work_curr, attn_data, q_t, k_t, v_t, g_t, beta_val, scale, S_v); } else { - const float gate = expf(g_t[0]); - uint32_t j = 0; - for (; j + 8 <= S_v; j += 8) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - float * row4 = s_work_curr + (uint64_t) (j + 4) * S_v; - float * row5 = s_work_curr + (uint64_t) (j + 5) * S_v; - float * row6 = s_work_curr + (uint64_t) (j + 6) * S_v; - float * row7 = s_work_curr + (uint64_t) (j + 7) * S_v; - gdn_mul_scalar_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 8 * sizeof(float), res_attn); - } - for (; j + 4 <= S_v; j += 4) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - gdn_mul_scalar_dot4_f32(row0, row1, row2, row3, gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot4_f32(row0, row1, row2, row3, local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 4 * sizeof(float), res_attn); - } - HVX_Vector vscale_splat = hvx_vec_splat_f32(scale); - for (; j < S_v; ++j) { - float * row = s_work_curr + (uint64_t) j * S_v; - HVX_Vector vsum = gdn_mul_scalar_dot_f32(row, gate, local_k, S_v); - HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); - HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), hvx_vec_splat_f32(beta_val)); - HVX_Vector vres = gdn_add_scaled_dot_f32(row, local_k, vdj, local_q, S_v); - attn_data[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale_splat)); - } + gdn_step_scalar_f32(s_work_curr, attn_data, q_t, k_t, v_t, g_t, beta_val, scale, S_v); } if (K > 1) { - // snapshot slot mapping: slot 0 = most recent state, slot s = s tokens back. const int64_t target_slot = (int64_t) n_tokens - 1 - (int64_t) t; - if (target_slot >= 0 && target_slot < (int64_t) K) { + if (target_slot > 0 && target_slot < (int64_t) K) { float * curr_state_o = state_out_base + (uint64_t) target_slot * state_size_per_snap + ((uint64_t) iv3 * H + iv1) * S_v * S_v; - if (curr_state_o != s_out) { - hvx_copy_f32_uu((uint8_t *) curr_state_o, (const uint8_t *) s_work_curr, S_v * S_v); - } + hvx_copy_f32_uu((uint8_t *) curr_state_o, (const uint8_t *) s_work_curr, S_v * S_v); } } attn_data += (uint64_t) S_v * H; } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); // Push real write-back - dma_queue_push(dma, dma_make_ptr(s_out, s_work_curr), + dma_queue_push(dma_q, dma_make_data(s_out, s_work_curr), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); // Prefetch next block (if any) - if (ir_prefetch < total_rows) { - const uint32_t piv1 = fastmodulo(ir_prefetch, H, &fd_H); - const uint32_t piv3 = fastdiv(ir_prefetch, &fd_H); - const float * ps_in = state_in_base + (uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v; + if (ir_prefetch < row_end) { + const uint32_t piv1 = fastmodulo(ir_prefetch, H, fd_H); + const uint32_t piv3 = fastdiv(ir_prefetch, fd_H); + dma_addr_t ps_in = state->data + ((uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v) * sizeof(float); - dma_queue_push(dma, dma_make_ptr(s_work[spad_idx], ps_in), + dma_queue_push(dma_q, dma_make_data(s_work[spad_idx], ps_in), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); @@ -827,13 +889,13 @@ static void gated_delta_net_f32_pp_thread(unsigned int nth, unsigned int ith, vo curr_spad_idx ^= 1; } - dma_queue_flush(dma); + dma_queue_flush(dma_q); } - static void gated_delta_net_f32_tg_thread(unsigned int nth, unsigned int ith, void * data) { struct htp_gdn_context * gctx = (struct htp_gdn_context *) data; struct htp_ops_context * octx = gctx->octx; + const struct htp_gdn_kernel_params * kparams = gctx->kparams; const struct htp_tensor * q = octx->src[0]; const struct htp_tensor * k = octx->src[1]; @@ -843,62 +905,51 @@ static void gated_delta_net_f32_tg_thread(unsigned int nth, unsigned int ith, vo const struct htp_tensor * state = octx->src[5]; const struct htp_tensor * dst = octx->dst; - const uint32_t S_v = v->ne[0]; - const uint32_t H = v->ne[1]; - const uint32_t n_seqs = v->ne[3]; + const uint32_t S_v = kparams->S_v; + const uint32_t H = kparams->H; + const uint32_t n_seqs = kparams->n_seqs; + const uint32_t row_end = gctx->row_start + gctx->nrows; - const uint32_t total_rows = H * n_seqs; - if (ith >= total_rows) { + if (ith >= gctx->nrows) { return; } - const uint32_t rq3 = n_seqs / q->ne[3]; - const uint32_t rk3 = n_seqs / k->ne[3]; - const float scale = 1.0f / sqrtf((float) S_v); + const struct htp_tensor * dst_cache = octx->dsts[1]; + const float scale = kparams->scale; + float * dst_base = (float *) (uintptr_t) dst->data; - float * dst_base = (float *) (uintptr_t) dst->data; - float * state_out_base = dst_base + (uint64_t) S_v * H * n_seqs; - const float * state_in_base = (const float *) (uintptr_t) state->data; - - const bool kda = (g->ne[0] == S_v); - float local_gate[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_q[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_k[HTP_GDN_MAX_SV] __attribute__((aligned(128))); - float local_sums[32] __attribute__((aligned(128))); - - dma_queue * dma = octx->ctx->dma[ith]; - size_t state_aligned = (size_t) S_v * S_v * sizeof(float); - state_aligned = (state_aligned + 127) & ~(size_t)127; + dma_queue * dma_q = octx->ctx->dma[ith]; + const struct htp_gdn_vtcm_layout * layout = &gctx->layout; float * s_work[2]; - s_work[0] = (float *) (gctx->vtcm_base + gctx->vtcm_per_thread * ith); - s_work[1] = s_work[0] + state_aligned / sizeof(float); + s_work[0] = (float *) (gctx->vtcm_base + layout->bytes_per_thread * ith); + s_work[1] = s_work[0] + layout->state_aligned / sizeof(float); - struct fastdiv_values fd_H = init_fastdiv_values(H); - struct fastdiv_values fd_q1 = init_fastdiv_values(q->ne[1]); - struct fastdiv_values fd_k1 = init_fastdiv_values(k->ne[1]); - struct fastdiv_values fd_rq3 = init_fastdiv_values(rq3); - struct fastdiv_values fd_rk3 = init_fastdiv_values(rk3); + const struct fastdiv_values * fd_H = &kparams->div_H; + const struct fastdiv_values * fd_q1 = &kparams->div_q1; + const struct fastdiv_values * fd_k1 = &kparams->div_k1; + const struct fastdiv_values * fd_rq3 = &kparams->div_rq3; + const struct fastdiv_values * fd_rk3 = &kparams->div_rk3; - const uint64_t state_seq_stride = state->nb[3] / sizeof(float); + const uint32_t state_seq_stride = kparams->state_seq_stride; + const dma_addr_t state_out_dma_base = dst_cache ? dst_cache->data : (dst->data + S_v * H * n_seqs * sizeof(float)); - uint32_t ir_prefetch = ith; + uint32_t ir_prefetch = gctx->row_start + ith; int spad_idx = 0; // Prefetch preamble (up to 2 steps) - for (int k = 0; k < 2 && ir_prefetch < total_rows; k++) { - const uint32_t piv1 = fastmodulo(ir_prefetch, H, &fd_H); - const uint32_t piv3 = fastdiv(ir_prefetch, &fd_H); - const float * ps_in = state_in_base + (uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v; - // final state lands in snapshot slot 0 (most-recent-first ordering) - float * ps_out = state_out_base + ((uint64_t) piv3 * H + piv1) * S_v * S_v; + for (int step = 0; step < 2 && ir_prefetch < row_end; step++) { + const uint32_t piv1 = fastmodulo(ir_prefetch, H, fd_H); + const uint32_t piv3 = fastdiv(ir_prefetch, fd_H); + dma_addr_t ps_in = state->data + ((uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v) * sizeof(float); + dma_addr_t ps_out = state_out_dma_base + ((uint64_t) piv3 * H + piv1) * S_v * S_v * sizeof(float); // Push dummy write-back - dma_queue_push(dma, dma_make_ptr(ps_out, s_work[spad_idx]), + dma_queue_push(dma_q, dma_make_data(ps_out, s_work[spad_idx]), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), 0); // Push fetch - dma_queue_push(dma, dma_make_ptr(s_work[spad_idx], ps_in), + dma_queue_push(dma_q, dma_make_data(s_work[spad_idx], ps_in), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); @@ -906,24 +957,24 @@ static void gated_delta_net_f32_tg_thread(unsigned int nth, unsigned int ith, vo spad_idx ^= 1; } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + int curr_spad_idx = 0; - for (uint32_t ir = ith; ir < total_rows; ir += nth) { - dma_queue_pop(dma); - dma_queue_pop(dma); + for (uint32_t ir = gctx->row_start + ith; ir < row_end; ir += nth) { + dma_queue_pop(dma_q); + dma_queue_pop(dma_q); float * s_work_curr = s_work[curr_spad_idx]; - const uint32_t iv1 = fastmodulo(ir, H, &fd_H); - const uint32_t iv3 = fastdiv(ir, &fd_H); - - const uint32_t iq1 = fastmodulo(iv1, q->ne[1], &fd_q1); - const uint32_t ik1 = fastmodulo(iv1, k->ne[1], &fd_k1); - const uint32_t iq3 = fastdiv(iv3, &fd_rq3); - const uint32_t ik3 = fastdiv(iv3, &fd_rk3); + const uint32_t iv1 = fastmodulo(ir, H, fd_H); + const uint32_t iv3 = fastdiv(ir, fd_H); - // final state lands in snapshot slot 0 (most-recent-first ordering) - float * s_out = state_out_base + ((uint64_t) iv3 * H + iv1) * S_v * S_v; + const uint32_t iq1 = fastmodulo(iv1, q->ne[1], fd_q1); + const uint32_t ik1 = fastmodulo(iv1, k->ne[1], fd_k1); + const uint32_t iq3 = fastdiv(iv3, fd_rq3); + const uint32_t ik3 = fastdiv(iv3, fd_rk3); + dma_addr_t s_out = state_out_dma_base + ((uint64_t) iv3 * H + iv1) * S_v * S_v * sizeof(float); float * attn_data = dst_base + ((uint64_t) iv3 * H + iv1) * S_v; const float * q_t = (const float *) ((const uint8_t *) (uintptr_t) q->data + @@ -937,132 +988,26 @@ static void gated_delta_net_f32_tg_thread(unsigned int nth, unsigned int ith, vo const float beta_val = *(const float *) ((const uint8_t *) (uintptr_t) beta->data + (uint64_t) iv3 * beta->nb[3] + (uint64_t) iv1 * beta->nb[1]); - hvx_copy_f32_au((uint8_t *) local_q, (const uint8_t *) q_t, S_v); - hvx_copy_f32_au((uint8_t *) local_k, (const uint8_t *) k_t, S_v); - - if (kda) { - hvx_exp_f32((uint8_t *) local_gate, (const uint8_t *) g_t, S_v, false); - - uint32_t j = 0; - for (; j + 8 <= S_v; j += 8) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - float * row4 = s_work_curr + (uint64_t) (j + 4) * S_v; - float * row5 = s_work_curr + (uint64_t) (j + 5) * S_v; - float * row6 = s_work_curr + (uint64_t) (j + 6) * S_v; - float * row7 = s_work_curr + (uint64_t) (j + 7) * S_v; - gdn_mul_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 8 * sizeof(float), res_attn); - } - for (; j + 4 <= S_v; j += 4) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - gdn_mul_dot4_f32(row0, row1, row2, row3, local_gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot4_f32(row0, row1, row2, row3, local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 4 * sizeof(float), res_attn); - } - HVX_Vector vscale_splat = hvx_vec_splat_f32(scale); - for (; j < S_v; ++j) { - float * row = s_work_curr + (uint64_t) j * S_v; - HVX_Vector vsum = gdn_mul_dot_f32(row, local_gate, local_k, S_v); - HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); - HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), hvx_vec_splat_f32(beta_val)); - HVX_Vector vres = gdn_add_scaled_dot_f32(row, local_k, vdj, local_q, S_v); - attn_data[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale_splat)); - } + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); + if (kparams->kda) { + gdn_step_kda_f32(s_work_curr, attn_data, q_t, k_t, v_t, g_t, beta_val, scale, S_v); } else { - const float gate = expf(g_t[0]); - uint32_t j = 0; - for (; j + 8 <= S_v; j += 8) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - float * row4 = s_work_curr + (uint64_t) (j + 4) * S_v; - float * row5 = s_work_curr + (uint64_t) (j + 5) * S_v; - float * row6 = s_work_curr + (uint64_t) (j + 6) * S_v; - float * row7 = s_work_curr + (uint64_t) (j + 7) * S_v; - gdn_mul_scalar_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot8_f32(row0, row1, row2, row3, row4, row5, row6, row7, - local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 8 * sizeof(float), res_attn); - } - for (; j + 4 <= S_v; j += 4) { - float * row0 = s_work_curr + (uint64_t) (j + 0) * S_v; - float * row1 = s_work_curr + (uint64_t) (j + 1) * S_v; - float * row2 = s_work_curr + (uint64_t) (j + 2) * S_v; - float * row3 = s_work_curr + (uint64_t) (j + 3) * S_v; - gdn_mul_scalar_dot4_f32(row0, row1, row2, row3, gate, local_k, S_v, local_sums); - - float local_delta_b[32] __attribute__((aligned(128))); - HVX_Vector vv_t = hvx_vmemu(v_t + j); - HVX_Vector v_local_sums = hvx_vmem(local_sums); - HVX_Vector diff = hvx_vec_sub_f32_f32(vv_t, v_local_sums); - hvx_vmem(local_delta_b) = hvx_vec_mul_f32_f32(diff, hvx_vec_splat_f32(beta_val)); - - gdn_add_scaled_dot4_f32(row0, row1, row2, row3, local_k, local_delta_b, local_q, S_v, local_sums); - - HVX_Vector res_attn = hvx_vec_mul_f32_f32(hvx_vmem(local_sums), hvx_vec_splat_f32(scale)); - hvx_vec_store_u(attn_data + j, 4 * sizeof(float), res_attn); - } - HVX_Vector vscale_splat = hvx_vec_splat_f32(scale); - for (; j < S_v; ++j) { - float * row = s_work_curr + (uint64_t) j * S_v; - HVX_Vector vsum = gdn_mul_scalar_dot_f32(row, gate, local_k, S_v); - HVX_Vector vv_t = hvx_vec_splat_f32(v_t[j]); - HVX_Vector vdj = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv_t, vsum), hvx_vec_splat_f32(beta_val)); - HVX_Vector vres = gdn_add_scaled_dot_f32(row, local_k, vdj, local_q, S_v); - attn_data[j] = hvx_vec_get_f32(hvx_vec_mul_f32_f32(vres, vscale_splat)); - } + gdn_step_scalar_f32(s_work_curr, attn_data, q_t, k_t, v_t, g_t, beta_val, scale, S_v); } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); // Push real write-back - dma_queue_push(dma, dma_make_ptr(s_out, s_work_curr), + dma_queue_push(dma_q, dma_make_data(s_out, s_work_curr), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); // Prefetch next block (if any) - if (ir_prefetch < total_rows) { - const uint32_t piv1 = fastmodulo(ir_prefetch, H, &fd_H); - const uint32_t piv3 = fastdiv(ir_prefetch, &fd_H); - const float * ps_in = state_in_base + (uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v; + if (ir_prefetch < row_end) { + const uint32_t piv1 = fastmodulo(ir_prefetch, H, fd_H); + const uint32_t piv3 = fastdiv(ir_prefetch, fd_H); + dma_addr_t ps_in = state->data + ((uint64_t) piv3 * state_seq_stride + (uint64_t) piv1 * S_v * S_v) * sizeof(float); - dma_queue_push(dma, dma_make_ptr(s_work[spad_idx], ps_in), + dma_queue_push(dma_q, dma_make_data(s_work[spad_idx], ps_in), S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); @@ -1072,9 +1017,1257 @@ static void gated_delta_net_f32_tg_thread(unsigned int nth, unsigned int ith, vo curr_spad_idx ^= 1; } - dma_queue_flush(dma); + dma_queue_flush(dma_q); +} + +struct htp_gdn_hmx_gemm_task { + const __fp16 * row_tiles; + const __fp16 * col_tiles; + __fp16 * out_tiles; + uint32_t n_row_tiles; + uint32_t n_col_tiles; + uint32_t n_dot_tiles; + uint32_t dot_stride; + uint8_t * hmx_scales; +}; + +static void htp_gdn_hmx_gemm_worker(void * data) { + struct htp_gdn_hmx_gemm_task * task = (struct htp_gdn_hmx_gemm_task *) data; + asm volatile(HMX_SET_BIAS("%0") :: "r"((unsigned int)task->hmx_scales)); + + const size_t dot_stride = task->dot_stride; + for (uint32_t r = 0; r < task->n_row_tiles; ++r) { + const __fp16 * r_tiles = task->row_tiles + r * dot_stride; + const __fp16 * c_tiles = task->col_tiles; + __fp16 * o_tile = task->out_tiles + r * task->n_col_tiles * HMX_FP16_TILE_N_ELMS; + + for (uint32_t c = 0; c < task->n_col_tiles; ++c) { + hmx_fa_qk_dot_tile(r_tiles, c_tiles, o_tile, task->n_dot_tiles); + c_tiles += dot_stride; + o_tile += HMX_FP16_TILE_N_ELMS; + } + } +} + +static inline void htp_gdn_push_hmx_gemm_task( + hmx_queue_t q, + struct htp_gdn_hmx_gemm_task * task, + const __fp16 * row_tiles, + const __fp16 * col_tiles, + __fp16 * out_tiles, + uint32_t n_row_tiles, + uint32_t n_col_tiles, + uint32_t n_dot_tiles, + uint8_t * scales +) { + task->row_tiles = row_tiles; + task->col_tiles = col_tiles; + task->out_tiles = out_tiles; + task->n_row_tiles = n_row_tiles; + task->n_col_tiles = n_col_tiles; + task->n_dot_tiles = n_dot_tiles; + task->dot_stride = n_dot_tiles * HMX_FP16_TILE_N_ELMS; + task->hmx_scales = scales; + + hmx_queue_push(q, hmx_queue_make_desc(htp_gdn_hmx_gemm_worker, task)); +} + +static inline void gdn_unpack_64x64_tiles_to_vectors( + HVX_Vector * restrict rows, + const __fp16 * restrict tiles +) { + const HVX_Vector * t00 = (const HVX_Vector *) (tiles + 0 * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t01 = (const HVX_Vector *) (tiles + 1 * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t10 = (const HVX_Vector *) (tiles + 2 * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t11 = (const HVX_Vector *) (tiles + 3 * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp0 = Q6_W_vdeal_VVR(t01[r], t00[r], -2); + rows[2 * r + 0] = Q6_V_lo_W(vp0); + rows[2 * r + 1] = Q6_V_hi_W(vp0); + + HVX_VectorPair vp1 = Q6_W_vdeal_VVR(t11[r], t10[r], -2); + rows[32 + 2 * r + 0] = Q6_V_lo_W(vp1); + rows[32 + 2 * r + 1] = Q6_V_hi_W(vp1); + } +} + +static inline void gdn_pack_64x64_vectors_to_tiles( + __fp16 * restrict tiles, + const HVX_Vector * restrict rows +) { + HVX_Vector * t00 = (HVX_Vector *) (tiles + 0 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * t01 = (HVX_Vector *) (tiles + 1 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * t10 = (HVX_Vector *) (tiles + 2 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * t11 = (HVX_Vector *) (tiles + 3 * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp0 = Q6_W_vshuff_VVR(rows[2 * r + 1], rows[2 * r + 0], -2); + t00[r] = Q6_V_lo_W(vp0); + t01[r] = Q6_V_hi_W(vp0); + + HVX_VectorPair vp1 = Q6_W_vshuff_VVR(rows[32 + 2 * r + 1], rows[32 + 2 * r + 0], -2); + t10[r] = Q6_V_lo_W(vp1); + t11[r] = Q6_V_hi_W(vp1); + } +} + +static inline void gdn_unpack_64xS_tiles_to_f32( + float * restrict dst_f32, + const __fp16 * restrict tiles, + uint32_t S_v +) { + const uint32_t n_col_tiles = S_v / 32; + for (uint32_t r0 = 0; r0 < 2; ++r0) { + for (uint32_t d = 0; d < S_v / 64; ++d) { + const HVX_Vector * t0 = (const HVX_Vector *) (tiles + (r0 * n_col_tiles + 2 * d + 0) * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t1 = (const HVX_Vector *) (tiles + (r0 * n_col_tiles + 2 * d + 1) * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp01 = Q6_W_vdeal_VVR(t1[r], t0[r], -2); + HVX_VectorPair p0 = hvx_vec_f16_to_f32(Q6_V_lo_W(vp01)); + HVX_VectorPair p1 = hvx_vec_f16_to_f32(Q6_V_hi_W(vp01)); + + float * out0 = dst_f32 + (r0 * 32 + 2 * r + 0) * S_v + d * 64; + float * out1 = dst_f32 + (r0 * 32 + 2 * r + 1) * S_v + d * 64; + + hvx_vmem(out0 + 0) = Q6_V_lo_W(p0); + hvx_vmem(out0 + 32) = Q6_V_hi_W(p0); + hvx_vmem(out1 + 0) = Q6_V_lo_W(p1); + hvx_vmem(out1 + 32) = Q6_V_hi_W(p1); + } + } + } +} + +static inline void gdn_unpack_64xS_tiles_to_f16( + __fp16 * restrict dst_f16, + const __fp16 * restrict tiles, + uint32_t S_v +) { + const uint32_t n_col_tiles = S_v / 32; + for (uint32_t r0 = 0; r0 < 2; ++r0) { + for (uint32_t d = 0; d < S_v / 64; ++d) { + const HVX_Vector * t0 = (const HVX_Vector *) (tiles + (r0 * n_col_tiles + 2 * d + 0) * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t1 = (const HVX_Vector *) (tiles + (r0 * n_col_tiles + 2 * d + 1) * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp01 = Q6_W_vdeal_VVR(t1[r], t0[r], -2); + __fp16 * out0 = dst_f16 + (r0 * 32 + 2 * r + 0) * S_v + d * 64; + __fp16 * out1 = dst_f16 + (r0 * 32 + 2 * r + 1) * S_v + d * 64; + + hvx_vmem(out0) = Q6_V_lo_W(vp01); + hvx_vmem(out1) = Q6_V_hi_W(vp01); + } + } + } +} + +static inline void gdn_unpack_SxS_tiles_to_f32( + float * restrict dst_f32, + const __fp16 * restrict tiles, + uint32_t S_v +) { + const uint32_t n_tiles = S_v / 32; + for (uint32_t r0 = 0; r0 < n_tiles; ++r0) { + for (uint32_t d = 0; d < S_v / 64; ++d) { + const HVX_Vector * t0 = (const HVX_Vector *) (tiles + (r0 * n_tiles + 2 * d + 0) * HMX_FP16_TILE_N_ELMS); + const HVX_Vector * t1 = (const HVX_Vector *) (tiles + (r0 * n_tiles + 2 * d + 1) * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp01 = Q6_W_vdeal_VVR(t1[r], t0[r], -2); + HVX_VectorPair p0 = hvx_vec_f16_to_f32(Q6_V_lo_W(vp01)); + HVX_VectorPair p1 = hvx_vec_f16_to_f32(Q6_V_hi_W(vp01)); + + float * out0 = dst_f32 + (r0 * 32 + 2 * r + 0) * S_v + d * 64; + float * out1 = dst_f32 + (r0 * 32 + 2 * r + 1) * S_v + d * 64; + + hvx_vmem(out0 + 0) = Q6_V_lo_W(p0); + hvx_vmem(out0 + 32) = Q6_V_hi_W(p0); + hvx_vmem(out1 + 0) = Q6_V_lo_W(p1); + hvx_vmem(out1 + 32) = Q6_V_hi_W(p1); + } + } + } +} + +static inline void gdn_f32_to_hmx_row_tiles_and_f16( + __fp16 * restrict dst_tiles, + __fp16 * restrict dst_prime_tiles, + __fp16 * restrict dst_f16, + const float * restrict src, + const __fp16 * restrict scale_per_row, + uint32_t n_rows, + uint32_t n_cols +) { + const uint32_t n_col_tiles = n_cols / 32; + const uint32_t * scale_pairs = (const uint32_t *) scale_per_row; + + for (uint32_t r = 0; r < n_rows; r += 2) { + uint32_t r0 = r / 32; + uint32_t r1 = (r % 32) / 2; + const float * p0 = src + (r + 0) * n_cols; + const float * p1 = src + (r + 1) * n_cols; + + HVX_Vector v_scale; + if (dst_prime_tiles) { + uint32_t scale_pair = scale_pairs ? scale_pairs[r / 2] : 0x3c003c00; + v_scale = Q6_V_vsplat_R(scale_pair); + } + + for (uint32_t c = 0; c < n_col_tiles; c += 2) { + HVX_Vector v0_0 = hvx_vmem(p0 + (c + 0) * 32); + HVX_Vector v1_0 = hvx_vmem(p1 + (c + 0) * 32); + HVX_Vector v0_1 = hvx_vmem(p0 + (c + 1) * 32); + HVX_Vector v1_1 = hvx_vmem(p1 + (c + 1) * 32); + + HVX_Vector vh0 = hvx_vec_f32_to_f16_shuff(v0_0, v1_0); + HVX_Vector vh1 = hvx_vec_f32_to_f16_shuff(v0_1, v1_1); + __fp16 * tile0 = dst_tiles + (r0 * n_col_tiles + c + 0) * HMX_FP16_TILE_N_ELMS; + __fp16 * tile1 = dst_tiles + (r0 * n_col_tiles + c + 1) * HMX_FP16_TILE_N_ELMS; + ((HVX_Vector *) tile0)[r1] = vh0; + ((HVX_Vector *) tile1)[r1] = vh1; + + if (dst_prime_tiles) { + HVX_Vector vh0_s = hvx_vec_mul_f16_f16(vh0, v_scale); + HVX_Vector vh1_s = hvx_vec_mul_f16_f16(vh1, v_scale); + __fp16 * tile0_s = dst_prime_tiles + (r0 * n_col_tiles + c + 0) * HMX_FP16_TILE_N_ELMS; + __fp16 * tile1_s = dst_prime_tiles + (r0 * n_col_tiles + c + 1) * HMX_FP16_TILE_N_ELMS; + ((HVX_Vector *) tile0_s)[r1] = vh0_s; + ((HVX_Vector *) tile1_s)[r1] = vh1_s; + } + + if (dst_f16) { + HVX_VectorPair vp01 = Q6_W_vdeal_VVR(vh1, vh0, -2); + hvx_vmem(dst_f16 + (r + 0) * n_cols + c * 32) = Q6_V_lo_W(vp01); + hvx_vmem(dst_f16 + (r + 1) * n_cols + c * 32) = Q6_V_hi_W(vp01); + } + } + } +} + +static inline void hvx_transpose_32x32_words(HVX_Vector * restrict m, HVX_Vector * restrict tmp) { + for (int i = 0; i < 16; ++i) { + HVX_VectorPair p = Q6_W_vshuff_VVR(m[2*i + 1], m[2*i], -4); + tmp[2*i + 0] = Q6_V_lo_W(p); + tmp[2*i + 1] = Q6_V_hi_W(p); + } + + for (int b = 0; b < 32; b += 4) { + HVX_VectorPair p0 = Q6_W_vshuff_VVR(tmp[b + 2], tmp[b + 0], -8); + HVX_VectorPair p1 = Q6_W_vshuff_VVR(tmp[b + 3], tmp[b + 1], -8); + m[b + 0] = Q6_V_lo_W(p0); m[b + 1] = Q6_V_hi_W(p0); + m[b + 2] = Q6_V_lo_W(p1); m[b + 3] = Q6_V_hi_W(p1); + } + + for (int b = 0; b < 32; b += 8) { + for (int i = 0; i < 4; ++i) { + HVX_VectorPair p = Q6_W_vshuff_VVR(m[b + i + 4], m[b + i], -16); + tmp[b + 2*i + 0] = Q6_V_lo_W(p); + tmp[b + 2*i + 1] = Q6_V_hi_W(p); + } + } + + for (int b = 0; b < 32; b += 16) { + for (int i = 0; i < 8; ++i) { + HVX_VectorPair p = Q6_W_vshuff_VVR(tmp[b + i + 8], tmp[b + i], -32); + m[b + 2*i + 0] = Q6_V_lo_W(p); + m[b + 2*i + 1] = Q6_V_hi_W(p); + } + } + + for (int i = 0; i < 16; ++i) { + HVX_VectorPair p = Q6_W_vshuff_VVR(m[i + 16], m[i], -64); + tmp[2 * i + 0] = Q6_V_lo_W(p); + tmp[2 * i + 1] = Q6_V_hi_W(p); + } + + for (int i = 0; i < 32; ++i) { + m[i] = tmp[i]; + } +} + +static inline void gdn_pack_d_t_row_tiles( + __fp16 * restrict dst_tiles, + const __fp16 * restrict src_d, + uint32_t S_v, + HVX_Vector * restrict m, + HVX_Vector * restrict tmp +) { + for (uint32_t col_half = 0; col_half < S_v / 64; ++col_half) { + uint32_t r0_base = col_half * 2; + for (uint32_t c0 = 0; c0 < 2; ++c0) { + for (uint32_t s_local = 0; s_local < 32; ++s_local) { + uint32_t s = c0 * 32 + s_local; + m[s_local] = hvx_vmem(src_d + s * S_v + col_half * 64); + } + + hvx_transpose_32x32_words(m, tmp); + + uint32_t tile0_idx = (r0_base + 0) * 2 + c0; + uint32_t tile1_idx = (r0_base + 1) * 2 + c0; + HVX_Vector * t0 = (HVX_Vector *)(dst_tiles + tile0_idx * HMX_FP16_TILE_N_ELMS); + HVX_Vector * t1 = (HVX_Vector *)(dst_tiles + tile1_idx * HMX_FP16_TILE_N_ELMS); + + for (uint32_t r = 0; r < 16; ++r) { + t0[r] = m[r]; + t1[r] = m[16 + r]; + } + } + } +} + +static __attribute__((noinline)) void gdn_build_inv_l_blocks( + __fp16 * restrict inv_row_tiles, + const HVX_Vector * restrict rows_kk, + const __fp16 * restrict decay_m, + const float * restrict beta, + __fp16 * restrict l10_tile, + __fp16 * restrict neg_a11_tile +) { + const HVX_Vector v_one_f16 = hvx_vec_splat_f16(1.0f); + const HVX_VectorPred q_mask64 = Q6_Q_vsetq2_R(64); + + uint16_t beta_u16[64] __attribute__((aligned(128))); + uint16_t l00[32][32] __attribute__((aligned(128))); + uint16_t l11[32][32] __attribute__((aligned(128))); + + HVX_Vector * restrict p_l00 = (HVX_Vector *) l00; + HVX_Vector * restrict p_l11 = (HVX_Vector *) l11; + HVX_Vector * restrict p_l10_tile = (HVX_Vector *) l10_tile; + + HVX_Vector * restrict tile00 = (HVX_Vector *) (inv_row_tiles + 0 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * restrict tile01 = (HVX_Vector *) (inv_row_tiles + 1 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * restrict tile11 = (HVX_Vector *) (inv_row_tiles + 3 * HMX_FP16_TILE_N_ELMS); + HVX_Vector * restrict p_neg_a11 = (HVX_Vector *) neg_a11_tile; + + hvx_vmem(beta_u16) = hvx_vec_f32_to_f16(hvx_vmem(beta + 0), hvx_vmem(beta + 32)); + + for (uint32_t r = 0; r < 16; ++r) { + tile01[r] = Q6_V_vzero(); + } + + for (uint32_t r = 0; r < 16; ++r) { + uint32_t t0 = 2 * r; + uint32_t t1 = t0 + 1; + + HVX_Vector v_d0 = hvx_vmem(decay_m + t0 * 64); + HVX_Vector v_d1 = hvx_vmem(decay_m + t1 * 64); + HVX_Vector v_b0 = Q6_Vh_vsplat_R(beta_u16[t0]); + HVX_Vector v_b1 = Q6_Vh_vsplat_R(beta_u16[t1]); + + HVX_Vector r0 = hvx_vec_mul_f16_f16(hvx_vec_mul_f16_f16(rows_kk[t0], v_d0), v_b0); + HVX_Vector r1 = hvx_vec_mul_f16_f16(hvx_vec_mul_f16_f16(rows_kk[t1], v_d1), v_b1); + + p_l00[r] = Q6_V_vmux_QVV(q_mask64, r0, Q6_V_vror_VR(r1, 64)); + } + + for (uint32_t r = 0; r < 16; ++r) { + uint32_t t0 = 32 + 2 * r; + uint32_t t1 = t0 + 1; + + HVX_Vector v_d0 = hvx_vmem(decay_m + t0 * 64); + HVX_Vector v_d1 = hvx_vmem(decay_m + t1 * 64); + HVX_Vector v_b0 = Q6_Vh_vsplat_R(beta_u16[t0]); + HVX_Vector v_b1 = Q6_Vh_vsplat_R(beta_u16[t1]); + + HVX_Vector r0 = hvx_vec_mul_f16_f16(hvx_vec_mul_f16_f16(rows_kk[t0], v_d0), v_b0); + HVX_Vector r1 = hvx_vec_mul_f16_f16(hvx_vec_mul_f16_f16(rows_kk[t1], v_d1), v_b1); + + HVX_VectorPair vp_l10 = Q6_W_vshuff_VVR(r1, r0, -2); + p_l10_tile[r] = Q6_V_lo_W(vp_l10); + p_l11[r] = Q6_V_vmux_QVV(q_mask64, Q6_V_vror_VR(r0, 64), r1); + } + + HVX_Vector a_rows[32]; + for (uint32_t t = 0; t < 32; ++t) { + HVX_Vector v_inv = Q6_V_vzero(); + for (uint32_t k = 0; k < t; ++k) { + HVX_Vector v_lk = Q6_Vh_vsplat_R(l00[t][k]); + v_inv = hvx_vec_sub_f16_f16(v_inv, hvx_vec_mul_f16_f16(v_lk, a_rows[k])); + } + HVX_VectorPred q_diag = (t == 0) ? Q6_Q_vsetq2_R(2) : Q6_Q_and_QQn(Q6_Q_vsetq2_R(2 * (t + 1)), Q6_Q_vsetq2_R(2 * t)); + a_rows[t] = Q6_V_vand_QV(q_mask64, Q6_V_vmux_QVV(q_diag, v_one_f16, v_inv)); + } + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp = Q6_W_vshuff_VVR(a_rows[2 * r + 1], a_rows[2 * r + 0], -2); + tile00[r] = Q6_V_lo_W(vp); + } + + for (uint32_t t = 0; t < 32; ++t) { + HVX_Vector v_inv = Q6_V_vzero(); + for (uint32_t k = 0; k < t; ++k) { + HVX_Vector v_lk = Q6_Vh_vsplat_R(l11[t][k]); + v_inv = hvx_vec_sub_f16_f16(v_inv, hvx_vec_mul_f16_f16(v_lk, a_rows[k])); + } + HVX_VectorPred q_diag = (t == 0) ? Q6_Q_vsetq2_R(2) : Q6_Q_and_QQn(Q6_Q_vsetq2_R(2 * (t + 1)), Q6_Q_vsetq2_R(2 * t)); + a_rows[t] = Q6_V_vand_QV(q_mask64, Q6_V_vmux_QVV(q_diag, v_one_f16, v_inv)); + } + + for (uint32_t r = 0; r < 16; ++r) { + HVX_VectorPair vp = Q6_W_vshuff_VVR(a_rows[2 * r + 1], a_rows[2 * r + 0], -2); + tile11[r] = Q6_V_lo_W(vp); + + HVX_Vector n0 = hvx_vec_sub_f16_f16(Q6_V_vzero(), a_rows[2 * r + 0]); + HVX_Vector n1 = hvx_vec_sub_f16_f16(Q6_V_vzero(), a_rows[2 * r + 1]); + HVX_VectorPair vp_neg = Q6_W_vshuff_VVR(n1, n0, -2); + p_neg_a11[r] = Q6_V_lo_W(vp_neg); + } +} + + +static inline void gdn_dma_push_chunk_inputs( + dma_queue * dma_q, + float * vtcm_q, + float * vtcm_k, + float * vtcm_v, + const struct htp_tensor * q, + const struct htp_tensor * k, + const struct htp_tensor * v, + uint32_t iq3, uint32_t iq1, + uint32_t ik3, uint32_t ik1, + uint32_t iv3, uint32_t iv1, + uint32_t t_chunk, + uint32_t chunk_size, + uint32_t S_v +) { + const dma_addr_t q_dma = q->data + (uint64_t) iq3 * q->nb[3] + (uint64_t) t_chunk * q->nb[2] + (uint64_t) iq1 * q->nb[1]; + const dma_addr_t k_dma = k->data + (uint64_t) ik3 * k->nb[3] + (uint64_t) t_chunk * k->nb[2] + (uint64_t) ik1 * k->nb[1]; + const dma_addr_t v_dma = v->data + (uint64_t) iv3 * v->nb[3] + (uint64_t) t_chunk * v->nb[2] + (uint64_t) iv1 * v->nb[1]; + + dma_queue_push(dma_q, dma_make_data(vtcm_q, q_dma), S_v * sizeof(float), q->nb[2], S_v * sizeof(float), chunk_size); + dma_queue_push(dma_q, dma_make_data(vtcm_k, k_dma), S_v * sizeof(float), k->nb[2], S_v * sizeof(float), chunk_size); + dma_queue_push(dma_q, dma_make_data(vtcm_v, v_dma), S_v * sizeof(float), v->nb[2], S_v * sizeof(float), chunk_size); +} + +static inline void gdn_dma_push_chunk_gb( + dma_queue * dma_q, + float * vtcm_g_raw, + float * vtcm_b_raw, + const struct htp_tensor * g, + const struct htp_tensor * beta, + uint32_t iv3, + uint32_t iv1, + uint32_t t_chunk, + uint32_t chunk_size, + uint32_t n_batch +) { + const dma_addr_t g_dma = g->data + (uint64_t) iv3 * g->nb[3] + (uint64_t) t_chunk * g->nb[2] + (uint64_t) iv1 * g->nb[1]; + const dma_addr_t beta_dma = beta->data + (uint64_t) iv3 * beta->nb[3] + (uint64_t) t_chunk * beta->nb[2] + (uint64_t) iv1 * beta->nb[1]; + const uint32_t row_bytes = n_batch * sizeof(float); + + dma_queue_push(dma_q, dma_make_data(vtcm_g_raw, g_dma), row_bytes, g->nb[2], row_bytes, chunk_size); + dma_queue_push(dma_q, dma_make_data(vtcm_b_raw, beta_dma), row_bytes, beta->nb[2], row_bytes, chunk_size); +} + +static inline void gdn_pack_s_col_tiles( + __fp16 * restrict vtcm_s_col_tiles, + __fp16 * restrict vtcm_s_f16, + const float * restrict vtcm_s_state, + uint32_t S_v +) { + for (uint32_t j = 0; j < S_v; ++j) { + for (uint32_t i = 0; i < S_v; i += 64) { + HVX_Vector v0 = hvx_vmem(vtcm_s_state + j * S_v + i + 0); + HVX_Vector v1 = (i + 32 < S_v) ? hvx_vmem(vtcm_s_state + j * S_v + i + 32) : Q6_V_vzero(); + hvx_vmem(vtcm_s_f16 + j * S_v + i) = hvx_vec_f32_to_f16(v0, v1); + } + } + hmx_interleave_rows_to_tiles(vtcm_s_col_tiles, vtcm_s_f16, S_v, S_v, S_v, 0, S_v); +} + +struct htp_gdn_head_ptrs { + float * s_state; + __fp16 * s_f16; + __fp16 * s_col_tiles; + float * s_update_f32; + __fp16 * s_update_tiles; + + float * q_f32[2]; + float * k_f32[2]; + float * v_f32[2]; + float * g_f32[2]; + float * b_f32[2]; + float * o_f32[2]; + + float * v_inter_f32; + float * o_inter_f32; + float * o_intra_f32; + + __fp16 * k_f16; + __fp16 * v_prime_f16; + __fp16 * delta_f16; + __fp16 * d_f16; + + __fp16 * q_row_tiles; + __fp16 * q_prime_row_tiles; + __fp16 * k_row_tiles; + __fp16 * k_col_tiles; + __fp16 * k_prime_row_tiles; + __fp16 * k_col_tiles_64x128; + __fp16 * kk_tiles; + __fp16 * qk_tiles; + __fp16 * v_inter_tiles; + __fp16 * o_inter_tiles; + __fp16 * inv_row_tiles; + __fp16 * a_row_tiles; + __fp16 * v_prime_col_tiles; + __fp16 * delta_tiles; + __fp16 * delta_col_tiles; + __fp16 * o_intra_tiles; + __fp16 * d_row_tiles; + + __fp16 * gamma; + float * lambda_init; + __fp16 * lambda_init_f16; + __fp16 * decay_m; + __fp16 * decay_a; + + HVX_Vector * rows_kk; + HVX_Vector * rows_qk; + HVX_Vector * rows_inv; + HVX_Vector * rows_a; + + HVX_Vector * vtcm_m; + HVX_Vector * vtcm_tmp; + + uint32_t iv1; + uint32_t iv3; + uint32_t iq1; + uint32_t ik1; + uint32_t iq3; + uint32_t ik3; + dma_addr_t state_in_dma; + dma_addr_t state_out_dma; +}; + +static inline void gdn_init_head_ptrs( + struct htp_gdn_head_ptrs * head, + const struct htp_gdn_hmx_vtcm_layout * L, + uint8_t * vtcm_base, + uint32_t h, + uint32_t base_iv1, + uint32_t iv3, + const struct htp_tensor * q, + const struct htp_tensor * k, + const struct htp_tensor * v, + const struct htp_tensor * state, + const struct htp_tensor * dst, + const struct htp_tensor * dst_cache, + const struct htp_gdn_kernel_params * kparams, + uint32_t S_v, + uint32_t H, + uint32_t n_tokens, + uint32_t chunk_size +) { + const size_t dma_scalar_sz = hex_round_up(chunk_size * sizeof(float), 128); + const size_t decay_sz = 64 * 64 * sizeof(__fp16); + const size_t row_vecs_sz = 64 * 128; + + head->s_state = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_s_state + h * L->state_f32_bytes); + head->s_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_s_f16 + h * L->state_f16_bytes); + head->s_col_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_s_col_tiles + h * L->state_tiles_bytes); + head->s_update_f32 = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_s_update_f32 + h * L->state_f32_bytes); + head->s_update_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_s_update_tiles + h * L->state_tiles_bytes); + + head->q_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_q_f32[0] + h * L->dma_chunk_bytes); + head->q_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_q_f32[1] + h * L->dma_chunk_bytes) : head->q_f32[0]; + head->k_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_k_f32[0] + h * L->dma_chunk_bytes); + head->k_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_k_f32[1] + h * L->dma_chunk_bytes) : head->k_f32[0]; + head->v_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_v_f32[0] + h * L->dma_chunk_bytes); + head->v_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_v_f32[1] + h * L->dma_chunk_bytes) : head->v_f32[0]; + head->g_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_g_f32[0] + h * dma_scalar_sz); + head->g_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_g_f32[1] + h * dma_scalar_sz) : head->g_f32[0]; + head->b_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_b_f32[0] + h * dma_scalar_sz); + head->b_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_b_f32[1] + h * dma_scalar_sz) : head->b_f32[0]; + head->o_f32[0] = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_o_f32[0] + h * L->dma_chunk_bytes); + head->o_f32[1] = L->pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L->off_o_f32[1] + h * L->dma_chunk_bytes) : head->o_f32[0]; + + head->v_inter_f32 = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_v_inter_f32 + h * L->dma_chunk_bytes); + head->o_inter_f32 = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_o_inter_f32 + h * L->dma_chunk_bytes); + head->o_intra_f32 = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_o_intra_f32 + h * L->dma_chunk_bytes); + + head->k_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_k_f16 + h * L->act_f16_bytes); + head->v_prime_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_v_prime_f16 + h * L->act_f16_bytes); + head->delta_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_delta_f16 + h * L->act_f16_bytes); + head->d_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_d_f16 + h * L->act_f16_bytes); + + head->q_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_q_row_tiles + h * L->tile_64xSv_bytes); + head->q_prime_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_q_prime_row_tiles + h * L->tile_64xSv_bytes); + head->k_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_k_row_tiles + h * L->tile_64xSv_bytes); + head->k_col_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_k_col_tiles + h * L->tile_64xSv_bytes); + head->k_prime_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_k_prime_row_tiles + h * L->tile_64xSv_bytes); + head->k_col_tiles_64x128 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_k_col_tiles_64x128 + h * L->tile_64xSv_bytes); + head->kk_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_kk_tiles + h * L->tile_64x64_bytes); + head->qk_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_qk_tiles + h * L->tile_64x64_bytes); + head->v_inter_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_v_inter_tiles + h * L->tile_64xSv_bytes); + head->o_inter_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_o_inter_tiles + h * L->tile_64xSv_bytes); + head->inv_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_inv_row_tiles + h * L->tile_64x64_bytes); + head->a_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_a_row_tiles + h * L->tile_64x64_bytes); + head->v_prime_col_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_v_prime_col_tiles + h * L->tile_64xSv_bytes); + head->delta_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_delta_tiles + h * L->tile_64xSv_bytes); + head->delta_col_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_delta_col_tiles + h * L->tile_64xSv_bytes); + head->o_intra_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_o_intra_tiles + h * L->tile_64xSv_bytes); + head->d_row_tiles = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_d_row_tiles + h * L->tile_64xSv_bytes); + + head->gamma = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_gamma + h * dma_scalar_sz); + head->lambda_init_f16 = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_gamma + h * dma_scalar_sz + 128); + head->lambda_init = VTCM_LAYOUT_PTR(float, vtcm_base, L->off_lambda_init + h * dma_scalar_sz); + head->decay_m = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_decay_m + h * decay_sz); + head->decay_a = VTCM_LAYOUT_PTR(__fp16, vtcm_base, L->off_decay_a + h * decay_sz); + + head->rows_kk = VTCM_LAYOUT_PTR(HVX_Vector, vtcm_base, L->off_rows_kk + h * row_vecs_sz); + head->rows_qk = VTCM_LAYOUT_PTR(HVX_Vector, vtcm_base, L->off_rows_qk + h * row_vecs_sz); + head->rows_inv = VTCM_LAYOUT_PTR(HVX_Vector, vtcm_base, L->off_rows_inv + h * row_vecs_sz); + head->rows_a = VTCM_LAYOUT_PTR(HVX_Vector, vtcm_base, L->off_rows_a + h * row_vecs_sz); + + head->vtcm_m = VTCM_LAYOUT_PTR(HVX_Vector, vtcm_base, L->off_thread_scratch + h * (64 * 128)); + head->vtcm_tmp = head->vtcm_m + 32; + + head->iv1 = base_iv1 + h; + head->iv3 = iv3; + head->iq1 = fastmodulo(head->iv1, q->ne[1], &kparams->div_q1); + head->ik1 = fastmodulo(head->iv1, k->ne[1], &kparams->div_k1); + head->iq3 = fastdiv(head->iv3, &kparams->div_rq3); + head->ik3 = fastdiv(head->iv3, &kparams->div_rk3); + + head->state_in_dma = state->data + + ((uint64_t) head->iv3 * kparams->state_seq_stride + (uint64_t) head->iv1 * S_v * S_v) * sizeof(float); + + head->state_out_dma = dst_cache ? + (dst_cache->data + ((uint64_t) head->iv3 * H + head->iv1) * S_v * S_v * sizeof(float)) : + (dst->data + ((uint64_t) S_v * H * n_tokens * kparams->n_seqs + (uint64_t) (head->iv3 * H + head->iv1) * S_v * S_v) * sizeof(float)); +} + +struct htp_gdn_batch_context { + struct htp_gdn_head_ptrs * heads; + const float * vtcm_g_raw; + const float * vtcm_b_raw; + uint32_t curr_buf; + uint32_t c; + uint32_t n_batch; + uint32_t S_v; + float scale; + struct htp_ops_context * octx; + const struct htp_gdn_kernel_params * kparams; +}; + +static void gdn_hvx_init_state_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, 0); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + gdn_pack_s_col_tiles(head->s_col_tiles, head->s_f16, head->s_state, bctx->S_v); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, 0); +} + +static inline __attribute__((unused)) HVX_Vector hvx_clamp_neg20_0(HVX_Vector v, HVX_Vector v_zero, HVX_Vector v_neg20) { + HVX_VectorPred p_gt = Q6_Q_vcmp_gt_VsfVsf(v, v_zero); + v = Q6_V_vmux_QVV(p_gt, v_zero, v); + HVX_VectorPred p_lt = Q6_Q_vcmp_gt_VsfVsf(v_neg20, v); + return Q6_V_vmux_QVV(p_lt, v_neg20, v); +} + +static inline HVX_Vector hvx_prefix_scan_f32(HVX_Vector v, HVX_Vector carry_in) { + const HVX_Vector zero = Q6_V_vzero(); + + v = hvx_vec_add_f32_f32(v, Q6_V_vlalign_VVR(v, zero, 4)); + v = hvx_vec_add_f32_f32(v, Q6_V_vlalign_VVR(v, zero, 8)); + v = hvx_vec_add_f32_f32(v, Q6_V_vlalign_VVR(v, zero, 16)); + v = hvx_vec_add_f32_f32(v, Q6_V_vlalign_VVR(v, zero, 32)); + v = hvx_vec_add_f32_f32(v, Q6_V_vlalign_VVR(v, zero, 64)); + v = hvx_vec_add_f32_f32(v, carry_in); + + return v; +} + +static inline HVX_Vector hvx_splat_last_f32(HVX_Vector v) { + return hvx_vec_repl4(Q6_V_vror_VR(v, 124)); +} + +static void gdn_hvx_phase1a_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_PREP, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t curr_buf = bctx->curr_buf; + const uint32_t S_v = bctx->S_v; + const uint32_t n_batch = bctx->n_batch; + + if (n_batch == 1) { + hvx_vmem(head->g_f32[curr_buf] + 0) = hvx_vmem(bctx->vtcm_g_raw + 0); + hvx_vmem(head->g_f32[curr_buf] + 32) = hvx_vmem(bctx->vtcm_g_raw + 32); + hvx_vmem(head->b_f32[curr_buf] + 0) = hvx_vmem(bctx->vtcm_b_raw + 0); + hvx_vmem(head->b_f32[curr_buf] + 32) = hvx_vmem(bctx->vtcm_b_raw + 32); + } else { + int32_t offsets[32] __attribute__((aligned(128))); + for (int k = 0; k < 32; ++k) { + offsets[k] = k * n_batch * sizeof(float); + } + HVX_Vector vv = *(const HVX_Vector *) offsets; + const size_t rt_g = (size_t) ((const uint8_t *) bctx->vtcm_g_raw + i * sizeof(float)); + const size_t rt_b = (size_t) ((const uint8_t *) bctx->vtcm_b_raw + i * sizeof(float)); + const size_t mu = 64 * n_batch * sizeof(float); + + Q6_vgather_ARMVw((HVX_Vector *) (head->g_f32[curr_buf] + 0), rt_g, mu, vv); + Q6_vgather_ARMVw((HVX_Vector *) (head->g_f32[curr_buf] + 32), rt_g + 32 * n_batch * sizeof(float), mu, vv); + Q6_vgather_ARMVw((HVX_Vector *) (head->b_f32[curr_buf] + 0), rt_b, mu, vv); + Q6_vgather_ARMVw((HVX_Vector *) (head->b_f32[curr_buf] + 32), rt_b + 32 * n_batch * sizeof(float), mu, vv); + } + + const uint32_t t_chunk = bctx->c * 64; + const uint32_t valid_tokens = hex_smin(64, bctx->kparams->n_tokens - t_chunk); + if (valid_tokens < 64) { + for (uint32_t t = valid_tokens; t < 64; ++t) { + head->g_f32[curr_buf][t] = 0.0f; + head->b_f32[curr_buf][t] = 0.0f; + } + const HVX_Vector vzero = Q6_V_vzero(); + for (uint32_t t = valid_tokens; t < 64; ++t) { + for (uint32_t j = 0; j < S_v; j += 32) { + hvx_vmem(head->q_f32[curr_buf] + t * S_v + j) = vzero; + hvx_vmem(head->k_f32[curr_buf] + t * S_v + j) = vzero; + hvx_vmem(head->v_f32[curr_buf] + t * S_v + j) = vzero; + } + } + } + + const HVX_Vector v_g0 = hvx_vmem(head->g_f32[curr_buf] + 0); + const HVX_Vector v_g1 = hvx_vmem(head->g_f32[curr_buf] + 32); + + HVX_Vector v_gamma0 = hvx_prefix_scan_f32(v_g0, Q6_V_vzero()); + HVX_Vector v_carry = hvx_splat_last_f32(v_gamma0); + HVX_Vector v_gamma1 = hvx_prefix_scan_f32(v_g1, v_carry); + + const HVX_Vector v_zero = Q6_V_vzero(); + const HVX_Vector v_neg20 = hvx_vec_splat_f32(-20.0f); + + hvx_vmem(head->gamma) = hvx_vec_f32_to_f16(v_gamma0, v_gamma1); + + HVX_Vector v_l0 = hvx_vec_exp_f32(hvx_clamp_neg20_0(v_gamma0, v_zero, v_neg20)); + HVX_Vector v_l1 = hvx_vec_exp_f32(hvx_clamp_neg20_0(v_gamma1, v_zero, v_neg20)); + + hvx_vmem(head->lambda_init + 0) = v_l0; + hvx_vmem(head->lambda_init + 32) = v_l1; + hvx_vmem(head->lambda_init_f16) = hvx_vec_f32_to_f16(v_l0, v_l1); + + gdn_f32_to_hmx_row_tiles_and_f16(head->k_row_tiles, head->k_prime_row_tiles, head->k_f16, + head->k_f32[curr_buf], head->lambda_init_f16, 64, S_v); + hmx_interleave_rows_to_tiles(head->k_col_tiles, head->k_f16, 64, S_v, S_v, 0, 64); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_PREP, info); +} + +static void gdn_hvx_phase1b_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_PREP, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t curr_buf = bctx->curr_buf; + const uint32_t S_v = bctx->S_v; + + gdn_f32_to_hmx_row_tiles_and_f16(head->q_row_tiles, head->q_prime_row_tiles, NULL, + head->q_f32[curr_buf], head->lambda_init_f16, 64, S_v); + + hmx_interleave_cols_to_tiles(head->k_col_tiles_64x128, head->k_f16, 64, S_v, S_v, 2, 0, 64); + + const uint16_t * gamma_u16 = (const uint16_t *) head->gamma; + const HVX_Vector v_gamma = hvx_vmem(head->gamma); + + const HVX_Vector v_zero_f16 = Q6_V_vzero(); + const HVX_Vector v_neg20_f16 = hvx_vec_splat_f16(-20.0f); + const HVX_Vector v_log2e_f16 = hvx_vec_splat_f16(1.4426950408889634f); + const HVX_Vector v_one_f16 = hvx_vec_splat_f16(1.0f); + + hvx_vmem(head->decay_m + 0) = Q6_V_vzero(); + hvx_vmem(head->decay_a + 0) = Q6_V_vand_QV(Q6_Q_vsetq2_R(2), v_one_f16); + + for (uint32_t t = 1; t < 63; t += 2) { + uint32_t t0 = t; + uint32_t t1 = t + 1; + + HVX_Vector v_gamma_t0 = Q6_Vh_vsplat_R(gamma_u16[t0]); + HVX_Vector v_gamma_t1 = Q6_Vh_vsplat_R(gamma_u16[t1]); + + HVX_Vector diff0 = hvx_vec_sub_f16_f16(v_gamma_t0, v_gamma); + HVX_Vector diff1 = hvx_vec_sub_f16_f16(v_gamma_t1, v_gamma); + + HVX_VectorPred p_gt0 = Q6_Q_vcmp_gt_VhfVhf(diff0, v_zero_f16); + HVX_VectorPred p_gt1 = Q6_Q_vcmp_gt_VhfVhf(diff1, v_zero_f16); + + diff0 = Q6_V_vmux_QVV(p_gt0, v_zero_f16, diff0); + diff1 = Q6_V_vmux_QVV(p_gt1, v_zero_f16, diff1); + + diff0 = Q6_Vhf_vmax_VhfVhf(v_neg20_f16, diff0); + diff1 = Q6_Vhf_vmax_VhfVhf(v_neg20_f16, diff1); + + HVX_Vector diff_log2e0 = hvx_vec_mul_f16_f16(diff0, v_log2e_f16); + HVX_Vector diff_log2e1 = hvx_vec_mul_f16_f16(diff1, v_log2e_f16); + + HVX_Vector v_exp0 = hvx_vec_exp2_f16(diff_log2e0); + HVX_Vector v_exp1 = hvx_vec_exp2_f16(diff_log2e1); + + HVX_VectorPred mask_lt0 = Q6_Q_vsetq2_R(2 * t0); + HVX_VectorPred mask_lt1 = Q6_Q_vsetq2_R(2 * t1); + + HVX_Vector v_m0 = Q6_V_vand_QV(mask_lt0, v_exp0); + HVX_Vector v_m1 = Q6_V_vand_QV(mask_lt1, v_exp1); + + HVX_VectorPred mask_le0 = Q6_Q_vsetq2_R(2 * (t0 + 1)); + HVX_VectorPred mask_le1 = Q6_Q_vsetq2_R(2 * (t1 + 1)); + + HVX_VectorPred mask_diag0 = Q6_Q_and_QQn(mask_le0, mask_lt0); + HVX_VectorPred mask_diag1 = Q6_Q_and_QQn(mask_le1, mask_lt1); + + HVX_Vector v_a0 = Q6_V_vmux_QVV(mask_diag0, v_one_f16, v_m0); + HVX_Vector v_a1 = Q6_V_vmux_QVV(mask_diag1, v_one_f16, v_m1); + + hvx_vmem(head->decay_m + t0 * 64) = v_m0; + hvx_vmem(head->decay_a + t0 * 64) = v_a0; + hvx_vmem(head->decay_m + t1 * 64) = v_m1; + hvx_vmem(head->decay_a + t1 * 64) = v_a1; + } + + { + HVX_Vector v_gamma_t = Q6_Vh_vsplat_R(gamma_u16[63]); + HVX_Vector diff = hvx_vec_sub_f16_f16(v_gamma_t, v_gamma); + HVX_VectorPred p_gt = Q6_Q_vcmp_gt_VhfVhf(diff, v_zero_f16); + diff = Q6_V_vmux_QVV(p_gt, v_zero_f16, diff); + diff = Q6_Vhf_vmax_VhfVhf(v_neg20_f16, diff); + + HVX_Vector diff_log2e = hvx_vec_mul_f16_f16(diff, v_log2e_f16); + HVX_Vector v_exp = hvx_vec_exp2_f16(diff_log2e); + + HVX_VectorPred mask_lt_t = Q6_Q_vsetq2_R(2 * 63); + HVX_Vector v_m = Q6_V_vand_QV(mask_lt_t, v_exp); + + HVX_VectorPred mask_le_t = Q6_Q_vcmp_eq_VhVh(v_zero_f16, v_zero_f16); + HVX_VectorPred mask_diag = Q6_Q_and_QQn(mask_le_t, mask_lt_t); + HVX_Vector v_a = Q6_V_vmux_QVV(mask_diag, v_one_f16, v_m); + + hvx_vmem(head->decay_m + 63 * 64) = v_m; + hvx_vmem(head->decay_a + 63 * 64) = v_a; + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_PREP, info); +} + +static void gdn_hvx_phase2_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_SOLVE, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t curr_buf = bctx->curr_buf; + + gdn_unpack_64x64_tiles_to_vectors(head->rows_kk, head->kk_tiles); + + gdn_build_inv_l_blocks( + head->inv_row_tiles, + head->rows_kk, + head->decay_m, + head->b_f32[curr_buf], + (__fp16 *) head->vtcm_m, + (__fp16 *) head->vtcm_m + HMX_FP16_TILE_N_ELMS + ); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_SOLVE, info); +} + +static void gdn_hvx_phase3_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_V_PREP, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t curr_buf = bctx->curr_buf; + const uint32_t S_v = bctx->S_v; + + gdn_unpack_64xS_tiles_to_f32(head->v_inter_f32, head->v_inter_tiles, S_v); + + HVX_VectorAlias local_b[2]; + local_b[0].v = hvx_vmem(head->b_f32[curr_buf] + 0); + local_b[1].v = hvx_vmem(head->b_f32[curr_buf] + 32); + + for (uint32_t t = 0; t < 64; ++t) { + HVX_Vector vb = hvx_vec_splat_f32(local_b[t / 32].fp32[t % 32]); + for (uint32_t j = 0; j < S_v; j += 64) { + HVX_Vector vv0 = hvx_vmem(head->v_f32[curr_buf] + t * S_v + j + 0); + HVX_Vector vv1 = (j + 32 < S_v) ? hvx_vmem(head->v_f32[curr_buf] + t * S_v + j + 32) : Q6_V_vzero(); + HVX_Vector vi0 = hvx_vmem(head->v_inter_f32 + t * S_v + j + 0); + HVX_Vector vi1 = (j + 32 < S_v) ? hvx_vmem(head->v_inter_f32 + t * S_v + j + 32) : Q6_V_vzero(); + + HVX_Vector vp0 = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv0, vi0), vb); + HVX_Vector vp1 = hvx_vec_mul_f32_f32(hvx_vec_sub_f32_f32(vv1, vi1), vb); + + hvx_vmem(head->v_prime_f16 + t * S_v + j) = hvx_vec_f32_to_f16(vp0, vp1); + } + } + + hmx_interleave_cols_to_tiles(head->v_prime_col_tiles, head->v_prime_f16, 64, S_v, S_v, 2, 0, 64); + + gdn_unpack_64x64_tiles_to_vectors(head->rows_qk, head->qk_tiles); + for (uint32_t t = 0; t < 64; ++t) { + HVX_Vector v_decay_a = hvx_vmem(head->decay_a + t * 64); + head->rows_a[t] = hvx_vec_mul_f16_f16(head->rows_qk[t], v_decay_a); + } + gdn_pack_64x64_vectors_to_tiles(head->a_row_tiles, head->rows_a); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_V_PREP, info); } +static void gdn_hvx_phase4_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_D_PREP, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t S_v = bctx->S_v; + + gdn_unpack_64xS_tiles_to_f16(head->delta_f16, head->delta_tiles, S_v); + hmx_interleave_cols_to_tiles(head->delta_col_tiles, head->delta_f16, 64, S_v, S_v, 2, 0, 64); + + const uint16_t * decay_last = (const uint16_t *) (head->decay_a + 63 * 64); + const HVX_Vector vzero = Q6_V_vzero(); + + for (uint32_t s = 0; s < 64; ++s) { + HVX_Vector vs = Q6_Vh_vsplat_R(decay_last[s]); + HVX_VectorPred p_zero = Q6_Q_vcmp_eq_VhVh(vs, vzero); + for (uint32_t j = 0; j < S_v; j += 64) { + HVX_Vector vd = hvx_vmem(head->delta_f16 + s * S_v + j); + HVX_Vector prod = hvx_vec_mul_f16_f16(vd, vs); + hvx_vmem(head->d_f16 + s * S_v + j) = Q6_V_vmux_QVV(p_zero, vzero, prod); + } + } + + gdn_pack_d_t_row_tiles(head->d_row_tiles, head->d_f16, S_v, head->vtcm_m, head->vtcm_tmp); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_D_PREP, info); +} + +static void gdn_hvx_phase5_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_OUT, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t curr_buf = bctx->curr_buf; + const uint32_t S_v = bctx->S_v; + const float scale = bctx->scale; + + gdn_unpack_64xS_tiles_to_f32(head->o_inter_f32, head->o_inter_tiles, S_v); + gdn_unpack_64xS_tiles_to_f32(head->o_intra_f32, head->o_intra_tiles, S_v); + + HVX_Vector vscale = hvx_vec_splat_f32(scale); + for (uint32_t j = 0; j < 64 * S_v / 32; ++j) { + HVX_Vector vi = hvx_vmem(head->o_inter_f32 + j * 32); + HVX_Vector va = hvx_vmem(head->o_intra_f32 + j * 32); + hvx_vmem(head->o_f32[curr_buf] + j * 32) = hvx_vec_mul_f32_f32(hvx_vec_add_f32_f32(vi, va), vscale); + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_OUT, info); +} + +static void gdn_hvx_phase6_worker(unsigned int n, unsigned int i, void * data) { + (void) n; + struct htp_gdn_batch_context * bctx = (struct htp_gdn_batch_context *) data; + struct htp_thread_trace * tr = &bctx->octx->ctx->trace[i]; + const uint16_t info = (uint16_t) bctx->c; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_GDN_STATE, info); + + struct htp_gdn_head_ptrs * head = &bctx->heads[i]; + const uint32_t S_v = bctx->S_v; + const uint32_t c = bctx->c; + const uint32_t n_chunks = bctx->kparams->n_chunks; + + gdn_unpack_SxS_tiles_to_f32(head->s_update_f32, head->s_update_tiles, S_v); + + HVX_VectorAlias last_lambda; + last_lambda.v = hvx_vmem(head->lambda_init + 32); + HVX_Vector v_l_final = hvx_vec_splat_f32(last_lambda.fp32[31]); + + for (uint32_t j = 0; j < S_v * S_v / 32; ++j) { + HVX_Vector vs_old = hvx_vmem(head->s_state + j * 32); + HVX_Vector vsu = hvx_vmem(head->s_update_f32 + j * 32); + hvx_vmem(head->s_state + j * 32) = hvx_vec_add_f32_f32(hvx_vec_mul_f32_f32(vs_old, v_l_final), vsu); + } + + if (c + 1 < n_chunks) { + gdn_pack_s_col_tiles(head->s_col_tiles, head->s_f16, head->s_state, S_v); + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_GDN_STATE, info); +} + + +static int gated_delta_net_f32_hmx_chunked( + struct htp_ops_context * octx, + const struct htp_gdn_kernel_params * kparams, + uint32_t row_start, + uint32_t nrows +) { + const struct htp_tensor * q = octx->src[0]; + const struct htp_tensor * k = octx->src[1]; + const struct htp_tensor * v = octx->src[2]; + const struct htp_tensor * g = octx->src[3]; + const struct htp_tensor * beta = octx->src[4]; + const struct htp_tensor * state = octx->src[5]; + const struct htp_tensor * dst = octx->dst; + const struct htp_tensor * dst_cache = octx->dsts[1]; + + const uint32_t S_v = kparams->S_v; + const uint32_t H = kparams->H; + const uint32_t n_tokens = kparams->n_tokens; + const float scale = kparams->scale; + const uint32_t chunk_size = kparams->chunk_size; + const uint32_t n_chunks = kparams->n_chunks; + const uint32_t n_sv_tiles = S_v / 32; + + struct htp_gdn_hmx_vtcm_layout L; + htp_gdn_hmx_vtcm_layout_build(&L, S_v, chunk_size, kparams->n_heads_batch, kparams->n_threads, kparams->pipeline != 0); + + if (L.total_bytes > octx->ctx->vtcm_size) { + return HTP_STATUS_VTCM_TOO_SMALL; + } + + uint8_t * const vtcm_base = (uint8_t *) octx->ctx->vtcm_base; + + float * vtcm_g_raw[2] = { + VTCM_LAYOUT_PTR(float, vtcm_base, L.off_g_raw[0]), + L.pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L.off_g_raw[1]) : VTCM_LAYOUT_PTR(float, vtcm_base, L.off_g_raw[0]) + }; + float * vtcm_b_raw[2] = { + VTCM_LAYOUT_PTR(float, vtcm_base, L.off_b_raw[0]), + L.pipeline ? VTCM_LAYOUT_PTR(float, vtcm_base, L.off_b_raw[1]) : VTCM_LAYOUT_PTR(float, vtcm_base, L.off_b_raw[0]) + }; + + uint8_t * vtcm_scales_1 = VTCM_LAYOUT_PTR(uint8_t, vtcm_base, L.off_scales_1); + hmx_init_column_scales(vtcm_scales_1, Q6_V_vsplat_R(0x3c00)); + + hmx_queue_t hmx_q = octx->ctx->hmx_queue; + dma_queue * dma_q = octx->ctx->dma[0]; + work_queue_t wp = octx->ctx->work_queue; + + struct htp_gdn_head_ptrs heads[8]; + struct htp_gdn_hmx_gemm_task gemm_tasks[8][9]; + + uint32_t n_batch = 1; + for (uint32_t r = row_start; r < row_start + nrows; r += n_batch) { + const uint32_t head_in_seq = fastmodulo(r, H, &kparams->div_H); + const uint32_t iv3 = fastdiv(r, &kparams->div_H); + const uint32_t heads_left_in_seq = H - head_in_seq; + const uint32_t heads_left_in_range = (row_start + nrows) - r; + n_batch = hex_smin((uint32_t) kparams->n_heads_batch, hex_smin(heads_left_in_seq, heads_left_in_range)); + + for (uint32_t h = 0; h < n_batch; ++h) { + gdn_init_head_ptrs(&heads[h], &L, vtcm_base, h, head_in_seq, iv3, + q, k, v, state, dst, dst_cache, kparams, S_v, H, n_tokens, chunk_size); + } + + struct htp_gdn_batch_context bctx; + bctx.heads = heads; + bctx.vtcm_g_raw = NULL; + bctx.vtcm_b_raw = NULL; + bctx.curr_buf = 0; + bctx.c = 0; + bctx.n_batch = n_batch; + bctx.S_v = S_v; + bctx.scale = scale; + bctx.octx = octx; + bctx.kparams = kparams; + + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_push(dma_q, dma_make_data(heads[h].s_state, heads[h].state_in_dma), + S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); + } + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_pop(dma_q); + } + + if (n_chunks > 0) { + work_queue_run(wp, gdn_hvx_init_state_worker, &bctx, n_batch); + + const uint32_t chunk0_tokens = hex_smin(chunk_size, n_tokens); + for (uint32_t h = 0; h < n_batch; ++h) { + gdn_dma_push_chunk_inputs(dma_q, heads[h].q_f32[0], heads[h].k_f32[0], heads[h].v_f32[0], + q, k, v, heads[h].iq3, heads[h].iq1, heads[h].ik3, heads[h].ik1, + heads[h].iv3, heads[h].iv1, 0, chunk0_tokens, S_v); + } + gdn_dma_push_chunk_gb(dma_q, vtcm_g_raw[0], vtcm_b_raw[0], g, beta, iv3, head_in_seq, 0, chunk0_tokens, n_batch); + } + + for (uint32_t c = 0; c < n_chunks; ++c) { + const uint32_t curr_buf = c & 1; + const uint32_t next_buf = (c + 1) & 1; + const uint32_t t_chunk = c * chunk_size; + + bctx.curr_buf = curr_buf; + bctx.c = c; + bctx.vtcm_g_raw = vtcm_g_raw[curr_buf]; + bctx.vtcm_b_raw = vtcm_b_raw[curr_buf]; + + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_pop(dma_q); + dma_queue_pop(dma_q); + dma_queue_pop(dma_q); + } + dma_queue_pop(dma_q); + dma_queue_pop(dma_q); + + if (c + 1 < n_chunks) { + const uint32_t next_t_chunk = (c + 1) * chunk_size; + const uint32_t next_tokens = hex_smin(chunk_size, n_tokens - next_t_chunk); + for (uint32_t h = 0; h < n_batch; ++h) { + gdn_dma_push_chunk_inputs(dma_q, heads[h].q_f32[next_buf], heads[h].k_f32[next_buf], heads[h].v_f32[next_buf], + q, k, v, heads[h].iq3, heads[h].iq1, heads[h].ik3, heads[h].ik1, + heads[h].iv3, heads[h].iv1, next_t_chunk, next_tokens, S_v); + } + gdn_dma_push_chunk_gb(dma_q, vtcm_g_raw[next_buf], vtcm_b_raw[next_buf], + g, beta, iv3, head_in_seq, next_t_chunk, next_tokens, n_batch); + } + + if (c > 0) { + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_pop(dma_q); + } + } + + work_queue_run(wp, gdn_hvx_phase1a_worker, &bctx, n_batch); + + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][0], heads[h].k_row_tiles, heads[h].k_col_tiles, heads[h].kk_tiles, 2, 2, n_sv_tiles, vtcm_scales_1); + } + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][2], heads[h].k_prime_row_tiles, heads[h].s_col_tiles, heads[h].v_inter_tiles, 2, n_sv_tiles, n_sv_tiles, vtcm_scales_1); + } + + work_queue_run(wp, gdn_hvx_phase1b_worker, &bctx, n_batch); + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][1], heads[h].q_row_tiles, heads[h].k_col_tiles, heads[h].qk_tiles, 2, 2, n_sv_tiles, vtcm_scales_1); + } + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][3], heads[h].q_prime_row_tiles, heads[h].s_col_tiles, heads[h].o_inter_tiles, 2, n_sv_tiles, n_sv_tiles, vtcm_scales_1); + } + + work_queue_run(wp, gdn_hvx_phase2_worker, &bctx, n_batch); + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task( + hmx_q, &gemm_tasks[h][7], + (__fp16 *) heads[h].vtcm_m, + heads[h].inv_row_tiles + 0 * HMX_FP16_TILE_N_ELMS, + (__fp16 *) heads[h].vtcm_tmp, + 1, 1, 1, vtcm_scales_1 + ); + } + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task( + hmx_q, &gemm_tasks[h][8], + (__fp16 *) heads[h].vtcm_m + HMX_FP16_TILE_N_ELMS, + (__fp16 *) heads[h].vtcm_tmp, + heads[h].inv_row_tiles + 2 * HMX_FP16_TILE_N_ELMS, + 1, 1, 1, vtcm_scales_1 + ); + } + + work_queue_run(wp, gdn_hvx_phase3_worker, &bctx, n_batch); + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][4], heads[h].inv_row_tiles, heads[h].v_prime_col_tiles, heads[h].delta_tiles, 2, n_sv_tiles, 2, vtcm_scales_1); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + work_queue_run(wp, gdn_hvx_phase4_worker, &bctx, n_batch); + + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][5], heads[h].a_row_tiles, heads[h].delta_col_tiles, heads[h].o_intra_tiles, 2, n_sv_tiles, 2, vtcm_scales_1); + } + for (uint32_t h = 0; h < n_batch; ++h) { + htp_gdn_push_hmx_gemm_task(hmx_q, &gemm_tasks[h][6], heads[h].d_row_tiles, heads[h].k_col_tiles_64x128, heads[h].s_update_tiles, n_sv_tiles, n_sv_tiles, 2, vtcm_scales_1); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + work_queue_run(wp, gdn_hvx_phase5_worker, &bctx, n_batch); + + const uint32_t valid_tokens = hex_smin(chunk_size, n_tokens - t_chunk); + for (uint32_t h = 0; h < n_batch; ++h) { + const dma_addr_t attn_chunk_dma = dst->data + + ((uint64_t) heads[h].iv3 * n_tokens * H + (uint64_t) t_chunk * H + heads[h].iv1) * S_v * sizeof(float); + dma_queue_push(dma_q, dma_make_data(attn_chunk_dma, heads[h].o_f32[curr_buf]), + dst->nb[1], S_v * sizeof(float), S_v * sizeof(float), valid_tokens); + } + + for (uint32_t h = 0; h < n_batch; ++h) { + hmx_queue_pop(hmx_q); + } + + work_queue_run(wp, gdn_hvx_phase6_worker, &bctx, n_batch); + } + + if (n_chunks > 0) { + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_pop(dma_q); + } + } + + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_push(dma_q, dma_make_data(heads[h].state_out_dma, heads[h].s_state), + S_v * sizeof(float), S_v * sizeof(float), S_v * sizeof(float), S_v); + } + for (uint32_t h = 0; h < n_batch; ++h) { + dma_queue_pop(dma_q); + } + } + + dma_queue_flush(dma_q); + return HTP_STATUS_OK; +} int op_gated_delta_net(struct htp_ops_context * octx) { const struct htp_tensor * q = octx->src[0]; @@ -1085,10 +2278,6 @@ int op_gated_delta_net(struct htp_ops_context * octx) { const struct htp_tensor * state = octx->src[5]; const struct htp_tensor * dst = octx->dst; - if (!q || !k || !v || !g || !beta || !state || !dst) { - return HTP_STATUS_INVAL_PARAMS; - } - if (q->type != HTP_TYPE_F32 || k->type != HTP_TYPE_F32 || v->type != HTP_TYPE_F32 || g->type != HTP_TYPE_F32 || beta->type != HTP_TYPE_F32 || state->type != HTP_TYPE_F32 || dst->type != HTP_TYPE_F32) { @@ -1120,28 +2309,141 @@ int op_gated_delta_net(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + for (int i = 0; i < 5; i++) { + if (htp_tensor_is_extended(octx->src[i])) { + return HTP_STATUS_NO_SUPPORT; + } + } + if (htp_tensor_is_extended(octx->dst)) { + return HTP_STATUS_NO_SUPPORT; + } + if (octx->dsts[1]) { + const struct htp_tensor * dst_cache = octx->dsts[1]; + if (dst_cache->type != HTP_TYPE_F32 || htp_tensor_is_extended(dst_cache)) { + return HTP_STATUS_NO_SUPPORT; + } + } + + const struct htp_gdn_kernel_params * kparams = (const struct htp_gdn_kernel_params *) octx->kernel_params; + struct htp_gdn_kernel_params kparams_local; + if (!kparams || kparams->S_v == 0) { + const uint32_t rq3 = n_seqs / q->ne[3]; + const uint32_t rk3 = n_seqs / k->ne[3]; + const uint32_t total_rows = H * n_seqs; + uint32_t n_threads = (total_rows < octx->n_threads) ? total_rows : octx->n_threads; + if (n_threads == 0) { + n_threads = 1; + } + + memset(&kparams_local, 0, sizeof(kparams_local)); + kparams_local.n_threads = n_threads; + kparams_local.S_v = S_v; + kparams_local.H = H; + kparams_local.n_tokens = n_tokens; + kparams_local.n_seqs = n_seqs; + kparams_local.K = K; + kparams_local.total_rows = total_rows; + kparams_local.rows_per_thread = (total_rows + n_threads - 1) / n_threads; + const bool can_use_hmx = (octx->ctx->hmx_enabled) && + (S_v % 64 == 0) && + (n_tokens >= HTP_GDN_MIN_TOKENS) && + (g->ne[0] == 1) && + (K == 1); + + struct htp_gdn_hmx_vtcm_layout hmx_layout_local; + struct htp_gdn_vtcm_layout hvx_layout_local; + uint32_t n_heads_batch = 1; + + if (can_use_hmx && htp_gdn_hmx_solve_layout(&hmx_layout_local, S_v, HTP_GDN_CHUNK_SIZE, total_rows, octx->ctx->vtcm_size, n_threads, true, &n_heads_batch)) { + kparams_local.kernel_type = HTP_GDN_KERNEL_HMX_CHUNKED; + kparams_local.pipeline = hmx_layout_local.pipeline ? 1 : 0; + kparams_local.chunk_size = HTP_GDN_CHUNK_SIZE; + kparams_local.n_chunks = (n_tokens + HTP_GDN_CHUNK_SIZE - 1) / HTP_GDN_CHUNK_SIZE; + kparams_local.n_heads_batch = (uint16_t) n_heads_batch; + kparams_local.vtcm_size = (uint32_t) hmx_layout_local.total_bytes; + kparams_local.state_aligned = (uint32_t) hmx_layout_local.state_f32_bytes; + kparams_local.vtcm_per_thread = (uint32_t) (hmx_layout_local.total_bytes / (n_threads > 0 ? n_threads : 1)); + } else { + htp_gdn_vtcm_layout_build(&hvx_layout_local, S_v, n_threads); + kparams_local.kernel_type = HTP_GDN_KERNEL_HVX_RECURRENT; + kparams_local.pipeline = 0; + kparams_local.n_heads_batch = 1; + kparams_local.state_aligned = (uint32_t) hvx_layout_local.state_aligned; + kparams_local.vtcm_per_thread = (uint32_t) hvx_layout_local.bytes_per_thread; + kparams_local.vtcm_size = (uint32_t) hvx_layout_local.total_bytes; + } + kparams_local.kda = (g->ne[0] == S_v) ? 1 : 0; + kparams_local.scale = 1.0f / sqrtf((float) S_v); + kparams_local.state_seq_stride = (uint32_t) (state->nb[3] / sizeof(float)); + kparams_local.state_size_per_snap = S_v * S_v * H * n_seqs; + + kparams_local.div_H = init_fastdiv_values(H); + kparams_local.div_q1 = init_fastdiv_values(q->ne[1]); + kparams_local.div_k1 = init_fastdiv_values(k->ne[1]); + kparams_local.div_rq3 = init_fastdiv_values(rq3); + kparams_local.div_rk3 = init_fastdiv_values(rk3); + kparams_local.div_n_threads = init_fastdiv_values(n_threads); + + kparams = &kparams_local; + } + + const uint32_t total_rows = kparams->total_rows; + uint32_t row_start = 0; + uint32_t nrows = total_rows; + + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && + ((dst->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition( + total_rows, + can_split ? 1 : 0, + octx->ctx->mdev.idx, + octx->ctx->mdev.count, + &octx->ctx->mdev.count_div + ); + row_start = range.start; + nrows = range.count; + } else if (octx->op_params[1] != 0) { + row_start = octx->op_params[1]; + nrows = octx->op_params[2]; + } + + if (nrows == 0) { return HTP_STATUS_OK; } + if (kparams->kernel_type == HTP_GDN_KERNEL_HMX_CHUNKED) { + return gated_delta_net_f32_hmx_chunked(octx, kparams, row_start, nrows); + } + + const uint32_t n_threads = (nrows < kparams->n_threads) ? nrows : kparams->n_threads; + struct htp_gdn_context gctx; - gctx.octx = octx; - gctx.rows_per_thread = (H * n_seqs + octx->n_threads - 1) / octx->n_threads; - gctx.state_bytes = (size_t) S_v * S_v * sizeof(float); + gctx.octx = octx; + gctx.kparams = kparams; + gctx.row_start = row_start; + gctx.nrows = nrows; + gctx.vtcm_base = octx->ctx->vtcm_base; - size_t state_aligned = (size_t) S_v * S_v * sizeof(float); - state_aligned = (state_aligned + 127) & ~(size_t)127; + htp_gdn_vtcm_layout_build(&gctx.layout, S_v, n_threads); - assert(octx->ctx->vtcm_base != NULL); - assert(octx->ctx->vtcm_size >= 2 * state_aligned * octx->n_threads); + if (gctx.layout.total_bytes > octx->ctx->vtcm_size) { + return HTP_STATUS_VTCM_TOO_SMALL; + } - gctx.vtcm_base = octx->ctx->vtcm_base; - gctx.vtcm_per_thread = 2 * state_aligned; + FARF(HIGH, "gated-delta-net-f32: q(%ux%ux%ux%u) k(%ux%ux%ux%u) v(%ux%ux%ux%u) state(%ux%ux%ux%u) -> (%ux%ux%ux%u) : " + "vtcm-size %zu n_threads %u\n", + q->ne[0], q->ne[1], q->ne[2], q->ne[3], + k->ne[0], k->ne[1], k->ne[2], k->ne[3], + v->ne[0], v->ne[1], v->ne[2], v->ne[3], + state->ne[0], state->ne[1], state->ne[2], state->ne[3], + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], + gctx.layout.total_bytes, n_threads); if (n_tokens == 1) { - worker_pool_run_func(octx->ctx->worker_pool, gated_delta_net_f32_tg_thread, &gctx, octx->n_threads); + work_queue_run(octx->ctx->work_queue, gated_delta_net_f32_tg_thread, &gctx, n_threads); } else { - worker_pool_run_func(octx->ctx->worker_pool, gated_delta_net_f32_pp_thread, &gctx, octx->n_threads); + work_queue_run(octx->ctx->work_queue, gated_delta_net_f32_pp_thread, &gctx, n_threads); } return HTP_STATUS_OK; diff --git a/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.h b/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.h new file mode 100644 index 00000000..32fb7d24 --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.h @@ -0,0 +1,301 @@ +#ifndef HTP_GATED_DELTA_NET_OPS_H +#define HTP_GATED_DELTA_NET_OPS_H + +#include +#include +#include + +#include "hex-fastdiv.h" +#include "hex-common.h" +#include "htp-vtcm.h" + +#define HTP_GDN_MAX_SV 128 +#define HTP_GDN_CHUNK_SIZE 64 +#define HTP_GDN_MIN_TOKENS 8 + +#ifndef HMX_FP16_TILE_SIZE +#define HMX_FP16_TILE_SIZE 2048 +#endif + +enum htp_gdn_kernel_type { + HTP_GDN_KERNEL_HVX_RECURRENT = 0, + HTP_GDN_KERNEL_HMX_CHUNKED = 1, +}; + +struct htp_gdn_kernel_params { + uint8_t kernel_type; + uint8_t pipeline; + uint16_t chunk_size; + uint16_t n_chunks; + uint16_t n_heads_batch; + + uint32_t n_threads; + uint32_t S_v; + uint32_t H; + uint32_t n_tokens; + uint32_t n_seqs; + uint32_t K; + + uint32_t total_rows; + uint32_t row_start; + uint32_t nrows; + uint32_t rows_per_thread; + + uint32_t kda; + uint32_t state_aligned; + uint32_t vtcm_per_thread; + uint32_t vtcm_size; + uint32_t state_seq_stride; + uint32_t state_size_per_snap; + + float scale; + + struct fastdiv_values div_H; + struct fastdiv_values div_q1; + struct fastdiv_values div_k1; + struct fastdiv_values div_rq3; + struct fastdiv_values div_rk3; + struct fastdiv_values div_n_threads; +}; + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_gdn_kernel_params) <= 128, "htp_gdn_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_gdn_kernel_params) <= 128, "htp_gdn_kernel_params is too large for kernel_params blob"); +#endif + +struct htp_gdn_vtcm_layout { + size_t state_aligned; + size_t bytes_per_thread; + size_t total_bytes; +}; + +static inline void htp_gdn_vtcm_layout_build( + struct htp_gdn_vtcm_layout * layout, + uint32_t S_v, + uint32_t n_threads +) { + size_t state_bytes = (size_t) S_v * S_v * sizeof(float); + layout->state_aligned = hex_round_up(state_bytes, 128); + layout->bytes_per_thread = 2 * layout->state_aligned; + layout->total_bytes = layout->bytes_per_thread * n_threads; +} + +struct htp_gdn_hmx_vtcm_layout { + size_t off_s_state; + size_t off_s_f16; + size_t off_s_col_tiles; + size_t off_s_update_f32; + size_t off_s_update_tiles; + + size_t off_q_f32[2]; + size_t off_k_f32[2]; + size_t off_v_f32[2]; + size_t off_g_f32[2]; + size_t off_b_f32[2]; + size_t off_g_raw[2]; + size_t off_b_raw[2]; + size_t off_o_f32[2]; + + size_t off_v_inter_f32; + size_t off_o_inter_f32; + size_t off_o_intra_f32; + size_t off_k_f16; + size_t off_v_prime_f16; + size_t off_delta_f16; + size_t off_d_f16; + + size_t off_q_row_tiles; + size_t off_q_prime_row_tiles; + size_t off_k_row_tiles; + size_t off_k_col_tiles; + size_t off_k_prime_row_tiles; + size_t off_k_col_tiles_64x128; + size_t off_kk_tiles; + size_t off_qk_tiles; + size_t off_v_inter_tiles; + size_t off_o_inter_tiles; + size_t off_inv_row_tiles; + size_t off_a_row_tiles; + size_t off_v_prime_col_tiles; + size_t off_delta_tiles; + size_t off_delta_col_tiles; + size_t off_o_intra_tiles; + size_t off_d_row_tiles; + + size_t off_gamma; + size_t off_lambda_init; + size_t off_decay_m; + size_t off_decay_a; + size_t off_rows_kk; + size_t off_rows_qk; + size_t off_rows_inv; + size_t off_rows_a; + + size_t off_thread_scratch; + size_t off_scales_1; + + size_t state_f32_bytes; + size_t state_f16_bytes; + size_t state_tiles_bytes; + size_t dma_chunk_bytes; + size_t act_f16_bytes; + size_t tile_64xSv_bytes; + size_t tile_64x64_bytes; + + uint32_t n_heads_batch; + uint32_t n_threads; + bool pipeline; + size_t total_bytes; +}; + +static inline void htp_gdn_hmx_vtcm_layout_build( + struct htp_gdn_hmx_vtcm_layout * L, + uint32_t S_v, + uint32_t chunk_size, + uint32_t n_heads_batch, + uint32_t n_threads, + bool pipeline +) { + memset(L, 0, sizeof(*L)); + L->n_heads_batch = n_heads_batch; + L->n_threads = n_threads; + L->pipeline = pipeline; + + const size_t bh = (size_t) n_heads_batch; + const size_t nth = (size_t) (n_threads > 0 ? n_threads : 1); + + const size_t state_f32_sz = hex_round_up(S_v * S_v * sizeof(float), 2048); + const size_t state_f16_sz = hex_round_up(S_v * S_v * sizeof(__fp16), 2048); + const size_t n_sv_tiles = S_v / 32; + const size_t state_tiles_sz = n_sv_tiles * n_sv_tiles * 2048; + + const size_t dma_chunk_sz = hex_round_up(chunk_size * S_v * sizeof(float), 2048); + const size_t dma_scalar_sz = hex_round_up(chunk_size * sizeof(float), 128); + + const size_t act_f16_sz = hex_round_up(chunk_size * S_v * sizeof(__fp16), 2048); + const size_t tile_64xSv_sz = 2 * n_sv_tiles * 2048; + const size_t tile_64x64_sz = 4 * 2048; + + const size_t decay_sz = 64 * 64 * sizeof(__fp16); + const size_t row_vecs_sz = 64 * 128; + + L->state_f32_bytes = state_f32_sz; + L->state_f16_bytes = state_f16_sz; + L->state_tiles_bytes = state_tiles_sz; + L->dma_chunk_bytes = dma_chunk_sz; + L->act_f16_bytes = act_f16_sz; + L->tile_64xSv_bytes = tile_64xSv_sz; + L->tile_64x64_bytes = tile_64x64_sz; + + size_t off = 0; + + VTCM_LAYOUT_ALLOC(off, off_s_state, bh * state_f32_sz); + VTCM_LAYOUT_ALLOC(off, off_s_f16, bh * state_f16_sz); + off = hex_align_up(off, HMX_FP16_TILE_SIZE); + VTCM_LAYOUT_ALLOC(off, off_s_col_tiles, bh * state_tiles_sz); + VTCM_LAYOUT_ALLOC(off, off_s_update_f32, bh * state_f32_sz); + off = hex_align_up(off, HMX_FP16_TILE_SIZE); + VTCM_LAYOUT_ALLOC(off, off_s_update_tiles, bh * state_tiles_sz); + + VTCM_LAYOUT_ALLOC(off, off_q_f32[0], bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_q_f32[1], bh * dma_chunk_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_k_f32[0], bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_k_f32[1], bh * dma_chunk_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_v_f32[0], bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_v_f32[1], bh * dma_chunk_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_g_f32[0], bh * dma_scalar_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_g_f32[1], bh * dma_scalar_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_b_f32[0], bh * dma_scalar_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_b_f32[1], bh * dma_scalar_sz, pipeline); + const size_t raw_gb_sz = hex_round_up(bh * chunk_size * sizeof(float), 128); + VTCM_LAYOUT_ALLOC(off, off_g_raw[0], raw_gb_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_g_raw[1], raw_gb_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_b_raw[0], raw_gb_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_b_raw[1], raw_gb_sz, pipeline); + VTCM_LAYOUT_ALLOC(off, off_o_f32[0], bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC_OPTIONAL(off, off_o_f32[1], bh * dma_chunk_sz, pipeline); + + VTCM_LAYOUT_ALLOC(off, off_v_inter_f32, bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC(off, off_o_inter_f32, bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC(off, off_o_intra_f32, bh * dma_chunk_sz); + VTCM_LAYOUT_ALLOC(off, off_k_f16, bh * act_f16_sz); + VTCM_LAYOUT_ALLOC(off, off_v_prime_f16, bh * act_f16_sz); + VTCM_LAYOUT_ALLOC(off, off_delta_f16, bh * act_f16_sz); + VTCM_LAYOUT_ALLOC(off, off_d_f16, bh * act_f16_sz); + + off = hex_align_up(off, HMX_FP16_TILE_SIZE); + VTCM_LAYOUT_ALLOC(off, off_q_row_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_q_prime_row_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_k_row_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_k_col_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_k_prime_row_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_k_col_tiles_64x128, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_kk_tiles, bh * tile_64x64_sz); + VTCM_LAYOUT_ALLOC(off, off_qk_tiles, bh * tile_64x64_sz); + VTCM_LAYOUT_ALLOC(off, off_v_inter_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_o_inter_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_inv_row_tiles, bh * tile_64x64_sz); + VTCM_LAYOUT_ALLOC(off, off_a_row_tiles, bh * tile_64x64_sz); + VTCM_LAYOUT_ALLOC(off, off_v_prime_col_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_delta_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_delta_col_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_o_intra_tiles, bh * tile_64xSv_sz); + VTCM_LAYOUT_ALLOC(off, off_d_row_tiles, bh * tile_64xSv_sz); + + VTCM_LAYOUT_ALLOC(off, off_gamma, bh * hex_round_up(chunk_size * sizeof(float), 128)); + VTCM_LAYOUT_ALLOC(off, off_lambda_init, bh * hex_round_up(chunk_size * sizeof(float), 128)); + VTCM_LAYOUT_ALLOC(off, off_decay_m, bh * decay_sz); + VTCM_LAYOUT_ALLOC(off, off_decay_a, bh * decay_sz); + VTCM_LAYOUT_ALLOC(off, off_rows_kk, bh * row_vecs_sz); + VTCM_LAYOUT_ALLOC(off, off_rows_qk, bh * row_vecs_sz); + VTCM_LAYOUT_ALLOC(off, off_rows_inv, bh * row_vecs_sz); + VTCM_LAYOUT_ALLOC(off, off_rows_a, bh * row_vecs_sz); + + const size_t thread_scratch_sz = 64 * 128; + off = hex_align_up(off, HMX_FP16_TILE_SIZE); + VTCM_LAYOUT_ALLOC(off, off_thread_scratch, nth * thread_scratch_sz); + off = hex_align_up(off, HMX_FP16_TILE_SIZE); + VTCM_LAYOUT_ALLOC(off, off_scales_1, HMX_FP16_TILE_SIZE); + + L->total_bytes = off; +} + +static inline bool htp_gdn_hmx_solve_layout( + struct htp_gdn_hmx_vtcm_layout * layout_out, + uint32_t S_v, + uint32_t chunk_size, + uint32_t total_rows, + size_t vtcm_budget, + uint32_t n_threads, + bool pipeline, + uint32_t * n_heads_batch_out +) { + uint32_t max_batch = 8; + if (max_batch > total_rows) { + max_batch = total_rows; + } + if (max_batch > n_threads) { + max_batch = n_threads; + } + static const uint32_t candidates[] = { 8, 6, 4, 2, 1 }; + for (size_t i = 0; i < sizeof(candidates) / sizeof(candidates[0]); ++i) { + uint32_t bh = candidates[i]; + if (bh > max_batch) { + continue; + } + struct htp_gdn_hmx_vtcm_layout L; + htp_gdn_hmx_vtcm_layout_build(&L, S_v, chunk_size, bh, n_threads, pipeline); + if (L.total_bytes <= vtcm_budget) { + *layout_out = L; + *n_heads_batch_out = bh; + return true; + } + } + if (pipeline) { + return htp_gdn_hmx_solve_layout(layout_out, S_v, chunk_size, total_rows, vtcm_budget, n_threads, false, n_heads_batch_out); + } + return false; +} + +#endif // HTP_GATED_DELTA_NET_OPS_H diff --git a/ggml/src/ggml-hexagon/htp/get-rows-ops.c b/ggml/src/ggml-hexagon/htp/get-rows-ops.c index bf7063e9..f51e00c1 100644 --- a/ggml/src/ggml-hexagon/htp/get-rows-ops.c +++ b/ggml/src/ggml-hexagon/htp/get-rows-ops.c @@ -10,23 +10,27 @@ #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "dma-queue.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "htp-tensor.h" #include "hvx-utils.h" +#include "hvx-quant.h" +#include "get-rows-ops.h" +#include "work-queue.h" struct get_rows_context { struct htp_ops_context * octx; + const struct htp_get_rows_kernel_params * kparams; + struct htp_get_rows_vtcm_layout vtcm_layout; + uint8_t * vtcm_base; + uint32_t task_start; + uint32_t tasks; uint32_t tasks_per_thread; - uint32_t total_tasks; - uint32_t chunks_per_row; - uint32_t chunk_size; - struct fastdiv_values get_rows_div_ne10; - struct fastdiv_values get_rows_div_ne10_ne11; - struct fastdiv_values get_rows_div_chunks_per_row; }; -#define get_rows_preamble \ +#define get_rows_preamble \ const uint32_t ne00 = octx->src[0]->ne[0]; \ const uint32_t ne01 = octx->src[0]->ne[1]; \ const uint32_t ne02 = octx->src[0]->ne[2]; \ @@ -56,106 +60,166 @@ struct get_rows_context { \ const uint32_t nr = ne10 * ne11 * ne12; -static void get_rows_thread_f32_f32_dma(unsigned int nth, unsigned int ith, void *data) { - struct get_rows_context * grctx = (struct get_rows_context *)data; - struct htp_ops_context * octx = grctx->octx; - get_rows_preamble; - - uint64_t qt = HAP_perf_get_qtimer_count(); - - const uint32_t dr = grctx->tasks_per_thread; - const uint32_t ir0 = dr * ith; - if (ir0 >= grctx->total_tasks) { - return; - } - const uint32_t ir1 = MIN(ir0 + dr, grctx->total_tasks); - - const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); - - dma_queue * dma_queue = octx->ctx->dma[ith]; - for (uint32_t i = ir0; i < ir1; ++i) { - const uint32_t i12 = fastdiv(i, &grctx->get_rows_div_ne10_ne11); - const uint32_t rem = i - i12 * ne11 * ne10; - const uint32_t i11 = fastdiv(rem, &grctx->get_rows_div_ne10); - const uint32_t i10 = rem - i11 * ne10; - - const uintptr_t src1_addr = octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12; - uint32_t i01 = is_i32 ? *(int32_t *)src1_addr : *(int64_t *)src1_addr; - - if (i01 >= ne01) { - continue; - } - - const uintptr_t src0_ptr = octx->src[0]->data + i01*nb01 + i11*nb02 + i12*nb03; - const uintptr_t dst_ptr = octx->dst->data + i10*nb1 + i11*nb2 + i12*nb3; - - while (!dma_queue_push(dma_queue, dma_make_ptr((void *)dst_ptr, (const void *)src0_ptr), nb1, nb01, ne00 * sizeof(float), 1)) { - dma_queue_pop(dma_queue); - } - } - dma_queue_flush(dma_queue); - - qt = HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - qt); - FARF(HIGH, "get-rows-f32-f32-dma %d/%d: %ux%ux%ux%u (%u:%u) x %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", ith, nth, - ne00, ne01, ne02, ne03, ir0, ir1, ne10, ne11, ne12, ne13, ne0, ne1, ne2, ne3, (unsigned) qt); +#define GET_ROWS_THREAD_ST_FN(IDX_TYPE) \ +static void get_rows_thread_st_##IDX_TYPE(unsigned int nth, unsigned int ith, void *data) { \ + struct get_rows_context * grctx = (struct get_rows_context *)data; \ + struct htp_ops_context * octx = grctx->octx; \ + const struct htp_get_rows_kernel_params * kparams = grctx->kparams; \ + get_rows_preamble; \ + const uint32_t dr = grctx->tasks_per_thread; \ + const uint32_t ir0 = grctx->task_start + dr * ith; \ + if (ir0 >= grctx->task_start + grctx->tasks) { \ + return; \ + } \ + const uint32_t ir1 = MIN(ir0 + dr, grctx->task_start + grctx->tasks); \ + const uint32_t row_size_bytes = htp_tensor_get_row_size(octx->src[0]->type, ne00); \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + for (uint32_t i = ir0; i < ir1; ++i) { \ + const uint32_t i12 = fastdiv(i, &kparams->div_ne10_ne11); \ + const uint32_t rem = i - i12 * ne11 * ne10; \ + const uint32_t i11 = fastdiv(rem, &kparams->div_ne10); \ + const uint32_t i10 = rem - i11 * ne10; \ + const IDX_TYPE * src1_ptr = (const IDX_TYPE *)(uintptr_t)(octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12); \ + const uint32_t i01 = (uint32_t)*src1_ptr; \ + assert(i01 < ne01); \ + const uint32_t q02 = fastdiv(i11, &kparams->div_ne02); \ + const uint32_t i02 = i11 - q02 * ne02; \ + const uint32_t q03 = fastdiv(i12, &kparams->div_ne03); \ + const uint32_t i03 = i12 - q03 * ne03; \ + const dma_addr_t src0_data = octx->src[0]->data + i01*nb01 + i02*nb02 + i03*nb03; \ + const dma_addr_t dst_data = octx->dst->data + i10*nb1 + i11*nb2 + i12*nb3; \ + while (!dma_queue_push(dma_q, dma_make_data(dst_data, src0_data), nb1, nb01, \ + row_size_bytes, 1)) { \ + dma_queue_pop(dma_q); \ + } \ + } \ + dma_queue_flush(dma_q); \ } -static void get_rows_thread_f32_f32_hvx(unsigned int nth, unsigned int ith, void *data) { - struct get_rows_context * grctx = (struct get_rows_context *)data; - struct htp_ops_context * octx = grctx->octx; - get_rows_preamble; - - uint64_t qt = HAP_perf_get_qtimer_count(); - - const uint32_t dr = grctx->tasks_per_thread; - const uint32_t ir0 = dr * ith; - if (ir0 >= grctx->total_tasks) { - return; - } - const uint32_t ir1 = MIN(ir0 + dr, grctx->total_tasks); - - const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); - - const uint32_t chunks_per_row = grctx->chunks_per_row; - const uint32_t chunk_size = grctx->chunk_size; - for (uint32_t i = ir0; i < ir1; ++i) { - const uint32_t row_idx = fastdiv(i, &grctx->get_rows_div_chunks_per_row); - const uint32_t chunk_idx = i - row_idx * chunks_per_row; +GET_ROWS_THREAD_ST_FN(int32_t) +GET_ROWS_THREAD_ST_FN(int64_t) + +#define GET_ROWS_THREAD_DT_FN(TYPE_NAME, SRC0_SIZE_EXPR, IDX_TYPE, COMPUTE_EXPR) \ +static void get_rows_thread_##TYPE_NAME##_##IDX_TYPE(unsigned int nth, unsigned int ith, void *data) { \ + struct get_rows_context * grctx = (struct get_rows_context *)data; \ + struct htp_ops_context * octx = grctx->octx; \ + const struct htp_get_rows_kernel_params * kparams = grctx->kparams; \ + get_rows_preamble; \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + const uint32_t dr = grctx->tasks_per_thread; \ + const uint32_t ir0 = grctx->task_start + dr * ith; \ + if (ir0 >= grctx->task_start + grctx->tasks) { \ + return; \ + } \ + const uint32_t ir1 = MIN(ir0 + dr, grctx->task_start + grctx->tasks); \ + const uint32_t chunks_per_row = kparams->chunks_per_row; \ + const uint32_t chunk_size = kparams->chunk_size; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + const struct htp_get_rows_vtcm_layout * vtcm_layout = &grctx->vtcm_layout; \ + uint8_t * vtcm_src0 = grctx->vtcm_base + vtcm_layout->off_src0 + ith * vtcm_layout->src0_bytes_per_thread; \ + uint8_t * vtcm_dst = grctx->vtcm_base + vtcm_layout->off_dst + ith * vtcm_layout->dst_bytes_per_thread; \ + for (uint32_t step = 0, spad_idx = 0; step < ir1 - ir0 && spad_idx < 2; ++step, spad_idx++) { \ + const uint32_t i = ir0 + step; \ + const uint32_t row_idx = fastdiv(i, &kparams->div_chunks_per_row); \ + const uint32_t chunk_idx = i - row_idx * chunks_per_row; \ + const uint32_t i12 = fastdiv(row_idx, &kparams->div_ne10_ne11); \ + const uint32_t rem = row_idx - i12 * ne11 * ne10; \ + const uint32_t i11 = fastdiv(rem, &kparams->div_ne10); \ + const uint32_t i10 = rem - i11 * ne10; \ + const IDX_TYPE * src1_ptr = (const IDX_TYPE *)(uintptr_t)(octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12); \ + const uint32_t i01 = (uint32_t)*src1_ptr; \ + assert(i01 < ne01); \ + const uint32_t q02 = fastdiv(i11, &kparams->div_ne02); \ + const uint32_t i02 = i11 - q02 * ne02; \ + const uint32_t q03 = fastdiv(i12, &kparams->div_ne03); \ + const uint32_t i03 = i12 - q03 * ne03; \ + const uint32_t offset = chunk_idx * chunk_size; \ + const uint32_t cur_elems = (offset < ne00) ? MIN(chunk_size, ne00 - offset) : 0; \ + const uint32_t cur_src0_bytes = SRC0_SIZE_EXPR(cur_elems); \ + const uint32_t cur_dst_bytes = cur_elems * sizeof(float); \ + const dma_addr_t src0_data = octx->src[0]->data + i01*nb01 + i02*nb02 + i03*nb03 + SRC0_SIZE_EXPR(offset); \ + dma_queue_push(dma_q, \ + dma_make_data(octx->dst->data, \ + vtcm_dst + spad_idx * vtcm_layout->dst_spad_half_size), \ + cur_dst_bytes, vtcm_layout->dst_spad_half_size, cur_dst_bytes, 0); \ + dma_queue_push(dma_q, \ + dma_make_data(vtcm_src0 + spad_idx * vtcm_layout->src0_spad_half_size, src0_data), \ + vtcm_layout->src0_spad_half_size, cur_src0_bytes, cur_src0_bytes, 1); \ + } \ + for (uint32_t step = 0; step < ir1 - ir0; ++step) { \ + const uint32_t i = ir0 + step; \ + void * dst_spad = (void *) dma_queue_pop(dma_q).src; \ + void * src_spad = (void *) dma_queue_pop(dma_q).dst; \ + const uint32_t row_idx = fastdiv(i, &kparams->div_chunks_per_row); \ + const uint32_t chunk_idx = i - row_idx * chunks_per_row; \ + const uint32_t i12 = fastdiv(row_idx, &kparams->div_ne10_ne11); \ + const uint32_t rem = row_idx - i12 * ne11 * ne10; \ + const uint32_t i11 = fastdiv(rem, &kparams->div_ne10); \ + const uint32_t i10 = rem - i11 * ne10; \ + const uint32_t offset = chunk_idx * chunk_size; \ + const uint32_t cur_elems = (offset < ne00) ? MIN(chunk_size, ne00 - offset) : 0; \ + const uint32_t cur_dst_bytes = cur_elems * sizeof(float); \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, i); \ + COMPUTE_EXPR; \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, i); \ + const dma_addr_t dst_data = octx->dst->data + i10*nb1 + i11*nb2 + i12*nb3 + offset * sizeof(float); \ + dma_queue_push(dma_q, \ + dma_make_data(dst_data, dst_spad), \ + cur_dst_bytes, vtcm_layout->dst_spad_half_size, cur_dst_bytes, 1); \ + const uint32_t next_step = step + 2; \ + if (next_step < ir1 - ir0) { \ + const uint32_t pi = ir0 + next_step; \ + const uint32_t prow_idx = fastdiv(pi, &kparams->div_chunks_per_row); \ + const uint32_t pchunk_idx = pi - prow_idx * chunks_per_row; \ + const uint32_t pi12 = fastdiv(prow_idx, &kparams->div_ne10_ne11); \ + const uint32_t prem = prow_idx - pi12 * ne11 * ne10; \ + const uint32_t pi11 = fastdiv(prem, &kparams->div_ne10); \ + const uint32_t pi10 = prem - pi11 * ne10; \ + const IDX_TYPE * psrc1_ptr = (const IDX_TYPE *)(uintptr_t)(octx->src[1]->data + pi10*nb10 + pi11*nb11 + pi12*nb12); \ + const uint32_t pi01 = (uint32_t)*psrc1_ptr; \ + assert(pi01 < ne01); \ + const uint32_t pq02 = fastdiv(pi11, &kparams->div_ne02); \ + const uint32_t pi02 = pi11 - pq02 * ne02; \ + const uint32_t pq03 = fastdiv(pi12, &kparams->div_ne03); \ + const uint32_t pi03 = pi12 - pq03 * ne03; \ + const uint32_t poffset = pchunk_idx * chunk_size; \ + const uint32_t pcur_elems = (poffset < ne00) ? MIN(chunk_size, ne00 - poffset) : 0; \ + const uint32_t pcur_src0_bytes = SRC0_SIZE_EXPR(pcur_elems); \ + const dma_addr_t psrc0_data = \ + octx->src[0]->data + pi01*nb01 + pi02*nb02 + pi03*nb03 + SRC0_SIZE_EXPR(poffset); \ + dma_queue_push(dma_q, \ + dma_make_data(src_spad, psrc0_data), \ + vtcm_layout->src0_spad_half_size, pcur_src0_bytes, pcur_src0_bytes, 1); \ + } \ + } \ + dma_queue_flush(dma_q); \ +} - const uint32_t i12 = fastdiv(row_idx, &grctx->get_rows_div_ne10_ne11); - const uint32_t rem = row_idx - i12 * ne11 * ne10; - const uint32_t i11 = fastdiv(rem, &grctx->get_rows_div_ne10); - const uint32_t i10 = rem - i11 * ne10; +#define F32_BYTES(n) ((n) * sizeof(float)) +#define F16_BYTES(n) ((n) * sizeof(__fp16)) +#define Q8_0_BYTES(n) (((n) / 32) * sizeof(block_q8_0)) - const uintptr_t src1_addr = octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12; - uint32_t i01 = is_i32 ? *(int32_t *)src1_addr : *(int64_t *)src1_addr; +GET_ROWS_THREAD_DT_FN(f32, F32_BYTES, int32_t, { if (cur_elems > 0) hvx_copy_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, cur_elems); }) +GET_ROWS_THREAD_DT_FN(f32, F32_BYTES, int64_t, { if (cur_elems > 0) hvx_copy_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, cur_elems); }) - if (i01 >= ne01) { - continue; - } - - const uint32_t offset = chunk_idx * chunk_size; - if (offset < ne00) { - const uint32_t copy_size = MIN(chunk_size, ne00 - offset); - const uintptr_t src0_ptr = octx->src[0]->data + i01*nb01 + i11*nb02 + i12*nb03 + offset * sizeof(float); - const uintptr_t dst_ptr = octx->dst->data + i10*nb1 + i11*nb2 + i12*nb3 + offset * sizeof(float); - hvx_copy_f32_uu((uint8_t *)dst_ptr, (const uint8_t *)src0_ptr, copy_size); - } - } +GET_ROWS_THREAD_DT_FN(f16, F16_BYTES, int32_t, { hvx_dequantize_row_f16_f32((float *)dst_spad, src_spad, ne00); }) +GET_ROWS_THREAD_DT_FN(f16, F16_BYTES, int64_t, { hvx_dequantize_row_f16_f32((float *)dst_spad, src_spad, ne00); }) - qt = HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - qt); - FARF(HIGH, "get-rows-f32-f32-hvx %d/%d: %ux%ux%ux%u (%u:%u) x %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", ith, nth, - ne00, ne01, ne02, ne03, ir0, ir1, ne10, ne11, ne12, ne13, ne0, ne1, ne2, ne3, (unsigned) qt); -} +GET_ROWS_THREAD_DT_FN(q8_0, Q8_0_BYTES, int32_t, { hvx_dequantize_row_q8_0_f32((float *)dst_spad, src_spad, ne00); }) +GET_ROWS_THREAD_DT_FN(q8_0, Q8_0_BYTES, int64_t, { hvx_dequantize_row_q8_0_f32((float *)dst_spad, src_spad, ne00); }) int op_get_rows(struct htp_ops_context * octx) { - get_rows_preamble; + const struct htp_get_rows_kernel_params * kparams = (const struct htp_get_rows_kernel_params *) octx->kernel_params; - if (octx->src[0]->type != HTP_TYPE_F32) { + if (octx->src[0]->type != HTP_TYPE_F32 && + octx->src[0]->type != HTP_TYPE_F16 && + octx->src[0]->type != HTP_TYPE_Q8_0 && + octx->src[0]->type != HTP_TYPE_I32) { return HTP_STATUS_NO_SUPPORT; } - if (octx->dst->type != HTP_TYPE_F32) { + if ((octx->src[0]->type == HTP_TYPE_I32 && octx->dst->type != HTP_TYPE_I32) || + (octx->src[0]->type != HTP_TYPE_I32 && octx->dst->type != HTP_TYPE_F32)) { return HTP_STATUS_NO_SUPPORT; } @@ -163,56 +227,69 @@ int op_get_rows(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + if (htp_tensor_is_extended(octx->src[1])) { + return HTP_STATUS_NO_SUPPORT; + } + + const struct htp_tensor * dst = octx->dst; + const uint32_t total_tasks = kparams->total_tasks; + const size_t dst_row_size = htp_tensor_get_row_size(dst->type, dst->ne[0]); + + uint32_t task_start = 0; + uint32_t tasks = total_tasks; + + if (octx->ctx->mdev.count > 1) { + uint32_t tasks_per_chunk = 1; + htp_tensor_mdev_rows_per_chunk(dst, dst_row_size / dst->ne[0], (uint32_t) dst_row_size, &tasks_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_tasks, tasks_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + task_start = range.start; + tasks = range.count; + } + + if (tasks == 0) { return HTP_STATUS_OK; } - const uint32_t nb00 = octx->src[0]->nb[0]; - const uint32_t nb0 = octx->dst->nb[0]; + if (!htp_ops_context_set_n_threads(octx, (uint32_t) kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } - const bool can_use_dma = (nb00 == sizeof(float)) && (nb0 == sizeof(float)); - const bool use_dma = can_use_dma && (ne00 >= 2048); + const uint32_t n_threads = octx->n_threads; struct get_rows_context grctx; grctx.octx = octx; - grctx.get_rows_div_ne10 = init_fastdiv_values(octx->src[1]->ne[0]); - grctx.get_rows_div_ne10_ne11 = init_fastdiv_values(octx->src[1]->ne[0] * octx->src[1]->ne[1]); + grctx.kparams = kparams; + grctx.vtcm_base = (uint8_t *)octx->ctx->vtcm_base; + grctx.task_start = task_start; + grctx.tasks = tasks; + grctx.tasks_per_thread = fastdiv(tasks + n_threads - 1, &octx->n_threads_div); - if (use_dma) { - grctx.chunks_per_row = 1; - grctx.chunk_size = ne00; - grctx.total_tasks = nr; - grctx.get_rows_div_chunks_per_row = init_fastdiv_values(1); + const uint32_t ne00 = octx->src[0]->ne[0]; + htp_get_rows_vtcm_layout_build(&grctx.vtcm_layout, octx->src[0]->type, ne00, n_threads); - const uint32_t n_threads = MIN(nr, octx->n_threads); - grctx.tasks_per_thread = (nr + n_threads - 1) / n_threads; + const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); - worker_pool_run_func(octx->ctx->worker_pool, get_rows_thread_f32_f32_dma, &grctx, n_threads); + work_queue_func_t q_func = NULL; + if (kparams->use_dma) { + q_func = (work_queue_func_t)(is_i32 ? get_rows_thread_st_int32_t : get_rows_thread_st_int64_t); } else { - uint32_t chunks_per_row = 1; - uint32_t chunk_size = ne00; - uint32_t total_tasks = nr; - - if (nr < octx->n_threads) { - const uint32_t min_chunk_size = 1024; - uint32_t max_chunks = ne00 / min_chunk_size; - if (max_chunks == 0) { - max_chunks = 1; - } - chunks_per_row = MIN((octx->n_threads + nr - 1) / nr, max_chunks); - chunk_size = (ne00 + chunks_per_row - 1) / chunks_per_row; - total_tasks = nr * chunks_per_row; + switch (octx->src[0]->type) { + case HTP_TYPE_F32: q_func = (work_queue_func_t)(is_i32 ? get_rows_thread_f32_int32_t : get_rows_thread_f32_int64_t); break; + case HTP_TYPE_F16: q_func = (work_queue_func_t)(is_i32 ? get_rows_thread_f16_int32_t : get_rows_thread_f16_int64_t); break; + case HTP_TYPE_Q8_0: q_func = (work_queue_func_t)(is_i32 ? get_rows_thread_q8_0_int32_t : get_rows_thread_q8_0_int64_t); break; + case HTP_TYPE_I32: q_func = (work_queue_func_t)(is_i32 ? get_rows_thread_st_int32_t : get_rows_thread_st_int64_t); break; + default: return HTP_STATUS_NO_SUPPORT; } + } - grctx.chunks_per_row = chunks_per_row; - grctx.chunk_size = chunk_size; - grctx.total_tasks = total_tasks; - grctx.get_rows_div_chunks_per_row = init_fastdiv_values(chunks_per_row); - - const uint32_t n_threads = MIN(total_tasks, octx->n_threads); - grctx.tasks_per_thread = (total_tasks + n_threads - 1) / n_threads; + FARF(HIGH, "get-rows: (%ux%ux%ux%u) x (%ux%ux%ux%u) -> (%ux%ux%ux%u) : src0-vtcm-size %zu dst-vtcm-size %zu use-dma %d n-threads %d\n", + octx->src[0]->ne[0], octx->src[0]->ne[1], octx->src[0]->ne[2], octx->src[0]->ne[3], + octx->src[1]->ne[0], octx->src[1]->ne[1], octx->src[1]->ne[2], octx->src[1]->ne[3], + octx->dst->ne[0], octx->dst->ne[1], octx->dst->ne[2], octx->dst->ne[3], + grctx.vtcm_layout.src0_bytes_per_thread * n_threads, + grctx.vtcm_layout.dst_bytes_per_thread * n_threads, + kparams->use_dma, n_threads); - worker_pool_run_func(octx->ctx->worker_pool, get_rows_thread_f32_f32_hvx, &grctx, n_threads); - } + work_queue_run(octx->ctx->work_queue, q_func, &grctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/get-rows-ops.h b/ggml/src/ggml-hexagon/htp/get-rows-ops.h new file mode 100644 index 00000000..0e7c2ca8 --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/get-rows-ops.h @@ -0,0 +1,77 @@ +#ifndef HTP_GET_ROWS_OPS_H +#define HTP_GET_ROWS_OPS_H + +#include "hex-fastdiv.h" + +struct htp_get_rows_kernel_params { + int32_t n_threads; + int32_t use_dma; + int32_t chunks_per_row; + int32_t chunk_size; + int32_t total_tasks; + int32_t tasks_per_thread; + int32_t vtcm_size; + + // Fastdiv helpers + struct fastdiv_values div_ne10; + struct fastdiv_values div_ne10_ne11; + struct fastdiv_values div_chunks_per_row; + struct fastdiv_values div_ne02; + struct fastdiv_values div_ne03; +}; + +struct htp_get_rows_vtcm_layout { + size_t total_bytes; + size_t off_src0; + size_t off_dst; + + size_t src0_bytes_per_thread; + size_t dst_bytes_per_thread; + + size_t src0_spad_half_size; + size_t dst_spad_half_size; +}; + +static inline void htp_get_rows_vtcm_layout_build( + struct htp_get_rows_vtcm_layout * vtcm_layout, + int type, + uint32_t ne00, + uint32_t n_threads) { + + uint32_t src0_row_size = 0; + switch (type) { + case 0: // HTP_TYPE_F32 + src0_row_size = ne00 * 4; + break; + case 1: // HTP_TYPE_F16 + src0_row_size = ne00 * 2; + break; + case 8: // HTP_TYPE_Q8_0 + src0_row_size = (ne00 / 32) * 34; + break; + default: + src0_row_size = 0; + break; + } + + size_t src0_row_size_aligned = (src0_row_size + 255) & ~255; + size_t dst_row_size_aligned = (ne00 * sizeof(float) + 255) & ~255; + + vtcm_layout->src0_spad_half_size = src0_row_size_aligned; + vtcm_layout->dst_spad_half_size = dst_row_size_aligned; + + vtcm_layout->src0_bytes_per_thread = src0_row_size_aligned * 2; + vtcm_layout->dst_bytes_per_thread = dst_row_size_aligned * 2; + + vtcm_layout->off_src0 = 0; + vtcm_layout->off_dst = vtcm_layout->off_src0 + vtcm_layout->src0_bytes_per_thread * n_threads; + vtcm_layout->total_bytes = vtcm_layout->off_dst + vtcm_layout->dst_bytes_per_thread * n_threads; +} + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_get_rows_kernel_params) <= 128, "htp_get_rows_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_get_rows_kernel_params) <= 128, "htp_get_rows_kernel_params is too large for kernel_params blob"); +#endif + +#endif // HTP_GET_ROWS_OPS_H diff --git a/ggml/src/ggml-hexagon/htp/hex-common.h b/ggml/src/ggml-hexagon/htp/hex-common.h index 4714486a..e6a52540 100644 --- a/ggml/src/ggml-hexagon/htp/hex-common.h +++ b/ggml/src/ggml-hexagon/htp/hex-common.h @@ -77,4 +77,13 @@ static inline bool hex_add_overflow(size_t a, size_t b, size_t *out) { return false; } +static inline uint32_t hex_gcd_u32(uint32_t a, uint32_t b) { + while (b != 0) { + uint32_t t = b; + b = a % b; + a = t; + } + return a; +} + #endif // HEX_COMMON_H diff --git a/ggml/src/ggml-hexagon/htp/hex-dma.h b/ggml/src/ggml-hexagon/htp/hex-dma.h deleted file mode 100644 index 9e9a5f95..00000000 --- a/ggml/src/ggml-hexagon/htp/hex-dma.h +++ /dev/null @@ -1,2 +0,0 @@ -#pragma once -#include "dma-queue.h" diff --git a/ggml/src/ggml-hexagon/htp/hex-utils.h b/ggml/src/ggml-hexagon/htp/hex-utils.h index 93e87efc..853f1c1b 100644 --- a/ggml/src/ggml-hexagon/htp/hex-utils.h +++ b/ggml/src/ggml-hexagon/htp/hex-utils.h @@ -45,11 +45,15 @@ static inline void hex_l2fetch_block(const void * addr, size_t size) { static inline void hex_l2flush(void * addr, size_t size) { const uint32_t s = ((uint32_t) addr) & ~(HEX_L2_LINE_SIZE - 1); const uint32_t e = (((uint32_t) addr) + size + HEX_L2_LINE_SIZE - 1) & ~(HEX_L2_LINE_SIZE - 1); - for (uint32_t i = s; i < e; i += HEX_L2_BLOCK_SIZE) { - Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 0); - Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 1); - Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 2); - Q6_dccleaninva_A((void *) i + HEX_L2_LINE_SIZE * 3); + const uint32_t eb = s + ((e - s) & ~(HEX_L2_BLOCK_SIZE - 1)); + for (uint32_t i = s; i < eb; i += HEX_L2_BLOCK_SIZE) { + Q6_dccleaninva_A((void *) (i + HEX_L2_LINE_SIZE * 0)); + Q6_dccleaninva_A((void *) (i + HEX_L2_LINE_SIZE * 1)); + Q6_dccleaninva_A((void *) (i + HEX_L2_LINE_SIZE * 2)); + Q6_dccleaninva_A((void *) (i + HEX_L2_LINE_SIZE * 3)); + } + for (uint32_t i = eb; i < e; i += HEX_L2_LINE_SIZE) { + Q6_dccleaninva_A((void *) i); } } diff --git a/ggml/src/ggml-hexagon/htp/hmx-fa-kernels.h b/ggml/src/ggml-hexagon/htp/hmx-fa-kernels.h index d6795bf0..d5fb48ad 100644 --- a/ggml/src/ggml-hexagon/htp/hmx-fa-kernels.h +++ b/ggml/src/ggml-hexagon/htp/hmx-fa-kernels.h @@ -48,7 +48,7 @@ static const int16_t d_tile_scatter_offsets[64] __attribute__((aligned(128))) = }; // Inner HMX tile computation kernels -static void hmx_fa_qk_dot_tile( +static inline void hmx_fa_qk_dot_tile( const __fp16 * row_tiles, const __fp16 * col_tiles, __fp16 * out_tile, @@ -116,7 +116,7 @@ static void hmx_fa_qk_dot_tile( ); } -static void hmx_fa_o_update_tile( +static inline void hmx_fa_o_update_tile( const __fp16 * d_diag, const __fp16 * o_rc, const __fp16 * p_tile_in, @@ -495,12 +495,140 @@ static inline void hmx_fa_q_prep_fp16( } +// Head-dim-padded Q-prep (f32). Used when DK is not a multiple of 64. +static inline void hmx_fa_q_prep_fp32_pad(__fp16 * vtcm_q_tiles, + const uint8_t * temp_q_vtcm, + size_t start, + size_t end, + size_t g_rows_end, + size_t dk_in, + size_t dk_out, + size_t G, + size_t n_rows_q, + const struct fastdiv_values * div_G, + bool q_transposed) { + const uint32_t n_out_tiles = (uint32_t) (dk_out / 32); + for (size_t r = start; r < end; r += 2) { + size_t r0 = r / HMX_FP16_TILE_N_ROWS; + size_t r1 = r % HMX_FP16_TILE_N_ROWS; + __fp16 * out_base = vtcm_q_tiles + r0 * HMX_FP16_TILE_N_ROWS * dk_out; + + if (r >= g_rows_end) { + for (uint32_t d = 0; d < n_out_tiles; ++d) { + ((HVX_Vector *) (out_base + d * HMX_FP16_TILE_N_ELMS))[r1 / 2] = Q6_V_vzero(); + } + continue; + } + + const size_t q_idx0 = fastdiv(r + 0, div_G); + const size_t h_idx0 = fastmodulo(r + 0, G, div_G); + const size_t q_idx1 = fastdiv(r + 1, div_G); + const size_t h_idx1 = fastmodulo(r + 1, G, div_G); + + const size_t offset0 = q_transposed ? (h_idx0 * n_rows_q + q_idx0) : (q_idx0 * G + h_idx0); + const size_t offset1 = q_transposed ? (h_idx1 * n_rows_q + q_idx1) : (q_idx1 * G + h_idx1); + + const HVX_UVector * pv_in0 = (const HVX_UVector *) (temp_q_vtcm + offset0 * dk_in * sizeof(float)); + const HVX_UVector * pv_in1 = (r + 1 < g_rows_end) ? (const HVX_UVector *) (temp_q_vtcm + offset1 * dk_in * sizeof(float)) : NULL; + + for (uint32_t d = 0; d < n_out_tiles; ++d) { + HVX_Vector * out_tile = (HVX_Vector *) (out_base + d * HMX_FP16_TILE_N_ELMS); + const size_t base_lane = (size_t) d * 32; + const size_t real_lanes = (base_lane < dk_in) ? hex_smin(32, dk_in - base_lane) : 0; + + if (real_lanes == 0) { + out_tile[r1 / 2] = Q6_V_vzero(); + continue; + } + + HVX_Vector v0 = pv_in0[d]; + HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); + if (real_lanes < 32) { + // Straddle tile: keep the first real_lanes floats, zero the padded tail so + // the packed f16 lanes beyond DK are zero. + const HVX_VectorPred keep = Q6_Q_vsetq_R((uint32_t) (real_lanes * sizeof(float))); + v0 = Q6_V_vmux_QVV(keep, v0, Q6_V_vzero()); + v1 = Q6_V_vmux_QVV(keep, v1, Q6_V_vzero()); + } + out_tile[r1 / 2] = hvx_vec_f32_to_f16_shuff(v0, v1); + } + } +} + +// Head-dim-padded Q-prep (f16). Used when DK is not a multiple of 64. +static inline void hmx_fa_q_prep_fp16_pad(__fp16 * vtcm_q_tiles, + const uint8_t * temp_q_vtcm, + size_t start, + size_t end, + size_t g_rows_end, + size_t dk_in, + size_t dk_out, + size_t G, + size_t n_rows_q, + const struct fastdiv_values * div_G, + bool q_transposed) { + const uint32_t n_out_pairs = (uint32_t) (dk_out / 64); + for (size_t r = start; r < end; r += 2) { + size_t r0 = r / HMX_FP16_TILE_N_ROWS; + size_t r1 = r % HMX_FP16_TILE_N_ROWS; + __fp16 * out_base = vtcm_q_tiles + r0 * HMX_FP16_TILE_N_ROWS * dk_out; + + if (r >= g_rows_end) { + for (uint32_t d = 0; d < n_out_pairs; ++d) { + __fp16 * out_dtile = out_base + d * HMX_FP16_TILE_N_ELMS * 2; + HVX_Vector * pv_out0 = ((HVX_Vector *) out_dtile) + r1 / 2; + HVX_Vector * pv_out1 = pv_out0 + 16; + *pv_out0 = Q6_V_vzero(); + *pv_out1 = Q6_V_vzero(); + } + continue; + } + + const size_t q_idx0 = fastdiv(r + 0, div_G); + const size_t h_idx0 = fastmodulo(r + 0, G, div_G); + const size_t q_idx1 = fastdiv(r + 1, div_G); + const size_t h_idx1 = fastmodulo(r + 1, G, div_G); + + const size_t offset0 = q_transposed ? (h_idx0 * n_rows_q + q_idx0) : (q_idx0 * G + h_idx0); + const size_t offset1 = q_transposed ? (h_idx1 * n_rows_q + q_idx1) : (q_idx1 * G + h_idx1); + + const HVX_UVector * pv_in0 = (const HVX_UVector *) (temp_q_vtcm + offset0 * dk_in * sizeof(__fp16)); + const HVX_UVector * pv_in1 = (r + 1 < g_rows_end) ? (const HVX_UVector *) (temp_q_vtcm + offset1 * dk_in * sizeof(__fp16)) : NULL; + + for (uint32_t d = 0; d < n_out_pairs; ++d) { + __fp16 * out_dtile = out_base + d * HMX_FP16_TILE_N_ELMS * 2; + HVX_Vector * pv_out0 = ((HVX_Vector *) out_dtile) + r1 / 2; + HVX_Vector * pv_out1 = pv_out0 + 16; + + const size_t base_lane = (size_t) d * 64; + const size_t real_lanes = (base_lane < dk_in) ? hex_smin(64, dk_in - base_lane) : 0; + + if (real_lanes == 0) { + *pv_out0 = Q6_V_vzero(); + *pv_out1 = Q6_V_vzero(); + continue; + } + + HVX_Vector v0 = pv_in0[d]; + HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); + if (real_lanes < 64) { + const HVX_VectorPred keep = Q6_Q_vsetq_R((uint32_t) (real_lanes * sizeof(__fp16))); + v0 = Q6_V_vmux_QVV(keep, v0, Q6_V_vzero()); + v1 = Q6_V_vmux_QVV(keep, v1, Q6_V_vzero()); + } + HVX_VectorPair vp = Q6_W_vshuff_VVR(v1, v0, -2); + *pv_out0 = Q6_V_lo_W(vp); + *pv_out1 = Q6_V_hi_W(vp); + } + } +} + static inline void hmx_fa_q_prep_fallback( __fp16 * vtcm_q_tiles, uintptr_t q_data, size_t q_nb1, size_t q_nb2, size_t q_nb3, uint32_t q_start, uint32_t kv_head, uint32_t ib3, size_t start, size_t end, size_t n_rows_g, - size_t G, size_t DK, bool is_q_fp32, + size_t G, size_t dk_in, size_t dk_out, bool is_q_fp32, const struct fastdiv_values * div_G ) { for (size_t r = start; r < end; r += 2) { @@ -518,33 +646,55 @@ static inline void hmx_fa_q_prep_fallback( size_t r0 = r / HMX_FP16_TILE_N_ROWS; size_t r1 = r % HMX_FP16_TILE_N_ROWS; - __fp16 * out_base = vtcm_q_tiles + r0 * HMX_FP16_TILE_N_ROWS * DK; + __fp16 * out_base = vtcm_q_tiles + r0 * HMX_FP16_TILE_N_ROWS * dk_out; if (is_q_fp32) { const HVX_UVector * pv_in0 = q_ptr0 ? (const HVX_UVector *) q_ptr0 : NULL; const HVX_UVector * pv_in1 = q_ptr1 ? (const HVX_UVector *) q_ptr1 : NULL; - for (uint32_t d = 0; d < DK / 32; ++d) { - HVX_Vector v0 = pv_in0 ? pv_in0[d] : Q6_V_vzero(); - HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); - HVX_Vector v_hf = hvx_vec_f32_to_f16_shuff(v0, v1); - - HVX_Vector * out_tile = (HVX_Vector *) (out_base + d * HMX_FP16_TILE_N_ELMS); - out_tile[r1 / 2] = v_hf; + for (uint32_t d = 0; d < dk_out / 32; ++d) { + HVX_Vector * out_tile = (HVX_Vector *) (out_base + d * HMX_FP16_TILE_N_ELMS); + const size_t base_lane = (size_t) d * 32; + const size_t real_lanes = (base_lane < dk_in) ? hex_smin(32, dk_in - base_lane) : 0; + + if (real_lanes == 0) { + out_tile[r1 / 2] = Q6_V_vzero(); + continue; + } + HVX_Vector v0 = pv_in0 ? pv_in0[d] : Q6_V_vzero(); + HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); + if (real_lanes < 32) { + const HVX_VectorPred keep = Q6_Q_vsetq_R((uint32_t) (real_lanes * sizeof(float))); + v0 = Q6_V_vmux_QVV(keep, v0, Q6_V_vzero()); + v1 = Q6_V_vmux_QVV(keep, v1, Q6_V_vzero()); + } + out_tile[r1 / 2] = hvx_vec_f32_to_f16_shuff(v0, v1); } } else { const HVX_UVector * pv_in0 = q_ptr0 ? (const HVX_UVector *) q_ptr0 : NULL; const HVX_UVector * pv_in1 = q_ptr1 ? (const HVX_UVector *) q_ptr1 : NULL; - for (uint32_t d = 0; d < DK / 64; ++d) { - HVX_Vector v0 = pv_in0 ? pv_in0[d] : Q6_V_vzero(); - HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); - HVX_VectorPair vp = Q6_W_vshuff_VVR(v1, v0, -2); - + for (uint32_t d = 0; d < dk_out / 64; ++d) { __fp16 * out_dtile = out_base + d * HMX_FP16_TILE_N_ELMS * 2; HVX_Vector * pv_out0 = ((HVX_Vector *) out_dtile) + r1 / 2; HVX_Vector * pv_out1 = pv_out0 + 16; + const size_t base_lane = (size_t) d * 64; + const size_t real_lanes = (base_lane < dk_in) ? hex_smin(64, dk_in - base_lane) : 0; + + if (real_lanes == 0) { + *pv_out0 = Q6_V_vzero(); + *pv_out1 = Q6_V_vzero(); + continue; + } + HVX_Vector v0 = pv_in0 ? pv_in0[d] : Q6_V_vzero(); + HVX_Vector v1 = pv_in1 ? pv_in1[d] : Q6_V_vzero(); + if (real_lanes < 64) { + const HVX_VectorPred keep = Q6_Q_vsetq_R((uint32_t) (real_lanes * sizeof(__fp16))); + v0 = Q6_V_vmux_QVV(keep, v0, Q6_V_vzero()); + v1 = Q6_V_vmux_QVV(keep, v1, Q6_V_vzero()); + } + HVX_VectorPair vp = Q6_W_vshuff_VVR(v1, v0, -2); *pv_out0 = Q6_V_lo_W(vp); *pv_out1 = Q6_V_hi_W(vp); } diff --git a/ggml/src/ggml-hexagon/htp/hmx-mm-kernels-tiled.h b/ggml/src/ggml-hexagon/htp/hmx-mm-kernels-tiled.h index 0011abba..d6d40586 100644 --- a/ggml/src/ggml-hexagon/htp/hmx-mm-kernels-tiled.h +++ b/ggml/src/ggml-hexagon/htp/hmx-mm-kernels-tiled.h @@ -506,6 +506,41 @@ static void dequantize_tiled_weight_to_fp16_task_q8_0( } } +// Q6_K stores 6-bit weights and one fp16 scale per 16 k, see HTP_MM_WEIGHT_TILE_SIZE_Q6_K. +// A k-group holds 4 k per row, the HMX tile holds 2, so each group is dealt into two tiles. +static void dequantize_tiled_weight_to_fp16_task_q6_k( + const tiled_dequantize_state_t *state, + uint32_t start_tile, uint32_t end_tile) { + + const HVX_Vector mask_0f = Q6_Vb_vsplat_R(0x0F); + const HVX_Vector mask_03 = Q6_Vb_vsplat_R(0x03); + const HVX_Vector i32 = Q6_Vb_vsplat_R(32); + + for (uint32_t t = start_tile; t < end_tile; t++) { + const HVX_Vector * vptr = (const HVX_Vector *) (state->src + t * state->aligned_tile_size); + __fp16 * dst_ptr = state->dst + t * HTP_MM_HMX_TILE_N_ELMS; + + HVX_Vector v_sc = vptr[6]; + HVX_Vector v_sc_k16 = Q6_V_vror_VR(v_sc, 64); + HVX_Vector v_scale_k0 = Q6_V_lo_W(Q6_W_vshuff_VVR(v_sc, v_sc, -2)); + HVX_Vector v_scale_k16 = Q6_V_lo_W(Q6_W_vshuff_VVR(v_sc_k16, v_sc_k16, -2)); + + #pragma unroll + for (int g = 0; g < 8; g++) { + const HVX_Vector v_scale = (g < 4) ? v_scale_k0 : v_scale_k16; + + HVX_Vector v_q = unpack_q6_k_group(vptr, g, mask_0f, mask_03, i32); + HVX_VectorPair vp16 = Q6_Wh_vunpack_Vb(v_q); + HVX_VectorPair vp_k = Q6_W_vdeal_VVR(Q6_V_hi_W(vp16), Q6_V_lo_W(vp16), -4); + + hvx_vmem(dst_ptr + (2 * g + 0) * 64) = + Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(Q6_Vhf_equals_Vh(Q6_V_lo_W(vp_k)), v_scale)); + hvx_vmem(dst_ptr + (2 * g + 1) * 64) = + Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(Q6_Vhf_equals_Vh(Q6_V_hi_W(vp_k)), v_scale)); + } + } +} + static __attribute__((noinline)) void convert_f16_weight_to_fp16_tiles_task( const tiled_dequantize_state_t *state, @@ -803,15 +838,12 @@ static void transfer_output_chunk_fp16_to_fp32_col_chunk( HVX_Vector v = ((const HVX_Vector *) tile)[r1]; HVX_VectorPair vp = Q6_Wqf32_vmpy_VhfVhf(v, one); - HVX_Vector *pv_out0 = (HVX_Vector *) (output_row_base + c + 0); - HVX_Vector *pv_out1 = (HVX_Vector *) (output_row_base + c + dst_stride); - HVX_Vector v_out0 = Q6_Vsf_equals_Vqf32(Q6_V_lo_W(vp)); if (src2_row_base) { HVX_Vector v_src2_0 = hvx_vmemu(src2_row_base + c + 0); v_out0 = hvx_vec_add_f32_f32(v_out0, v_src2_0); } - *pv_out0 = v_out0; + hvx_vmemu(output_row_base + c + 0) = v_out0; if (r + 1 < n_rows) { HVX_Vector v_out1 = Q6_Vsf_equals_Vqf32(Q6_V_hi_W(vp)); @@ -819,7 +851,7 @@ static void transfer_output_chunk_fp16_to_fp32_col_chunk( HVX_Vector v_src2_1 = hvx_vmemu(src2_row_base + c + src2_stride); v_out1 = hvx_vec_add_f32_f32(v_out1, v_src2_1); } - *pv_out1 = v_out1; + hvx_vmemu(output_row_base + c + dst_stride) = v_out1; } } @@ -1366,12 +1398,9 @@ static void transfer_output_chunk_fp16_to_fp32_scattered( HVX_Vector v = ((const HVX_Vector *) tile)[r1]; HVX_VectorPair vp = Q6_Wqf32_vmpy_VhfVhf(v, one); - HVX_Vector *pv_out0 = (HVX_Vector *) (output_row0 + c); - HVX_Vector *pv_out1 = output_row1 ? (HVX_Vector *) (output_row1 + c) : NULL; - - *pv_out0 = Q6_Vsf_equals_Vqf32(Q6_V_lo_W(vp)); - if (pv_out1) { - *pv_out1 = Q6_Vsf_equals_Vqf32(Q6_V_hi_W(vp)); + hvx_vmemu(output_row0 + c) = Q6_Vsf_equals_Vqf32(Q6_V_lo_W(vp)); + if (output_row1) { + hvx_vmemu(output_row1 + c) = Q6_Vsf_equals_Vqf32(Q6_V_hi_W(vp)); } } } diff --git a/ggml/src/ggml-hexagon/htp/hmx-utils.h b/ggml/src/ggml-hexagon/htp/hmx-utils.h index 2a61ca73..ad295cb7 100644 --- a/ggml/src/ggml-hexagon/htp/hmx-utils.h +++ b/ggml/src/ggml-hexagon/htp/hmx-utils.h @@ -27,7 +27,7 @@ static inline void hmx_init_column_scales(void *out_scales, HVX_Vector v_scale) // vscatter offsets for fused dequant+transpose: write K-values directly to [K][N] tile. // word[i] = i*128 maps K-row-pair i to byte offset i*128. // Column offset (n*4) is added at runtime. Entries 0..15 cover one tile (region 2047); -// entries 16..31 cover the next adjacent tile (region 4095) — pick region size at the +// entries 16..31 cover the next adjacent tile (region 4095) - pick region size at the // call site to scatter into one tile (masked) or two contiguous tiles (unmasked). static const int32_t hmx_transpose_scatter_offsets[32] __attribute__((aligned(VLEN))) = { 0 * 128, 1 * 128, 2 * 128, 3 * 128, 4 * 128, 5 * 128, 6 * 128, 7 * 128, 8 * 128, 9 * 128, 10 * 128, @@ -198,16 +198,16 @@ static inline void hmx_interleave_cols_to_tiles(__fp16 * restrict tiles_out, } // --- HMX inline asm macros for load-store packetization --- -#define HMX_LOAD_MPY_F16(act, wt, range) \ - "{\n" \ +#define HMX_LOAD_MPY_F16(act, wt, range) \ + "{\n" \ " activation.hf = mxmem(" act ", " range ")\n" \ - " weight.hf = mxmem(" wt ", " range ")\n" \ + " weight.hf = mxmem(" wt ", " range ")\n" \ "}\n" -#define HMX_LOAD_MPY_DEEP_F16(act, wt, range) \ - "{\n" \ +#define HMX_LOAD_MPY_DEEP_F16(act, wt, range) \ + "{\n" \ " activation.hf = mxmem(" act ", " range "):deep\n" \ - " weight.hf = mxmem(" wt ", " range ")\n" \ + " weight.hf = mxmem(" wt ", " range ")\n" \ "}\n" #define HMX_STORE_AFTER_F16(out, scale_reg) \ diff --git a/ggml/src/ggml-hexagon/htp/htp-ctx.h b/ggml/src/ggml-hexagon/htp/htp-ctx.h index e0f9a0c4..814feef7 100644 --- a/ggml/src/ggml-hexagon/htp/htp-ctx.h +++ b/ggml/src/ggml-hexagon/htp/htp-ctx.h @@ -1,7 +1,7 @@ #ifndef HTP_CTX_H #define HTP_CTX_H -#include "hex-dma.h" +#include "dma-queue.h" #include "hmx-queue.h" #include "htp-ops.h" #include "hex-profile.h" @@ -17,16 +17,21 @@ #ifndef HTP_MAX_NTHREADS #define HTP_MAX_NTHREADS 10 #endif -#define HTP_MAX_MMAPS 16 -#define HTP_MAX_DIRTY_RANGES 16 +#define HTP_MAX_MMAPS 64 +#define HTP_MAX_DIRTY_RANGES 64 // Memory mapping struct htp_mmap { uint64_t size; uint64_t base; uint32_t fd; - uint32_t reserved; + uint32_t flags; +}; + +struct htp_dirty_range { + uint32_t start; + uint32_t end; }; // Scratchpad state @@ -38,6 +43,14 @@ struct htp_spad { uint32_t size_per_thread; // size per thread }; +struct htp_mdev_group { + uint16_t idx; + uint16_t count; + struct fastdiv_values count_div; + uint8_t * fence_base; + uint32_t fence_seq; +}; + struct htp_context; // Context while processing an Op @@ -55,9 +68,6 @@ struct htp_ops_context { const struct htp_tensor * dsts[HTP_OP_MAX_OUTPUTS]; }; - dma_queue ** src_dma[HTP_OP_MAX_INPUTS]; - dma_queue ** dst_dma[HTP_OP_MAX_OUTPUTS]; - // TODO convert these to an array struct htp_spad src0_spad; struct htp_spad src1_spad; @@ -65,8 +75,10 @@ struct htp_ops_context { struct htp_spad src3_spad; struct htp_spad dst_spad; - uint32_t n_threads; - uint32_t flags; + uint32_t flags; + uint32_t n_threads; + struct fastdiv_values n_threads_div; + int status; }; // Main context for htp DSP backend @@ -75,7 +87,7 @@ struct htp_context { struct htp_mmap mmap[HTP_MAX_MMAPS]; dma_queue_t dma[HTP_MAX_NTHREADS]; - dma_queue_t dma_cached[HTP_MAX_NTHREADS]; + struct htp_thread_trace trace[HTP_MAX_NTHREADS + 1]; work_queue_t work_queue; hmx_queue_t hmx_queue; @@ -88,7 +100,6 @@ struct htp_context { bool hmx_enabled; bool etm; uint32_t profiler; - struct htp_thread_trace trace[HTP_MAX_NTHREADS + 1]; uint8_t * vtcm_base; size_t vtcm_size; @@ -97,16 +108,13 @@ struct htp_context { atomic_bool vtcm_needs_release; uint64_t max_vmem; - struct htp_dirty_range { - uint32_t start; - uint32_t end; - uint32_t bi; - } dirty_ranges[HTP_MAX_DIRTY_RANGES]; + struct htp_dirty_range dirty_ranges[HTP_MAX_DIRTY_RANGES]; // Persistent DDR scratchpad for MUL_MAT_ID mappings void * ddr_spad_base; size_t ddr_spad_size; + struct htp_mdev_group mdev; struct htp_ops_context octx; qurt_thread_t main_thread; @@ -115,10 +123,31 @@ struct htp_context { size_t footprint; }; +static inline bool htp_ops_context_set_n_threads(struct htp_ops_context * octx, uint32_t n_threads) { + if (n_threads == 0 || n_threads > octx->ctx->n_threads) { + return false; + } + + if (n_threads != octx->n_threads) { + octx->n_threads = n_threads; + octx->n_threads_div = n_threads == octx->ctx->n_threads + ? octx->ctx->n_threads_div + : init_fastdiv_values(n_threads); + } + + return true; +} + +static inline void htp_ops_context_set_status(struct htp_ops_context * octx, int status) { + if (status > HTP_STATUS_OK && octx->status == HTP_STATUS_OK) { + octx->status = status; + } +} + int op_matmul(struct htp_ops_context * octx); int op_matmul_id(struct htp_ops_context * octx); -int op_matmul_qkv(struct htp_ops_context * octx); -int op_matmul_ffn(struct htp_ops_context * octx); +int op_matmul_nx(struct htp_ops_context * octx); +int op_matmul_id_nx(struct htp_ops_context * octx); int op_binary(struct htp_ops_context * octx); int op_unary(struct htp_ops_context * octx); int op_sum_rows(struct htp_ops_context * octx); @@ -132,6 +161,7 @@ int op_get_rows(struct htp_ops_context * octx); int op_cpy(struct htp_ops_context * octx); int op_repeat(struct htp_ops_context * octx); int op_argsort(struct htp_ops_context * octx); +int op_top_k(struct htp_ops_context * octx); int op_ssm_conv(struct htp_ops_context * octx); int op_cumsum(struct htp_ops_context * octx); int op_fill(struct htp_ops_context * octx); @@ -141,5 +171,7 @@ int op_solve_tri(struct htp_ops_context * octx); int op_gated_delta_net(struct htp_ops_context * octx); int op_pad(struct htp_ops_context * octx); int op_im2col(struct htp_ops_context * octx); +int op_allreduce(struct htp_ops_context * octx); +int op_roll(struct htp_ops_context * octx); #endif /* HTP_CTX_H */ diff --git a/ggml/src/ggml-hexagon/htp/htp-fence.h b/ggml/src/ggml-hexagon/htp/htp-fence.h new file mode 100644 index 00000000..7450b5de --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/htp-fence.h @@ -0,0 +1,89 @@ +#ifndef HTP_FENCE_H +#define HTP_FENCE_H + +#include +#include + +#include + +#include "hex-utils.h" +#include "htp-ops.h" +#include "htp-ctx.h" + +static inline atomic_uint * htp_mdev_fence_slot(const void * fence_base, uint32_t idx) { + return (atomic_uint *) ((const uint8_t *) fence_base + (size_t) idx * HTP_FENCE_SLOT_SIZE); +} + +static inline void htp_fence_write(void * fence_ptr, uint32_t seq, uint32_t status) { + atomic_uint * fence = (atomic_uint *) fence_ptr; + atomic_store(&fence[1], status); + atomic_store(&fence[0], seq); + asm volatile ("syncht" : : : "memory"); + Q6_dccleaninva_A((void *) fence); +} + +static inline void htp_fence_read(const void * fence_ptr, uint32_t * seq, uint32_t * status) { + const atomic_uint * fence = (const atomic_uint *) fence_ptr; + Q6_dccleaninva_A((void *) fence); + asm volatile ("syncht" : : : "memory"); + *seq = atomic_load(&fence[0]); + *status = atomic_load(&fence[1]); +} + +static inline void htp_mdev_group_barrier(struct htp_ops_context * octx) { + struct htp_context * ctx = octx->ctx; + if (ctx->mdev.count <= 1) { + return; + } + + const uint32_t seq = ++ctx->mdev.fence_seq; + + struct htp_thread_trace * tr = &ctx->trace[0]; + htp_trace_event_start(tr, HTP_TRACE_EVT_FENCE, (uint16_t) seq); + + const uint32_t mdev_idx = ctx->mdev.idx; + const uint32_t mdev_count = ctx->mdev.count; + + uint8_t * fence_base = ctx->mdev.fence_base; + atomic_uint * my_fence = htp_mdev_fence_slot(fence_base, mdev_idx); + htp_fence_write(my_fence, seq, octx->status); + + for (uint32_t d = 0; d < mdev_count; d++) { + if (d == mdev_idx) continue; + atomic_uint * peer_fence = htp_mdev_fence_slot(fence_base, d); + uint64_t spins = 0; + while (1) { + uint32_t peer_seq; + uint32_t peer_status; + htp_fence_read(peer_fence, &peer_seq, &peer_status); + if ((int32_t)(peer_seq - seq) >= 0) { + if (peer_status > HTP_STATUS_OK) { + FARF(ERROR, "ggml-hex: mdev %u peer %u failed with status %u : seq 0x%08x\n", + mdev_idx, d, peer_status, seq); + htp_ops_context_set_status(octx, peer_status); + } + break; + } + if (++spins == 10000) { + FARF(ALWAYS, "ggml-hex: mdev %u waiting for mdev %u : seq 0x%08x (b %u op %u) my-fence %p peer-fence %p peer-seq 0x%08x (diff %d)\n", + mdev_idx, d, seq, seq >> 12, seq & 0xfff, my_fence, peer_fence, peer_seq, (int32_t)(peer_seq - seq)); + } + if (spins > HTP_FENCE_TIMEOUT) { + FARF(ERROR, "ggml-hex: mdev %u timeout waiting for mdev %u : seq 0x%08x (b %u op %u) peer-fence %p peer-seq 0x%08x\n", + mdev_idx, d, seq, seq >> 12, seq & 0xfff, peer_fence, peer_seq); + htp_ops_context_set_status(octx, HTP_STATUS_INTERNAL_ERR); + break; + } + hex_pause(); + } + } + asm volatile ("syncht" : : : "memory"); + + if (octx->status > HTP_STATUS_OK) { + htp_fence_write(my_fence, seq, octx->status); + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_FENCE, (uint16_t) seq); +} + +#endif // HTP_FENCE_H diff --git a/ggml/src/ggml-hexagon/htp/htp-ops.h b/ggml/src/ggml-hexagon/htp/htp-ops.h index a138f062..ee5b9244 100644 --- a/ggml/src/ggml-hexagon/htp/htp-ops.h +++ b/ggml/src/ggml-hexagon/htp/htp-ops.h @@ -22,6 +22,8 @@ enum htp_data_type { HTP_TYPE_Q4_0 = 2, HTP_TYPE_Q4_1 = 3, HTP_TYPE_Q8_0 = 8, + HTP_TYPE_Q4_K = 12, + HTP_TYPE_Q6_K = 14, HTP_TYPE_IQ4_NL = 20, HTP_TYPE_I32 = 26, HTP_TYPE_I64 = 27, @@ -43,13 +45,6 @@ enum htp_data_type { -// Mask to enable various stages of the Ops. -// Used for debugging and profiling. -enum htp_op_stage { - HTP_OPSTAGE_QUEUE = (1 << 0), // Enable Queueing (ie calls into NPU) - HTP_OPSTAGE_COMPUTE = (1 << 1), // Enable Compute -}; - // Do not reorder first 4 (used as an index) enum htp_op_code { HTP_OP_MUL = 0, @@ -58,8 +53,8 @@ enum htp_op_code { HTP_OP_DIV = 3, HTP_OP_MUL_MAT, HTP_OP_MUL_MAT_ID, - HTP_OP_MUL_MAT_QKV, - HTP_OP_MUL_MAT_FFN, + HTP_OP_MUL_MAT_NX, + HTP_OP_MUL_MAT_ID_NX, HTP_OP_MUL_MAT_ADD, HTP_OP_RMS_NORM, HTP_OP_RMS_NORM_MUL, @@ -70,9 +65,13 @@ enum htp_op_code { HTP_OP_UNARY_NEG, HTP_OP_UNARY_SOFTPLUS, HTP_OP_UNARY_TANH, + HTP_OP_UNARY_ABS, + HTP_OP_UNARY_LOG, + HTP_OP_UNARY_RELU, HTP_OP_GLU_SWIGLU, HTP_OP_GLU_SWIGLU_OAI, HTP_OP_GLU_GEGLU, + HTP_OP_GLU_GEGLU_QUICK, HTP_OP_SOFTMAX, HTP_OP_ADD_ID, HTP_OP_ROPE, @@ -81,7 +80,9 @@ enum htp_op_code { HTP_OP_GET_ROWS, HTP_OP_SCALE, HTP_OP_CPY, + HTP_OP_CPY_FENCE, HTP_OP_ARGSORT, + HTP_OP_TOP_K, HTP_OP_SQR, HTP_OP_SQRT, HTP_OP_SUM_ROWS, @@ -98,13 +99,20 @@ enum htp_op_code { HTP_OP_NORM, HTP_OP_CONCAT, HTP_OP_CLAMP, + HTP_OP_LEAKY_RELU, HTP_OP_IM2COL, + HTP_OP_FENCE, + HTP_OP_ALLREDUCE, + HTP_OP_ALLREDUCE_ADD, + HTP_OP_GLU_SWIGLU_CLAMP, + HTP_OP_MDEV_GROUP, + HTP_OP_ROLL, HTP_OP_INVALID }; #define HTP_OP_MAX_DIMS 4 // aka GGML_MAX_DIMS -#define HTP_OP_MAX_INPUTS 6 // aka GGML_MAX_SRCS +#define HTP_OP_MAX_INPUTS 10 // aka GGML_MAX_SRCS #define HTP_OP_MAX_OUTPUTS 4 #define HTP_OP_MAX_PARAMS 16 // aka GGML_MAX_OP_PARAMS #define HTP_OP_MAX_KERN_PARAMS 32 @@ -112,19 +120,26 @@ enum htp_op_code { #define HTP_OP_MAX_BUFS 16 #define HTP_OP_MAX_TENSORS 8192 // must stay under 64K (uint16) +#define HTP_FENCE_TIMEOUT (1000000000ULL) +#define HTP_FENCE_SLOT_SIZE 128 + #define HTP_OP_MAX_VMEM_DEFAULT (3355443200u) #define HTP_MMAP_MAX_VMEM (2147483648u) enum htp_tensor_flags { - HTP_TENSOR_COMPUTE = (1U << 0), // Tensor buffer temporal compute data (not weights) - HTP_TENSOR_DIRTY = (1U << 1) // Tensor buffer is dirty and needs to be flushed + HTP_TENSOR_WEIGHT = (1U << 0), // Tensor buffer model weight data (not compute) + HTP_TENSOR_REPACK = (1U << 1), // Tensor is in repacked tiled format + HTP_TENSOR_FENCE = (1U << 2) // Tensor is synchronization fence (explicitly managed) +}; + +enum htp_buf_flags { + HTP_BUF_EXTENDED = (1U << 0), }; // Tensor descriptor struct htp_tensor { - uint32_t data; // Buffer offset in the messages, and data pointer on the NPU - uint32_t reserved; // Reserved for alignment padding (must be multiple of 8) + uint64_t data; // Buffer offset in the messages, and data pointer on the NPU uint32_t size; // Data size in bytes uint32_t flags; // Buffer / tensor flags uint32_t type; // Data type @@ -138,12 +153,12 @@ struct htp_tensor { struct htp_buf_desc { uint64_t base; // base address uint64_t size; // total size - uint32_t flags; // buffer flags (unused) + uint32_t flags; // HTP_BUF_* uint32_t fd; // file descriptor }; enum htp_op_flags { - HTP_OPFLAGS_SKIP_COMPUTE = (1U << 0), // Skip actual computation (used for profiling) + HTP_OPFLAGS_STUB = (1U << 0), }; // Op descriptor @@ -175,6 +190,7 @@ enum htp_trace_event_id { HTP_TRACE_EVT_L2FLUSH = 1, HTP_TRACE_EVT_INIT = 2, HTP_TRACE_EVT_BUFF = 3, + HTP_TRACE_EVT_FENCE = 4, HTP_TRACE_EVT_HVX_COMP = 20, HTP_TRACE_EVT_HVX_A_QUANT = 21, @@ -188,6 +204,14 @@ enum htp_trace_event_id { HTP_TRACE_EVT_HVX_FA_K_PREP = 29, HTP_TRACE_EVT_HVX_FA_V_PREP = 30, + HTP_TRACE_EVT_HVX_GDN_PREP = 31, + HTP_TRACE_EVT_HVX_GDN_SOLVE = 32, + HTP_TRACE_EVT_HVX_GDN_V_PREP = 33, + HTP_TRACE_EVT_HVX_GDN_D_PREP = 34, + HTP_TRACE_EVT_HVX_GDN_OUT = 35, + HTP_TRACE_EVT_HVX_GDN_STATE = 36, + HTP_TRACE_EVT_HVX_GDN_REM = 37, + HTP_TRACE_EVT_HMX_COMP = 40, }; @@ -209,28 +233,26 @@ struct htp_prof_desc { }; struct htp_opbatch_req { - uint32_t id; // Batch id + uint64_t seq; // Sequence number uint32_t n_bufs; // Number of buffers uint32_t n_tensors; // Number of tensors uint32_t n_ops; // Number of ops uint32_t n_traces; // Number of trace descriptors per thread - uint32_t pad; // unused // struct htp_buf_desc bufs[]; -- dspqueue buf 0 // struct htp_tensor tensors[]; -- dspqueue buf 0 // struct htp_op_desc ops[]; -- dspqueue buf 0 }; struct htp_opbatch_rsp { - uint32_t id; // Batch id - uint32_t status; // HTP_STATUS_... - uint32_t n_bufs; // Number of buffers - uint32_t n_tensors; // Number of tensors - uint32_t n_ops; // Number of op profile descriptors - uint32_t n_traces[HTP_MAX_NTHREADS + 1]; - uint32_t usecs; // Number of usec - uint32_t pad; // align to 8 bytes + uint64_t seq; // Sequence number uint64_t cycles_start; // Start cycle counter uint64_t cycles_stop; // Stop cycle counter + uint32_t status; // HTP_STATUS_... + uint32_t n_bufs; // Number of buffers + uint32_t n_tensors; // Number of tensors + uint32_t n_ops; // Number of op profile descriptors + uint32_t usecs; // Number of usec + uint32_t n_traces[HTP_MAX_NTHREADS + 1]; // struct htp_prof_desc profs[]; -- dspqueue buf 0 }; diff --git a/ggml/src/ggml-hexagon/htp/htp-tensor.c b/ggml/src/ggml-hexagon/htp/htp-tensor.c index 39436e26..03b0070d 100644 --- a/ggml/src/ggml-hexagon/htp/htp-tensor.c +++ b/ggml/src/ggml-hexagon/htp/htp-tensor.c @@ -20,7 +20,7 @@ struct l2flush_range { struct l2flush_multi_task { struct htp_thread_trace * trace; - struct l2flush_range ranges[HTP_OP_MAX_INPUTS]; + struct l2flush_range ranges[HTP_MAX_DIRTY_RANGES]; uint32_t n_ranges; uint32_t total_blocks; uint32_t blocks_per_thread; @@ -73,13 +73,36 @@ static void l2flush_multi_worker(unsigned int n, unsigned int i, void * data) { htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, gb_first); } +static void merge_dirty_ranges(struct htp_context * ctx) { + for (uint32_t i = 0; i < HTP_MAX_DIRTY_RANGES; i++) { + struct htp_dirty_range * r = &ctx->dirty_ranges[i]; + if (!r->start) continue; + + for (uint32_t j = 0; j < HTP_MAX_DIRTY_RANGES;) { + struct htp_dirty_range * s = &ctx->dirty_ranges[j]; + if (i == j || !s->start || r->end < s->start || s->end < r->start) { + j++; + continue; + } + + r->start = MIN(r->start, s->start); + r->end = MAX(r->end, s->end); + s->start = 0; + s->end = 0; + j = 0; + } + } +} + void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n) { const struct htp_tensor * pending[HTP_OP_MAX_OUTPUTS]; uint32_t n_pending = 0; for (uint32_t i = 0; i < n; i++) { const struct htp_tensor * t = tensors[i]; - if (!t) continue; + if (!t || (t->flags & (HTP_TENSOR_WEIGHT | HTP_TENSOR_FENCE))) { + continue; + } uint32_t t_start = t->data; uint32_t t_end = t_start + t->size; @@ -103,6 +126,8 @@ void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * co } } + merge_dirty_ranges(ctx); + if (n_pending == 0) { return; } @@ -125,8 +150,8 @@ void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * co struct htp_dirty_range * r = &ctx->dirty_ranges[idx]; r->start = pending[i]->data; r->end = pending[i]->data + pending[i]->size; - r->bi = pending[i]->bi; } + merge_dirty_ranges(ctx); return; } @@ -144,12 +169,12 @@ void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * co struct htp_dirty_range * r = &ctx->dirty_ranges[i]; r->start = pending[i]->data; r->end = pending[i]->data + pending[i]->size; - r->bi = pending[i]->bi; } + merge_dirty_ranges(ctx); return; } - if (total_evict_size > HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1 && n_evict <= HTP_OP_MAX_INPUTS) { + if (total_evict_size > HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1 && n_evict <= HTP_MAX_DIRTY_RANGES) { struct l2flush_multi_task task; task.trace = ctx->trace; task.n_ranges = n_evict; @@ -188,7 +213,6 @@ void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * co struct htp_dirty_range * r = &ctx->dirty_ranges[idx]; r->start = pending[i]->data; r->end = pending[i]->data + pending[i]->size; - r->bi = pending[i]->bi; } for (uint32_t i = 0; i < n_empty; i++) { @@ -196,11 +220,16 @@ void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * co struct htp_dirty_range * r = &ctx->dirty_ranges[idx]; r->start = pending[n_evict + i]->data; r->end = pending[n_evict + i]->data + pending[n_evict + i]->size; - r->bi = pending[n_evict + i]->bi; } + + merge_dirty_ranges(ctx); } static void make_tensor_clean(struct htp_context * ctx, const struct htp_tensor * t) { + if (!t || (t->flags & (HTP_TENSOR_WEIGHT | HTP_TENSOR_FENCE))) { + return; + } + uint32_t t_start = t->data; uint32_t t_end = t_start + t->size; @@ -211,6 +240,7 @@ static void make_tensor_clean(struct htp_context * ctx, const struct htp_tensor if (r->start < t_end && t_start < r->end) { if (t_start <= r->start && r->end <= t_end) { r->start = 0; + r->end = 0; } else if (t_start <= r->start) { r->start = t_end; } else if (r->end <= t_end) { @@ -221,6 +251,10 @@ static void make_tensor_clean(struct htp_context * ctx, const struct htp_tensor } static inline bool is_tensor_dirty(struct htp_context * ctx, const struct htp_tensor * t) { + if (!t || (t->flags & (HTP_TENSOR_WEIGHT | HTP_TENSOR_FENCE))) { + return false; + } + uint32_t t_start = t->data; uint32_t t_end = t_start + t->size; @@ -235,17 +269,50 @@ static inline bool is_tensor_dirty(struct htp_context * ctx, const struct htp_te return false; } -void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n) { - const struct htp_tensor * dirty_tensors[HTP_OP_MAX_INPUTS]; - uint32_t n_dirty = 0; +static void flush_dirty_ranges(struct htp_context * ctx, const struct htp_dirty_range * ranges, uint32_t n_ranges, uint64_t total_dirty) { + if (total_dirty >= HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1) { + struct l2flush_multi_task task; + task.trace = ctx->trace; + task.n_ranges = n_ranges; + + uint32_t block_acc = 0; + for (uint32_t i = 0; i < n_ranges; i++) { + const struct htp_dirty_range * r = &ranges[i]; + struct l2flush_range * rg = &task.ranges[i]; + rg->start = hex_align_down((size_t) r->start, HEX_L2_LINE_SIZE); + rg->end = hex_align_up((size_t) r->end, HEX_L2_LINE_SIZE); + rg->block_first = block_acc; + rg->n_blocks = (rg->end - rg->start + HEX_L2_BLOCK_SIZE - 1) / HEX_L2_BLOCK_SIZE; + block_acc += rg->n_blocks; + } + + task.total_blocks = block_acc; + task.blocks_per_thread = fastdiv(block_acc + ctx->n_threads - 1, &ctx->n_threads_div); + + work_queue_run(ctx->work_queue, l2flush_multi_worker, &task, ctx->n_threads); + } else { + struct htp_thread_trace * tr = &ctx->trace[0]; + htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, 0); + for (uint32_t i = 0; i < n_ranges; i++) { + const struct htp_dirty_range * r = &ranges[i]; + hex_l2flush((void *) (uintptr_t) r->start, r->end - r->start); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, 0); + } +} + +void htp_flush_dirty_ranges(struct htp_context * ctx) { + struct htp_dirty_range ranges[HTP_MAX_DIRTY_RANGES]; + uint32_t n_ranges = 0; uint64_t total_dirty = 0; - for (uint32_t i = 0; i < n; i++) { - const struct htp_tensor * t = tensors[i]; - if (t && (t->flags & HTP_TENSOR_COMPUTE) && is_tensor_dirty(ctx, t)) { - dirty_tensors[n_dirty++] = t; - total_dirty += t->size; + for (uint32_t i = 0; i < HTP_MAX_DIRTY_RANGES; i++) { + const struct htp_dirty_range * r = &ctx->dirty_ranges[i]; + if (!r->start) { + continue; } + ranges[n_ranges++] = *r; + total_dirty += r->end - r->start; } if (total_dirty == 0) { @@ -257,37 +324,37 @@ void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * co return; } - if (total_dirty >= HEX_L2_FLUSH_WQ_THRESHOLD && ctx->n_threads > 1) { - struct l2flush_multi_task task; - task.trace = ctx->trace; - task.n_ranges = 0; + flush_dirty_ranges(ctx, ranges, n_ranges, total_dirty); + memset(ctx->dirty_ranges, 0, sizeof(ctx->dirty_ranges)); +} - uint32_t block_acc = 0; - for (uint32_t i = 0; i < n_dirty; i++) { - const struct htp_tensor * t = dirty_tensors[i]; - make_tensor_clean(ctx, t); +void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n) { + const struct htp_tensor * dirty_tensors[HTP_OP_MAX_INPUTS]; + struct htp_dirty_range ranges[HTP_OP_MAX_INPUTS]; + uint32_t n_dirty = 0; + uint64_t total_dirty = 0; - struct l2flush_range * rg = &task.ranges[task.n_ranges++]; - rg->start = hex_align_down((size_t) t->data, HEX_L2_LINE_SIZE); - rg->end = hex_align_up((size_t) t->data + t->size, HEX_L2_LINE_SIZE); - rg->block_first = block_acc; - rg->n_blocks = (rg->end - rg->start + HEX_L2_BLOCK_SIZE - 1) / HEX_L2_BLOCK_SIZE; - block_acc += rg->n_blocks; + for (uint32_t i = 0; i < n; i++) { + const struct htp_tensor * t = tensors[i]; + if (is_tensor_dirty(ctx, t)) { + dirty_tensors[n_dirty++] = t; + ranges[n_dirty - 1].start = t->data; + ranges[n_dirty - 1].end = t->data + t->size; + total_dirty += t->size; } + } - task.total_blocks = block_acc; - task.blocks_per_thread = fastdiv(block_acc + ctx->n_threads - 1, &ctx->n_threads_div); + if (total_dirty == 0) { + return; + } - work_queue_run(ctx->work_queue, l2flush_multi_worker, &task, ctx->n_threads); + if (total_dirty > HEX_L2_FLUSH_ALL_THRESHOLD) { + flush_all_dcache(ctx); return; } - struct htp_thread_trace * tr = &ctx->trace[0]; + flush_dirty_ranges(ctx, ranges, n_dirty, total_dirty); for (uint32_t i = 0; i < n_dirty; i++) { - const struct htp_tensor * t = dirty_tensors[i]; - htp_trace_event_start(tr, HTP_TRACE_EVT_L2FLUSH, t->ti); - hex_l2flush((void *) (uintptr_t) t->data, t->size); - htp_trace_event_stop(tr, HTP_TRACE_EVT_L2FLUSH, t->ti); - make_tensor_clean(ctx, t); + make_tensor_clean(ctx, dirty_tensors[i]); } } diff --git a/ggml/src/ggml-hexagon/htp/htp-tensor.h b/ggml/src/ggml-hexagon/htp/htp-tensor.h index 2c3fc54c..f7a96683 100644 --- a/ggml/src/ggml-hexagon/htp/htp-tensor.h +++ b/ggml/src/ggml-hexagon/htp/htp-tensor.h @@ -2,18 +2,141 @@ #define HTP_TENSOR_H #include +#include #include "htp-ops.h" #include "hex-bitmap.h" +#include "hex-common.h" +#include "hex-fastdiv.h" + +enum { + HTP_TENSOR_MDEV_LINE_SIZE = 128, +}; + +struct htp_tensor_mdev_range { + uint32_t start; + uint32_t count; +}; static inline void * htp_tensor_data(const struct htp_tensor * t) { return (void *) (uintptr_t) t->data; } +static inline bool htp_tensor_is_extended(const struct htp_tensor * t) { + return t && (t->data >> 32) != 0; +} + static inline uint32_t * htp_tensor_flags(const struct htp_tensor * t) { return (uint32_t *) &t->flags; } +static inline bool htp_tensor_is_contiguous(const struct htp_tensor * t, uint32_t type_size) { + uint32_t next_nb = type_size; + if (t->ne[0] != 1 && t->nb[0] != next_nb) { + return false; + } + next_nb *= t->ne[0]; + for (int i = 1; i < HTP_OP_MAX_DIMS; i++) { + if (t->ne[i] != 1 && t->nb[i] != next_nb) { + return false; + } + next_nb *= t->ne[i]; + } + return true; +} + +static inline bool htp_tensor_is_permuted(const struct htp_tensor * t) { + return t->nb[0] > t->nb[1] || t->nb[1] > t->nb[2] || t->nb[2] > t->nb[3]; +} + +static inline bool htp_tensor_mdev_data_aligned(const struct htp_tensor * t) { + return ((uintptr_t) t->data & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0; +} + +static inline bool htp_tensor_can_row_partition(const struct htp_tensor * t, uint32_t elem_size) { + if (!htp_tensor_mdev_data_aligned(t)) { + return false; + } + if (t->ne[0] != 1 && t->nb[0] != elem_size) { + return false; + } + if (htp_tensor_is_permuted(t)) { + return false; + } + if (t->ne[1] > 1 && (t->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) != 0) return false; + if (t->ne[2] > 1 && (t->nb[2] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) != 0) return false; + if (t->ne[3] > 1 && (t->nb[3] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) != 0) return false; + return true; +} + +static inline bool htp_tensor_mdev_rows_per_chunk(const struct htp_tensor * t, uint32_t elem_size, uint32_t row_size, uint32_t * rows_per_chunk) { + *rows_per_chunk = 0; + + if (!htp_tensor_mdev_data_aligned(t)) { + return false; + } + if (t->ne[0] != 1 && t->nb[0] != elem_size) { + return false; + } + if (htp_tensor_is_permuted(t)) { + return false; + } + if (t->ne[1] > 1 && (t->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0 && + (t->ne[2] <= 1 || (t->nb[2] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0) && + (t->ne[3] <= 1 || (t->nb[3] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0)) { + *rows_per_chunk = 1; + return true; + } + if (t->nb[1] == row_size && + (t->ne[2] <= 1 || t->nb[2] == t->nb[1] * t->ne[1]) && + (t->ne[3] <= 1 || t->nb[3] == t->nb[2] * t->ne[2])) { + *rows_per_chunk = (row_size > 0) ? (HTP_TENSOR_MDEV_LINE_SIZE / hex_gcd_u32(row_size, HTP_TENSOR_MDEV_LINE_SIZE)) : 1; + return true; + } + return false; +} + +static inline struct htp_tensor_mdev_range htp_tensor_mdev_partition(uint32_t total_units, uint32_t units_per_chunk, uint32_t mdev_idx, uint32_t mdev_count, const struct fastdiv_values * mdev_count_div) { + struct htp_tensor_mdev_range range = { 0, total_units }; + + if (mdev_count <= 1) { + return range; + } + + if (units_per_chunk == 0) { + range.start = (mdev_idx == 0) ? 0 : total_units; + range.count = (mdev_idx == 0) ? total_units : 0; + return range; + } + + const uint32_t total_chunks = total_units / units_per_chunk; + if (total_chunks < mdev_count) { + range.start = (mdev_idx == 0) ? 0 : total_units; + range.count = (mdev_idx == 0) ? total_units : 0; + return range; + } + + const uint32_t chunks_per_mdev = fastdiv(total_chunks + mdev_count - 1, mdev_count_div); + range.start = MIN(mdev_idx * chunks_per_mdev * units_per_chunk, total_units); + if (mdev_idx == mdev_count - 1) { + range.count = total_units - range.start; + } else { + range.count = MIN(chunks_per_mdev * units_per_chunk, total_units - range.start); + } + return range; +} + +static inline uint32_t htp_tensor_get_row_size(int type, uint32_t ne00) { + switch (type) { + case HTP_TYPE_F32: return ne00 * 4; + case HTP_TYPE_F16: return ne00 * 2; + case HTP_TYPE_Q8_0: return (ne00 / 32) * 34; + case HTP_TYPE_I32: return ne00 * 4; + default: return 0; + } +} + struct htp_context; +void htp_flush_dirty_ranges(struct htp_context * ctx); void htp_tensor_flush_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n); void htp_tensor_dirty_all(struct htp_context * ctx, const struct htp_tensor * const * tensors, uint32_t n); diff --git a/ggml/src/ggml-hexagon/htp/htp_iface.idl b/ggml/src/ggml-hexagon/htp/htp_iface.idl index 47693d8b..b46e2529 100644 --- a/ggml/src/ggml-hexagon/htp/htp_iface.idl +++ b/ggml/src/ggml-hexagon/htp/htp_iface.idl @@ -13,7 +13,7 @@ struct htp_iface_pmu_conf { interface htp_iface : remote_handle64 { AEEResult start(in uint32 sess_id, in uint64 dsp_queue_id, in uint32 n_hvx, in uint32 n_hmx, in uint64 max_vmem); AEEResult stop(); - AEEResult mmap(in uint32 fd, in uint32 size); + AEEResult mmap(in uint32 fd, in uint64 size); AEEResult munmap(in uint32 fd); AEEResult profiler(in uint32 mode, in htp_iface_pmu_conf pmu); AEEResult etm(in uint32 enable); diff --git a/ggml/src/ggml-hexagon/htp/hvx-arith.h b/ggml/src/ggml-hexagon/htp/hvx-arith.h index 82e34169..6cbead74 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-arith.h +++ b/ggml/src/ggml-hexagon/htp/hvx-arith.h @@ -16,25 +16,25 @@ #define UNUSED(x) (void)(x) #define hvx_arith_loop_body(dst_type, src0_type, src1_type, elem_size, vec_store, vec_op) \ - do { \ - dst_type * restrict vdst = (dst_type *) dst; \ - src0_type * restrict vsrc0 = (src0_type *) src0; \ - src1_type * restrict vsrc1 = (src1_type *) src1; \ - \ - const uint32_t epv = 128 / (elem_size); \ - const uint32_t nvec = n / epv; \ - const uint32_t nloe = n % epv; \ - \ - uint32_t i = 0; \ - \ - _Pragma("unroll(4)") \ - for (; i < nvec; i++) { \ - vdst[i] = vec_op(vsrc0[i], vsrc1[i]); \ - } \ - if (nloe) { \ - HVX_Vector v = vec_op(vsrc0[i], vsrc1[i]); \ - vec_store((void *) &vdst[i], nloe * (elem_size), v); \ - } \ + do { \ + dst_type * vdst = (dst_type *) dst; \ + src0_type * vsrc0 = (src0_type *) src0; \ + src1_type * vsrc1 = (src1_type *) src1; \ + \ + const uint32_t epv = 128 / (elem_size); \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + vdst[i] = vec_op(vsrc0[i], vsrc1[i]); \ + } \ + if (nloe) { \ + HVX_Vector v = vec_op(vsrc0[i], vsrc1[i]); \ + vec_store((void *) &vdst[i], nloe * (elem_size), v); \ + } \ } while(0) #if __HVX_ARCH__ < 79 @@ -56,43 +56,43 @@ #define HVX_OP_MUL_F16(a, b) hvx_vec_mul_f16_f16(a, b) // Generic macro to define alignment permutations for an op -#define DEFINE_HVX_BINARY_OP_VARIANTS(OP_NAME, OP_MACRO, ELEM_TYPE) \ -static inline void OP_NAME##_aaa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src0 % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - hvx_arith_loop_body(HVX_Vector, HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ -static inline void OP_NAME##_aau(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src0 % 128 == 0); \ - hvx_arith_loop_body(HVX_Vector, HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ -static inline void OP_NAME##_aua(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - hvx_arith_loop_body(HVX_Vector, HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ -static inline void OP_NAME##_auu(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - hvx_arith_loop_body(HVX_Vector, HVX_UVector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ -static inline void OP_NAME##_uaa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src0 % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - hvx_arith_loop_body(HVX_UVector, HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ -static inline void OP_NAME##_uau(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src0 % 128 == 0); \ - hvx_arith_loop_body(HVX_UVector, HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ -static inline void OP_NAME##_uua(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src1 % 128 == 0); \ - hvx_arith_loop_body(HVX_UVector, HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ -static inline void OP_NAME##_uuu(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ +#define DEFINE_HVX_BINARY_OP_VARIANTS(OP_NAME, OP_MACRO, ELEM_TYPE) \ +static inline void OP_NAME##_aaa(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src0 % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + hvx_arith_loop_body(HVX_Vector, HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ +static inline void OP_NAME##_aau(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src0 % 128 == 0); \ + hvx_arith_loop_body(HVX_Vector, HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ +static inline void OP_NAME##_aua(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + hvx_arith_loop_body(HVX_Vector, HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ +static inline void OP_NAME##_auu(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) dst % 128 == 0); \ + hvx_arith_loop_body(HVX_Vector, HVX_UVector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ +static inline void OP_NAME##_uaa(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) src0 % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + hvx_arith_loop_body(HVX_UVector, HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ +} \ +static inline void OP_NAME##_uau(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) src0 % 128 == 0); \ + hvx_arith_loop_body(HVX_UVector, HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ +} \ +static inline void OP_NAME##_uua(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ + assert((uintptr_t) src1 % 128 == 0); \ + hvx_arith_loop_body(HVX_UVector, HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ +} \ +static inline void OP_NAME##_uuu(uint8_t * dst, const uint8_t * src0, const uint8_t * src1, uint32_t n) { \ hvx_arith_loop_body(HVX_UVector, HVX_UVector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ +} \ DEFINE_HVX_BINARY_OP_VARIANTS(hvx_add_f32, HVX_OP_ADD_F32, float) DEFINE_HVX_BINARY_OP_VARIANTS(hvx_sub_f32, HVX_OP_SUB_F32, float) @@ -103,25 +103,25 @@ DEFINE_HVX_BINARY_OP_VARIANTS(hvx_sub_f16, HVX_OP_SUB_F16, _Float16) DEFINE_HVX_BINARY_OP_VARIANTS(hvx_mul_f16, HVX_OP_MUL_F16, _Float16) // Dispatcher logic -#define HVX_BINARY_DISPATCHER(OP_NAME) \ +#define HVX_BINARY_DISPATCHER(OP_NAME) \ static inline void OP_NAME(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, const uint32_t num_elems) { \ - if (hex_is_aligned((void *) dst, 128)) { \ - if (hex_is_aligned((void *) src0, 128)) { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aaa(dst, src0, src1, num_elems); \ - else OP_NAME##_aau(dst, src0, src1, num_elems); \ - } else { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aua(dst, src0, src1, num_elems); \ - else OP_NAME##_auu(dst, src0, src1, num_elems); \ - } \ - } else { \ - if (hex_is_aligned((void *) src0, 128)) { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uaa(dst, src0, src1, num_elems); \ - else OP_NAME##_uau(dst, src0, src1, num_elems); \ - } else { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uua(dst, src0, src1, num_elems); \ - else OP_NAME##_uuu(dst, src0, src1, num_elems); \ - } \ - } \ + if (hex_is_aligned((void *) dst, 128)) { \ + if (hex_is_aligned((void *) src0, 128)) { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aaa(dst, src0, src1, num_elems); \ + else OP_NAME##_aau(dst, src0, src1, num_elems); \ + } else { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aua(dst, src0, src1, num_elems); \ + else OP_NAME##_auu(dst, src0, src1, num_elems); \ + } \ + } else { \ + if (hex_is_aligned((void *) src0, 128)) { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uaa(dst, src0, src1, num_elems); \ + else OP_NAME##_uau(dst, src0, src1, num_elems); \ + } else { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uua(dst, src0, src1, num_elems); \ + else OP_NAME##_uuu(dst, src0, src1, num_elems); \ + } \ + } \ } HVX_BINARY_DISPATCHER(hvx_add_f32) @@ -166,44 +166,44 @@ static inline void hvx_mul_mul_f32_aa(uint8_t * restrict dst, const uint8_t * re // Scalar Operations -#define hvx_scalar_loop_body(dst_type, src_type, elem_size, vec_store, scalar_op_macro) \ - do { \ - dst_type * restrict vdst = (dst_type *) dst; \ - src_type * restrict vsrc = (src_type *) src; \ - \ - const uint32_t epv = 128 / (elem_size); \ - const uint32_t nvec = n / epv; \ - const uint32_t nloe = n % epv; \ - \ - uint32_t i = 0; \ - \ - _Pragma("unroll(4)") \ - for (; i < nvec; i++) { \ - HVX_Vector v = vsrc[i]; \ - vdst[i] = scalar_op_macro(v); \ - } \ - if (nloe) { \ - HVX_Vector v = vsrc[i]; \ - v = scalar_op_macro(v); \ - vec_store((void *) &vdst[i], nloe * (elem_size), v); \ - } \ +#define hvx_scalar_loop_body(dst_type, src_type, elem_size, vec_store, scalar_op_macro) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const uint32_t epv = 128 / (elem_size); \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + HVX_Vector v = vsrc[i]; \ + vdst[i] = scalar_op_macro(v); \ + } \ + if (nloe) { \ + HVX_Vector v = vsrc[i]; \ + v = scalar_op_macro(v); \ + vec_store((void *) &vdst[i], nloe * (elem_size), v); \ + } \ } while(0) -#define HVX_OP_ADD_SCALAR_F32(v) \ - ({ \ +#define HVX_OP_ADD_SCALAR_F32(v) \ + ({ \ const HVX_VectorPred pred_inf = Q6_Q_vcmp_eq_VwVw(inf, v); \ - HVX_Vector out = HVX_OP_ADD_F32(v, val_vec); \ - Q6_V_vmux_QVV(pred_inf, inf, out); \ + HVX_Vector out = HVX_OP_ADD_F32(v, val_vec); \ + Q6_V_vmux_QVV(pred_inf, inf, out); \ }) #define HVX_OP_MUL_SCALAR_F32(v) HVX_OP_MUL_F32(v, val_vec) #define HVX_OP_SUB_SCALAR_F32(v) HVX_OP_SUB_F32(v, val_vec) -#define HVX_OP_ADD_SCALAR_F16(v) \ - ({ \ +#define HVX_OP_ADD_SCALAR_F16(v) \ + ({ \ const HVX_VectorPred pred_inf = Q6_Q_vcmp_eq_VhVh(inf, v); \ - HVX_Vector out = HVX_OP_ADD_F16(v, val_vec); \ - Q6_V_vmux_QVV(pred_inf, inf, out); \ + HVX_Vector out = HVX_OP_ADD_F16(v, val_vec); \ + Q6_V_vmux_QVV(pred_inf, inf, out); \ }) #define HVX_OP_MUL_SCALAR_F16(v) HVX_OP_MUL_F16(v, val_vec) @@ -212,31 +212,31 @@ static inline void hvx_mul_mul_f32_aa(uint8_t * restrict dst, const uint8_t * re // Scalar Variants // Generic macro to define alignment permutations for an op -#define DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(OP_NAME, OP_MACRO, SPLAT_MACRO, ELEM_TYPE) \ +#define DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(OP_NAME, OP_MACRO, SPLAT_MACRO, ELEM_TYPE) \ static inline void OP_NAME##_aa(uint8_t * restrict dst, const uint8_t * restrict src, const ELEM_TYPE val, uint32_t n) { \ - const HVX_Vector val_vec = SPLAT_MACRO(val); \ - const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src % 128 == 0); \ - hvx_scalar_loop_body(HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ + const HVX_Vector val_vec = SPLAT_MACRO(val); \ + const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src % 128 == 0); \ + hvx_scalar_loop_body(HVX_Vector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ static inline void OP_NAME##_au(uint8_t * restrict dst, const uint8_t * restrict src, const ELEM_TYPE val, uint32_t n) { \ - const HVX_Vector val_vec = SPLAT_MACRO(val); \ - const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ - assert((uintptr_t) dst % 128 == 0); \ - hvx_scalar_loop_body(HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ -} \ + const HVX_Vector val_vec = SPLAT_MACRO(val); \ + const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ + assert((uintptr_t) dst % 128 == 0); \ + hvx_scalar_loop_body(HVX_Vector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_a, OP_MACRO); \ +} \ static inline void OP_NAME##_ua(uint8_t * restrict dst, const uint8_t * restrict src, const ELEM_TYPE val, uint32_t n) { \ - const HVX_Vector val_vec = SPLAT_MACRO(val); \ - const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ - assert((uintptr_t) src % 128 == 0); \ - hvx_scalar_loop_body(HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ + const HVX_Vector val_vec = SPLAT_MACRO(val); \ + const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ + assert((uintptr_t) src % 128 == 0); \ + hvx_scalar_loop_body(HVX_UVector, HVX_Vector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ +} \ static inline void OP_NAME##_uu(uint8_t * restrict dst, const uint8_t * restrict src, const ELEM_TYPE val, uint32_t n) { \ - const HVX_Vector val_vec = SPLAT_MACRO(val); \ - const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ - hvx_scalar_loop_body(HVX_UVector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ -} \ + const HVX_Vector val_vec = SPLAT_MACRO(val); \ + const HVX_Vector inf = SPLAT_MACRO((ELEM_TYPE)INFINITY); UNUSED(inf); \ + hvx_scalar_loop_body(HVX_UVector, HVX_UVector, sizeof(ELEM_TYPE), hvx_vec_store_u, OP_MACRO); \ +} \ DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(hvx_add_scalar_f32, HVX_OP_ADD_SCALAR_F32, hvx_vec_splat_f32, float) DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(hvx_sub_scalar_f32, HVX_OP_SUB_SCALAR_F32, hvx_vec_splat_f32, float) @@ -247,17 +247,17 @@ DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(hvx_sub_scalar_f16, HVX_OP_SUB_SCALAR_F16, DEFINE_HVX_BINARY_SCALAR_OP_VARIANTS(hvx_mul_scalar_f16, HVX_OP_MUL_SCALAR_F16, hvx_vec_splat_f16, _Float16) // Dispatcher logic -#define HVX_BINARY_SCALAR_DISPATCHER(OP_NAME, ELEM_TYPE) \ +#define HVX_BINARY_SCALAR_DISPATCHER(OP_NAME, ELEM_TYPE) \ static inline void OP_NAME(uint8_t * restrict dst, const uint8_t * restrict src, const ELEM_TYPE val, const uint32_t num_elems) { \ - if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { \ - OP_NAME##_aa(dst, src, val, num_elems); \ - } else if (hex_is_aligned((void *) dst, 128)) { \ - OP_NAME##_au(dst, src, val, num_elems); \ - } else if (hex_is_aligned((void *) src, 128)) { \ - OP_NAME##_ua(dst, src, val, num_elems); \ - } else { \ - OP_NAME##_uu(dst, src, val, num_elems); \ - } \ + if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { \ + OP_NAME##_aa(dst, src, val, num_elems); \ + } else if (hex_is_aligned((void *) dst, 128)) { \ + OP_NAME##_au(dst, src, val, num_elems); \ + } else if (hex_is_aligned((void *) src, 128)) { \ + OP_NAME##_ua(dst, src, val, num_elems); \ + } else { \ + OP_NAME##_uu(dst, src, val, num_elems); \ + } \ } HVX_BINARY_SCALAR_DISPATCHER(hvx_add_scalar_f32, float) @@ -308,14 +308,54 @@ static inline void hvx_min_scalar_f32(uint8_t * restrict dst, const uint8_t * re } } +// MAX Scalar variants + +#define HVX_OP_MAX_SCALAR(v) Q6_Vsf_vmax_VsfVsf(val_vec, v) + +static inline void hvx_max_scalar_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, const float val, uint32_t n) { + const HVX_Vector val_vec = hvx_vec_splat_f32(val); + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_Vector, sizeof(float), hvx_vec_store_a, HVX_OP_MAX_SCALAR); +} + +static inline void hvx_max_scalar_f32_au(uint8_t * restrict dst, const uint8_t * restrict src, const float val, uint32_t n) { + const HVX_Vector val_vec = hvx_vec_splat_f32(val); + assert((unsigned long) dst % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_UVector, sizeof(float), hvx_vec_store_a, HVX_OP_MAX_SCALAR); +} + +static inline void hvx_max_scalar_f32_ua(uint8_t * restrict dst, const uint8_t * restrict src, const float val, uint32_t n) { + const HVX_Vector val_vec = hvx_vec_splat_f32(val); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_UVector, HVX_Vector, sizeof(float), hvx_vec_store_u, HVX_OP_MAX_SCALAR); +} + +static inline void hvx_max_scalar_f32_uu(uint8_t * restrict dst, const uint8_t * restrict src, const float val, uint32_t n) { + const HVX_Vector val_vec = hvx_vec_splat_f32(val); + hvx_scalar_loop_body(HVX_UVector, HVX_UVector, sizeof(float), hvx_vec_store_u, HVX_OP_MAX_SCALAR); +} + +static inline void hvx_max_scalar_f32(uint8_t * restrict dst, const uint8_t * restrict src, const float val, const int num_elems) { + if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { + hvx_max_scalar_f32_aa(dst, src, val, num_elems); + } else if (hex_is_aligned((void *) dst, 128)) { + hvx_max_scalar_f32_au(dst, src, val, num_elems); + } else if (hex_is_aligned((void *) src, 128)) { + hvx_max_scalar_f32_ua(dst, src, val, num_elems); + } else { + hvx_max_scalar_f32_uu(dst, src, val, num_elems); + } +} + // CLAMP Scalar variants -#define HVX_OP_CLAMP_SCALAR(v) \ - ({ \ +#define HVX_OP_CLAMP_SCALAR(v) \ + ({ \ HVX_VectorPred pred_cap_right = Q6_Q_vcmp_gt_VsfVsf(v, max_vec); \ HVX_VectorPred pred_cap_left = Q6_Q_vcmp_gt_VsfVsf(min_vec, v); \ - HVX_Vector tmp = Q6_V_vmux_QVV(pred_cap_right, max_vec, v); \ - Q6_V_vmux_QVV(pred_cap_left, min_vec, tmp); \ + HVX_Vector tmp = Q6_V_vmux_QVV(pred_cap_right, max_vec, v); \ + Q6_V_vmux_QVV(pred_cap_left, min_vec, tmp); \ }) static inline void hvx_clamp_scalar_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, const float min, const float max, uint32_t n) { @@ -358,11 +398,192 @@ static inline void hvx_clamp_scalar_f32(uint8_t * restrict dst, const uint8_t * } } +#define HVX_OP_CLAMP_SCALAR_F16(v) \ + ({ \ + HVX_VectorPred pred_cap_right = Q6_Q_vcmp_gt_VhfVhf(v, max_vec); \ + HVX_VectorPred pred_cap_left = Q6_Q_vcmp_gt_VhfVhf(min_vec, v); \ + HVX_Vector tmp = Q6_V_vmux_QVV(pred_cap_right, max_vec, v); \ + Q6_V_vmux_QVV(pred_cap_left, min_vec, tmp); \ + }) + +static inline void hvx_clamp_scalar_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, const _Float16 min, const _Float16 max, uint32_t n) { + const HVX_Vector min_vec = hvx_vec_splat_f16(min); + const HVX_Vector max_vec = hvx_vec_splat_f16(max); + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_Vector, sizeof(_Float16), hvx_vec_store_a, HVX_OP_CLAMP_SCALAR_F16); +} + +static inline void hvx_clamp_scalar_f16_au(uint8_t * restrict dst, const uint8_t * restrict src, const _Float16 min, const _Float16 max, uint32_t n) { + const HVX_Vector min_vec = hvx_vec_splat_f16(min); + const HVX_Vector max_vec = hvx_vec_splat_f16(max); + assert((unsigned long) dst % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_UVector, sizeof(_Float16), hvx_vec_store_a, HVX_OP_CLAMP_SCALAR_F16); +} + +static inline void hvx_clamp_scalar_f16_ua(uint8_t * restrict dst, const uint8_t * restrict src, const _Float16 min, const _Float16 max, uint32_t n) { + const HVX_Vector min_vec = hvx_vec_splat_f16(min); + const HVX_Vector max_vec = hvx_vec_splat_f16(max); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_UVector, HVX_Vector, sizeof(_Float16), hvx_vec_store_u, HVX_OP_CLAMP_SCALAR_F16); +} + +static inline void hvx_clamp_scalar_f16_uu(uint8_t * restrict dst, const uint8_t * restrict src, const _Float16 min, const _Float16 max, uint32_t n) { + const HVX_Vector min_vec = hvx_vec_splat_f16(min); + const HVX_Vector max_vec = hvx_vec_splat_f16(max); + hvx_scalar_loop_body(HVX_UVector, HVX_UVector, sizeof(_Float16), hvx_vec_store_u, HVX_OP_CLAMP_SCALAR_F16); +} + +static inline void hvx_clamp_scalar_f16(uint8_t * restrict dst, const uint8_t * restrict src, const _Float16 min, const _Float16 max, const int num_elems) { + if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { + hvx_clamp_scalar_f16_aa(dst, src, min, max, num_elems); + } else if (hex_is_aligned((void *) dst, 128)) { + hvx_clamp_scalar_f16_au(dst, src, min, max, num_elems); + } else if (hex_is_aligned((void *) src, 128)) { + hvx_clamp_scalar_f16_ua(dst, src, min, max, num_elems); + } else { + hvx_clamp_scalar_f16_uu(dst, src, min, max, num_elems); + } +} + +#define HVX_OP_LEAKY_RELU_SCALAR(v) \ + ({ \ + HVX_VectorPred pred_neg = Q6_Q_vcmp_gt_VsfVsf(zero_vec, v); \ + HVX_Vector scaled = HVX_OP_MUL_F32(v, ns_vec); \ + Q6_V_vmux_QVV(pred_neg, scaled, v); \ + }) + +static inline void hvx_leaky_relu_scalar_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, const float ns, uint32_t n) { + const HVX_Vector zero_vec = hvx_vec_splat_f32(0.0f); + const HVX_Vector ns_vec = hvx_vec_splat_f32(ns); + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_Vector, sizeof(float), hvx_vec_store_a, HVX_OP_LEAKY_RELU_SCALAR); +} + +static inline void hvx_leaky_relu_scalar_f32_au(uint8_t * restrict dst, const uint8_t * restrict src, const float ns, uint32_t n) { + const HVX_Vector zero_vec = hvx_vec_splat_f32(0.0f); + const HVX_Vector ns_vec = hvx_vec_splat_f32(ns); + assert((unsigned long) dst % 128 == 0); + hvx_scalar_loop_body(HVX_Vector, HVX_UVector, sizeof(float), hvx_vec_store_a, HVX_OP_LEAKY_RELU_SCALAR); +} + +static inline void hvx_leaky_relu_scalar_f32_ua(uint8_t * restrict dst, const uint8_t * restrict src, const float ns, uint32_t n) { + const HVX_Vector zero_vec = hvx_vec_splat_f32(0.0f); + const HVX_Vector ns_vec = hvx_vec_splat_f32(ns); + assert((unsigned long) src % 128 == 0); + hvx_scalar_loop_body(HVX_UVector, HVX_Vector, sizeof(float), hvx_vec_store_u, HVX_OP_LEAKY_RELU_SCALAR); +} + +static inline void hvx_leaky_relu_scalar_f32_uu(uint8_t * restrict dst, const uint8_t * restrict src, const float ns, uint32_t n) { + const HVX_Vector zero_vec = hvx_vec_splat_f32(0.0f); + const HVX_Vector ns_vec = hvx_vec_splat_f32(ns); + hvx_scalar_loop_body(HVX_UVector, HVX_UVector, sizeof(float), hvx_vec_store_u, HVX_OP_LEAKY_RELU_SCALAR); +} + +static inline void hvx_leaky_relu_scalar_f32(uint8_t * restrict dst, const uint8_t * restrict src, const float ns, const int num_elems) { + if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { + hvx_leaky_relu_scalar_f32_aa(dst, src, ns, num_elems); + } else if (hex_is_aligned((void *) dst, 128)) { + hvx_leaky_relu_scalar_f32_au(dst, src, ns, num_elems); + } else if (hex_is_aligned((void *) src, 128)) { + hvx_leaky_relu_scalar_f32_ua(dst, src, ns, num_elems); + } else { + hvx_leaky_relu_scalar_f32_uu(dst, src, ns, num_elems); + } +} + +// +// Abs +// + +static inline void hvx_abs_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + + HVX_Vector * restrict vdst = (HVX_Vector *) dst; + HVX_Vector * restrict vsrc = (HVX_Vector *) src; + + const uint32_t elem_size = sizeof(float); + const uint32_t epv = 128 / elem_size; + const uint32_t nvec = n / epv; + const uint32_t nloe = n % epv; + + uint32_t i = 0; + + _Pragma("unroll(4)") + for (; i < nvec; i++) { + vdst[i] = hvx_vec_abs_f32(vsrc[i]); + } + if (nloe) { + HVX_Vector v = hvx_vec_abs_f32(vsrc[i]); + hvx_vec_store_a((void *) &vdst[i], nloe * elem_size, v); + } +} + +#define hvx_abs_f16_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const uint32_t elem_size = sizeof(_Float16); \ + const uint32_t epv = 128 / elem_size; \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + vdst[i] = hvx_vec_abs_f16(vsrc[i]); \ + } \ + if (nloe) { \ + HVX_Vector v = hvx_vec_abs_f16(vsrc[i]); \ + vec_store((void *) &vdst[i], nloe * elem_size, v); \ + } \ + } while(0) + +static inline void hvx_abs_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_abs_f16_loop_body(HVX_Vector, HVX_Vector, hvx_vec_store_a); +} + +static inline void hvx_abs_f16_au(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + hvx_abs_f16_loop_body(HVX_Vector, HVX_UVector, hvx_vec_store_a); +} + +static inline void hvx_abs_f16_ua(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) src % 128 == 0); + hvx_abs_f16_loop_body(HVX_UVector, HVX_Vector, hvx_vec_store_u); +} + +static inline void hvx_abs_f16_uu(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + hvx_abs_f16_loop_body(HVX_UVector, HVX_UVector, hvx_vec_store_u); +} + +static inline void hvx_abs_f16(uint8_t * restrict dst, const uint8_t * restrict src, const uint32_t num_elems) { + if (hex_is_aligned((void *) dst, 128)) { + if (hex_is_aligned((void *) src, 128)) { + hvx_abs_f16_aa(dst, src, num_elems); + } else { + hvx_abs_f16_au(dst, src, num_elems); + } + } else { + if (hex_is_aligned((void *) src, 128)) { + hvx_abs_f16_ua(dst, src, num_elems); + } else { + hvx_abs_f16_uu(dst, src, num_elems); + } + } +} + // // Square // -#define hvx_sqr_f32_loop_body(dst_type, src_type, vec_store) \ +#define hvx_sqr_f32_loop_body(dst_type, src_type, vec_store) \ do { \ dst_type * restrict vdst = (dst_type *) dst; \ src_type * restrict vsrc = (src_type *) src; \ @@ -376,10 +597,10 @@ static inline void hvx_clamp_scalar_f32(uint8_t * restrict dst, const uint8_t * \ _Pragma("unroll(4)") \ for (; i < nvec; i++) { \ - vdst[i] = HVX_OP_MUL_F32(vsrc[i], vsrc[i]); \ + vdst[i] = HVX_OP_MUL_F32(vsrc[i], vsrc[i]); \ } \ if (nloe) { \ - HVX_Vector v = HVX_OP_MUL_F32(vsrc[i], vsrc[i]); \ + HVX_Vector v = HVX_OP_MUL_F32(vsrc[i], vsrc[i]); \ vec_store((void *) &vdst[i], nloe * elem_size, v); \ } \ } while(0) @@ -420,6 +641,64 @@ static inline void hvx_sqr_f32(uint8_t * restrict dst, const uint8_t * restrict } } +#define hvx_sqr_f16_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const uint32_t elem_size = sizeof(_Float16); \ + const uint32_t epv = 128 / elem_size; \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + vdst[i] = HVX_OP_MUL_F16(vsrc[i], vsrc[i]); \ + } \ + if (nloe) { \ + HVX_Vector v = HVX_OP_MUL_F16(vsrc[i], vsrc[i]); \ + vec_store((void *) &vdst[i], nloe * elem_size, v); \ + } \ + } while(0) + +static inline void hvx_sqr_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_sqr_f16_loop_body(HVX_Vector, HVX_Vector, hvx_vec_store_a); +} + +static inline void hvx_sqr_f16_au(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + hvx_sqr_f16_loop_body(HVX_Vector, HVX_UVector, hvx_vec_store_a); +} + +static inline void hvx_sqr_f16_ua(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) src % 128 == 0); + hvx_sqr_f16_loop_body(HVX_UVector, HVX_Vector, hvx_vec_store_u); +} + +static inline void hvx_sqr_f16_uu(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + hvx_sqr_f16_loop_body(HVX_UVector, HVX_UVector, hvx_vec_store_u); +} + +static inline void hvx_sqr_f16(uint8_t * restrict dst, const uint8_t * restrict src, const uint32_t num_elems) { + if (hex_is_aligned((void *) dst, 128)) { + if (hex_is_aligned((void *) src, 128)) { + hvx_sqr_f16_aa(dst, src, num_elems); + } else { + hvx_sqr_f16_au(dst, src, num_elems); + } + } else { + if (hex_is_aligned((void *) src, 128)) { + hvx_sqr_f16_ua(dst, src, num_elems); + } else { + hvx_sqr_f16_uu(dst, src, num_elems); + } + } +} + #undef HVX_OP_ADD_F32 #undef HVX_OP_SUB_F32 #undef HVX_OP_MUL_F32 @@ -435,7 +714,10 @@ static inline void hvx_sqr_f32(uint8_t * restrict dst, const uint8_t * restrict #undef HVX_OP_MUL_SCALAR_F16 #undef hvx_scalar_loop_body #undef HVX_OP_MIN_SCALAR +#undef HVX_OP_MAX_SCALAR #undef HVX_OP_CLAMP_SCALAR +#undef HVX_OP_CLAMP_SCALAR_F16 +#undef HVX_OP_LEAKY_RELU_SCALAR #undef DEFINE_HVX_BINARY_OP_VARIANTS #undef HVX_BINARY_DISPATCHER #undef UNUSED diff --git a/ggml/src/ggml-hexagon/htp/hvx-div.h b/ggml/src/ggml-hexagon/htp/hvx-div.h index 53ee304e..bb7ab051 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-div.h +++ b/ggml/src/ggml-hexagon/htp/hvx-div.h @@ -219,64 +219,64 @@ static inline HVX_Vector hvx_vec_hybrid_div_f16(HVX_Vector vec1, HVX_Vector vec2 } while(0) // Generic macro to define alignment permutations for an op -#define DEFINE_HVX_DIV_OP_VARIANTS(OP_NAME, OP_LOOP_BODY) \ +#define DEFINE_HVX_DIV_OP_VARIANTS(OP_NAME, OP_LOOP_BODY) \ static inline void OP_NAME##_aaa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src0 % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_Vector, HVX_Vector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src0 % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_Vector, HVX_Vector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_aau(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src0 % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_Vector, HVX_UVector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src0 % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_Vector, HVX_UVector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_aua(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_UVector, HVX_Vector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_UVector, HVX_Vector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_auu(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_UVector, HVX_UVector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_UVector, HVX_UVector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_uaa(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src0 % 128 == 0); \ - assert((uintptr_t) src1 % 128 == 0); \ - OP_LOOP_BODY(HVX_UVector, HVX_Vector, HVX_Vector, hvx_vec_store_u); \ -} \ + assert((uintptr_t) src0 % 128 == 0); \ + assert((uintptr_t) src1 % 128 == 0); \ + OP_LOOP_BODY(HVX_UVector, HVX_Vector, HVX_Vector, hvx_vec_store_u); \ +} \ static inline void OP_NAME##_uau(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src0 % 128 == 0); \ - OP_LOOP_BODY(HVX_UVector, HVX_Vector, HVX_UVector, hvx_vec_store_u); \ -} \ + assert((uintptr_t) src0 % 128 == 0); \ + OP_LOOP_BODY(HVX_UVector, HVX_Vector, HVX_UVector, hvx_vec_store_u); \ +} \ static inline void OP_NAME##_uua(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - assert((uintptr_t) src1 % 128 == 0); \ - OP_LOOP_BODY(HVX_UVector, HVX_UVector, HVX_Vector, hvx_vec_store_u); \ -} \ + assert((uintptr_t) src1 % 128 == 0); \ + OP_LOOP_BODY(HVX_UVector, HVX_UVector, HVX_Vector, hvx_vec_store_u); \ +} \ static inline void OP_NAME##_uuu(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, uint32_t n) { \ - OP_LOOP_BODY(HVX_UVector, HVX_UVector, HVX_UVector, hvx_vec_store_u); \ -} \ + OP_LOOP_BODY(HVX_UVector, HVX_UVector, HVX_UVector, hvx_vec_store_u); \ +} \ // Dispatcher logic -#define HVX_DIV_DISPATCHER(OP_NAME) \ +#define HVX_DIV_DISPATCHER(OP_NAME) \ static inline void OP_NAME(uint8_t * restrict dst, const uint8_t * restrict src0, const uint8_t * restrict src1, const uint32_t num_elems) { \ - if (hex_is_aligned((void *) dst, 128)) { \ - if (hex_is_aligned((void *) src0, 128)) { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aaa(dst, src0, src1, num_elems); \ - else OP_NAME##_aau(dst, src0, src1, num_elems); \ - } else { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aua(dst, src0, src1, num_elems); \ - else OP_NAME##_auu(dst, src0, src1, num_elems); \ - } \ - } else { \ - if (hex_is_aligned((void *) src0, 128)) { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uaa(dst, src0, src1, num_elems); \ - else OP_NAME##_uau(dst, src0, src1, num_elems); \ - } else { \ - if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uua(dst, src0, src1, num_elems); \ - else OP_NAME##_uuu(dst, src0, src1, num_elems); \ - } \ - } \ + if (hex_is_aligned((void *) dst, 128)) { \ + if (hex_is_aligned((void *) src0, 128)) { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aaa(dst, src0, src1, num_elems); \ + else OP_NAME##_aau(dst, src0, src1, num_elems); \ + } else { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_aua(dst, src0, src1, num_elems); \ + else OP_NAME##_auu(dst, src0, src1, num_elems); \ + } \ + } else { \ + if (hex_is_aligned((void *) src0, 128)) { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uaa(dst, src0, src1, num_elems); \ + else OP_NAME##_uau(dst, src0, src1, num_elems); \ + } else { \ + if (hex_is_aligned((void *) src1, 128)) OP_NAME##_uua(dst, src0, src1, num_elems); \ + else OP_NAME##_uuu(dst, src0, src1, num_elems); \ + } \ + } \ } DEFINE_HVX_DIV_OP_VARIANTS(hvx_div_f32, hvx_div_f32_loop_body) diff --git a/ggml/src/ggml-hexagon/htp/hvx-exp.h b/ggml/src/ggml-hexagon/htp/hvx-exp.h index bcd3d2d3..93ca8cf5 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-exp.h +++ b/ggml/src/ggml-hexagon/htp/hvx-exp.h @@ -173,7 +173,7 @@ static inline void hvx_exp_f32(uint8_t * restrict dst, const uint8_t * restrict HVX_Vector * p_vec_in1 = (HVX_Vector *) src; HVX_Vector * p_vec_out = (HVX_Vector *) dst; - #pragma unroll(4) + #pragma unroll(2) for (int i = 0; i < num_elems_whole; i += VLEN_FP32) { if (true == negate) { HVX_Vector neg_vec_in = hvx_vec_neg_f32(*p_vec_in1++); @@ -183,7 +183,7 @@ static inline void hvx_exp_f32(uint8_t * restrict dst, const uint8_t * restrict } } } else { - #pragma unroll(4) + #pragma unroll(2) for (int i = 0; i < num_elems_whole; i += VLEN_FP32) { HVX_Vector in = *(HVX_UVector *) (src + i * SIZEOF_FP32); diff --git a/ggml/src/ggml-hexagon/htp/hvx-inverse.h b/ggml/src/ggml-hexagon/htp/hvx-inverse.h index f2054f45..256a8843 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-inverse.h +++ b/ggml/src/ggml-hexagon/htp/hvx-inverse.h @@ -169,36 +169,36 @@ static inline HVX_Vector hvx_vec_inverse_f16_guard(HVX_Vector v_sf, HVX_Vector n } while(0) // Generic macro to define alignment permutations for an op -#define DEFINE_HVX_INV_OP_VARIANTS(OP_NAME, OP_LOOP_BODY) \ +#define DEFINE_HVX_INV_OP_VARIANTS(OP_NAME, OP_LOOP_BODY) \ static inline void OP_NAME##_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - assert((uintptr_t) src % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_Vector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + assert((uintptr_t) src % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_Vector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_au(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { \ - assert((uintptr_t) dst % 128 == 0); \ - OP_LOOP_BODY(HVX_Vector, HVX_UVector, hvx_vec_store_a); \ -} \ + assert((uintptr_t) dst % 128 == 0); \ + OP_LOOP_BODY(HVX_Vector, HVX_UVector, hvx_vec_store_a); \ +} \ static inline void OP_NAME##_ua(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { \ - assert((uintptr_t) src % 128 == 0); \ - OP_LOOP_BODY(HVX_UVector, HVX_Vector, hvx_vec_store_u); \ -} \ + assert((uintptr_t) src % 128 == 0); \ + OP_LOOP_BODY(HVX_UVector, HVX_Vector, hvx_vec_store_u); \ +} \ static inline void OP_NAME##_uu(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { \ - OP_LOOP_BODY(HVX_UVector, HVX_UVector, hvx_vec_store_u); \ -} \ + OP_LOOP_BODY(HVX_UVector, HVX_UVector, hvx_vec_store_u); \ +} \ // Dispatcher logic -#define HVX_INV_DISPATCHER(OP_NAME) \ +#define HVX_INV_DISPATCHER(OP_NAME) \ static inline void OP_NAME(uint8_t * restrict dst, const uint8_t * restrict src, const uint32_t num_elems) { \ - if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { \ - OP_NAME##_aa(dst, src, num_elems); \ - } else if (hex_is_aligned((void *) dst, 128)) { \ - OP_NAME##_au(dst, src, num_elems); \ - } else if (hex_is_aligned((void *) src, 128)) { \ - OP_NAME##_ua(dst, src, num_elems); \ - } else { \ - OP_NAME##_uu(dst, src, num_elems); \ - } \ + if (hex_is_aligned((void *) dst, 128) && hex_is_aligned((void *) src, 128)) { \ + OP_NAME##_aa(dst, src, num_elems); \ + } else if (hex_is_aligned((void *) dst, 128)) { \ + OP_NAME##_au(dst, src, num_elems); \ + } else if (hex_is_aligned((void *) src, 128)) { \ + OP_NAME##_ua(dst, src, num_elems); \ + } else { \ + OP_NAME##_uu(dst, src, num_elems); \ + } \ } DEFINE_HVX_INV_OP_VARIANTS(hvx_inverse_f32, hvx_inverse_f32_loop_body) diff --git a/ggml/src/ggml-hexagon/htp/hvx-log.h b/ggml/src/ggml-hexagon/htp/hvx-log.h index 7013dae7..491041d5 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-log.h +++ b/ggml/src/ggml-hexagon/htp/hvx-log.h @@ -62,4 +62,57 @@ static inline HVX_Vector hvx_vec_log_f32(HVX_Vector x) { return hvx_vec_add_f32_f32(term_e, res); } +static inline void hvx_log_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + + HVX_Vector * restrict vdst = (HVX_Vector *) dst; + HVX_Vector * restrict vsrc = (HVX_Vector *) src; + + const uint32_t elem_size = sizeof(float); + const uint32_t epv = 128 / elem_size; + const uint32_t nvec = n / epv; + const uint32_t nloe = n % epv; + + uint32_t i = 0; + + _Pragma("unroll(4)") + for (; i < nvec; i++) { + vdst[i] = hvx_vec_log_f32(vsrc[i]); + } + if (nloe) { + HVX_Vector v = hvx_vec_log_f32(vsrc[i]); + hvx_vec_store_a((void *) &vdst[i], nloe * elem_size, v); + } +} + +// Compute log(x) for f16 by promoting to f32, applying hvx_vec_log_f32, and narrowing back. +static inline void hvx_log_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + + HVX_Vector * restrict vdst = (HVX_Vector *) dst; + HVX_Vector * restrict vsrc = (HVX_Vector *) src; + + const uint32_t nvec = n / VLEN_FP16; + const uint32_t nloe = n % VLEN_FP16; + + uint32_t i = 0; + + _Pragma("unroll(4)") + for (; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); + HVX_Vector r0 = hvx_vec_log_f32(Q6_V_lo_W(p)); + HVX_Vector r1 = hvx_vec_log_f32(Q6_V_hi_W(p)); + vdst[i] = hvx_vec_f32_to_f16(r0, r1); + } + if (nloe) { + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); + HVX_Vector r0 = hvx_vec_log_f32(Q6_V_lo_W(p)); + HVX_Vector r1 = hvx_vec_log_f32(Q6_V_hi_W(p)); + HVX_Vector v = hvx_vec_f32_to_f16(r0, r1); + hvx_vec_store_a((void *) &vdst[i], nloe * SIZEOF_FP16, v); + } +} + #endif /* HVX_LOG_H */ diff --git a/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-flat.h b/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-flat.h deleted file mode 100644 index 328a8311..00000000 --- a/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-flat.h +++ /dev/null @@ -1,1511 +0,0 @@ -// Dynamic quantizers that produce flat (non-tiled) activations - -static inline void quantize_block_f32_q8_0_flat( - float * restrict x, - uint8_t * restrict y_quants, - __fp16 * restrict y_scales, - uint32_t block_idx -) { - HVX_Vector * vx = (HVX_Vector *) x; - HVX_Vector zero = Q6_V_vzero(); - - HVX_Vector vmax0_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[0])); - HVX_Vector vmax1_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[1])); - HVX_Vector vmax2_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[2])); - HVX_Vector vmax3_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[3])); - - HVX_Vector vx0_qf = Q6_Vqf32_vsub_VsfVsf(vx[0], zero); - HVX_Vector vx1_qf = Q6_Vqf32_vsub_VsfVsf(vx[1], zero); - HVX_Vector vx2_qf = Q6_Vqf32_vsub_VsfVsf(vx[2], zero); - HVX_Vector vx3_qf = Q6_Vqf32_vsub_VsfVsf(vx[3], zero); - - HVX_Vector vmax0_qf = Q6_Vqf32_vsub_VsfVsf(vmax0_sf, zero); - HVX_Vector vmax1_qf = Q6_Vqf32_vsub_VsfVsf(vmax1_sf, zero); - HVX_Vector vmax2_qf = Q6_Vqf32_vsub_VsfVsf(vmax2_sf, zero); - HVX_Vector vmax3_qf = Q6_Vqf32_vsub_VsfVsf(vmax3_sf, zero); - - HVX_Vector vmax01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax1_qf, vmax0_qf))); - HVX_Vector vmax23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax3_qf, vmax2_qf))); - - HVX_Vector vx01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx1_qf, vx0_qf))); - HVX_Vector vx23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx3_qf, vx2_qf))); - - HVX_Vector vd01_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax01_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 - HVX_Vector vd23_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax23_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 - HVX_Vector vd01_hf = Q6_Vhf_equals_Vqf16(vd01_qf16); - HVX_Vector vd23_hf = Q6_Vhf_equals_Vqf16(vd23_qf16); - - HVX_Vector vd01_inv_hf = hvx_vec_inverse_f16(vd01_hf); - HVX_Vector vd23_inv_hf = hvx_vec_inverse_f16(vd23_hf); - vx01_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx01_hf, vd01_inv_hf)); - vx23_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx23_hf, vd23_inv_hf)); - - HVX_Vector vx01_i16 = hvx_vec_i16_from_hf_rnd_sat(vx01_hf); - HVX_Vector vx23_i16 = hvx_vec_i16_from_hf_rnd_sat(vx23_hf); - HVX_Vector vx_i8 = Q6_Vb_vpack_VhVh_sat(vx23_i16, vx01_i16); - - * (HVX_Vector *) (y_quants + block_idx * 128) = vx_i8; - - HVX_VectorPair vp1 = Q6_W_vshuff_VVR(vd23_hf, vd01_hf, -2); - HVX_VectorPair vp2 = Q6_W_vshuff_VVR(Q6_V_hi_W(vp1), Q6_V_lo_W(vp1), -2); - HVX_Vector v_scales = Q6_V_lo_W(vp2); - hvx_vec_store_u(y_scales + block_idx * 4, 8, v_scales); -} - -static inline void quantize_block_f32_q8_1_flat( - float * restrict x, - uint8_t * restrict y_quants, - __fp16 * restrict y_scales, - uint32_t block_idx -) { - HVX_Vector * vx = (HVX_Vector *) x; - HVX_Vector zero = Q6_V_vzero(); - - HVX_Vector vmax0_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[0])); - HVX_Vector vmax1_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[1])); - HVX_Vector vmax2_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[2])); - HVX_Vector vmax3_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[3])); - - HVX_Vector vx0_qf = Q6_Vqf32_vsub_VsfVsf(vx[0], zero); - HVX_Vector vx1_qf = Q6_Vqf32_vsub_VsfVsf(vx[1], zero); - HVX_Vector vx2_qf = Q6_Vqf32_vsub_VsfVsf(vx[2], zero); - HVX_Vector vx3_qf = Q6_Vqf32_vsub_VsfVsf(vx[3], zero); - - HVX_Vector vmax0_qf = Q6_Vqf32_vsub_VsfVsf(vmax0_sf, zero); - HVX_Vector vmax1_qf = Q6_Vqf32_vsub_VsfVsf(vmax1_sf, zero); - HVX_Vector vmax2_qf = Q6_Vqf32_vsub_VsfVsf(vmax2_sf, zero); - HVX_Vector vmax3_qf = Q6_Vqf32_vsub_VsfVsf(vmax3_sf, zero); - - HVX_Vector vmax01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax1_qf, vmax0_qf))); - HVX_Vector vmax23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax3_qf, vmax2_qf))); - - HVX_Vector vx01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx1_qf, vx0_qf))); - HVX_Vector vx23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx3_qf, vx2_qf))); - - HVX_Vector vd01_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax01_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 - HVX_Vector vd23_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax23_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 - HVX_Vector vd01_hf = Q6_Vhf_equals_Vqf16(vd01_qf16); - HVX_Vector vd23_hf = Q6_Vhf_equals_Vqf16(vd23_qf16); - - HVX_Vector vd01_inv_hf = hvx_vec_inverse_f16(vd01_hf); - HVX_Vector vd23_inv_hf = hvx_vec_inverse_f16(vd23_hf); - vx01_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx01_hf, vd01_inv_hf)); - vx23_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx23_hf, vd23_inv_hf)); - - HVX_Vector vx01_i16 = hvx_vec_i16_from_hf_rnd_sat(vx01_hf); - HVX_Vector vx23_i16 = hvx_vec_i16_from_hf_rnd_sat(vx23_hf); - HVX_Vector vx_i8 = Q6_Vb_vpack_VhVh_sat(vx23_i16, vx01_i16); - - const HVX_Vector ones = Q6_Vb_vsplat_R(1); - HVX_Vector v_sums = Q6_Vw_vrmpy_VbVb(vx_i8, ones); - v_sums = Q6_Vw_vadd_VwVw(v_sums, Q6_V_vror_VR(v_sums, 4)); - v_sums = Q6_Vw_vadd_VwVw(v_sums, Q6_V_vror_VR(v_sums, 8)); - v_sums = Q6_Vw_vadd_VwVw(v_sums, Q6_V_vror_VR(v_sums, 16)); - - * (HVX_Vector *) (y_quants + block_idx * 128) = vx_i8; - - HVX_VectorPair vp1 = Q6_W_vshuff_VVR(vd23_hf, vd01_hf, -2); - HVX_VectorPair vp2 = Q6_W_vshuff_VVR(Q6_V_hi_W(vp1), Q6_V_lo_W(vp1), -2); - HVX_Vector v_scales = Q6_V_lo_W(vp2); - - HVX_VectorPair v_deal1 = Q6_W_vdeal_VVR(v_sums, v_sums, -4); - HVX_Vector v_even1 = Q6_V_lo_W(v_deal1); - HVX_VectorPair v_deal2 = Q6_W_vdeal_VVR(v_even1, v_even1, -4); - HVX_Vector v_even2 = Q6_V_lo_W(v_deal2); - HVX_VectorPair v_deal3 = Q6_W_vdeal_VVR(v_even2, v_even2, -4); - HVX_Vector v_sums_shuffled = Q6_V_lo_W(v_deal3); - - HVX_Vector v_sums_sf = Q6_Vsf_equals_Vw(v_sums_shuffled); - HVX_Vector v_sums_hf = hvx_vec_f32_to_f16(v_sums_sf, Q6_V_vzero()); - - HVX_Vector v_prod = hvx_vec_mul_f16_f16(v_scales, v_sums_hf); - - HVX_VectorPair vp_scales = Q6_W_vshuff_VVR(v_prod, v_scales, -2); - HVX_Vector v_final = Q6_V_lo_W(vp_scales); - - hvx_vec_store_u(y_scales + block_idx * 8, 16, v_final); -} - -static inline void quantize_row_f32_q8_0_flat(float * restrict x, uint8_t * restrict y, uint32_t k) { - assert(k % 32 == 0); - const uint32_t quants_size = hex_round_up(k, 128); - uint8_t * restrict y_quants = y; - __fp16 * restrict y_scales = (__fp16 *) (y + quants_size); - - const uint32_t nb = (k + 127) / 128; - for (uint32_t i = 0; i < nb; i++) { - quantize_block_f32_q8_0_flat(x + i * 128, y_quants, y_scales, i); - } -} - -static inline void quantize_row_f32_q8_1_flat(float * restrict x, uint8_t * restrict y, uint32_t k) { - assert(k % 32 == 0); - const uint32_t quants_size = hex_round_up(k, 128); - uint8_t * restrict y_quants = y; - __fp16 * restrict y_scales = (__fp16 *) (y + quants_size); - - const uint32_t nb = (k + 127) / 128; - for (uint32_t i = 0; i < nb; i++) { - quantize_block_f32_q8_1_flat(x + i * 128, y_quants, y_scales, i); - } -} - -static inline void quantize_f32_q8_0_flat_kernel( - const uint8_t * restrict src_data, - uint8_t * restrict dst_data, - uint8_t * restrict tmp_data, - uint32_t ne0, - uint32_t nrows, - size_t src_row_size, - size_t dst_row_size -) { - const size_t src_row_size_padded = hex_round_up(src_row_size, QK_Q8_0_TILED * sizeof(float)); - hvx_splat_f32_a(tmp_data, 0.0f, src_row_size_padded / sizeof(float)); - - for (uint32_t i = 0; i < nrows; ++i) { - hex_l2fetch(src_data, src_row_size, src_row_size, 2); - hvx_copy_f32_aa(tmp_data, src_data, ne0); - - quantize_row_f32_q8_0_flat((float *) tmp_data, dst_data, ne0); - dst_data += dst_row_size; - src_data += src_row_size; - } -} - -static inline void quantize_f32_q8_1_flat_kernel( - const uint8_t * restrict src_data, - uint8_t * restrict dst_data, - uint8_t * restrict tmp_data, - uint32_t ne0, - uint32_t nrows, - size_t src_row_size, - size_t dst_row_size -) { - const size_t src_row_size_padded = hex_round_up(src_row_size, QK_Q8_0_TILED * sizeof(float)); - hvx_splat_f32_a(tmp_data, 0.0f, src_row_size_padded / sizeof(float)); - - for (uint32_t i = 0; i < nrows; ++i) { - hex_l2fetch(src_data, src_row_size, src_row_size, 2); - hvx_copy_f32_aa(tmp_data, src_data, ne0); - - quantize_row_f32_q8_1_flat((float *) tmp_data, dst_data, ne0); - dst_data += dst_row_size; - src_data += src_row_size; - } -} - -static inline void quantize_f32_f32_flat_kernel( - const uint8_t * restrict src_data, - uint8_t * restrict dst_data, - uint8_t * restrict tmp_data, - uint32_t ne0, - uint32_t nrows, - size_t src_stride, - size_t dst_stride -) { - (void) tmp_data; - const size_t src_row_size = ne0 * sizeof(float); - for (uint32_t i = 0; i < nrows; ++i) { - hex_l2fetch(src_data, src_row_size, src_stride, 2); - hvx_copy_f32_au(dst_data, src_data, ne0); - - dst_data += dst_stride; - src_data += src_stride; - } -} - -static inline void quantize_f32_f16_flat_kernel( - const uint8_t * restrict src_data, - uint8_t * restrict dst_data, - uint8_t * restrict tmp_data, - uint32_t ne0, - uint32_t nrows, - size_t src_stride, - size_t dst_stride -) { - (void) tmp_data; - const size_t src_row_size = ne0 * sizeof(float); - for (uint32_t i = 0; i < nrows; ++i) { - hex_l2fetch(src_data, src_row_size, src_stride, 2); - hvx_copy_f16_f32_au(dst_data, src_data, ne0); - - dst_data += dst_stride; - src_data += src_stride; - } -} - -static inline void quantize_f16_f16_flat_kernel( - const uint8_t * restrict src_data, - uint8_t * restrict dst_data, - uint8_t * restrict tmp_data, - uint32_t ne0, - uint32_t nrows, - size_t src_stride, - size_t dst_stride -) { - (void) tmp_data; - const size_t src_row_size = ne0 * sizeof(float); - for (uint32_t i = 0; i < nrows; ++i) { - hex_l2fetch(src_data, src_row_size, src_stride, 2); - hvx_copy_f16_au(dst_data, src_data, ne0); - - dst_data += dst_stride; - src_data += src_stride; - } -} - -// Dot kernels that consume flat (non-tiled) activations - -static void flat_vec_dot_q4_0_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y_q = vy; - - HVX_Vector v_sum_float = Q6_V_vzero(); - HVX_Vector i8 = Q6_Vb_vsplat_R(8); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y_scales = (const __fp16 *) (y_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx_i8 = * (const HVX_Vector *) (y_q + block_idx * 128); - HVX_Vector v_act_raw = Q6_V_vror_VR(vx_i8, sub_idx * 32); - - HVX_Vector v_act_rep[8]; - v_act_rep[0] = Q6_V_vdelta_VV(v_act_raw, v_repl_ctrl); - v_act_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 4), v_repl_ctrl); - v_act_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 8), v_repl_ctrl); - v_act_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 12), v_repl_ctrl); - v_act_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 16), v_repl_ctrl); - v_act_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 20), v_repl_ctrl); - v_act_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 24), v_repl_ctrl); - v_act_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 28), v_repl_ctrl); - - HVX_Vector v_sum = accum_4bit_32x1(vptr, v_act_rep, i8); - HVX_Vector v_sum_sf = Q6_Vsf_equals_Vw(v_sum); - - HVX_Vector v_scale_w = vptr[4]; - - __fp16 scale_a_val = y_scales[kt]; - HVX_Vector v_scale_a = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a_val)); - - HVX_Vector v_scale_comb = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a); - HVX_Vector v_sum_scaled = hvx_vec_mul_f32_f32(v_sum_sf, v_scale_comb); - - v_sum_float = hvx_vec_add_f32_f32(v_sum_float, v_sum_scaled); - } - - if (sz) { - hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); - } else { - hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); - } -} - -static void flat_vec_dot_q4_0_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y0_q = vy0; - const uint8_t * restrict y1_q = vy1; - - HVX_Vector v_sum_float_c0 = Q6_V_vzero(); - HVX_Vector v_sum_float_c1 = Q6_V_vzero(); - HVX_Vector i8 = Q6_Vb_vsplat_R(8); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y0_scales = (const __fp16 *) (y0_q + quants_size); - const __fp16 * restrict y1_scales = (const __fp16 *) (y1_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx0_i8 = * (const HVX_Vector *) (y0_q + block_idx * 128); - HVX_Vector vx1_i8 = * (const HVX_Vector *) (y1_q + block_idx * 128); - - HVX_Vector v_act0_raw = Q6_V_vror_VR(vx0_i8, sub_idx * 32); - HVX_Vector v_act1_raw = Q6_V_vror_VR(vx1_i8, sub_idx * 32); - - HVX_Vector v_act0_rep[8]; - v_act0_rep[0] = Q6_V_vdelta_VV(v_act0_raw, v_repl_ctrl); - v_act0_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 4), v_repl_ctrl); - v_act0_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 8), v_repl_ctrl); - v_act0_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 12), v_repl_ctrl); - v_act0_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 16), v_repl_ctrl); - v_act0_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 20), v_repl_ctrl); - v_act0_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 24), v_repl_ctrl); - v_act0_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 28), v_repl_ctrl); - - HVX_Vector v_act1_rep[8]; - v_act1_rep[0] = Q6_V_vdelta_VV(v_act1_raw, v_repl_ctrl); - v_act1_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 4), v_repl_ctrl); - v_act1_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 8), v_repl_ctrl); - v_act1_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 12), v_repl_ctrl); - v_act1_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 16), v_repl_ctrl); - v_act1_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 20), v_repl_ctrl); - v_act1_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 24), v_repl_ctrl); - v_act1_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 28), v_repl_ctrl); - - HVX_VectorPair v_sums = accum_4bit_32x2(vptr, v_act0_rep, v_act1_rep, i8); - HVX_Vector v_sum_c0 = Q6_V_lo_W(v_sums); - HVX_Vector v_sum_c1 = Q6_V_hi_W(v_sums); - - HVX_Vector v_sum_sf_c0 = Q6_Vsf_equals_Vw(v_sum_c0); - HVX_Vector v_sum_sf_c1 = Q6_Vsf_equals_Vw(v_sum_c1); - - HVX_Vector v_scale_w = vptr[4]; - - __fp16 scale_a0_val = y0_scales[kt]; - __fp16 scale_a1_val = y1_scales[kt]; - HVX_Vector v_scale_a0 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a0_val)); - HVX_Vector v_scale_a1 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a1_val)); - - HVX_Vector v_scale_comb_c0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a0); - HVX_Vector v_scale_comb_c1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a1); - - HVX_Vector v_sum_scaled_c0 = hvx_vec_mul_f32_f32(v_sum_sf_c0, v_scale_comb_c0); - HVX_Vector v_sum_scaled_c1 = hvx_vec_mul_f32_f32(v_sum_sf_c1, v_scale_comb_c1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, v_sum_scaled_c0); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, v_sum_scaled_c1); - } - - if (sz0) { - hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); - } else { - hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); - } - if (sz1) { - hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); - } else { - hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); - } -} - -static void flat_vec_dot_q4_1_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y_q = vy; - - HVX_Vector v_sum_float = Q6_V_vzero(); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y_scales = (const __fp16 *) (y_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx_i8 = * (const HVX_Vector *) (y_q + block_idx * 128); - HVX_Vector v_act_raw = Q6_V_vror_VR(vx_i8, sub_idx * 32); - - HVX_Vector v_act_rep[8]; - v_act_rep[0] = Q6_V_vdelta_VV(v_act_raw, v_repl_ctrl); - v_act_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 4), v_repl_ctrl); - v_act_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 8), v_repl_ctrl); - v_act_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 12), v_repl_ctrl); - v_act_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 16), v_repl_ctrl); - v_act_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 20), v_repl_ctrl); - v_act_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 24), v_repl_ctrl); - v_act_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 28), v_repl_ctrl); - - HVX_Vector v_sum = accum_4bit_32x1(vptr, v_act_rep, Q6_V_vzero()); - HVX_Vector v_sum_sf = Q6_Vsf_equals_Vw(v_sum); - - HVX_Vector v_scale_offset = vptr[4]; - HVX_VectorPair p_deal = Q6_W_vdeal_VVR(v_scale_offset, v_scale_offset, -2); - HVX_Vector v_scale = Q6_V_lo_W(p_deal); - HVX_Vector v_offset = Q6_V_hi_W(p_deal); - - __fp16 scale_a_val = y_scales[kt * 2 + 0]; - __fp16 sum_a_val = y_scales[kt * 2 + 1]; - HVX_Vector v_scale_a = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a_val)); - HVX_Vector v_sum_a = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&sum_a_val)); - - HVX_Vector v_scale_comb = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale, v_scale_a); - HVX_Vector v_offset_comb = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset, v_sum_a); - - HVX_Vector v_scaled_dot = hvx_vec_mul_f32_f32(v_sum_sf, v_scale_comb); - HVX_Vector v_sum_scaled = hvx_vec_add_f32_f32(v_scaled_dot, v_offset_comb); - - v_sum_float = hvx_vec_add_f32_f32(v_sum_float, v_sum_scaled); - } - - if (sz) { - hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); - } else { - hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); - } -} - -static void flat_vec_dot_q4_1_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y0_q = vy0; - const uint8_t * restrict y1_q = vy1; - - HVX_Vector v_sum_float_c0 = Q6_V_vzero(); - HVX_Vector v_sum_float_c1 = Q6_V_vzero(); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y0_scales = (const __fp16 *) (y0_q + quants_size); - const __fp16 * restrict y1_scales = (const __fp16 *) (y1_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx0_i8 = * (const HVX_Vector *) (y0_q + block_idx * 128); - HVX_Vector vx1_i8 = * (const HVX_Vector *) (y1_q + block_idx * 128); - - HVX_Vector v_act0_raw = Q6_V_vror_VR(vx0_i8, sub_idx * 32); - HVX_Vector v_act1_raw = Q6_V_vror_VR(vx1_i8, sub_idx * 32); - - HVX_Vector v_act0_rep[8]; - v_act0_rep[0] = Q6_V_vdelta_VV(v_act0_raw, v_repl_ctrl); - v_act0_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 4), v_repl_ctrl); - v_act0_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 8), v_repl_ctrl); - v_act0_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 12), v_repl_ctrl); - v_act0_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 16), v_repl_ctrl); - v_act0_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 20), v_repl_ctrl); - v_act0_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 24), v_repl_ctrl); - v_act0_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 28), v_repl_ctrl); - - HVX_Vector v_act1_rep[8]; - v_act1_rep[0] = Q6_V_vdelta_VV(v_act1_raw, v_repl_ctrl); - v_act1_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 4), v_repl_ctrl); - v_act1_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 8), v_repl_ctrl); - v_act1_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 12), v_repl_ctrl); - v_act1_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 16), v_repl_ctrl); - v_act1_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 20), v_repl_ctrl); - v_act1_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 24), v_repl_ctrl); - v_act1_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 28), v_repl_ctrl); - - HVX_VectorPair v_sums = accum_4bit_32x2(vptr, v_act0_rep, v_act1_rep, Q6_V_vzero()); - HVX_Vector v_sum_c0 = Q6_V_lo_W(v_sums); - HVX_Vector v_sum_c1 = Q6_V_hi_W(v_sums); - - HVX_Vector v_sum_sf_c0 = Q6_Vsf_equals_Vw(v_sum_c0); - HVX_Vector v_sum_sf_c1 = Q6_Vsf_equals_Vw(v_sum_c1); - - HVX_Vector v_scale_offset = vptr[4]; - HVX_VectorPair p_deal = Q6_W_vdeal_VVR(v_scale_offset, v_scale_offset, -2); - HVX_Vector v_scale = Q6_V_lo_W(p_deal); - HVX_Vector v_offset = Q6_V_hi_W(p_deal); - - __fp16 scale_a0_val = y0_scales[kt * 2 + 0]; - __fp16 sum_a0_val = y0_scales[kt * 2 + 1]; - __fp16 scale_a1_val = y1_scales[kt * 2 + 0]; - __fp16 sum_a1_val = y1_scales[kt * 2 + 1]; - - HVX_Vector v_scale_a0 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a0_val)); - HVX_Vector v_sum_a0 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&sum_a0_val)); - HVX_Vector v_scale_a1 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a1_val)); - HVX_Vector v_sum_a1 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&sum_a1_val)); - - HVX_Vector v_scale_comb_c0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale, v_scale_a0); - HVX_Vector v_offset_comb_c0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset, v_sum_a0); - HVX_Vector v_scale_comb_c1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale, v_scale_a1); - HVX_Vector v_offset_comb_c1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset, v_sum_a1); - - HVX_Vector v_scaled_dot_c0 = hvx_vec_mul_f32_f32(v_sum_sf_c0, v_scale_comb_c0); - HVX_Vector v_sum_scaled_c0 = hvx_vec_add_f32_f32(v_scaled_dot_c0, v_offset_comb_c0); - - HVX_Vector v_scaled_dot_c1 = hvx_vec_mul_f32_f32(v_sum_sf_c1, v_scale_comb_c1); - HVX_Vector v_sum_scaled_c1 = hvx_vec_add_f32_f32(v_scaled_dot_c1, v_offset_comb_c1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, v_sum_scaled_c0); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, v_sum_scaled_c1); - } - - if (sz0) { - hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); - } else { - hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); - } - if (sz1) { - hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); - } else { - hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); - } -} - -static void flat_vec_dot_q8_0_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y_q = vy; - - HVX_Vector v_sum_float = Q6_V_vzero(); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y_scales = (const __fp16 *) (y_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 1152); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx_i8 = * (const HVX_Vector *) (y_q + block_idx * 128); - HVX_Vector v_act_raw = Q6_V_vror_VR(vx_i8, sub_idx * 32); - - HVX_Vector v_act_rep[8]; - v_act_rep[0] = Q6_V_vdelta_VV(v_act_raw, v_repl_ctrl); - v_act_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 4), v_repl_ctrl); - v_act_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 8), v_repl_ctrl); - v_act_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 12), v_repl_ctrl); - v_act_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 16), v_repl_ctrl); - v_act_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 20), v_repl_ctrl); - v_act_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 24), v_repl_ctrl); - v_act_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 28), v_repl_ctrl); - - HVX_Vector v_sum = accum_q8_0_32x1(vptr, v_act_rep); - HVX_Vector v_sum_sf = Q6_Vsf_equals_Vw(v_sum); - - HVX_Vector v_scale_w = vptr[8]; - - __fp16 scale_a_val = y_scales[kt]; - HVX_Vector v_scale_a = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a_val)); - - HVX_Vector v_scale_comb = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a); - HVX_Vector v_sum_scaled = hvx_vec_mul_f32_f32(v_sum_sf, v_scale_comb); - - v_sum_float = hvx_vec_add_f32_f32(v_sum_float, v_sum_scaled); - } - - if (sz) { - hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); - } else { - hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); - } -} - -static void flat_vec_dot_q8_0_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y0_q = vy0; - const uint8_t * restrict y1_q = vy1; - - HVX_Vector v_sum_float_c0 = Q6_V_vzero(); - HVX_Vector v_sum_float_c1 = Q6_V_vzero(); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y0_scales = (const __fp16 *) (y0_q + quants_size); - const __fp16 * restrict y1_scales = (const __fp16 *) (y1_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 1152); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx0_i8 = * (const HVX_Vector *) (y0_q + block_idx * 128); - HVX_Vector vx1_i8 = * (const HVX_Vector *) (y1_q + block_idx * 128); - - HVX_Vector v_act0_raw = Q6_V_vror_VR(vx0_i8, sub_idx * 32); - HVX_Vector v_act1_raw = Q6_V_vror_VR(vx1_i8, sub_idx * 32); - - HVX_Vector v_act0_rep[8]; - v_act0_rep[0] = Q6_V_vdelta_VV(v_act0_raw, v_repl_ctrl); - v_act0_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 4), v_repl_ctrl); - v_act0_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 8), v_repl_ctrl); - v_act0_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 12), v_repl_ctrl); - v_act0_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 16), v_repl_ctrl); - v_act0_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 20), v_repl_ctrl); - v_act0_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 24), v_repl_ctrl); - v_act0_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 28), v_repl_ctrl); - - HVX_Vector v_act1_rep[8]; - v_act1_rep[0] = Q6_V_vdelta_VV(v_act1_raw, v_repl_ctrl); - v_act1_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 4), v_repl_ctrl); - v_act1_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 8), v_repl_ctrl); - v_act1_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 12), v_repl_ctrl); - v_act1_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 16), v_repl_ctrl); - v_act1_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 20), v_repl_ctrl); - v_act1_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 24), v_repl_ctrl); - v_act1_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 28), v_repl_ctrl); - - HVX_VectorPair v_sums = accum_q8_0_32x2(vptr, v_act0_rep, v_act1_rep); - HVX_Vector v_sum_c0 = Q6_V_lo_W(v_sums); - HVX_Vector v_sum_c1 = Q6_V_hi_W(v_sums); - - HVX_Vector v_sum_sf_c0 = Q6_Vsf_equals_Vw(v_sum_c0); - HVX_Vector v_sum_sf_c1 = Q6_Vsf_equals_Vw(v_sum_c1); - - HVX_Vector v_scale_w = vptr[8]; - - __fp16 scale_a0_val = y0_scales[kt]; - __fp16 scale_a1_val = y1_scales[kt]; - HVX_Vector v_scale_a0 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a0_val)); - HVX_Vector v_scale_a1 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a1_val)); - - HVX_Vector v_scale_comb_c0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a0); - HVX_Vector v_scale_comb_c1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a1); - - HVX_Vector v_sum_scaled_c0 = hvx_vec_mul_f32_f32(v_sum_sf_c0, v_scale_comb_c0); - HVX_Vector v_sum_scaled_c1 = hvx_vec_mul_f32_f32(v_sum_sf_c1, v_scale_comb_c1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, v_sum_scaled_c0); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, v_sum_scaled_c1); - } - - if (sz0) { - hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); - } else { - hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); - } - if (sz1) { - hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); - } else { - hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); - } -} - -static void flat_vec_dot_iq4nl_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y_q = vy; - - HVX_Vector v_sum_float = Q6_V_vzero(); - HVX_Vector mask_h4 = Q6_Vb_vsplat_R(0x0F); - HVX_Vector lut = *(const HVX_Vector *) kvalues_iq4nl_lut; - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y_scales = (const __fp16 *) (y_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx = * (const HVX_Vector *) (y_q + block_idx * 128); - HVX_Vector v_act_raw = Q6_V_vror_VR(vx, sub_idx * 32); - - HVX_Vector v_act_rep[8]; - v_act_rep[0] = Q6_V_vdelta_VV(v_act_raw, v_repl_ctrl); - v_act_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 4), v_repl_ctrl); - v_act_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 8), v_repl_ctrl); - v_act_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 12), v_repl_ctrl); - v_act_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 16), v_repl_ctrl); - v_act_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 20), v_repl_ctrl); - v_act_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 24), v_repl_ctrl); - v_act_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 28), v_repl_ctrl); - - HVX_Vector v_sum = accum_4bit_32x1_lut(vptr, v_act_rep, mask_h4, lut); - HVX_Vector v_sum_sf = Q6_Vsf_equals_Vw(v_sum); - - HVX_Vector v_scale_w = vptr[4]; - - __fp16 scale_a_val = y_scales[kt]; - HVX_Vector v_scale_a = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a_val)); - - HVX_Vector v_scale_comb = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a); - HVX_Vector v_sum_scaled = hvx_vec_mul_f32_f32(v_sum_sf, v_scale_comb); - - v_sum_float = hvx_vec_add_f32_f32(v_sum_float, v_sum_scaled); - } - - if (sz) { - hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); - } else { - hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); - } -} - -static void flat_vec_dot_iq4nl_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y0_q = vy0; - const uint8_t * restrict y1_q = vy1; - - HVX_Vector v_sum_float_c0 = Q6_V_vzero(); - HVX_Vector v_sum_float_c1 = Q6_V_vzero(); - HVX_Vector mask_h4 = Q6_Vb_vsplat_R(0x0F); - HVX_Vector lut = *(const HVX_Vector *) kvalues_iq4nl_lut; - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y0_scales = (const __fp16 *) (y0_q + quants_size); - const __fp16 * restrict y1_scales = (const __fp16 *) (y1_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx0 = * (const HVX_Vector *) (y0_q + block_idx * 128); - HVX_Vector vx1 = * (const HVX_Vector *) (y1_q + block_idx * 128); - - HVX_Vector v_act0_raw = Q6_V_vror_VR(vx0, sub_idx * 32); - HVX_Vector v_act1_raw = Q6_V_vror_VR(vx1, sub_idx * 32); - - HVX_Vector v_act0_rep[8]; - v_act0_rep[0] = Q6_V_vdelta_VV(v_act0_raw, v_repl_ctrl); - v_act0_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 4), v_repl_ctrl); - v_act0_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 8), v_repl_ctrl); - v_act0_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 12), v_repl_ctrl); - v_act0_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 16), v_repl_ctrl); - v_act0_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 20), v_repl_ctrl); - v_act0_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 24), v_repl_ctrl); - v_act0_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 28), v_repl_ctrl); - - HVX_Vector v_act1_rep[8]; - v_act1_rep[0] = Q6_V_vdelta_VV(v_act1_raw, v_repl_ctrl); - v_act1_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 4), v_repl_ctrl); - v_act1_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 8), v_repl_ctrl); - v_act1_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 12), v_repl_ctrl); - v_act1_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 16), v_repl_ctrl); - v_act1_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 20), v_repl_ctrl); - v_act1_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 24), v_repl_ctrl); - v_act1_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 28), v_repl_ctrl); - - HVX_VectorPair v_sums = accum_4bit_32x2_lut(vptr, v_act0_rep, v_act1_rep, mask_h4, lut); - HVX_Vector v_sum_c0 = Q6_V_lo_W(v_sums); - HVX_Vector v_sum_c1 = Q6_V_hi_W(v_sums); - - HVX_Vector v_sum_sf_c0 = Q6_Vsf_equals_Vw(v_sum_c0); - HVX_Vector v_sum_sf_c1 = Q6_Vsf_equals_Vw(v_sum_c1); - - HVX_Vector v_scale_w = vptr[4]; - - __fp16 scale_a0_val = y0_scales[kt]; - __fp16 scale_a1_val = y1_scales[kt]; - HVX_Vector v_scale_a0 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a0_val)); - HVX_Vector v_scale_a1 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a1_val)); - - HVX_Vector v_scale_comb_c0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a0); - HVX_Vector v_scale_comb_c1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a1); - - HVX_Vector v_sum_scaled_c0 = hvx_vec_mul_f32_f32(v_sum_sf_c0, v_scale_comb_c0); - HVX_Vector v_sum_scaled_c1 = hvx_vec_mul_f32_f32(v_sum_sf_c1, v_scale_comb_c1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, v_sum_scaled_c0); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, v_sum_scaled_c1); - } - - if (sz0) { - hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); - } else { - hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); - } - if (sz1) { - hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); - } else { - hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); - } -} - -static void flat_vec_dot_mxfp4_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y_q = vy; - - HVX_Vector v_sum_float = Q6_V_vzero(); - HVX_Vector mask_h4 = Q6_Vb_vsplat_R(0x0F); - HVX_Vector lut = *(const HVX_Vector *) kvalues_mxfp4_lut; - HVX_Vector expand = *(const HVX_Vector *) expand_x32_e8m0; - HVX_Vector e8m0_mask = Q6_V_vsplat_R(0x000000ff); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y_scales = (const __fp16 *) (y_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx = * (const HVX_Vector *) (y_q + block_idx * 128); - HVX_Vector v_act_raw = Q6_V_vror_VR(vx, sub_idx * 32); - - HVX_Vector v_act_rep[8]; - v_act_rep[0] = Q6_V_vdelta_VV(v_act_raw, v_repl_ctrl); - v_act_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 4), v_repl_ctrl); - v_act_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 8), v_repl_ctrl); - v_act_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 12), v_repl_ctrl); - v_act_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 16), v_repl_ctrl); - v_act_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 20), v_repl_ctrl); - v_act_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 24), v_repl_ctrl); - v_act_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act_raw, 28), v_repl_ctrl); - - HVX_Vector v_sum = accum_4bit_32x1_lut(vptr, v_act_rep, mask_h4, lut); - HVX_Vector v_sum_sf = Q6_Vsf_equals_Vw(v_sum); - - HVX_Vector v_scale_w = hvx_vmem(tile_ptr + kt * 640 + 512); - HVX_Vector r0_d = Q6_V_vdelta_VV(v_scale_w, expand); - r0_d = Q6_V_vand_VV(r0_d, e8m0_mask); - HVX_Vector v_scale_w_f32 = Q6_Vw_vasl_VwR(r0_d, 23); - - __fp16 scale_a_val = y_scales[kt]; - HVX_Vector v_scale_a_f16 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a_val)); - HVX_VectorPair p_scale_a_f32 = hvx_vec_f16_to_f32(v_scale_a_f16); - HVX_Vector v_scale_a = Q6_V_lo_W(p_scale_a_f32); - - HVX_Vector v_scale_comb = hvx_vec_mul_f32_f32(v_scale_w_f32, v_scale_a); - HVX_Vector v_sum_scaled = hvx_vec_mul_f32_f32(v_sum_sf, v_scale_comb); - - v_sum_float = hvx_vec_add_f32_f32(v_sum_float, v_sum_scaled); - } - - v_sum_float = hvx_vec_mul_f32_f32(v_sum_float, hvx_vec_splat_f32(0.5f)); - - if (sz) { - hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); - } else { - hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); - } -} - -static void flat_vec_dot_mxfp4_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { - const uint8_t * restrict tile_ptr = vx; - const uint8_t * restrict y0_q = vy0; - const uint8_t * restrict y1_q = vy1; - - HVX_Vector v_sum_float_c0 = Q6_V_vzero(); - HVX_Vector v_sum_float_c1 = Q6_V_vzero(); - HVX_Vector mask_h4 = Q6_Vb_vsplat_R(0x0F); - HVX_Vector lut = *(const HVX_Vector *) kvalues_mxfp4_lut; - HVX_Vector expand = *(const HVX_Vector *) expand_x32_e8m0; - HVX_Vector e8m0_mask = Q6_V_vsplat_R(0x000000ff); - - static const uint8_t __attribute__((aligned(128))) repl[128] = { - 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x40, 0x40, 0x40, 0x40, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x20, 0x20, 0x20, 0x20, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - 0x10, 0x10, 0x10, 0x10, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, - }; - HVX_Vector v_repl_ctrl = * (const HVX_Vector *) repl; - - const uint32_t quants_size = hex_round_up(n, 128); - const __fp16 * restrict y0_scales = (const __fp16 *) (y0_q + quants_size); - const __fp16 * restrict y1_scales = (const __fp16 *) (y1_q + quants_size); - - uint32_t n_k_tiles = n / 32; - for (uint32_t kt = 0; kt < n_k_tiles; kt++) { - const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); - - uint32_t block_idx = kt / 4; - uint32_t sub_idx = kt % 4; - - HVX_Vector vx0 = * (const HVX_Vector *) (y0_q + block_idx * 128); - HVX_Vector vx1 = * (const HVX_Vector *) (y1_q + block_idx * 128); - - HVX_Vector v_act0_raw = Q6_V_vror_VR(vx0, sub_idx * 32); - HVX_Vector v_act1_raw = Q6_V_vror_VR(vx1, sub_idx * 32); - - HVX_Vector v_act0_rep[8]; - v_act0_rep[0] = Q6_V_vdelta_VV(v_act0_raw, v_repl_ctrl); - v_act0_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 4), v_repl_ctrl); - v_act0_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 8), v_repl_ctrl); - v_act0_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 12), v_repl_ctrl); - v_act0_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 16), v_repl_ctrl); - v_act0_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 20), v_repl_ctrl); - v_act0_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 24), v_repl_ctrl); - v_act0_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act0_raw, 28), v_repl_ctrl); - - HVX_Vector v_act1_rep[8]; - v_act1_rep[0] = Q6_V_vdelta_VV(v_act1_raw, v_repl_ctrl); - v_act1_rep[1] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 4), v_repl_ctrl); - v_act1_rep[2] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 8), v_repl_ctrl); - v_act1_rep[3] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 12), v_repl_ctrl); - v_act1_rep[4] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 16), v_repl_ctrl); - v_act1_rep[5] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 20), v_repl_ctrl); - v_act1_rep[6] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 24), v_repl_ctrl); - v_act1_rep[7] = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act1_raw, 28), v_repl_ctrl); - - HVX_VectorPair v_sums = accum_4bit_32x2_lut(vptr, v_act0_rep, v_act1_rep, mask_h4, lut); - HVX_Vector v_sum_c0 = Q6_V_lo_W(v_sums); - HVX_Vector v_sum_c1 = Q6_V_hi_W(v_sums); - - HVX_Vector v_sum_sf_c0 = Q6_Vsf_equals_Vw(v_sum_c0); - HVX_Vector v_sum_sf_c1 = Q6_Vsf_equals_Vw(v_sum_c1); - - HVX_Vector v_scale_w = hvx_vmem(tile_ptr + kt * 640 + 512); - HVX_Vector r0_d = Q6_V_vdelta_VV(v_scale_w, expand); - r0_d = Q6_V_vand_VV(r0_d, e8m0_mask); - HVX_Vector v_scale_w_f32 = Q6_Vw_vasl_VwR(r0_d, 23); - - __fp16 scale_a0_val = y0_scales[kt]; - __fp16 scale_a1_val = y1_scales[kt]; - HVX_Vector v_scale_a0_f16 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a0_val)); - HVX_Vector v_scale_a1_f16 = hvx_vec_repl_f16(Q6_Vh_vsplat_R(*(const int16_t *)&scale_a1_val)); - HVX_VectorPair p_scale_a0_f32 = hvx_vec_f16_to_f32(v_scale_a0_f16); - HVX_VectorPair p_scale_a1_f32 = hvx_vec_f16_to_f32(v_scale_a1_f16); - HVX_Vector v_scale_a0 = Q6_V_lo_W(p_scale_a0_f32); - HVX_Vector v_scale_a1 = Q6_V_lo_W(p_scale_a1_f32); - - HVX_Vector v_scale_comb_c0 = hvx_vec_mul_f32_f32(v_scale_w_f32, v_scale_a0); - HVX_Vector v_scale_comb_c1 = hvx_vec_mul_f32_f32(v_scale_w_f32, v_scale_a1); - - HVX_Vector v_sum_scaled_c0 = hvx_vec_mul_f32_f32(v_sum_sf_c0, v_scale_comb_c0); - HVX_Vector v_sum_scaled_c1 = hvx_vec_mul_f32_f32(v_sum_sf_c1, v_scale_comb_c1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, v_sum_scaled_c0); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, v_sum_scaled_c1); - } - - v_sum_float_c0 = hvx_vec_mul_f32_f32(v_sum_float_c0, hvx_vec_splat_f32(0.5f)); - v_sum_float_c1 = hvx_vec_mul_f32_f32(v_sum_float_c1, hvx_vec_splat_f32(0.5f)); - - if (sz0) { - hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); - } else { - hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); - } - if (sz1) { - hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); - } else { - hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); - } -} - -#if __HVX_ARCH__ < 79 -#define HVX_OP_ADD_F32(a, b) Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(a, b)) -#define HVX_OP_MUL_F32(a, b) Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(a, b)) -#else -#define HVX_OP_ADD_F32(a, b) Q6_Vsf_vadd_VsfVsf(a, b) -#define HVX_OP_MUL_F32(a, b) Q6_Vsf_vmpy_VsfVsf(a, b) -#endif - -static inline void vec_dot_f32_f32_aa_1x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy) { - const HVX_Vector * restrict x = (const HVX_Vector *) vx; - const HVX_Vector * restrict y = (const HVX_Vector *) vy; - - uint32_t nvec = n / VLEN_FP32; // num full fp32 hvx vectors - uint32_t nloe = n % VLEN_FP32; // leftover elements - - HVX_Vector rsum = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(4) - for (i = 0; i < nvec; i++) { - HVX_Vector prod = HVX_OP_MUL_F32(x[i], y[i]); - rsum = HVX_OP_ADD_F32(rsum, prod); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); - HVX_Vector x_sf = Q6_V_vand_QV(bmask, x[i]); - HVX_Vector y_sf = Q6_V_vand_QV(bmask, y[i]); - HVX_Vector prod = HVX_OP_MUL_F32(x_sf, y_sf); - rsum = HVX_OP_ADD_F32(rsum, prod); - } - - *s = hvx_vec_get_f32(hvx_vec_reduce_sum_f32(rsum)); -} - -static inline void vec_dot_f32_f32_aa_2x1(const uint32_t n, float * restrict s0, - const void * restrict vx0, const void * restrict vx1, - const void * restrict vy0) { - const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; - const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; - const HVX_Vector * restrict y = (const HVX_Vector *) vy0; - - uint32_t nvec = n / VLEN_FP32; - uint32_t nloe = n % VLEN_FP32; - - HVX_Vector rsum0 = Q6_V_vzero(); - HVX_Vector rsum1 = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - HVX_Vector y_sf = y[i]; - HVX_Vector prod0 = HVX_OP_MUL_F32(x0[i], y_sf); - HVX_Vector prod1 = HVX_OP_MUL_F32(x1[i], y_sf); - rsum0 = HVX_OP_ADD_F32(rsum0, prod0); - rsum1 = HVX_OP_ADD_F32(rsum1, prod1); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); - HVX_Vector y_sf = Q6_V_vand_QV(bmask, y[i]); - HVX_Vector x0_sf = Q6_V_vand_QV(bmask, x0[i]); - HVX_Vector x1_sf = Q6_V_vand_QV(bmask, x1[i]); - HVX_Vector prod0 = HVX_OP_MUL_F32(x0_sf, y_sf); - HVX_Vector prod1 = HVX_OP_MUL_F32(x1_sf, y_sf); - rsum0 = HVX_OP_ADD_F32(rsum0, prod0); - rsum1 = HVX_OP_ADD_F32(rsum1, prod1); - } - - HVX_Vector rsum = hvx_vec_reduce_sum_f32x2(rsum0, rsum1); - hvx_vec_store_u(s0, 8, rsum); -} - -static inline void vec_dot_f32_f32_aa_2x2(const uint32_t n, float * restrict s0, float * restrict s1, - const void * restrict vx0, const void * restrict vx1, - const void * restrict vy0, const void * restrict vy1) { - const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; - const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; - const HVX_Vector * restrict y0 = (const HVX_Vector *) vy0; - const HVX_Vector * restrict y1 = (const HVX_Vector *) vy1; - - uint32_t nvec = n / VLEN_FP32; - uint32_t nloe = n % VLEN_FP32; - - HVX_Vector r0_c0_sum = Q6_V_vzero(); - HVX_Vector r0_c1_sum = Q6_V_vzero(); - HVX_Vector r1_c0_sum = Q6_V_vzero(); - HVX_Vector r1_c1_sum = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - HVX_Vector r0_sf = x0[i]; - HVX_Vector r1_sf = x1[i]; - HVX_Vector c0_sf = y0[i]; - HVX_Vector c1_sf = y1[i]; - - r0_c0_sum = HVX_OP_ADD_F32(r0_c0_sum, HVX_OP_MUL_F32(r0_sf, c0_sf)); - r0_c1_sum = HVX_OP_ADD_F32(r0_c1_sum, HVX_OP_MUL_F32(r0_sf, c1_sf)); - r1_c0_sum = HVX_OP_ADD_F32(r1_c0_sum, HVX_OP_MUL_F32(r1_sf, c0_sf)); - r1_c1_sum = HVX_OP_ADD_F32(r1_c1_sum, HVX_OP_MUL_F32(r1_sf, c1_sf)); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); - - HVX_Vector r0_sf = Q6_V_vand_QV(bmask, x0[i]); - HVX_Vector r1_sf = Q6_V_vand_QV(bmask, x1[i]); - HVX_Vector c0_sf = Q6_V_vand_QV(bmask, y0[i]); - HVX_Vector c1_sf = Q6_V_vand_QV(bmask, y1[i]); - - r0_c0_sum = HVX_OP_ADD_F32(r0_c0_sum, HVX_OP_MUL_F32(r0_sf, c0_sf)); - r0_c1_sum = HVX_OP_ADD_F32(r0_c1_sum, HVX_OP_MUL_F32(r0_sf, c1_sf)); - r1_c0_sum = HVX_OP_ADD_F32(r1_c0_sum, HVX_OP_MUL_F32(r1_sf, c0_sf)); - r1_c1_sum = HVX_OP_ADD_F32(r1_c1_sum, HVX_OP_MUL_F32(r1_sf, c1_sf)); - } - - // Reduce and store results - HVX_Vector r0_r1_c0_sum = hvx_vec_reduce_sum_f32x2(r0_c0_sum, r1_c0_sum); - HVX_Vector r0_r1_c1_sum = hvx_vec_reduce_sum_f32x2(r0_c1_sum, r1_c1_sum); - - hvx_vec_store_u(s0, 8, r0_r1_c0_sum); - hvx_vec_store_u(s1, 8, r0_r1_c1_sum); -} - -static inline void vec_dot_f32_f32_uu_1x1(const uint32_t n, float * restrict s, const void * restrict x, const void * restrict y) { - const HVX_UVector * restrict vx = (const HVX_UVector * restrict) x; - const HVX_UVector * restrict vy = (const HVX_UVector * restrict) y; - - uint32_t nvec = n / VLEN_FP32; // num full fp32 hvx vectors - uint32_t nloe = n % VLEN_FP32; // leftover elements - - HVX_Vector rsum = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - HVX_Vector x_sf = vx[i]; - HVX_Vector y_sf = vy[i]; - - rsum = HVX_OP_ADD_F32(rsum, HVX_OP_MUL_F32(x_sf, y_sf)); - } - - if (nloe) { - HVX_Vector x_sf = vx[i]; - HVX_Vector y_sf = vy[i]; - - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); - x_sf = Q6_V_vand_QV(bmask, x_sf); - y_sf = Q6_V_vand_QV(bmask, y_sf); - - rsum = HVX_OP_ADD_F32(rsum, HVX_OP_MUL_F32(x_sf, y_sf)); - } - - rsum = hvx_vec_reduce_sum_f32(rsum); - hvx_vec_store_u(&s[0], 4, rsum); -} - -#undef HVX_OP_ADD_F32 -#undef HVX_OP_MUL_F32 - -static inline void vec_dot_f16_f16_aa_1x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy) { - const HVX_Vector * restrict x = (const HVX_Vector *) vx; - const HVX_Vector * restrict y = (const HVX_Vector *) vy; - - uint32_t nvec = n / VLEN_FP16; // num full fp16 hvx vectors - uint32_t nloe = n % VLEN_FP16; // leftover elements - - HVX_VectorPair rsum_p = Q6_W_vzero(); - - uint32_t i = 0; - - #pragma unroll(4) - for (i = 0; i < nvec; i++) { - rsum_p = hvx_vec_mpyacc_f32_f16(rsum_p, x[i], y[i]); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); - HVX_Vector x_hf = Q6_V_vand_QV(bmask, x[i]); - HVX_Vector y_hf = Q6_V_vand_QV(bmask, y[i]); - rsum_p = hvx_vec_mpyacc_f32_f16(rsum_p, x_hf, y_hf); - } - - HVX_Vector rsum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum_p), Q6_V_hi_W(rsum_p))); - hvx_vec_store_u(s, 4, hvx_vec_reduce_sum_f32(rsum)); -} - -static inline void vec_dot_f16_f16_aa_2x1(const uint32_t n, float * restrict s0, - const void * restrict vx0, const void * restrict vx1, - const void * restrict vy0) { - const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; - const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; - const HVX_Vector * restrict y = (const HVX_Vector *) vy0; - - uint32_t nvec = n / VLEN_FP16; - uint32_t nloe = n % VLEN_FP16; - - HVX_VectorPair rsum0_p = Q6_W_vzero(); - HVX_VectorPair rsum1_p = Q6_W_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - HVX_Vector y_hf = y[i]; - rsum0_p = hvx_vec_mpyacc_f32_f16(rsum0_p, x0[i], y_hf); - rsum1_p = hvx_vec_mpyacc_f32_f16(rsum1_p, x1[i], y_hf); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); - HVX_Vector y_hf = Q6_V_vand_QV(bmask, y[i]); - HVX_Vector x0_hf = Q6_V_vand_QV(bmask, x0[i]); - HVX_Vector x1_hf = Q6_V_vand_QV(bmask, x1[i]); - rsum0_p = hvx_vec_mpyacc_f32_f16(rsum0_p, x0_hf, y_hf); - rsum1_p = hvx_vec_mpyacc_f32_f16(rsum1_p, x1_hf, y_hf); - } - - HVX_Vector rsum0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum0_p), Q6_V_hi_W(rsum0_p))); - HVX_Vector rsum1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum1_p), Q6_V_hi_W(rsum1_p))); - HVX_Vector rsum = hvx_vec_reduce_sum_f32x2(rsum0, rsum1); - hvx_vec_store_u(s0, 8, rsum); -} - -static inline void vec_dot_f16_f16_aa_2x2(const uint32_t n, float * restrict s0, float * restrict s1, - const void * restrict vx0, const void * restrict vx1, - const void * restrict vy0, const void * restrict vy1) { - const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; - const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; - const HVX_Vector * restrict y0 = (const HVX_Vector *) vy0; - const HVX_Vector * restrict y1 = (const HVX_Vector *) vy1; - - uint32_t nvec = n / VLEN_FP16; - uint32_t nloe = n % VLEN_FP16; - - // Row sums (sf) - 4 accumulators for 2x2 tile - HVX_VectorPair r0_c0_sum_p = Q6_W_vzero(); - HVX_VectorPair r0_c1_sum_p = Q6_W_vzero(); - HVX_VectorPair r1_c0_sum_p = Q6_W_vzero(); - HVX_VectorPair r1_c1_sum_p = Q6_W_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - HVX_Vector r0_hf = x0[i]; - HVX_Vector r1_hf = x1[i]; - HVX_Vector c0_hf = y0[i]; - HVX_Vector c1_hf = y1[i]; - - // Compute 4 dot products: r0xc0, r0xc1, r1xc0, r1xc1 - r0_c0_sum_p = hvx_vec_mpyacc_f32_f16(r0_c0_sum_p, r0_hf, c0_hf); - r0_c1_sum_p = hvx_vec_mpyacc_f32_f16(r0_c1_sum_p, r0_hf, c1_hf); - r1_c0_sum_p = hvx_vec_mpyacc_f32_f16(r1_c0_sum_p, r1_hf, c0_hf); - r1_c1_sum_p = hvx_vec_mpyacc_f32_f16(r1_c1_sum_p, r1_hf, c1_hf); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); - - HVX_Vector r0_hf = Q6_V_vand_QV(bmask, x0[i]); - HVX_Vector r1_hf = Q6_V_vand_QV(bmask, x1[i]); - HVX_Vector c0_hf = Q6_V_vand_QV(bmask, y0[i]); - HVX_Vector c1_hf = Q6_V_vand_QV(bmask, y1[i]); - - r0_c0_sum_p = hvx_vec_mpyacc_f32_f16(r0_c0_sum_p, r0_hf, c0_hf); - r0_c1_sum_p = hvx_vec_mpyacc_f32_f16(r0_c1_sum_p, r0_hf, c1_hf); - r1_c0_sum_p = hvx_vec_mpyacc_f32_f16(r1_c0_sum_p, r1_hf, c0_hf); - r1_c1_sum_p = hvx_vec_mpyacc_f32_f16(r1_c1_sum_p, r1_hf, c1_hf); - } - - HVX_Vector r0_c0_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r0_c0_sum_p), Q6_V_hi_W(r0_c0_sum_p))); - HVX_Vector r0_c1_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r0_c1_sum_p), Q6_V_hi_W(r0_c1_sum_p))); - HVX_Vector r1_c0_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r1_c0_sum_p), Q6_V_hi_W(r1_c0_sum_p))); - HVX_Vector r1_c1_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r1_c1_sum_p), Q6_V_hi_W(r1_c1_sum_p))); - - // Reduce and store results - HVX_Vector r0_r1_c0_sum = hvx_vec_reduce_sum_f32x2(r0_c0_sum, r1_c0_sum); - HVX_Vector r0_r1_c1_sum = hvx_vec_reduce_sum_f32x2(r0_c1_sum, r1_c1_sum); - - hvx_vec_store_u(&s0[0], 8, r0_r1_c0_sum); // row0,col0 row1,col0 - hvx_vec_store_u(&s1[0], 8, r0_r1_c1_sum); // row0,col1 row1,col1 -} - -static inline void vec_dot_f16_f16_uu_1x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy) { - const HVX_UVector * restrict x = (const HVX_UVector *) vx; - const HVX_UVector * restrict y = (const HVX_UVector *) vy; - - uint32_t nvec = n / VLEN_FP16; // num full fp16 hvx vectors - uint32_t nloe = n % VLEN_FP16; // leftover elements - - HVX_Vector rsum = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(4) - for (i = 0; i < nvec; i++) { - HVX_VectorPair xy_qf = Q6_Wqf32_vmpy_VhfVhf(x[i], y[i]); - rsum = Q6_Vqf32_vadd_Vqf32Vqf32(rsum, Q6_Vqf32_vadd_Vqf32Vqf32(Q6_V_lo_W(xy_qf), Q6_V_hi_W(xy_qf))); - } - - if (nloe) { - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); - HVX_Vector x_hf = Q6_V_vand_QV(bmask, x[i]); - HVX_Vector y_hf = Q6_V_vand_QV(bmask, y[i]); - - HVX_VectorPair xy_qf = Q6_Wqf32_vmpy_VhfVhf(x_hf, y_hf); - rsum = Q6_Vqf32_vadd_Vqf32Vqf32(rsum, Q6_Vqf32_vadd_Vqf32Vqf32(Q6_V_lo_W(xy_qf), Q6_V_hi_W(xy_qf))); - } - - rsum = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(rsum)); - hvx_vec_store_u(&s[0], 4, rsum); -} - -static inline void vec_dot_f16_f32_uu_1x1(const uint32_t n, float * restrict s, const void * restrict x, const void * restrict y) { - const HVX_UVector * restrict vx = (const HVX_UVector * restrict) x; - const HVX_UVector * restrict vy = (const HVX_UVector * restrict) y; - - uint32_t nvec = n / VLEN_FP16; // num full fp16 hvx vectors - uint32_t nloe = n % VLEN_FP16; // leftover elements - - const HVX_Vector zero = Q6_V_vzero(); - - HVX_Vector rsum = Q6_V_vzero(); - - uint32_t i = 0; - - #pragma unroll(2) - for (i = 0; i < nvec; i++) { - // Load y (fp32) and convert into fp16 - HVX_Vector y0_qf = Q6_Vqf32_vsub_VsfVsf(vy[i*2+0], zero); // 32 elements - HVX_Vector y1_qf = Q6_Vqf32_vsub_VsfVsf(vy[i*2+1], zero); // 32 elements - HVX_Vector y_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(y1_qf, y0_qf))); - - // Load x (fp16) - HVX_Vector x_hf = vx[i]; - - HVX_VectorPair xy_qf = Q6_Wqf32_vmpy_VhfVhf(x_hf, y_hf); - - rsum = Q6_Vqf32_vadd_Vqf32Vqf32(rsum, Q6_Vqf32_vadd_Vqf32Vqf32(Q6_V_lo_W(xy_qf), Q6_V_hi_W(xy_qf))); - } - - if (nloe) { - // Load y (fp32) and convert into fp16 - HVX_Vector y0_qf = Q6_Vqf32_vsub_VsfVsf(vy[i*2+0], zero); // 32 elements - HVX_Vector y1_qf = Q6_Vqf32_vsub_VsfVsf(vy[i*2+1], zero); // 32 elements - HVX_Vector y_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(y1_qf, y0_qf))); - - // Load x (fp16) - HVX_Vector x_hf = vx[i]; - - // Zero-out unused elements - // Note that we need to clear both x and y because they may contain NANs - HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); - x_hf = Q6_V_vand_QV(bmask, x_hf); - y_hf = Q6_V_vand_QV(bmask, y_hf); - - HVX_VectorPair xy_qf = Q6_Wqf32_vmpy_VhfVhf(x_hf, y_hf); - - rsum = Q6_Vqf32_vadd_Vqf32Vqf32(rsum, Q6_Vqf32_vadd_Vqf32Vqf32(Q6_V_lo_W(xy_qf), Q6_V_hi_W(xy_qf))); - } - - // Convert into fp32 and reduce - rsum = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(rsum)); - hvx_vec_store_u(&s[0], 4, rsum); -} - -static inline void hvx_tensor_add_f32_grid( - const struct htp_tensor * restrict dst, - const struct htp_tensor * restrict src2, - uint32_t start_row, - uint32_t end_row, - uint32_t start_col, - uint32_t end_col, - const struct fastdiv_values * div_ne11_12, - const struct fastdiv_values * div_ne11 -) { - if (start_row >= end_row || start_col >= end_col) return; - const uint32_t nb1 = dst->nb[1]; // row stride in bytes - - const uint32_t ne11 = dst->ne[1]; - const uint32_t ne12 = dst->ne[2]; - const uint32_t ne11_12 = ne11 * ne12; - - const bool is_broadcast1 = (src2->ne[1] == 1); - const bool is_broadcast2 = (src2->ne[2] == 1); - const bool is_broadcast3 = (src2->ne[3] == 1); - - for (uint32_t r = start_row; r < end_row; r++) { - float * dst_row = (float *) ((uint8_t *) dst->data + r * nb1); - - uint32_t i13 = fastdiv(r, div_ne11_12); - uint32_t i12 = fastdiv(r - i13 * ne11_12, div_ne11); - uint32_t i11 = r - i13 * ne11_12 - i12 * ne11; - - uint32_t i23 = is_broadcast3 ? 0 : i13; - uint32_t i22 = is_broadcast2 ? 0 : i12; - uint32_t i21 = is_broadcast1 ? 0 : i11; - - const float * src2_row = (const float *) ((const uint8_t *) src2->data + - i21 * src2->nb[1] + i22 * src2->nb[2] + i23 * src2->nb[3]); - - float * dst_ptr = &dst_row[start_col]; - const float * src2_ptr = &src2_row[start_col]; - int remaining = end_col - start_col; - while (remaining >= 32) { - HVX_Vector v_out = hvx_vmemu(dst_ptr); - HVX_Vector v_z = hvx_vmemu(src2_ptr); - hvx_vmemu(dst_ptr) = hvx_vec_add_f32_f32(v_out, v_z); - dst_ptr += 32; - src2_ptr += 32; - remaining -= 32; - } - if (remaining > 0) { - HVX_Vector v_out = hvx_vmemu(dst_ptr); - HVX_Vector v_z = hvx_vmemu(src2_ptr); - hvx_vec_store_u(dst_ptr, remaining * sizeof(float), hvx_vec_add_f32_f32(v_out, v_z)); - } - } -} - diff --git a/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-float.h b/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-float.h new file mode 100644 index 00000000..605892aa --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-float.h @@ -0,0 +1,382 @@ +#ifndef HVX_MM_KERNELS_FLOAT_H +#define HVX_MM_KERNELS_FLOAT_H + +#include "hvx-utils.h" +#include "htp-tensor.h" + +// Float activation copy/quantization kernels (DDR -> VTCM) + +static inline void quantize_f32_f32_kernel( + const uint8_t * restrict src_data, + uint8_t * restrict dst_data, + uint8_t * restrict tmp_data, + uint32_t ne0, + uint32_t nrows, + size_t src_stride, + size_t dst_stride +) { + (void) tmp_data; + const size_t src_row_size = ne0 * sizeof(float); + for (uint32_t i = 0; i < nrows; ++i) { + hex_l2fetch(src_data, src_row_size, src_stride, 2); + hvx_copy_f32_au(dst_data, src_data, ne0); + + dst_data += dst_stride; + src_data += src_stride; + } +} + +static inline void quantize_f32_f16_kernel( + const uint8_t * restrict src_data, + uint8_t * restrict dst_data, + uint8_t * restrict tmp_data, + uint32_t ne0, + uint32_t nrows, + size_t src_stride, + size_t dst_stride +) { + (void) tmp_data; + const size_t src_row_size = ne0 * sizeof(float); + for (uint32_t i = 0; i < nrows; ++i) { + hex_l2fetch(src_data, src_row_size, src_stride, 2); + hvx_copy_f16_f32_au(dst_data, src_data, ne0); + + dst_data += dst_stride; + src_data += src_stride; + } +} + +static inline void quantize_f16_f16_kernel( + const uint8_t * restrict src_data, + uint8_t * restrict dst_data, + uint8_t * restrict tmp_data, + uint32_t ne0, + uint32_t nrows, + size_t src_stride, + size_t dst_stride +) { + (void) tmp_data; + const size_t src_row_size = ne0 * sizeof(float); + for (uint32_t i = 0; i < nrows; ++i) { + hex_l2fetch(src_data, src_row_size, src_stride, 2); + hvx_copy_f16_au(dst_data, src_data, ne0); + + dst_data += dst_stride; + src_data += src_stride; + } +} + +// Float dot product kernels (HVX) + +#if __HVX_ARCH__ < 79 +#define HVX_OP_ADD_F32(a, b) Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(a, b)) +#define HVX_OP_MUL_F32(a, b) Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(a, b)) +#else +#define HVX_OP_ADD_F32(a, b) Q6_Vsf_vadd_VsfVsf(a, b) +#define HVX_OP_MUL_F32(a, b) Q6_Vsf_vmpy_VsfVsf(a, b) +#endif + +static inline void vec_dot_f32_f32_aa_1x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy) { + const HVX_Vector * restrict x = (const HVX_Vector *) vx; + const HVX_Vector * restrict y = (const HVX_Vector *) vy; + + uint32_t nvec = n / VLEN_FP32; // num full fp32 hvx vectors + uint32_t nloe = n % VLEN_FP32; // leftover elements + + HVX_Vector rsum = Q6_V_vzero(); + + uint32_t i = 0; + + #pragma unroll(4) + for (i = 0; i < nvec; i++) { + HVX_Vector prod = HVX_OP_MUL_F32(x[i], y[i]); + rsum = HVX_OP_ADD_F32(rsum, prod); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); + HVX_Vector x_sf = Q6_V_vand_QV(bmask, x[i]); + HVX_Vector y_sf = Q6_V_vand_QV(bmask, y[i]); + HVX_Vector prod = HVX_OP_MUL_F32(x_sf, y_sf); + rsum = HVX_OP_ADD_F32(rsum, prod); + } + + *s = hvx_vec_get_f32(hvx_vec_reduce_sum_f32(rsum)); +} + +static inline void vec_dot_f32_f32_aa_2x1(const uint32_t n, float * restrict s0, + const void * restrict vx0, const void * restrict vx1, + const void * restrict vy0) { + const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; + const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; + const HVX_Vector * restrict y = (const HVX_Vector *) vy0; + + uint32_t nvec = n / VLEN_FP32; + uint32_t nloe = n % VLEN_FP32; + + HVX_Vector rsum0 = Q6_V_vzero(); + HVX_Vector rsum1 = Q6_V_vzero(); + + uint32_t i = 0; + + #pragma unroll(2) + for (i = 0; i < nvec; i++) { + HVX_Vector y_sf = y[i]; + HVX_Vector prod0 = HVX_OP_MUL_F32(x0[i], y_sf); + HVX_Vector prod1 = HVX_OP_MUL_F32(x1[i], y_sf); + rsum0 = HVX_OP_ADD_F32(rsum0, prod0); + rsum1 = HVX_OP_ADD_F32(rsum1, prod1); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); + HVX_Vector y_sf = Q6_V_vand_QV(bmask, y[i]); + HVX_Vector x0_sf = Q6_V_vand_QV(bmask, x0[i]); + HVX_Vector x1_sf = Q6_V_vand_QV(bmask, x1[i]); + HVX_Vector prod0 = HVX_OP_MUL_F32(x0_sf, y_sf); + HVX_Vector prod1 = HVX_OP_MUL_F32(x1_sf, y_sf); + rsum0 = HVX_OP_ADD_F32(rsum0, prod0); + rsum1 = HVX_OP_ADD_F32(rsum1, prod1); + } + + HVX_Vector rsum = hvx_vec_reduce_sum_f32x2(rsum0, rsum1); + hvx_vec_store_u(s0, 8, rsum); +} + +static inline void vec_dot_f32_f32_aa_2x2(const uint32_t n, float * restrict s0, float * restrict s1, + const void * restrict vx0, const void * restrict vx1, + const void * restrict vy0, const void * restrict vy1) { + const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; + const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; + const HVX_Vector * restrict y0 = (const HVX_Vector *) vy0; + const HVX_Vector * restrict y1 = (const HVX_Vector *) vy1; + + uint32_t nvec = n / VLEN_FP32; + uint32_t nloe = n % VLEN_FP32; + + HVX_Vector r0_c0_sum = Q6_V_vzero(); + HVX_Vector r0_c1_sum = Q6_V_vzero(); + HVX_Vector r1_c0_sum = Q6_V_vzero(); + HVX_Vector r1_c1_sum = Q6_V_vzero(); + + uint32_t i = 0; + + #pragma unroll(2) + for (i = 0; i < nvec; i++) { + HVX_Vector r0_sf = x0[i]; + HVX_Vector r1_sf = x1[i]; + HVX_Vector c0_sf = y0[i]; + HVX_Vector c1_sf = y1[i]; + + r0_c0_sum = HVX_OP_ADD_F32(r0_c0_sum, HVX_OP_MUL_F32(r0_sf, c0_sf)); + r0_c1_sum = HVX_OP_ADD_F32(r0_c1_sum, HVX_OP_MUL_F32(r0_sf, c1_sf)); + r1_c0_sum = HVX_OP_ADD_F32(r1_c0_sum, HVX_OP_MUL_F32(r1_sf, c0_sf)); + r1_c1_sum = HVX_OP_ADD_F32(r1_c1_sum, HVX_OP_MUL_F32(r1_sf, c1_sf)); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 4); + + HVX_Vector r0_sf = Q6_V_vand_QV(bmask, x0[i]); + HVX_Vector r1_sf = Q6_V_vand_QV(bmask, x1[i]); + HVX_Vector c0_sf = Q6_V_vand_QV(bmask, y0[i]); + HVX_Vector c1_sf = Q6_V_vand_QV(bmask, y1[i]); + + r0_c0_sum = HVX_OP_ADD_F32(r0_c0_sum, HVX_OP_MUL_F32(r0_sf, c0_sf)); + r0_c1_sum = HVX_OP_ADD_F32(r0_c1_sum, HVX_OP_MUL_F32(r0_sf, c1_sf)); + r1_c0_sum = HVX_OP_ADD_F32(r1_c0_sum, HVX_OP_MUL_F32(r1_sf, c0_sf)); + r1_c1_sum = HVX_OP_ADD_F32(r1_c1_sum, HVX_OP_MUL_F32(r1_sf, c1_sf)); + } + + // Reduce and store results + HVX_Vector r0_r1_c0_sum = hvx_vec_reduce_sum_f32x2(r0_c0_sum, r1_c0_sum); + HVX_Vector r0_r1_c1_sum = hvx_vec_reduce_sum_f32x2(r0_c1_sum, r1_c1_sum); + + hvx_vec_store_u(s0, 8, r0_r1_c0_sum); + hvx_vec_store_u(s1, 8, r0_r1_c1_sum); +} + +#undef HVX_OP_ADD_F32 +#undef HVX_OP_MUL_F32 + +static inline void vec_dot_f16_f16_aa_1x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy) { + const HVX_Vector * restrict x = (const HVX_Vector *) vx; + const HVX_Vector * restrict y = (const HVX_Vector *) vy; + + uint32_t nvec = n / VLEN_FP16; // num full fp16 hvx vectors + uint32_t nloe = n % VLEN_FP16; // leftover elements + + HVX_VectorPair rsum_p = Q6_W_vzero(); + + uint32_t i = 0; + + #pragma unroll(4) + for (i = 0; i < nvec; i++) { + rsum_p = hvx_vec_mpyacc_f32_f16(rsum_p, x[i], y[i]); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); + HVX_Vector x_hf = Q6_V_vand_QV(bmask, x[i]); + HVX_Vector y_hf = Q6_V_vand_QV(bmask, y[i]); + rsum_p = hvx_vec_mpyacc_f32_f16(rsum_p, x_hf, y_hf); + } + + HVX_Vector rsum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum_p), Q6_V_hi_W(rsum_p))); + hvx_vec_store_u(s, 4, hvx_vec_reduce_sum_f32(rsum)); +} + +static inline void vec_dot_f16_f16_aa_2x1(const uint32_t n, float * restrict s0, + const void * restrict vx0, const void * restrict vx1, + const void * restrict vy0) { + const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; + const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; + const HVX_Vector * restrict y = (const HVX_Vector *) vy0; + + uint32_t nvec = n / VLEN_FP16; + uint32_t nloe = n % VLEN_FP16; + + HVX_VectorPair rsum0_p = Q6_W_vzero(); + HVX_VectorPair rsum1_p = Q6_W_vzero(); + + uint32_t i = 0; + + #pragma unroll(2) + for (i = 0; i < nvec; i++) { + HVX_Vector y_hf = y[i]; + rsum0_p = hvx_vec_mpyacc_f32_f16(rsum0_p, x0[i], y_hf); + rsum1_p = hvx_vec_mpyacc_f32_f16(rsum1_p, x1[i], y_hf); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); + HVX_Vector y_hf = Q6_V_vand_QV(bmask, y[i]); + HVX_Vector x0_hf = Q6_V_vand_QV(bmask, x0[i]); + HVX_Vector x1_hf = Q6_V_vand_QV(bmask, x1[i]); + rsum0_p = hvx_vec_mpyacc_f32_f16(rsum0_p, x0_hf, y_hf); + rsum1_p = hvx_vec_mpyacc_f32_f16(rsum1_p, x1_hf, y_hf); + } + + HVX_Vector rsum0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum0_p), Q6_V_hi_W(rsum0_p))); + HVX_Vector rsum1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(rsum1_p), Q6_V_hi_W(rsum1_p))); + HVX_Vector rsum = hvx_vec_reduce_sum_f32x2(rsum0, rsum1); + hvx_vec_store_u(s0, 8, rsum); +} + +static inline void vec_dot_f16_f16_aa_2x2(const uint32_t n, float * restrict s0, float * restrict s1, + const void * restrict vx0, const void * restrict vx1, + const void * restrict vy0, const void * restrict vy1) { + const HVX_Vector * restrict x0 = (const HVX_Vector *) vx0; + const HVX_Vector * restrict x1 = (const HVX_Vector *) vx1; + const HVX_Vector * restrict y0 = (const HVX_Vector *) vy0; + const HVX_Vector * restrict y1 = (const HVX_Vector *) vy1; + + uint32_t nvec = n / VLEN_FP16; + uint32_t nloe = n % VLEN_FP16; + + // Row sums (sf) - 4 accumulators for 2x2 tile + HVX_VectorPair r0_c0_sum_p = Q6_W_vzero(); + HVX_VectorPair r0_c1_sum_p = Q6_W_vzero(); + HVX_VectorPair r1_c0_sum_p = Q6_W_vzero(); + HVX_VectorPair r1_c1_sum_p = Q6_W_vzero(); + + uint32_t i = 0; + + #pragma unroll(2) + for (i = 0; i < nvec; i++) { + HVX_Vector r0_hf = x0[i]; + HVX_Vector r1_hf = x1[i]; + HVX_Vector c0_hf = y0[i]; + HVX_Vector c1_hf = y1[i]; + + // Compute 4 dot products: r0xc0, r0xc1, r1xc0, r1xc1 + r0_c0_sum_p = hvx_vec_mpyacc_f32_f16(r0_c0_sum_p, r0_hf, c0_hf); + r0_c1_sum_p = hvx_vec_mpyacc_f32_f16(r0_c1_sum_p, r0_hf, c1_hf); + r1_c0_sum_p = hvx_vec_mpyacc_f32_f16(r1_c0_sum_p, r1_hf, c0_hf); + r1_c1_sum_p = hvx_vec_mpyacc_f32_f16(r1_c1_sum_p, r1_hf, c1_hf); + } + + if (nloe) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * 2); + + HVX_Vector r0_hf = Q6_V_vand_QV(bmask, x0[i]); + HVX_Vector r1_hf = Q6_V_vand_QV(bmask, x1[i]); + HVX_Vector c0_hf = Q6_V_vand_QV(bmask, y0[i]); + HVX_Vector c1_hf = Q6_V_vand_QV(bmask, y1[i]); + + r0_c0_sum_p = hvx_vec_mpyacc_f32_f16(r0_c0_sum_p, r0_hf, c0_hf); + r0_c1_sum_p = hvx_vec_mpyacc_f32_f16(r0_c1_sum_p, r0_hf, c1_hf); + r1_c0_sum_p = hvx_vec_mpyacc_f32_f16(r1_c0_sum_p, r1_hf, c0_hf); + r1_c1_sum_p = hvx_vec_mpyacc_f32_f16(r1_c1_sum_p, r1_hf, c1_hf); + } + + HVX_Vector r0_c0_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r0_c0_sum_p), Q6_V_hi_W(r0_c0_sum_p))); + HVX_Vector r0_c1_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r0_c1_sum_p), Q6_V_hi_W(r0_c1_sum_p))); + HVX_Vector r1_c0_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r1_c0_sum_p), Q6_V_hi_W(r1_c0_sum_p))); + HVX_Vector r1_c1_sum = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(Q6_V_lo_W(r1_c1_sum_p), Q6_V_hi_W(r1_c1_sum_p))); + + // Reduce and store results + HVX_Vector r0_r1_c0_sum = hvx_vec_reduce_sum_f32x2(r0_c0_sum, r1_c0_sum); + HVX_Vector r0_r1_c1_sum = hvx_vec_reduce_sum_f32x2(r0_c1_sum, r1_c1_sum); + + hvx_vec_store_u(&s0[0], 8, r0_r1_c0_sum); // row0,col0 row1,col0 + hvx_vec_store_u(&s1[0], 8, r0_r1_c1_sum); // row0,col1 row1,col1 +} + + + +static inline void hvx_tensor_add_f32_grid( + const struct htp_tensor * restrict dst, + const struct htp_tensor * restrict src2, + uint32_t start_row, + uint32_t end_row, + uint32_t start_col, + uint32_t end_col, + const struct fastdiv_values * div_ne11_12, + const struct fastdiv_values * div_ne11 +) { + if (start_row >= end_row || start_col >= end_col) return; + const uint32_t nb1 = dst->nb[1]; // row stride in bytes + + const uint32_t ne11 = dst->ne[1]; + const uint32_t ne12 = dst->ne[2]; + const uint32_t ne11_12 = ne11 * ne12; + + const bool is_broadcast1 = (src2->ne[1] == 1); + const bool is_broadcast2 = (src2->ne[2] == 1); + const bool is_broadcast3 = (src2->ne[3] == 1); + + for (uint32_t r = start_row; r < end_row; r++) { + float * dst_row = (float *) ((uint8_t *) dst->data + (size_t) r * nb1); + + uint32_t i13 = fastdiv(r, div_ne11_12); + uint32_t i12 = fastdiv(r - i13 * ne11_12, div_ne11); + uint32_t i11 = r - i13 * ne11_12 - i12 * ne11; + + uint32_t i23 = is_broadcast3 ? 0 : i13; + uint32_t i22 = is_broadcast2 ? 0 : i12; + uint32_t i21 = is_broadcast1 ? 0 : i11; + + const float * src2_row = (const float *) ((const uint8_t *) src2->data + + (size_t) i21 * src2->nb[1] + (size_t) i22 * src2->nb[2] + (size_t) i23 * src2->nb[3]); + + float * dst_ptr = &dst_row[start_col]; + const float * src2_ptr = &src2_row[start_col]; + int remaining = end_col - start_col; + while (remaining >= 32) { + HVX_Vector v_out = hvx_vmemu(dst_ptr); + HVX_Vector v_z = hvx_vmemu(src2_ptr); + hvx_vmemu(dst_ptr) = hvx_vec_add_f32_f32(v_out, v_z); + dst_ptr += 32; + src2_ptr += 32; + remaining -= 32; + } + if (remaining > 0) { + HVX_Vector v_out = hvx_vmemu(dst_ptr); + HVX_Vector v_z = hvx_vmemu(src2_ptr); + hvx_vec_store_u(dst_ptr, remaining * sizeof(float), hvx_vec_add_f32_f32(v_out, v_z)); + } + } +} + +#endif // HVX_MM_KERNELS_FLOAT_H diff --git a/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-tiled.h b/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-tiled.h index 40b65aa3..4d6110ff 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-tiled.h +++ b/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-tiled.h @@ -48,22 +48,33 @@ static inline void quantize_block_f32_q8_1_tiled(float * restrict x, uint8_t * r v_sums = Q6_Vw_vadd_VwVw(v_sums, Q6_V_vror_VR(v_sums, 8)); v_sums = Q6_Vw_vadd_VwVw(v_sums, Q6_V_vror_VR(v_sums, 16)); - float vmax0[32] __attribute__((aligned(128))); - float vmax1[32] __attribute__((aligned(128))); - float vmax2[32] __attribute__((aligned(128))); - float vmax3[32] __attribute__((aligned(128))); - int32_t sums[32] __attribute__((aligned(128))); - - hvx_vec_store_u(vmax0, 128, vmax0_sf); - hvx_vec_store_u(vmax1, 128, vmax1_sf); - hvx_vec_store_u(vmax2, 128, vmax2_sf); - hvx_vec_store_u(vmax3, 128, vmax3_sf); - hvx_vec_store_u(sums, 128, v_sums); - - float d0 = vmax0[0] / 127.0f; - float d1 = vmax1[0] / 127.0f; - float d2 = vmax2[0] / 127.0f; - float d3 = vmax3[0] / 127.0f; + const HVX_Vector v_inv127 = hvx_vec_splat_f32(1.0f / 127.0f); + HVX_Vector vd0_sf = hvx_vec_mul_f32_f32(vmax0_sf, v_inv127); + HVX_Vector vd1_sf = hvx_vec_mul_f32_f32(vmax1_sf, v_inv127); + HVX_Vector vd2_sf = hvx_vec_mul_f32_f32(vmax2_sf, v_inv127); + HVX_Vector vd3_sf = hvx_vec_mul_f32_f32(vmax3_sf, v_inv127); + + HVX_Vector v_sums_sf = Q6_Vsf_equals_Vw(v_sums); + HVX_Vector voff0_sf = hvx_vec_mul_f32_f32(vd0_sf, v_sums_sf); + HVX_Vector voff1_sf = hvx_vec_mul_f32_f32(vd1_sf, Q6_V_vror_VR(v_sums_sf, 32)); + HVX_Vector voff2_sf = hvx_vec_mul_f32_f32(vd2_sf, Q6_V_vror_VR(v_sums_sf, 64)); + HVX_Vector voff3_sf = hvx_vec_mul_f32_f32(vd3_sf, Q6_V_vror_VR(v_sums_sf, 96)); + + HVX_Vector voff01_hf = hvx_vec_f32_to_f16(voff0_sf, voff1_sf); + HVX_Vector voff23_hf = hvx_vec_f32_to_f16(voff2_sf, voff3_sf); + + HVX_Vector r_scale[4] = { + hvx_vec_repl_f16(vd01_hf), + hvx_vec_repl_f16(Q6_V_vror_VR(vd01_hf, 64)), + hvx_vec_repl_f16(vd23_hf), + hvx_vec_repl_f16(Q6_V_vror_VR(vd23_hf, 64)), + }; + HVX_Vector r_offset[4] = { + hvx_vec_repl_f16(voff01_hf), + hvx_vec_repl_f16(Q6_V_vror_VR(voff01_hf, 64)), + hvx_vec_repl_f16(voff23_hf), + hvx_vec_repl_f16(Q6_V_vror_VR(voff23_hf, 64)), + }; static const uint8_t __attribute__((aligned(128))) repl[128] = { 0x00, 0x00, 0x00, 0x00, 0x04, 0x04, 0x04, 0x04, 0x08, 0x08, 0x08, 0x08, 0x04, 0x04, 0x04, 0x04, @@ -89,24 +100,6 @@ static inline void quantize_block_f32_q8_1_tiled(float * restrict x, uint8_t * r HVX_Vector r6 = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act, 24), v_repl_ctrl); HVX_Vector r7 = Q6_V_vdelta_VV(Q6_V_vror_VR(v_act, 28), v_repl_ctrl); - __fp16 scale_h, offset_h; - if (b == 0) { - scale_h = (__fp16) d0; - offset_h = (__fp16) (sums[0] * d0); - } else if (b == 1) { - scale_h = (__fp16) d1; - offset_h = (__fp16) (sums[8] * d1); - } else if (b == 2) { - scale_h = (__fp16) d2; - offset_h = (__fp16) (sums[16] * d2); - } else { - scale_h = (__fp16) d3; - offset_h = (__fp16) (sums[24] * d3); - } - - HVX_Vector r_scale = Q6_Vh_vsplat_R(*(int16_t *)&scale_h); - HVX_Vector r_offset = Q6_Vh_vsplat_R(*(int16_t *)&offset_h); - HVX_Vector * restrict dst = (HVX_Vector *) (y_block + b * 1280); dst[0] = r0; dst[1] = r1; @@ -116,8 +109,8 @@ static inline void quantize_block_f32_q8_1_tiled(float * restrict x, uint8_t * r dst[5] = r5; dst[6] = r6; dst[7] = r7; - dst[8] = r_scale; - dst[9] = r_offset; + dst[8] = r_scale[b]; + dst[9] = r_offset[b]; } } @@ -378,6 +371,74 @@ static inline HVX_VectorPair accum_q8_0_32x2( return Q6_W_vcombine_VV(v_sum1, v_sum0); } +// Q6_K weights are stored unsigned (0..63), see HTP_MM_WEIGHT_TILE_SIZE_Q6_K. Unpack k-group g of a tile to signed bytes (q - 32) +static inline HVX_Vector unpack_q6_k_group(const HVX_Vector * restrict vptr, int g, HVX_Vector mask_0f, HVX_Vector mask_03, HVX_Vector i32) { + HVX_Vector v_lo = (g & 1) ? Q6_Vub_vlsr_VubR(vptr[g >> 1], 4) : Q6_V_vand_VV(vptr[g >> 1], mask_0f); + HVX_Vector v_hi = (g & 3) ? Q6_Vub_vlsr_VubR(vptr[4 + (g >> 2)], 2 * (g & 3)) : vptr[4 + (g >> 2)]; + HVX_Vector v_q = Q6_V_vor_VV(v_lo, Q6_Vw_vasl_VwR(Q6_V_vand_VV(v_hi, mask_03), 4)); + return Q6_Vb_vsub_VbVb(v_q, i32); +} + +// k 0..15 and k 16..31 of a Q6_K tile have different scales: lo half of the pair sums k 0..15, hi half sums k 16..31 +static inline HVX_VectorPair accum_q6_k_32x1( + const HVX_Vector * restrict vptr, + const HVX_Vector * restrict v_act, + HVX_Vector i32 +) { + HVX_Vector v_sum_lo = Q6_V_vzero(); + HVX_Vector v_sum_hi = Q6_V_vzero(); + HVX_Vector mask_0f = Q6_Vb_vsplat_R(0x0F); + HVX_Vector mask_03 = Q6_Vb_vsplat_R(0x03); + + #pragma unroll + for (int g = 0; g < 4; g++) { + HVX_Vector v_W_lo = unpack_q6_k_group(vptr, g, mask_0f, mask_03, i32); + HVX_Vector v_W_hi = unpack_q6_k_group(vptr, g + 4, mask_0f, mask_03, i32); + v_sum_lo = Q6_Vw_vrmpyacc_VwVbVb(v_sum_lo, v_W_lo, v_act[g]); + v_sum_hi = Q6_Vw_vrmpyacc_VwVbVb(v_sum_hi, v_W_hi, v_act[g + 4]); + } + + return Q6_W_vcombine_VV(v_sum_hi, v_sum_lo); +} + +static inline void accum_q6_k_32x2( + const HVX_Vector * restrict vptr, + const HVX_Vector * restrict v_act0, + const HVX_Vector * restrict v_act1, + HVX_Vector i32, + HVX_VectorPair * v_sums0, + HVX_VectorPair * v_sums1 +) { + HVX_Vector v_sum0_lo = Q6_V_vzero(); + HVX_Vector v_sum0_hi = Q6_V_vzero(); + HVX_Vector v_sum1_lo = Q6_V_vzero(); + HVX_Vector v_sum1_hi = Q6_V_vzero(); + HVX_Vector mask_0f = Q6_Vb_vsplat_R(0x0F); + HVX_Vector mask_03 = Q6_Vb_vsplat_R(0x03); + + #pragma unroll + for (int g = 0; g < 4; g++) { + HVX_Vector v_W_lo = unpack_q6_k_group(vptr, g, mask_0f, mask_03, i32); + HVX_Vector v_W_hi = unpack_q6_k_group(vptr, g + 4, mask_0f, mask_03, i32); + v_sum0_lo = Q6_Vw_vrmpyacc_VwVbVb(v_sum0_lo, v_W_lo, v_act0[g]); + v_sum0_hi = Q6_Vw_vrmpyacc_VwVbVb(v_sum0_hi, v_W_hi, v_act0[g + 4]); + v_sum1_lo = Q6_Vw_vrmpyacc_VwVbVb(v_sum1_lo, v_W_lo, v_act1[g]); + v_sum1_hi = Q6_Vw_vrmpyacc_VwVbVb(v_sum1_hi, v_W_hi, v_act1[g + 4]); + } + + *v_sums0 = Q6_W_vcombine_VV(v_sum0_hi, v_sum0_lo); + *v_sums1 = Q6_W_vcombine_VV(v_sum1_hi, v_sum1_lo); +} + +// scale the two half sums with the per-row tile scales (v_scale_w = vptr[6]) and the activation scale +static inline HVX_Vector scale_q6_k_32x1(HVX_VectorPair v_sums, HVX_Vector v_scale_w, HVX_Vector v_scale_a) { + HVX_Vector v_scale_lo = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w, v_scale_a); + HVX_Vector v_scale_hi = hvx_vec_mul_f16_f16_to_f32_lower32(Q6_V_vror_VR(v_scale_w, 64), v_scale_a); + HVX_Vector v_lo = hvx_vec_mul_f32_f32(Q6_Vsf_equals_Vw(Q6_V_lo_W(v_sums)), v_scale_lo); + HVX_Vector v_hi = hvx_vec_mul_f32_f32(Q6_Vsf_equals_Vw(Q6_V_hi_W(v_sums)), v_scale_hi); + return hvx_vec_add_f32_f32(v_lo, v_hi); +} + static void tiled_vec_dot_q4_0_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { const uint8_t * restrict tile_ptr = vx; const uint8_t * restrict y_q = vy; @@ -418,51 +479,7 @@ static void tiled_vec_dot_q4_0_32x2(const uint32_t n, float * restrict s0, float HVX_Vector i8 = Q6_Vb_vsplat_R(8); uint32_t n_k_tiles = n / 32; - uint32_t kt = 0; - for (; kt + 1 < n_k_tiles; kt += 2) { - const HVX_Vector * restrict vptr0 = (const HVX_Vector *) (tile_ptr + (kt + 0) * 640); - const HVX_Vector * restrict v_act0_0 = (const HVX_Vector *) (y0_q + (kt + 0) * 1152); - const HVX_Vector * restrict v_act1_0 = (const HVX_Vector *) (y1_q + (kt + 0) * 1152); - - const HVX_Vector * restrict vptr1 = (const HVX_Vector *) (tile_ptr + (kt + 1) * 640); - const HVX_Vector * restrict v_act0_1 = (const HVX_Vector *) (y0_q + (kt + 1) * 1152); - const HVX_Vector * restrict v_act1_1 = (const HVX_Vector *) (y1_q + (kt + 1) * 1152); - - HVX_VectorPair v_sums0 = accum_4bit_32x2(vptr0, v_act0_0, v_act1_0, i8); - HVX_VectorPair v_sums1 = accum_4bit_32x2(vptr1, v_act0_1, v_act1_1, i8); - - HVX_Vector v_sum_c0_0 = Q6_V_lo_W(v_sums0); - HVX_Vector v_sum_c1_0 = Q6_V_hi_W(v_sums0); - HVX_Vector v_sum_c0_1 = Q6_V_lo_W(v_sums1); - HVX_Vector v_sum_c1_1 = Q6_V_hi_W(v_sums1); - - HVX_Vector v_sum_sf_c0_0 = Q6_Vsf_equals_Vw(v_sum_c0_0); - HVX_Vector v_sum_sf_c1_0 = Q6_Vsf_equals_Vw(v_sum_c1_0); - HVX_Vector v_sum_sf_c0_1 = Q6_Vsf_equals_Vw(v_sum_c0_1); - HVX_Vector v_sum_sf_c1_1 = Q6_Vsf_equals_Vw(v_sum_c1_1); - - HVX_Vector v_scale_w0 = vptr0[4]; - HVX_Vector v_scale_w1 = vptr1[4]; - HVX_Vector v_scale_a_c0_0 = v_act0_0[8]; - HVX_Vector v_scale_a_c1_0 = v_act1_0[8]; - HVX_Vector v_scale_a_c0_1 = v_act0_1[8]; - HVX_Vector v_scale_a_c1_1 = v_act1_1[8]; - - HVX_Vector v_scale_comb_c0_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c0_0); - HVX_Vector v_scale_comb_c1_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c1_0); - HVX_Vector v_scale_comb_c0_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c0_1); - HVX_Vector v_scale_comb_c1_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c1_1); - - HVX_Vector v_sum_scaled_c0_0 = hvx_vec_mul_f32_f32(v_sum_sf_c0_0, v_scale_comb_c0_0); - HVX_Vector v_sum_scaled_c1_0 = hvx_vec_mul_f32_f32(v_sum_sf_c1_0, v_scale_comb_c1_0); - HVX_Vector v_sum_scaled_c0_1 = hvx_vec_mul_f32_f32(v_sum_sf_c0_1, v_scale_comb_c0_1); - HVX_Vector v_sum_scaled_c1_1 = hvx_vec_mul_f32_f32(v_sum_sf_c1_1, v_scale_comb_c1_1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vec_add_f32_f32(v_sum_scaled_c0_0, v_sum_scaled_c0_1)); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vec_add_f32_f32(v_sum_scaled_c1_0, v_sum_scaled_c1_1)); - } - - for (; kt < n_k_tiles; kt++) { + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1152); const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1152); @@ -547,76 +564,7 @@ static void tiled_vec_dot_q4_1_32x2(const uint32_t n, float * restrict s0, float HVX_Vector v_sum_float_c1 = Q6_V_vzero(); uint32_t n_k_tiles = n / 32; - uint32_t kt = 0; - for (; kt + 1 < n_k_tiles; kt += 2) { - const HVX_Vector * restrict vptr0 = (const HVX_Vector *) (tile_ptr + (kt + 0) * 640); - const HVX_Vector * restrict v_act0_0 = (const HVX_Vector *) (y0_q + (kt + 0) * 1280); - const HVX_Vector * restrict v_act1_0 = (const HVX_Vector *) (y1_q + (kt + 0) * 1280); - - const HVX_Vector * restrict vptr1 = (const HVX_Vector *) (tile_ptr + (kt + 1) * 640); - const HVX_Vector * restrict v_act0_1 = (const HVX_Vector *) (y0_q + (kt + 1) * 1280); - const HVX_Vector * restrict v_act1_1 = (const HVX_Vector *) (y1_q + (kt + 1) * 1280); - - HVX_VectorPair v_sums0 = accum_4bit_32x2(vptr0, v_act0_0, v_act1_0, Q6_V_vzero()); - HVX_VectorPair v_sums1 = accum_4bit_32x2(vptr1, v_act0_1, v_act1_1, Q6_V_vzero()); - - HVX_Vector v_sum_c0_0 = Q6_V_lo_W(v_sums0); - HVX_Vector v_sum_c1_0 = Q6_V_hi_W(v_sums0); - HVX_Vector v_sum_c0_1 = Q6_V_lo_W(v_sums1); - HVX_Vector v_sum_c1_1 = Q6_V_hi_W(v_sums1); - - HVX_Vector v_sum_sf_c0_0 = Q6_Vsf_equals_Vw(v_sum_c0_0); - HVX_Vector v_sum_sf_c1_0 = Q6_Vsf_equals_Vw(v_sum_c1_0); - HVX_Vector v_sum_sf_c0_1 = Q6_Vsf_equals_Vw(v_sum_c0_1); - HVX_Vector v_sum_sf_c1_1 = Q6_Vsf_equals_Vw(v_sum_c1_1); - - HVX_Vector v_scale_offset0 = vptr0[4]; - HVX_VectorPair p_deal0 = Q6_W_vdeal_VVR(v_scale_offset0, v_scale_offset0, -2); - HVX_Vector v_scale0 = Q6_V_lo_W(p_deal0); - HVX_Vector v_offset0 = Q6_V_hi_W(p_deal0); - - HVX_Vector v_scale_offset1 = vptr1[4]; - HVX_VectorPair p_deal1 = Q6_W_vdeal_VVR(v_scale_offset1, v_scale_offset1, -2); - HVX_Vector v_scale1 = Q6_V_lo_W(p_deal1); - HVX_Vector v_offset1 = Q6_V_hi_W(p_deal1); - - HVX_Vector v_scale_a_c0_0 = v_act0_0[8]; - HVX_Vector v_sum_a_c0_0 = v_act0_0[9]; - HVX_Vector v_scale_a_c1_0 = v_act1_0[8]; - HVX_Vector v_sum_a_c1_0 = v_act1_0[9]; - - HVX_Vector v_scale_a_c0_1 = v_act0_1[8]; - HVX_Vector v_sum_a_c0_1 = v_act0_1[9]; - HVX_Vector v_scale_a_c1_1 = v_act1_1[8]; - HVX_Vector v_sum_a_c1_1 = v_act1_1[9]; - - HVX_Vector v_scale_comb_c0_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale0, v_scale_a_c0_0); - HVX_Vector v_offset_comb_c0_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset0, v_sum_a_c0_0); - HVX_Vector v_scale_comb_c1_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale0, v_scale_a_c1_0); - HVX_Vector v_offset_comb_c1_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset0, v_sum_a_c1_0); - - HVX_Vector v_scale_comb_c0_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale1, v_scale_a_c0_1); - HVX_Vector v_offset_comb_c0_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset1, v_sum_a_c0_1); - HVX_Vector v_scale_comb_c1_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale1, v_scale_a_c1_1); - HVX_Vector v_offset_comb_c1_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_offset1, v_sum_a_c1_1); - - HVX_Vector v_scaled_dot_c0_0 = hvx_vec_mul_f32_f32(v_sum_sf_c0_0, v_scale_comb_c0_0); - HVX_Vector v_sum_scaled_c0_0 = hvx_vec_add_f32_f32(v_scaled_dot_c0_0, v_offset_comb_c0_0); - - HVX_Vector v_scaled_dot_c1_0 = hvx_vec_mul_f32_f32(v_sum_sf_c1_0, v_scale_comb_c1_0); - HVX_Vector v_sum_scaled_c1_0 = hvx_vec_add_f32_f32(v_scaled_dot_c1_0, v_offset_comb_c1_0); - - HVX_Vector v_scaled_dot_c0_1 = hvx_vec_mul_f32_f32(v_sum_sf_c0_1, v_scale_comb_c0_1); - HVX_Vector v_sum_scaled_c0_1 = hvx_vec_add_f32_f32(v_scaled_dot_c0_1, v_offset_comb_c0_1); - - HVX_Vector v_scaled_dot_c1_1 = hvx_vec_mul_f32_f32(v_sum_sf_c1_1, v_scale_comb_c1_1); - HVX_Vector v_sum_scaled_c1_1 = hvx_vec_add_f32_f32(v_scaled_dot_c1_1, v_offset_comb_c1_1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vec_add_f32_f32(v_sum_scaled_c0_0, v_sum_scaled_c0_1)); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vec_add_f32_f32(v_sum_scaled_c1_0, v_sum_scaled_c1_1)); - } - - for (; kt < n_k_tiles; kt++) { + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1280); const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1280); @@ -703,51 +651,7 @@ static void tiled_vec_dot_q8_0_32x2(const uint32_t n, float * restrict s0, float HVX_Vector v_sum_float_c1 = Q6_V_vzero(); uint32_t n_k_tiles = n / 32; - uint32_t kt = 0; - for (; kt + 1 < n_k_tiles; kt += 2) { - const HVX_Vector * restrict vptr0 = (const HVX_Vector *) (tile_ptr + (kt + 0) * 1152); - const HVX_Vector * restrict v_act0_0 = (const HVX_Vector *) (y0_q + (kt + 0) * 1152); - const HVX_Vector * restrict v_act1_0 = (const HVX_Vector *) (y1_q + (kt + 0) * 1152); - - const HVX_Vector * restrict vptr1 = (const HVX_Vector *) (tile_ptr + (kt + 1) * 1152); - const HVX_Vector * restrict v_act0_1 = (const HVX_Vector *) (y0_q + (kt + 1) * 1152); - const HVX_Vector * restrict v_act1_1 = (const HVX_Vector *) (y1_q + (kt + 1) * 1152); - - HVX_VectorPair v_sums0 = accum_q8_0_32x2(vptr0, v_act0_0, v_act1_0); - HVX_VectorPair v_sums1 = accum_q8_0_32x2(vptr1, v_act0_1, v_act1_1); - - HVX_Vector v_sum_c0_0 = Q6_V_lo_W(v_sums0); - HVX_Vector v_sum_c1_0 = Q6_V_hi_W(v_sums0); - HVX_Vector v_sum_c0_1 = Q6_V_lo_W(v_sums1); - HVX_Vector v_sum_c1_1 = Q6_V_hi_W(v_sums1); - - HVX_Vector v_sum_sf_c0_0 = Q6_Vsf_equals_Vw(v_sum_c0_0); - HVX_Vector v_sum_sf_c1_0 = Q6_Vsf_equals_Vw(v_sum_c1_0); - HVX_Vector v_sum_sf_c0_1 = Q6_Vsf_equals_Vw(v_sum_c0_1); - HVX_Vector v_sum_sf_c1_1 = Q6_Vsf_equals_Vw(v_sum_c1_1); - - HVX_Vector v_scale_w0 = vptr0[8]; - HVX_Vector v_scale_w1 = vptr1[8]; - HVX_Vector v_scale_a_c0_0 = v_act0_0[8]; - HVX_Vector v_scale_a_c1_0 = v_act1_0[8]; - HVX_Vector v_scale_a_c0_1 = v_act0_1[8]; - HVX_Vector v_scale_a_c1_1 = v_act1_1[8]; - - HVX_Vector v_scale_comb_c0_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c0_0); - HVX_Vector v_scale_comb_c1_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c1_0); - HVX_Vector v_scale_comb_c0_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c0_1); - HVX_Vector v_scale_comb_c1_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c1_1); - - HVX_Vector v_sum_scaled_c0_0 = hvx_vec_mul_f32_f32(v_sum_sf_c0_0, v_scale_comb_c0_0); - HVX_Vector v_sum_scaled_c1_0 = hvx_vec_mul_f32_f32(v_sum_sf_c1_0, v_scale_comb_c1_0); - HVX_Vector v_sum_scaled_c0_1 = hvx_vec_mul_f32_f32(v_sum_sf_c0_1, v_scale_comb_c0_1); - HVX_Vector v_sum_scaled_c1_1 = hvx_vec_mul_f32_f32(v_sum_sf_c1_1, v_scale_comb_c1_1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vec_add_f32_f32(v_sum_scaled_c0_0, v_sum_scaled_c0_1)); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vec_add_f32_f32(v_sum_scaled_c1_0, v_sum_scaled_c1_1)); - } - - for (; kt < n_k_tiles; kt++) { + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 1152); const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1152); const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1152); @@ -785,6 +689,63 @@ static void tiled_vec_dot_q8_0_32x2(const uint32_t n, float * restrict s0, float } } +static void tiled_vec_dot_q6_k_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { + const uint8_t * restrict tile_ptr = vx; + const uint8_t * restrict y_q = vy; + + HVX_Vector v_sum_float = Q6_V_vzero(); + HVX_Vector i32 = Q6_Vb_vsplat_R(32); + + uint32_t n_k_tiles = n / 32; + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { + const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 896); + const HVX_Vector * restrict v_act = (const HVX_Vector *) (y_q + kt * 1152); + + HVX_VectorPair v_sums = accum_q6_k_32x1(vptr, v_act, i32); + v_sum_float = hvx_vec_add_f32_f32(v_sum_float, scale_q6_k_32x1(v_sums, vptr[6], v_act[8])); + } + + if (sz) { + hvx_vec_store_u(s, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float, hvx_vmemu(sz))); + } else { + hvx_vec_store_u(s, valid_rows * sizeof(float), v_sum_float); + } +} + +static void tiled_vec_dot_q6_k_32x2(const uint32_t n, float * restrict s0, float * restrict s1, const void * restrict vx, const void * restrict vy0, const void * restrict vy1, uint32_t valid_rows, const float * restrict sz0, const float * restrict sz1) { + const uint8_t * restrict tile_ptr = vx; + const uint8_t * restrict y0_q = vy0; + const uint8_t * restrict y1_q = vy1; + + HVX_Vector v_sum_float_c0 = Q6_V_vzero(); + HVX_Vector v_sum_float_c1 = Q6_V_vzero(); + HVX_Vector i32 = Q6_Vb_vsplat_R(32); + + uint32_t n_k_tiles = n / 32; + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { + const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 896); + const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1152); + const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1152); + + HVX_VectorPair v_sums0, v_sums1; + accum_q6_k_32x2(vptr, v_act0, v_act1, i32, &v_sums0, &v_sums1); + + v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, scale_q6_k_32x1(v_sums0, vptr[6], v_act0[8])); + v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, scale_q6_k_32x1(v_sums1, vptr[6], v_act1[8])); + } + + if (sz0) { + hvx_vec_store_u(s0, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vmemu(sz0))); + } else { + hvx_vec_store_u(s0, valid_rows * sizeof(float), v_sum_float_c0); + } + if (sz1) { + hvx_vec_store_u(s1, valid_rows * sizeof(float), hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vmemu(sz1))); + } else { + hvx_vec_store_u(s1, valid_rows * sizeof(float), v_sum_float_c1); + } +} + static void tiled_vec_dot_iq4nl_32x1(const uint32_t n, float * restrict s, const void * restrict vx, const void * restrict vy, uint32_t valid_rows, const float * restrict sz) { const uint8_t * restrict tile_ptr = vx; const uint8_t * restrict y_q = vy; @@ -827,51 +788,7 @@ static void tiled_vec_dot_iq4nl_32x2(const uint32_t n, float * restrict s0, floa HVX_Vector lut = *(const HVX_Vector *) kvalues_iq4nl_lut; uint32_t n_k_tiles = n / 32; - uint32_t kt = 0; - for (; kt + 1 < n_k_tiles; kt += 2) { - const HVX_Vector * restrict vptr0 = (const HVX_Vector *) (tile_ptr + (kt + 0) * 640); - const HVX_Vector * restrict v_act0_0 = (const HVX_Vector *) (y0_q + (kt + 0) * 1152); - const HVX_Vector * restrict v_act1_0 = (const HVX_Vector *) (y1_q + (kt + 0) * 1152); - - const HVX_Vector * restrict vptr1 = (const HVX_Vector *) (tile_ptr + (kt + 1) * 640); - const HVX_Vector * restrict v_act0_1 = (const HVX_Vector *) (y0_q + (kt + 1) * 1152); - const HVX_Vector * restrict v_act1_1 = (const HVX_Vector *) (y1_q + (kt + 1) * 1152); - - HVX_VectorPair v_sums0 = accum_4bit_32x2_lut(vptr0, v_act0_0, v_act1_0, mask_h4, lut); - HVX_VectorPair v_sums1 = accum_4bit_32x2_lut(vptr1, v_act0_1, v_act1_1, mask_h4, lut); - - HVX_Vector v_sum_c0_0 = Q6_V_lo_W(v_sums0); - HVX_Vector v_sum_c1_0 = Q6_V_hi_W(v_sums0); - HVX_Vector v_sum_c0_1 = Q6_V_lo_W(v_sums1); - HVX_Vector v_sum_c1_1 = Q6_V_hi_W(v_sums1); - - HVX_Vector v_sum_sf_c0_0 = Q6_Vsf_equals_Vw(v_sum_c0_0); - HVX_Vector v_sum_sf_c1_0 = Q6_Vsf_equals_Vw(v_sum_c1_0); - HVX_Vector v_sum_sf_c0_1 = Q6_Vsf_equals_Vw(v_sum_c0_1); - HVX_Vector v_sum_sf_c1_1 = Q6_Vsf_equals_Vw(v_sum_c1_1); - - HVX_Vector v_scale_w0 = vptr0[4]; - HVX_Vector v_scale_w1 = vptr1[4]; - HVX_Vector v_scale_a_c0_0 = v_act0_0[8]; - HVX_Vector v_scale_a_c1_0 = v_act1_0[8]; - HVX_Vector v_scale_a_c0_1 = v_act0_1[8]; - HVX_Vector v_scale_a_c1_1 = v_act1_1[8]; - - HVX_Vector v_scale_comb_c0_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c0_0); - HVX_Vector v_scale_comb_c1_0 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w0, v_scale_a_c1_0); - HVX_Vector v_scale_comb_c0_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c0_1); - HVX_Vector v_scale_comb_c1_1 = hvx_vec_mul_f16_f16_to_f32_lower32(v_scale_w1, v_scale_a_c1_1); - - HVX_Vector v_sum_scaled_c0_0 = hvx_vec_mul_f32_f32(v_sum_sf_c0_0, v_scale_comb_c0_0); - HVX_Vector v_sum_scaled_c1_0 = hvx_vec_mul_f32_f32(v_sum_sf_c1_0, v_scale_comb_c1_0); - HVX_Vector v_sum_scaled_c0_1 = hvx_vec_mul_f32_f32(v_sum_sf_c0_1, v_scale_comb_c0_1); - HVX_Vector v_sum_scaled_c1_1 = hvx_vec_mul_f32_f32(v_sum_sf_c1_1, v_scale_comb_c1_1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vec_add_f32_f32(v_sum_scaled_c0_0, v_sum_scaled_c0_1)); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vec_add_f32_f32(v_sum_scaled_c1_0, v_sum_scaled_c1_1)); - } - - for (; kt < n_k_tiles; kt++) { + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1152); const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1152); @@ -964,69 +881,7 @@ static void tiled_vec_dot_mxfp4_32x2(const uint32_t n, float * restrict s0, floa HVX_Vector e8m0_mask = Q6_V_vsplat_R(0x000000ff); uint32_t n_k_tiles = n / 32; - uint32_t kt = 0; - for (; kt + 1 < n_k_tiles; kt += 2) { - const HVX_Vector * restrict vptr0 = (const HVX_Vector *) (tile_ptr + (kt + 0) * 640); - const HVX_Vector * restrict v_act0_0 = (const HVX_Vector *) (y0_q + (kt + 0) * 1152); - const HVX_Vector * restrict v_act1_0 = (const HVX_Vector *) (y1_q + (kt + 0) * 1152); - - const HVX_Vector * restrict vptr1 = (const HVX_Vector *) (tile_ptr + (kt + 1) * 640); - const HVX_Vector * restrict v_act0_1 = (const HVX_Vector *) (y0_q + (kt + 1) * 1152); - const HVX_Vector * restrict v_act1_1 = (const HVX_Vector *) (y1_q + (kt + 1) * 1152); - - HVX_VectorPair v_sums0 = accum_4bit_32x2_lut(vptr0, v_act0_0, v_act1_0, mask_h4, lut); - HVX_VectorPair v_sums1 = accum_4bit_32x2_lut(vptr1, v_act0_1, v_act1_1, mask_h4, lut); - - HVX_Vector v_sum_c0_0 = Q6_V_lo_W(v_sums0); - HVX_Vector v_sum_c1_0 = Q6_V_hi_W(v_sums0); - HVX_Vector v_sum_c0_1 = Q6_V_lo_W(v_sums1); - HVX_Vector v_sum_c1_1 = Q6_V_hi_W(v_sums1); - - HVX_Vector v_sum_sf_c0_0 = Q6_Vsf_equals_Vw(v_sum_c0_0); - HVX_Vector v_sum_sf_c1_0 = Q6_Vsf_equals_Vw(v_sum_c1_0); - HVX_Vector v_sum_sf_c0_1 = Q6_Vsf_equals_Vw(v_sum_c0_1); - HVX_Vector v_sum_sf_c1_1 = Q6_Vsf_equals_Vw(v_sum_c1_1); - - HVX_Vector v_scale_w0 = hvx_vmem(tile_ptr + (kt + 0) * 640 + 512); - HVX_Vector r0_d0 = Q6_V_vdelta_VV(v_scale_w0, expand); - r0_d0 = Q6_V_vand_VV(r0_d0, e8m0_mask); - HVX_Vector v_scale_w_f32_0 = Q6_Vw_vasl_VwR(r0_d0, 23); - - HVX_Vector v_scale_w1 = hvx_vmem(tile_ptr + (kt + 1) * 640 + 512); - HVX_Vector r0_d1 = Q6_V_vdelta_VV(v_scale_w1, expand); - r0_d1 = Q6_V_vand_VV(r0_d1, e8m0_mask); - HVX_Vector v_scale_w_f32_1 = Q6_Vw_vasl_VwR(r0_d1, 23); - - HVX_Vector v_scale_a_c0_f16_0 = v_act0_0[8]; - HVX_Vector v_scale_a_c1_f16_0 = v_act1_0[8]; - HVX_Vector v_scale_a_c0_f16_1 = v_act0_1[8]; - HVX_Vector v_scale_a_c1_f16_1 = v_act1_1[8]; - - HVX_VectorPair p_scale_a_c0_f32_0 = hvx_vec_f16_to_f32_shuff(v_scale_a_c0_f16_0); - HVX_VectorPair p_scale_a_c1_f32_0 = hvx_vec_f16_to_f32_shuff(v_scale_a_c1_f16_0); - HVX_VectorPair p_scale_a_c0_f32_1 = hvx_vec_f16_to_f32_shuff(v_scale_a_c0_f16_1); - HVX_VectorPair p_scale_a_c1_f32_1 = hvx_vec_f16_to_f32_shuff(v_scale_a_c1_f16_1); - - HVX_Vector v_scale_a_c0_0 = Q6_V_lo_W(p_scale_a_c0_f32_0); - HVX_Vector v_scale_a_c1_0 = Q6_V_lo_W(p_scale_a_c1_f32_0); - HVX_Vector v_scale_a_c0_1 = Q6_V_lo_W(p_scale_a_c0_f32_1); - HVX_Vector v_scale_a_c1_1 = Q6_V_lo_W(p_scale_a_c1_f32_1); - - HVX_Vector v_scale_comb_c0_0 = hvx_vec_mul_f32_f32(v_scale_w_f32_0, v_scale_a_c0_0); - HVX_Vector v_scale_comb_c1_0 = hvx_vec_mul_f32_f32(v_scale_w_f32_0, v_scale_a_c1_0); - HVX_Vector v_scale_comb_c0_1 = hvx_vec_mul_f32_f32(v_scale_w_f32_1, v_scale_a_c0_1); - HVX_Vector v_scale_comb_c1_1 = hvx_vec_mul_f32_f32(v_scale_w_f32_1, v_scale_a_c1_1); - - HVX_Vector v_sum_scaled_c0_0 = hvx_vec_mul_f32_f32(v_sum_sf_c0_0, v_scale_comb_c0_0); - HVX_Vector v_sum_scaled_c1_0 = hvx_vec_mul_f32_f32(v_sum_sf_c1_0, v_scale_comb_c1_0); - HVX_Vector v_sum_scaled_c0_1 = hvx_vec_mul_f32_f32(v_sum_sf_c0_1, v_scale_comb_c0_1); - HVX_Vector v_sum_scaled_c1_1 = hvx_vec_mul_f32_f32(v_sum_sf_c1_1, v_scale_comb_c1_1); - - v_sum_float_c0 = hvx_vec_add_f32_f32(v_sum_float_c0, hvx_vec_add_f32_f32(v_sum_scaled_c0_0, v_sum_scaled_c0_1)); - v_sum_float_c1 = hvx_vec_add_f32_f32(v_sum_float_c1, hvx_vec_add_f32_f32(v_sum_scaled_c1_0, v_sum_scaled_c1_1)); - } - - for (; kt < n_k_tiles; kt++) { + for (uint32_t kt = 0; kt < n_k_tiles; kt++) { const HVX_Vector * restrict vptr = (const HVX_Vector *) (tile_ptr + kt * 640); const HVX_Vector * restrict v_act0 = (const HVX_Vector *) (y0_q + kt * 1152); const HVX_Vector * restrict v_act1 = (const HVX_Vector *) (y1_q + kt * 1152); diff --git a/ggml/src/ggml-hexagon/htp/hvx-norm.h b/ggml/src/ggml-hexagon/htp/hvx-norm.h index a8645e41..7ea945a3 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-norm.h +++ b/ggml/src/ggml-hexagon/htp/hvx-norm.h @@ -254,4 +254,201 @@ static inline void hvx_fast_l2_norm_f32(const uint8_t * restrict src, } } +// F16 norm kernels: reduce and scale in f32 (via promote/narrow), matching the +// precision-preserving pattern used by the flash-attn f16 kernels. + +static inline void hvx_fast_rms_norm_f16(const uint8_t * restrict src, + uint8_t * restrict dst, + const int num_elems, + float epsilon) { + + const HVX_Vector * restrict v_src = (HVX_Vector *) src; + HVX_Vector * restrict v_dst = (HVX_Vector *) dst; + + const int nvec = num_elems / VLEN_FP16; // number of full f16 vectors + const int nloe = num_elems % VLEN_FP16; // leftover elements + + HVX_Vector sum_v = Q6_V_vsplat_R(0x00000000); + HVX_Vector epsilon_v = hvx_vec_splat_f32(epsilon); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + } + + sum_v = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_v)); + + HVX_Vector t_v = hvx_vec_splat_f32((float) num_elems); + HVX_Vector denom_v = hvx_vec_inverse_f32(t_v); + HVX_Vector mean_v = Q6_Vqf32_vmpy_VsfVsf(sum_v, denom_v); + HVX_Vector mean_epsilon_v = Q6_Vqf32_vadd_Vqf32Vsf(mean_v, epsilon_v); + + HVX_Vector scale_v = hvx_vec_rsqrt_f32(Q6_Vsf_equals_Vqf32(mean_epsilon_v)); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), scale_v)); + v_dst[i] = hvx_vec_f32_to_f16(r0, r1); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), scale_v)); + HVX_Vector result = hvx_vec_f32_to_f16(r0, r1); + hvx_vec_store_a(&v_dst[nvec], nloe * SIZEOF_FP16, result); + } +} + +static inline void hvx_fast_norm_f16(const uint8_t * restrict src, + uint8_t * restrict dst, + const int num_elems, + float epsilon) { + + const HVX_Vector * restrict v_src = (HVX_Vector *) src; + HVX_Vector * restrict v_dst = (HVX_Vector *) dst; + + const int nvec = num_elems / VLEN_FP16; + const int nloe = num_elems % VLEN_FP16; + + HVX_Vector sum_sq_v = Q6_V_vsplat_R(0x00000000); + HVX_Vector sum_x_v = Q6_V_vsplat_R(0x00000000); + HVX_Vector epsilon_v = hvx_vec_splat_f32(epsilon); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_sq_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_sq_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_sq_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_sq_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + sum_x_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_x_v, Q6_Vqf32_vadd_VsfVsf(p0, Q6_V_vzero())); + sum_x_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_x_v, Q6_Vqf32_vadd_VsfVsf(p1, Q6_V_vzero())); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_sq_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_sq_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_sq_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_sq_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + sum_x_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_x_v, Q6_Vqf32_vadd_VsfVsf(p0, Q6_V_vzero())); + sum_x_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_x_v, Q6_Vqf32_vadd_VsfVsf(p1, Q6_V_vzero())); + } + + sum_sq_v = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_sq_v)); + sum_x_v = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_x_v)); + + HVX_Vector t_v = hvx_vec_splat_f32((float) num_elems); + HVX_Vector denom_v = hvx_vec_inverse_f32(t_v); + HVX_Vector mean_sq_v = Q6_Vqf32_vmpy_VsfVsf(sum_sq_v, denom_v); + HVX_Vector mean_x_v = Q6_Vqf32_vmpy_VsfVsf(sum_x_v, denom_v); + HVX_Vector mean_x_sq_v = Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(mean_x_v), Q6_Vsf_equals_Vqf32(mean_x_v)); + HVX_Vector var_v = Q6_Vqf32_vsub_Vqf32Vqf32(mean_sq_v, mean_x_sq_v); + HVX_Vector var_epsilon_v = Q6_Vqf32_vadd_Vqf32Vsf(var_v, epsilon_v); + + HVX_Vector scale_v = hvx_vec_rsqrt_f32(Q6_Vsf_equals_Vqf32(var_epsilon_v)); + HVX_Vector mean_x_b = hvx_vec_repl_f32(Q6_Vsf_equals_Vqf32(mean_x_v)); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector d0 = Q6_Vqf32_vsub_VsfVsf(Q6_V_lo_W(p), mean_x_b); + HVX_Vector d1 = Q6_Vqf32_vsub_VsfVsf(Q6_V_hi_W(p), mean_x_b); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(d0), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(d1), scale_v)); + v_dst[i] = hvx_vec_f32_to_f16(r0, r1); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector d0 = Q6_Vqf32_vsub_VsfVsf(Q6_V_lo_W(p), mean_x_b); + HVX_Vector d1 = Q6_Vqf32_vsub_VsfVsf(Q6_V_hi_W(p), mean_x_b); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(d0), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_Vsf_equals_Vqf32(d1), scale_v)); + HVX_Vector result = hvx_vec_f32_to_f16(r0, r1); + hvx_vec_store_a(&v_dst[nvec], nloe * SIZEOF_FP16, result); + } +} + +static inline void hvx_fast_l2_norm_f16(const uint8_t * restrict src, + uint8_t * restrict dst, + const int num_elems, + float epsilon) { + + const HVX_Vector * restrict v_src = (HVX_Vector *) src; + HVX_Vector * restrict v_dst = (HVX_Vector *) dst; + + const int nvec = num_elems / VLEN_FP16; + const int nloe = num_elems % VLEN_FP16; + + HVX_Vector sum_v = hvx_vec_splat_f32(0.0f); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector p0 = Q6_V_lo_W(p); + HVX_Vector p1 = Q6_V_hi_W(p); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p0, p0)); + sum_v = Q6_Vqf32_vadd_Vqf32Vqf32(sum_v, Q6_Vqf32_vmpy_VsfVsf(p1, p1)); + } + + HVX_Vector sum_sf = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_v)); + HVX_Vector rsqrt_v = hvx_vec_rsqrt_f32(sum_sf); + HVX_Vector sqrt_v = hvx_vec_inverse_f32(rsqrt_v); + HVX_Vector epsilon_v = hvx_vec_splat_f32(epsilon); + HVX_Vector denom_v = Q6_Vsf_vmax_VsfVsf(sqrt_v, epsilon_v); + HVX_Vector scale_v = hvx_vec_inverse_f32(denom_v); + + #pragma unroll(4) + for (int i = 0; i < nvec; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_src[i]); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), scale_v)); + v_dst[i] = hvx_vec_f32_to_f16(r0, r1); + } + + if (nloe > 0) { + HVX_VectorPred bmask = Q6_Q_vsetq_R(nloe * SIZEOF_FP16); + HVX_Vector v1 = Q6_V_vand_QV(bmask, v_src[nvec]); + HVX_VectorPair p = hvx_vec_f16_to_f32(v1); + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), scale_v)); + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), scale_v)); + HVX_Vector result = hvx_vec_f32_to_f16(r0, r1); + hvx_vec_store_a(&v_dst[nvec], nloe * SIZEOF_FP16, result); + } +} + #endif // HVX_NORM_H diff --git a/ggml/src/ggml-hexagon/htp/hvx-quant.h b/ggml/src/ggml-hexagon/htp/hvx-quant.h new file mode 100644 index 00000000..6b172cd6 --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/hvx-quant.h @@ -0,0 +1,165 @@ +#ifndef HVX_QUANT_H +#define HVX_QUANT_H + +#include +#include +#include + +#include "hvx-arith.h" +#include "hvx-base.h" +#include "hvx-reduce.h" +#include "hvx-repl.h" +#include "hvx-utils.h" + +#ifndef GGML_COMMON_DECL_C +#define GGML_COMMON_DECL_C +#endif +#include "ggml-common.h" +#include "ggml-impl.h" + +static inline void hvx_quantize_row_q8_0_f32(void * restrict dst_ptr, const float * restrict src_ptr, int n) { + const int nb = n / QK8_0; + block_q8_0 * dst = (block_q8_0 *) dst_ptr; + HVX_Vector zero = Q6_V_vzero(); + + int i = 0; + for (; i + 3 < nb; i += 4) { + HVX_Vector * vx = (HVX_Vector *) (src_ptr + i * QK8_0); + + HVX_Vector vmax0_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[0])); + HVX_Vector vmax1_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[1])); + HVX_Vector vmax2_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[2])); + HVX_Vector vmax3_sf = hvx_vec_reduce_max_f32(hvx_vec_abs_f32(vx[3])); + + HVX_Vector vx0_qf = Q6_Vqf32_vsub_VsfVsf(vx[0], zero); + HVX_Vector vx1_qf = Q6_Vqf32_vsub_VsfVsf(vx[1], zero); + HVX_Vector vx2_qf = Q6_Vqf32_vsub_VsfVsf(vx[2], zero); + HVX_Vector vx3_qf = Q6_Vqf32_vsub_VsfVsf(vx[3], zero); + + HVX_Vector vmax0_qf = Q6_Vqf32_vsub_VsfVsf(vmax0_sf, zero); + HVX_Vector vmax1_qf = Q6_Vqf32_vsub_VsfVsf(vmax1_sf, zero); + HVX_Vector vmax2_qf = Q6_Vqf32_vsub_VsfVsf(vmax2_sf, zero); + HVX_Vector vmax3_qf = Q6_Vqf32_vsub_VsfVsf(vmax3_sf, zero); + + HVX_Vector vmax01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax1_qf, vmax0_qf))); + HVX_Vector vmax23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vmax3_qf, vmax2_qf))); + + HVX_Vector vx01_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx1_qf, vx0_qf))); + HVX_Vector vx23_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(vx3_qf, vx2_qf))); + + HVX_Vector vd01_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax01_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 + HVX_Vector vd23_qf16 = Q6_Vqf16_vmpy_VhfVhf(vmax23_hf, Q6_Vh_vsplat_R(0x2008)); // 1.0 / 127.0 + HVX_Vector vd01_hf = Q6_Vhf_equals_Vqf16(vd01_qf16); + HVX_Vector vd23_hf = Q6_Vhf_equals_Vqf16(vd23_qf16); + + HVX_Vector vd01_inv_hf = hvx_vec_inverse_f16(vd01_hf); + HVX_Vector vd23_inv_hf = hvx_vec_inverse_f16(vd23_hf); + vx01_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx01_hf, vd01_inv_hf)); + vx23_hf = Q6_Vhf_equals_Vqf16(Q6_Vqf16_vmpy_VhfVhf(vx23_hf, vd23_inv_hf)); + + HVX_Vector vx01_i16 = hvx_vec_i16_from_hf_rnd_sat(vx01_hf); + HVX_Vector vx23_i16 = hvx_vec_i16_from_hf_rnd_sat(vx23_hf); + HVX_Vector vx_i8 = Q6_Vb_vpack_VhVh_sat(vx23_i16, vx01_i16); + + hvx_vec_store_u(&dst[i + 0].d, 2, vd01_hf); + hvx_vec_store_u(dst[i + 0].qs, 32, vx_i8); + + hvx_vec_store_u(&dst[i + 1].d, 2, Q6_V_vror_VR(vd01_hf, 64)); + hvx_vec_store_u(dst[i + 1].qs, 32, Q6_V_vror_VR(vx_i8, 32)); + + hvx_vec_store_u(&dst[i + 2].d, 2, vd23_hf); + hvx_vec_store_u(dst[i + 2].qs, 32, Q6_V_vror_VR(vx_i8, 64)); + + hvx_vec_store_u(&dst[i + 3].d, 2, Q6_V_vror_VR(vd23_hf, 64)); + hvx_vec_store_u(dst[i + 3].qs, 32, Q6_V_vror_VR(vx_i8, 96)); + } + + for (; i < nb; i++) { + const float * block_src = src_ptr + i * QK8_0; + HVX_Vector vx = *(const HVX_UVector *) block_src; + HVX_Vector v_abs = hvx_vec_abs_f32(vx); + HVX_Vector v_max = hvx_vec_reduce_max_f32(v_abs); + float amax = hvx_vec_get_f32(v_max); + + const float d = amax / 127.0f; + const float id = d ? (1.0f / d) : 0.0f; + dst[i].d = GGML_FP32_TO_FP16(d); + + HVX_Vector vid = hvx_vec_splat_f32(id); + HVX_Vector v_scaled = hvx_vec_mul_f32_f32(vx, vid); + HVX_Vector v_scaled_qf = Q6_Vqf32_vsub_VsfVsf(v_scaled, zero); + HVX_Vector v_scaled_hf = Q6_Vh_vdeal_Vh(Q6_Vhf_equals_Wqf32(Q6_W_vcombine_VV(zero, v_scaled_qf))); + HVX_Vector v_i16 = hvx_vec_i16_from_hf_rnd_sat(v_scaled_hf); + HVX_Vector v_i8 = Q6_Vb_vpack_VhVh_sat(zero, v_i16); + + hvx_vec_store_u(dst[i].qs, 32, v_i8); + } +} + +static inline void hvx_dequantize_row_q8_0_f32(float * restrict dst_ptr, const void * restrict src_ptr, int n) { + const int nb = n / QK8_0; + const block_q8_0 * src = (const block_q8_0 *) src_ptr; + + for (int i = 0; i < nb; i++) { + HVX_Vector vd_f16 = Q6_Vh_vsplat_R(*(const int16_t *) &src[i].d); + HVX_VectorPair vp_f32 = hvx_vec_f16_to_f32(vd_f16); + HVX_Vector vd = Q6_V_lo_W(vp_f32); + + HVX_Vector vq_i8 = *(const HVX_UVector *) src[i].qs; + + HVX_VectorPair p16 = Q6_Wh_vunpack_Vb(vq_i8); + HVX_Vector v_i16 = Q6_V_lo_W(p16); + HVX_VectorPair p32 = Q6_Ww_vunpack_Vh(v_i16); + HVX_Vector v_i32 = Q6_V_lo_W(p32); + + HVX_Vector v_f32 = Q6_Vsf_equals_Vw(v_i32); + HVX_Vector res = hvx_vec_mul_f32_f32(v_f32, vd); + + float * block_dst = dst_ptr + i * QK8_0; + hvx_vmem(block_dst) = res; + } +} + +static inline void hvx_dequantize_row_q8_0_f16(__fp16 * restrict dst_ptr, const void * restrict src_ptr, int n) { + const int nb = n / QK8_0; + const block_q8_0 * src = (const block_q8_0 *) src_ptr; + + for (int i = nb - 1; i >= 0; i--) { + HVX_Vector vd_f16 = Q6_Vh_vsplat_R(*(const int16_t *) &src[i].d); + HVX_VectorPair vp_f32 = hvx_vec_f16_to_f32(vd_f16); + HVX_Vector vd = Q6_V_lo_W(vp_f32); + + HVX_Vector vq_i8 = *(const HVX_UVector *) src[i].qs; + + HVX_VectorPair p16 = Q6_Wh_vunpack_Vb(vq_i8); + HVX_Vector v_i16 = Q6_V_lo_W(p16); + HVX_VectorPair p32 = Q6_Ww_vunpack_Vh(v_i16); + HVX_Vector v_i32 = Q6_V_lo_W(p32); + + HVX_Vector v_f32 = Q6_Vsf_equals_Vw(v_i32); + HVX_Vector res_f32 = hvx_vec_mul_f32_f32(v_f32, vd); + + HVX_Vector res_f16 = hvx_vec_f32_to_f16(res_f32, Q6_V_vzero()); + + __fp16 * block_dst = dst_ptr + i * QK8_0; + hvx_vec_store_u(block_dst, QK8_0 * sizeof(__fp16), res_f16); + } +} + +static inline void hvx_dequantize_row_f16_f32(float * restrict dst_ptr, const void * restrict src_ptr, int n) { + const int nb = n / 32; + const _Float16 * src = (const _Float16 *) src_ptr; + + for (int i = 0; i < nb; i++) { + HVX_Vector v_f16 = *(const HVX_UVector *) (src + i * 32); + HVX_VectorPair vp_f32 = hvx_vec_f16_to_f32(v_f16); + HVX_Vector res = Q6_V_lo_W(vp_f32); + + float * block_dst = dst_ptr + i * 32; + hvx_vmem(block_dst) = res; + } +} + + + +#endif // HVX_QUANT_H diff --git a/ggml/src/ggml-hexagon/htp/hvx-scale.h b/ggml/src/ggml-hexagon/htp/hvx-scale.h index c65c9863..5d065030 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-scale.h +++ b/ggml/src/ggml-hexagon/htp/hvx-scale.h @@ -68,30 +68,30 @@ static inline void hvx_scale_f32(uint8_t * restrict dst, const uint8_t * restric } } -#define hvx_scale_offset_f32_loop_body(dst_type, src_type, vec_store) \ - do { \ - dst_type * restrict vdst = (dst_type *) dst; \ - src_type * restrict vsrc = (src_type *) src; \ - \ - HVX_Vector vs = hvx_vec_splat_f32(scale); \ - HVX_Vector vo = hvx_vec_splat_f32(offset); \ - \ - const uint32_t elem_size = sizeof(float); \ - const uint32_t epv = 128 / elem_size; \ - const uint32_t nvec = n / epv; \ - const uint32_t nloe = n % epv; \ - \ - uint32_t i = 0; \ - \ - _Pragma("unroll(4)") \ - for (; i < nvec; ++i) { \ +#define hvx_scale_offset_f32_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + HVX_Vector vs = hvx_vec_splat_f32(scale); \ + HVX_Vector vo = hvx_vec_splat_f32(offset); \ + \ + const uint32_t elem_size = sizeof(float); \ + const uint32_t epv = 128 / elem_size; \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; ++i) { \ HVX_Vector v = Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(vsrc[i], vs), vo); \ - vdst[i] = Q6_Vsf_equals_Vqf32(v); \ - } \ - if (nloe) { \ + vdst[i] = Q6_Vsf_equals_Vqf32(v); \ + } \ + if (nloe) { \ HVX_Vector v = Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(vsrc[i], vs), vo); \ - vec_store((void *) &vdst[i], nloe * elem_size, Q6_Vsf_equals_Vqf32(v)); \ - } \ + vec_store((void *) &vdst[i], nloe * elem_size, Q6_Vsf_equals_Vqf32(v)); \ + } \ } while(0) static inline void hvx_scale_offset_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { @@ -130,4 +130,70 @@ static inline void hvx_scale_offset_f32(uint8_t * restrict dst, const uint8_t * } } +// Scale+offset computed by promoting f16 -> f32, then narrowing the result back to f16. +#define hvx_scale_offset_f16_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + HVX_Vector vs = hvx_vec_splat_f32(scale); \ + HVX_Vector vo = hvx_vec_splat_f32(offset); \ + \ + const uint32_t nvec = n / VLEN_FP16; \ + const uint32_t nloe = n % VLEN_FP16; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; ++i) { \ + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); \ + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), vs), vo)); \ + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), vs), vo)); \ + vdst[i] = hvx_vec_f32_to_f16(r0, r1); \ + } \ + if (nloe) { \ + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); \ + HVX_Vector r0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(p), vs), vo)); \ + HVX_Vector r1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vsf(Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(p), vs), vo)); \ + HVX_Vector v = hvx_vec_f32_to_f16(r0, r1); \ + vec_store((void *) &vdst[i], nloe * SIZEOF_FP16, v); \ + } \ + } while(0) + +static inline void hvx_scale_offset_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { + assert((size_t) dst % 128 == 0); + assert((size_t) src % 128 == 0); + hvx_scale_offset_f16_loop_body(HVX_Vector, HVX_Vector, hvx_vec_store_a); +} + +static inline void hvx_scale_offset_f16_au(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { + assert((size_t) dst % 128 == 0); + hvx_scale_offset_f16_loop_body(HVX_Vector, HVX_UVector, hvx_vec_store_a); +} + +static inline void hvx_scale_offset_f16_ua(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { + assert((size_t) src % 128 == 0); + hvx_scale_offset_f16_loop_body(HVX_UVector, HVX_Vector, hvx_vec_store_u); +} + +static inline void hvx_scale_offset_f16_uu(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { + hvx_scale_offset_f16_loop_body(HVX_UVector, HVX_UVector, hvx_vec_store_u); +} + +static inline void hvx_scale_offset_f16(uint8_t * restrict dst, const uint8_t * restrict src, const int n, const float scale, const float offset) { + if (((size_t) dst & 127) == 0) { + if (((size_t) src & 127) == 0) { + hvx_scale_offset_f16_aa(dst, src, n, scale, offset); + } else { + hvx_scale_offset_f16_au(dst, src, n, scale, offset); + } + } else { + if (((size_t) src & 127) == 0) { + hvx_scale_offset_f16_ua(dst, src, n, scale, offset); + } else { + hvx_scale_offset_f16_uu(dst, src, n, scale, offset); + } + } +} + #endif // HVX_SCALE_H diff --git a/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h b/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h index dd66dd84..55201730 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h +++ b/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h @@ -68,50 +68,50 @@ static inline HVX_Vector hvx_vec_tanh_f32(HVX_Vector x) { return Q6_Vsf_equals_Vqf32(res); } -#define hvx_sigmoid_loop_body(dst_type, src_type, vec_store) \ - do { \ - dst_type * restrict vdst = (dst_type *) dst; \ - src_type * restrict vsrc = (src_type *) src; \ - \ - const HVX_Vector one = hvx_vec_splat_f32(1.f); \ - const HVX_Vector max_exp = hvx_vec_splat_f32(87.f); \ - const HVX_Vector min_exp = hvx_vec_splat_f32(-87.f); \ - \ - const uint32_t epv = 128 / sizeof(float); \ - const uint32_t nvec = n / epv; \ - const uint32_t nloe = n % epv; \ - \ - uint32_t i = 0; \ - \ - _Pragma("unroll(4)") \ - for (; i < nvec; i++) { \ - vdst[i] = hvx_vec_fast_sigmoid_f32_guard(vsrc[i], one, max_exp, min_exp); \ - } \ - if (nloe) { \ +#define hvx_sigmoid_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const HVX_Vector one = hvx_vec_splat_f32(1.f); \ + const HVX_Vector max_exp = hvx_vec_splat_f32(87.f); \ + const HVX_Vector min_exp = hvx_vec_splat_f32(-87.f); \ + \ + const uint32_t epv = 128 / sizeof(float); \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + vdst[i] = hvx_vec_fast_sigmoid_f32_guard(vsrc[i], one, max_exp, min_exp); \ + } \ + if (nloe) { \ HVX_Vector tmp = hvx_vec_fast_sigmoid_f32_guard(vsrc[i], one, max_exp, min_exp); \ - vec_store((void *) &vdst[i], nloe * sizeof(float), tmp); \ - } \ + vec_store((void *) &vdst[i], nloe * sizeof(float), tmp); \ + } \ } while(0) -#define hvx_tanh_loop_body(dst_type, src_type, vec_store) \ - do { \ - dst_type * restrict vdst = (dst_type *) dst; \ - src_type * restrict vsrc = (src_type *) src; \ - \ - const uint32_t epv = 128 / sizeof(float); \ - const uint32_t nvec = n / epv; \ - const uint32_t nloe = n % epv; \ - \ - uint32_t i = 0; \ - \ - _Pragma("unroll(4)") \ - for (; i < nvec; i++) { \ - vdst[i] = hvx_vec_tanh_f32(vsrc[i]); \ - } \ - if (nloe) { \ - HVX_Vector tmp = hvx_vec_tanh_f32(vsrc[i]); \ +#define hvx_tanh_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const uint32_t epv = 128 / sizeof(float); \ + const uint32_t nvec = n / epv; \ + const uint32_t nloe = n % epv; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + vdst[i] = hvx_vec_tanh_f32(vsrc[i]); \ + } \ + if (nloe) { \ + HVX_Vector tmp = hvx_vec_tanh_f32(vsrc[i]); \ vec_store((void *) &vdst[i], nloe * sizeof(float), tmp); \ - } \ + } \ } while(0) static inline void hvx_sigmoid_f32_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { diff --git a/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h b/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h index c5b9a5d4..8648af0e 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h +++ b/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h @@ -4,87 +4,75 @@ #include "hvx-base.h" #include "hvx-floor.h" -static inline HVX_Vector hvx_vec_cos_f32(HVX_Vector x) { - HVX_Vector const_inv_pi = hvx_vec_splat_f32(0.3183098861837907f); - HVX_Vector const_half = hvx_vec_splat_f32(0.5f); - HVX_Vector const_pi = hvx_vec_splat_f32(3.141592653589793f); - HVX_Vector const_one = hvx_vec_splat_f32(1.0f); +// Range-reduce x to y in [-pi/2, pi/2] and the quadrant sign (-1)^n. +// Floor/truncate need IEEE bits, so convert qf32 back to sf before them. +static inline void hvx_vec_sincos_reduce_f32(HVX_Vector x, HVX_Vector * y, HVX_Vector * sign) { + HVX_Vector const_inv_pi = hvx_vec_splat_f32(0.3183098861837907f); + HVX_Vector const_half = hvx_vec_splat_f32(0.5f); + HVX_Vector const_pi = hvx_vec_splat_f32(3.141592653589793f); + HVX_Vector const_one = hvx_vec_splat_f32(1.0f); HVX_Vector const_neg_one = hvx_vec_splat_f32(-1.0f); + HVX_Vector const_one_i = Q6_V_vsplat_R(1); + + HVX_Vector x_over_pi = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(x, const_inv_pi)); + x_over_pi = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(x_over_pi, const_half)); - // n = floor(x * (1/pi) + 0.5) - HVX_Vector n_float = hvx_vec_floor_f32(hvx_vec_add_f32_f32(hvx_vec_mul_f32_f32(x, const_inv_pi), const_half)); + HVX_Vector n_float = hvx_vec_floor_f32(x_over_pi); + HVX_Vector n_int = hvx_vec_truncate_f32(n_float); - // y = x - n * pi - HVX_Vector y = hvx_vec_sub_f32_f32(x, hvx_vec_mul_f32_f32(n_float, const_pi)); + HVX_Vector n_pi = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(n_float, const_pi)); + *y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_VsfVsf(x, n_pi)); - // Sign determination: if n is odd, sign is -1.0f, else 1.0f - // half_n = n * 0.5f - HVX_Vector half_n = hvx_vec_mul_f32_f32(n_float, const_half); - // floor_half_n = floor(half_n) - HVX_Vector floor_half_n = hvx_vec_floor_f32(half_n); - // is_odd = half_n > floor_half_n - HVX_VectorPred is_odd = Q6_Q_vcmp_gt_VsfVsf(half_n, floor_half_n); - // sign = vmux(is_odd, -1.0f, 1.0f) - HVX_Vector sign = Q6_V_vmux_QVV(is_odd, const_neg_one, const_one); + HVX_VectorPred is_odd = Q6_Q_vcmp_eq_VwVw(Q6_V_vand_VV(n_int, const_one_i), const_one_i); + *sign = Q6_V_vmux_QVV(is_odd, const_neg_one, const_one); +} - // z = y^2 - HVX_Vector z = hvx_vec_mul_f32_f32(y, y); +static inline void hvx_vec_sincos_f32(HVX_Vector x, HVX_Vector * vcos, HVX_Vector * vsin) { + HVX_Vector y; + HVX_Vector sign; + hvx_vec_sincos_reduce_f32(x, &y, &sign); + + HVX_Vector z = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(y, y)); - // Chebyshev approximation for cos(y) HVX_Vector c4 = hvx_vec_splat_f32(2.3557242013849433e-05f); HVX_Vector c3 = hvx_vec_splat_f32(-0.0013871428263450528f); HVX_Vector c2 = hvx_vec_splat_f32(0.041665895266688284f); HVX_Vector c1 = hvx_vec_splat_f32(-0.4999999360426369f); HVX_Vector c0 = hvx_vec_splat_f32(0.9999999999071725f); - HVX_Vector cos_y = hvx_vec_add_f32_f32(c3, hvx_vec_mul_f32_f32(z, c4)); - cos_y = hvx_vec_add_f32_f32(c2, hvx_vec_mul_f32_f32(z, cos_y)); - cos_y = hvx_vec_add_f32_f32(c1, hvx_vec_mul_f32_f32(z, cos_y)); - cos_y = hvx_vec_add_f32_f32(c0, hvx_vec_mul_f32_f32(z, cos_y)); - - return hvx_vec_mul_f32_f32(cos_y, sign); -} - -static inline HVX_Vector hvx_vec_sin_f32(HVX_Vector x) { - HVX_Vector const_inv_pi = hvx_vec_splat_f32(0.3183098861837907f); - HVX_Vector const_half = hvx_vec_splat_f32(0.5f); - HVX_Vector const_pi = hvx_vec_splat_f32(3.141592653589793f); - HVX_Vector const_one = hvx_vec_splat_f32(1.0f); - HVX_Vector const_neg_one = hvx_vec_splat_f32(-1.0f); - - // n = floor(x * (1/pi) + 0.5) - HVX_Vector n_float = hvx_vec_floor_f32(hvx_vec_add_f32_f32(hvx_vec_mul_f32_f32(x, const_inv_pi), const_half)); + HVX_Vector cos_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(c3, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, c4)))); + cos_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(c2, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, cos_y)))); + cos_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(c1, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, cos_y)))); + cos_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(c0, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, cos_y)))); - // y = x - n * pi - HVX_Vector y = hvx_vec_sub_f32_f32(x, hvx_vec_mul_f32_f32(n_float, const_pi)); - - // Sign determination: if n is odd, sign is -1.0f, else 1.0f - // half_n = n * 0.5f - HVX_Vector half_n = hvx_vec_mul_f32_f32(n_float, const_half); - // floor_half_n = floor(half_n) - HVX_Vector floor_half_n = hvx_vec_floor_f32(half_n); - // is_odd = half_n > floor_half_n - HVX_VectorPred is_odd = Q6_Q_vcmp_gt_VsfVsf(half_n, floor_half_n); - // sign = vmux(is_odd, -1.0f, 1.0f) - HVX_Vector sign = Q6_V_vmux_QVV(is_odd, const_neg_one, const_one); - - // z = y^2 - HVX_Vector z = hvx_vec_mul_f32_f32(y, y); - - // Chebyshev approximation for sin(y) HVX_Vector s4 = hvx_vec_splat_f32(2.642186986152672e-06f); HVX_Vector s3 = hvx_vec_splat_f32(-0.00019825318964070864f); HVX_Vector s2 = hvx_vec_splat_f32(0.00833326283319605f); HVX_Vector s1 = hvx_vec_splat_f32(-0.16666666082087775f); HVX_Vector s0 = hvx_vec_splat_f32(0.999999999915155f); - HVX_Vector sin_y = hvx_vec_add_f32_f32(s3, hvx_vec_mul_f32_f32(z, s4)); - sin_y = hvx_vec_add_f32_f32(s2, hvx_vec_mul_f32_f32(z, sin_y)); - sin_y = hvx_vec_add_f32_f32(s1, hvx_vec_mul_f32_f32(z, sin_y)); - sin_y = hvx_vec_add_f32_f32(s0, hvx_vec_mul_f32_f32(z, sin_y)); - sin_y = hvx_vec_mul_f32_f32(y, sin_y); + HVX_Vector sin_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(s3, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, s4)))); + sin_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(s2, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, sin_y)))); + sin_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(s1, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, sin_y)))); + sin_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_VsfVsf(s0, Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(z, sin_y)))); + sin_y = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(y, sin_y)); + + *vcos = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(cos_y, sign)); + *vsin = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vmpy_VsfVsf(sin_y, sign)); +} - return hvx_vec_mul_f32_f32(sin_y, sign); +static inline HVX_Vector hvx_vec_cos_f32(HVX_Vector x) { + HVX_Vector vcos; + HVX_Vector vsin; + hvx_vec_sincos_f32(x, &vcos, &vsin); + return vcos; +} + +static inline HVX_Vector hvx_vec_sin_f32(HVX_Vector x) { + HVX_Vector vcos; + HVX_Vector vsin; + hvx_vec_sincos_f32(x, &vcos, &vsin); + return vsin; } #endif /* HVX_SIN_COS_H */ diff --git a/ggml/src/ggml-hexagon/htp/hvx-sqrt.h b/ggml/src/ggml-hexagon/htp/hvx-sqrt.h index e31a1006..abdded5c 100644 --- a/ggml/src/ggml-hexagon/htp/hvx-sqrt.h +++ b/ggml/src/ggml-hexagon/htp/hvx-sqrt.h @@ -123,4 +123,67 @@ static inline void hvx_sqrt_f32(uint8_t * restrict dst, const uint8_t * restrict } } +// Compute sqrt(x) for f16 by promoting to f32, applying hvx_vec_rsqrt_f32, and narrowing back. +#define hvx_sqrt_f16_loop_body(dst_type, src_type, vec_store) \ + do { \ + dst_type * restrict vdst = (dst_type *) dst; \ + src_type * restrict vsrc = (src_type *) src; \ + \ + const uint32_t nvec = n / VLEN_FP16; \ + const uint32_t nloe = n % VLEN_FP16; \ + \ + uint32_t i = 0; \ + \ + _Pragma("unroll(4)") \ + for (; i < nvec; i++) { \ + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); \ + HVX_Vector r0 = HVX_OP_MUL(hvx_vec_rsqrt_f32(Q6_V_lo_W(p)), Q6_V_lo_W(p)); \ + HVX_Vector r1 = HVX_OP_MUL(hvx_vec_rsqrt_f32(Q6_V_hi_W(p)), Q6_V_hi_W(p)); \ + vdst[i] = hvx_vec_f32_to_f16(r0, r1); \ + } \ + if (nloe) { \ + HVX_VectorPair p = hvx_vec_f16_to_f32(vsrc[i]); \ + HVX_Vector r0 = HVX_OP_MUL(hvx_vec_rsqrt_f32(Q6_V_lo_W(p)), Q6_V_lo_W(p)); \ + HVX_Vector r1 = HVX_OP_MUL(hvx_vec_rsqrt_f32(Q6_V_hi_W(p)), Q6_V_hi_W(p)); \ + HVX_Vector v = hvx_vec_f32_to_f16(r0, r1); \ + vec_store((void *) &vdst[i], nloe * SIZEOF_FP16, v); \ + } \ + } while(0) + +static inline void hvx_sqrt_f16_aa(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + assert((unsigned long) src % 128 == 0); + hvx_sqrt_f16_loop_body(HVX_Vector, HVX_Vector, hvx_vec_store_a); +} + +static inline void hvx_sqrt_f16_au(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) dst % 128 == 0); + hvx_sqrt_f16_loop_body(HVX_Vector, HVX_UVector, hvx_vec_store_a); +} + +static inline void hvx_sqrt_f16_ua(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + assert((unsigned long) src % 128 == 0); + hvx_sqrt_f16_loop_body(HVX_UVector, HVX_Vector, hvx_vec_store_u); +} + +static inline void hvx_sqrt_f16_uu(uint8_t * restrict dst, const uint8_t * restrict src, uint32_t n) { + hvx_sqrt_f16_loop_body(HVX_UVector, HVX_UVector, hvx_vec_store_u); +} + +static inline void hvx_sqrt_f16(uint8_t * restrict dst, const uint8_t * restrict src, const int num_elems) { + if ((unsigned long) dst % 128 == 0) { + if ((unsigned long) src % 128 == 0) { + hvx_sqrt_f16_aa(dst, src, num_elems); + } else { + hvx_sqrt_f16_au(dst, src, num_elems); + } + } else { + if ((unsigned long) src % 128 == 0) { + hvx_sqrt_f16_ua(dst, src, num_elems); + } else { + hvx_sqrt_f16_uu(dst, src, num_elems); + } + } +} + #endif /* HVX_SQRT_H */ diff --git a/ggml/src/ggml-hexagon/htp/im2col-ops.c b/ggml/src/ggml-hexagon/htp/im2col-ops.c index 35fc103d..2e05cf3e 100644 --- a/ggml/src/ggml-hexagon/htp/im2col-ops.c +++ b/ggml/src/ggml-hexagon/htp/im2col-ops.c @@ -3,24 +3,30 @@ #pragma clang diagnostic ignored "-Wunused-but-set-variable" #include -#include #include #include #include +#include "hex-common.h" + #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" #include "hvx-utils.h" -#include "hex-dma.h" +#include "dma-queue.h" #include "hex-profile.h" #include "htp-vtcm.h" +#include "htp-tensor.h" struct htp_im2col_context { struct htp_ops_context * octx; + uint32_t patch_base; // first patch index assigned to this dev + uint32_t npatches; // number of patches assigned to this dev uint32_t npatches_per_thread; // patches = N*OH*OW (pure-DDR kernel) + uint32_t pe_row_base; // first N*OH row index assigned to this dev (DMA path) + uint32_t pe_nrows; // number of N*OH rows assigned to this dev (DMA path) uint32_t pe_rows_per_thread; // N*OH rows per worker uint32_t pe_src_row_bytes; // one output row's source: IC*KH*IW*4, rounded 256 uint32_t pe_dst_row_bytes; // one output row's dst: OW*patch_stride*2, rounded 256 @@ -30,6 +36,9 @@ struct htp_im2col_context { uint8_t * pe_vtcm_dst; // base of the 2x dst buffers region uint32_t pe_src_size_per_thread; // 2 * pe_src_row_bytes uint32_t pe_dst_size_per_thread; // 2 * pe_dst_row_bytes + + uint32_t pe_owb; // output-col block size + uint32_t pe_wb; // staged source window width }; // Per-op VTCM layout for the patch-embed DMA path @@ -53,152 +62,324 @@ static inline void htp_im2col_vtcm_layout_build(struct htp_im2col_vtcm_layout * L->total_bytes = L->off_dst + L->dst_bytes_per_thread * n_threads; } -#define IM2COL_PATCHEMBED_BODY(FNAME, DST_CTYPE, COPY_FN, SPLAT_FN, DST_ELEM, TAG) \ - static void FNAME(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_im2col_context * ictx = (struct htp_im2col_context *) data; \ - struct htp_ops_context * octx = ictx->octx; \ - struct htp_thread_trace * restrict tr = &octx->ctx->trace[ith]; \ - const struct htp_tensor * restrict src1 = octx->src[1]; \ - const struct htp_tensor * restrict dst = octx->dst; \ - const int32_t s0 = octx->op_params[0]; \ - const int32_t s1 = octx->op_params[1]; \ - const int32_t p0 = octx->op_params[2]; \ - const int32_t p1 = octx->op_params[3]; \ - const int32_t d0 = octx->op_params[4]; \ - const int32_t d1 = octx->op_params[5]; \ - const uint32_t N = src1->ne[3]; \ - const uint32_t IC = src1->ne[2]; \ - const uint32_t IH = src1->ne[1]; \ - const uint32_t IW = src1->ne[0]; \ - const uint32_t KH = octx->src[0]->ne[1]; \ - const uint32_t KW = octx->src[0]->ne[0]; \ - const uint32_t OH = dst->ne[2]; \ - const uint32_t OW = dst->ne[1]; \ - const uint32_t patch_stride = IC * KH * KW; \ - const float * restrict src_data = (const float *) src1->data; \ - DST_CTYPE * restrict dst_data = (DST_CTYPE *) dst->data; \ - const uint32_t npatches = N * OH * OW; \ - const uint32_t patch_start = ictx->npatches_per_thread * ith; \ - const uint32_t patch_end = MIN(patch_start + ictx->npatches_per_thread, npatches); \ - if (patch_start >= patch_end) { \ - return; \ - } \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, patch_start); \ - for (uint32_t p = patch_start; p < patch_end; p++) { \ - const uint32_t iow = p % OW; \ - const uint32_t ioh = (p / OW) % OH; \ - const uint32_t in = p / (OW * OH); \ - DST_CTYPE * restrict dst_patch = dst_data + (uint64_t) p * patch_stride; \ - for (uint32_t iic = 0; iic < IC; iic++) { \ - const float * restrict src_plane = src_data + ((uint64_t) in * IC + iic) * IH * IW; \ - for (uint32_t ikh = 0; ikh < KH; ikh++) { \ - const int32_t iih = (int32_t) ioh * s1 + (int32_t) ikh * d1 - p1; \ - DST_CTYPE * restrict out_run = dst_patch + iic * (KH * KW) + ikh * KW; \ - if (iih < 0 || iih >= (int32_t) IH) { \ - SPLAT_FN(out_run, 0.0f, KW); \ - continue; \ - } \ - const int32_t iiw0 = (int32_t) iow * s0 - p0; \ - const float * restrict src_run = src_plane + (uint64_t) iih * IW + iiw0; \ - if (d0 == 1) { \ - /* contiguous source run: [lo,hi) is in-bounds, tails are zero pad */ \ - const int32_t lo = iiw0 < 0 ? -iiw0 : 0; \ - int32_t hi = (int32_t) IW - iiw0; \ - if (hi > (int32_t) KW) { \ - hi = (int32_t) KW; \ - } \ - if (hi <= lo) { \ - SPLAT_FN(out_run, 0.0f, KW); \ - } else { \ - if (lo > 0) { \ - SPLAT_FN(out_run, 0.0f, (uint32_t) lo); \ - } \ - COPY_FN((uint8_t *) (out_run + lo), (const uint8_t *) (src_run + lo), \ - (uint32_t) (hi - lo)); \ - if (hi < (int32_t) KW) { \ - SPLAT_FN(out_run + hi, 0.0f, (KW - (uint32_t) hi)); \ - } \ - } \ - continue; \ - } \ - for (uint32_t ikw = 0; ikw < KW; ikw++) { \ - const int32_t iiw = (int32_t) iow * s0 + (int32_t) ikw * d0 - p0; \ - out_run[ikw] = (iiw < 0 || iiw >= (int32_t) IW) ? \ - (DST_CTYPE) 0.0f : \ - (DST_CTYPE) src_plane[(uint64_t) iih * IW + iiw]; \ - } \ - } \ - } \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, patch_start); \ +#define IM2COL_PATCHEMBED_BODY(FNAME, DST_CTYPE, COPY_FN, SPLAT_FN, DST_ELEM, TAG) \ + static void FNAME(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_im2col_context * ictx = (struct htp_im2col_context *) data; \ + struct htp_ops_context * octx = ictx->octx; \ + struct htp_thread_trace * restrict tr = &octx->ctx->trace[ith]; \ + const struct htp_tensor * restrict src0 = octx->src[0]; \ + const struct htp_tensor * restrict src1 = octx->src[1]; \ + const struct htp_tensor * restrict dst = octx->dst; \ + const int32_t s0 = octx->op_params[0]; \ + const int32_t s1 = octx->op_params[1]; \ + const int32_t p0 = octx->op_params[2]; \ + const int32_t p1 = octx->op_params[3]; \ + const int32_t d0 = octx->op_params[4]; \ + const int32_t d1 = octx->op_params[5]; \ + const int32_t is_2D = octx->op_params[6] == 1; \ + const uint32_t N = is_2D ? src1->ne[3] : src1->ne[2]; \ + const uint32_t IC = is_2D ? src1->ne[2] : src1->ne[1]; \ + const uint32_t IH = is_2D ? src1->ne[1] : 1; \ + const uint32_t IW = src1->ne[0]; \ + const uint32_t KH = is_2D ? src0->ne[1] : 1; \ + const uint32_t KW = src0->ne[0]; \ + const uint32_t OH = is_2D ? dst->ne[2] : 1; \ + const uint32_t OW = dst->ne[1]; \ + const uint32_t patch_stride = IC * KH * KW; \ + const float * restrict src_data = (const float *) (uintptr_t) src1->data; \ + DST_CTYPE * restrict dst_data = (DST_CTYPE *) (uintptr_t) dst->data; \ + const uint32_t patch_end = ictx->patch_base + ictx->npatches; \ + const uint32_t patch_start = ictx->patch_base + ictx->npatches_per_thread * ith; \ + const uint32_t patch_stop = MIN(patch_start + ictx->npatches_per_thread, patch_end); \ + if (patch_start >= patch_stop) { \ + return; \ + } \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, patch_start); \ + for (uint32_t p = patch_start; p < patch_stop; p++) { \ + const uint32_t iow = p % OW; \ + const uint32_t ioh = (p / OW) % OH; \ + const uint32_t in = p / (OW * OH); \ + DST_CTYPE * restrict dst_patch = dst_data + (uint64_t) p * patch_stride; \ + for (uint32_t iic = 0; iic < IC; iic++) { \ + const float * restrict src_plane = src_data + ((uint64_t) in * IC + iic) * IH * IW; \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + const int32_t iih = (int32_t) ioh * s1 + (int32_t) ikh * d1 - p1; \ + DST_CTYPE * restrict out_run = dst_patch + iic * (KH * KW) + ikh * KW; \ + if (iih < 0 || iih >= (int32_t) IH) { \ + SPLAT_FN(out_run, 0.0f, KW); \ + continue; \ + } \ + const int32_t iiw0 = (int32_t) iow * s0 - p0; \ + const float * restrict src_run = src_plane + (uint64_t) iih * IW + iiw0; \ + if (d0 == 1) { \ + /* contiguous source run: [lo,hi) is in-bounds, tails are zero pad */ \ + const int32_t lo = iiw0 < 0 ? -iiw0 : 0; \ + int32_t hi = (int32_t) IW - iiw0; \ + if (hi > (int32_t) KW) { \ + hi = (int32_t) KW; \ + } \ + if (hi <= lo) { \ + SPLAT_FN(out_run, 0.0f, KW); \ + } else { \ + if (lo > 0) { \ + SPLAT_FN(out_run, 0.0f, (uint32_t) lo); \ + } \ + COPY_FN((uint8_t *) (out_run + lo), (const uint8_t *) (src_run + lo), \ + (uint32_t) (hi - lo)); \ + if (hi < (int32_t) KW) { \ + SPLAT_FN(out_run + hi, 0.0f, (KW - (uint32_t) hi)); \ + } \ + } \ + continue; \ + } \ + for (uint32_t ikw = 0; ikw < KW; ikw++) { \ + const int32_t iiw = (int32_t) iow * s0 + (int32_t) ikw * d0 - p0; \ + out_run[ikw] = (iiw < 0 || iiw >= (int32_t) IW) ? \ + (DST_CTYPE) 0.0f : \ + (DST_CTYPE) src_plane[(uint64_t) iih * IW + iiw]; \ + } \ + } \ + } \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, patch_start); \ } IM2COL_PATCHEMBED_BODY(im2col_patchembed_thread, __fp16, hvx_copy_f16_f32_uu, hvx_splat_f16_u, sizeof(__fp16), "f32-f16") IM2COL_PATCHEMBED_BODY(im2col_patchembed_f32_thread, float, hvx_copy_f32_uu, hvx_splat_f32_u, sizeof(float), "f32-f32") -#define IM2COL_PATCHEMBED_DMA_BODY(FNAME, DST_CTYPE, COPY_FN, SPLAT_FN, DST_ELEM, TAG) \ - static void FNAME(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_im2col_context * ictx = (struct htp_im2col_context *) data; \ - struct htp_ops_context * octx = ictx->octx; \ - struct htp_thread_trace * restrict tr = &octx->ctx->trace[ith]; \ - const struct htp_tensor * restrict src1 = octx->src[1]; \ - const struct htp_tensor * restrict dst = octx->dst; \ - const uint32_t N = src1->ne[3], IC = src1->ne[2], IH = src1->ne[1], IW = src1->ne[0]; \ - const uint32_t KH = octx->src[0]->ne[1], KW = octx->src[0]->ne[0]; \ - const uint32_t OH = dst->ne[2], OW = dst->ne[1]; \ - const uint32_t patch_stride = IC * KH * KW; \ - const float * restrict src_data = (const float *) src1->data; \ - DST_CTYPE * restrict dst_data = (DST_CTYPE *) dst->data; \ - dma_queue * dmaq = octx->ctx->dma[ith]; \ - uint8_t * src_base = ictx->pe_vtcm_src + ith * ictx->pe_src_size_per_thread; \ - uint8_t * dst_base = ictx->pe_vtcm_dst + ith * ictx->pe_dst_size_per_thread; \ - float * srcb = (float *) src_base; \ - DST_CTYPE * dstb = (DST_CTYPE *) dst_base; \ - const uint32_t nrows = N * OH; \ - const uint32_t per_thread = ictx->pe_rows_per_thread; \ - const uint32_t row_start = per_thread * ith; \ - const uint32_t row_end = MIN(row_start + per_thread, nrows); \ - if (row_start >= row_end) \ - return; \ - for (uint32_t r = row_start; r < row_end; r++) { \ - const uint32_t in = r / OH; \ - const uint32_t ioh = r % OH; \ - for (uint32_t ikh = 0; ikh < KH; ikh++) { \ - int32_t iih = (int32_t) ioh * (int32_t) KH + (int32_t) ikh; \ - int ok = (iih >= 0 && iih < (int32_t) IH); \ - for (uint32_t iic = 0; iic < IC; iic++) { \ - float * vdst = srcb + ((uint64_t) (iic * KH + ikh)) * IW; \ - const float * _vsrc = \ - ok ? (src_data + ((uint64_t) (in * IC + iic) * IH + iih) * IW) : (const float *) vdst; \ - dma_queue_push_ddr_to_vtcm( \ - dmaq, dma_make_ptr((uint8_t *) vdst, ok ? (const uint8_t *) _vsrc : (const uint8_t *) vdst), \ - IW * sizeof(float), IW * sizeof(float), ok ? 1 : 0); \ - } \ - } \ - for (uint32_t i = 0; i < IC * KH; i++) \ - dma_queue_pop(dmaq); \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); \ - for (uint32_t iow = 0; iow < OW; iow++) { \ - DST_CTYPE * dst_patch = dstb + (uint64_t) iow * patch_stride; \ - for (uint32_t ikh = 0; ikh < KH; ikh++) { \ - int32_t iih = (int32_t) ioh * (int32_t) KH + (int32_t) ikh; \ - for (uint32_t iic = 0; iic < IC; iic++) { \ - DST_CTYPE * out_run = dst_patch + iic * (KH * KW) + ikh * KW; \ - if (iih < 0 || iih >= (int32_t) IH) { \ - SPLAT_FN(out_run, 0.0f, KW); \ - continue; \ - } \ - const float * src_run = srcb + ((uint64_t) (iic * KH + ikh)) * IW + (uint64_t) iow * KW; \ - COPY_FN((uint8_t *) out_run, (const uint8_t *) src_run, KW); \ - } \ - } \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, r); \ - DST_CTYPE * ddr_row = dst_data + ((uint64_t) (in * OH + ioh) * OW) * patch_stride; \ - dma_queue_push_vtcm_to_ddr(dmaq, dma_make_ptr((uint8_t *) ddr_row, (uint8_t *) dstb), \ - OW * patch_stride * (DST_ELEM), OW * patch_stride * (DST_ELEM), 1); \ - dma_queue_flush(dmaq); \ - } \ +// Software-pipelined 2-deep: while HVX computes block bi from buffer slot +// (bi&1), the DMA engine stages block bi+1 into the other slot concurrently. +// A single dma_queue_flush per iteration (after issuing the next stage-in and +// this block's store-out) waits for both - safe because the ring is strict +// FIFO and each buffer slot is only reused after its prior consumer (compute +// or store-out) already finished in program order. +#define IM2COL_BLOCKED_DMA_BODY(FNAME, DST_CTYPE, COPY_FN, SPLAT_FN, DST_ELEM, TAG) \ + static void FNAME(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_im2col_context * ictx = (struct htp_im2col_context *) data; \ + struct htp_ops_context * octx = ictx->octx; \ + struct htp_thread_trace * restrict tr = &octx->ctx->trace[ith]; \ + const struct htp_tensor * restrict src1 = octx->src[1]; \ + const struct htp_tensor * restrict dst = octx->dst; \ + const int32_t s0 = octx->op_params[0], s1 = octx->op_params[1]; \ + const int32_t p0 = octx->op_params[2], p1 = octx->op_params[3]; \ + const int32_t d0 = octx->op_params[4], d1 = octx->op_params[5]; \ + const int32_t is_2D = octx->op_params[6] == 1; \ + const uint32_t N = is_2D ? src1->ne[3] : src1->ne[2]; \ + const uint32_t IC = is_2D ? src1->ne[2] : src1->ne[1]; \ + const uint32_t IH = is_2D ? src1->ne[1] : 1; \ + const uint32_t IW = src1->ne[0]; \ + const uint32_t KH = is_2D ? octx->src[0]->ne[1] : 1; \ + const uint32_t KW = octx->src[0]->ne[0]; \ + const uint32_t OH = is_2D ? dst->ne[2] : 1; \ + const uint32_t OW = dst->ne[1]; \ + const uint32_t owb = ictx->pe_owb, Wb = ictx->pe_wb; \ + const uint32_t patch_stride = IC * KH * KW; \ + const dma_addr_t src_data = src1->data; \ + const dma_addr_t dst_data = dst->data; \ + dma_queue * dmaq = octx->ctx->dma[ith]; \ + uint8_t * srcb_base = ictx->pe_vtcm_src + ith * ictx->pe_src_size_per_thread; \ + uint8_t * dstb_base = ictx->pe_vtcm_dst + ith * ictx->pe_dst_size_per_thread; \ + float * srcb2[2] = { (float *) srcb_base, (float *) (srcb_base + ictx->pe_src_row_bytes) }; \ + DST_CTYPE * dstb2[2] = { (DST_CTYPE *) dstb_base, (DST_CTYPE *) (dstb_base + ictx->pe_dst_row_bytes) }; \ + const uint32_t nrows = N * OH; \ + const uint32_t per_thread = ictx->pe_rows_per_thread; \ + const uint32_t row_start = per_thread * ith; \ + const uint32_t row_end = MIN(row_start + per_thread, nrows); \ + if (row_start >= row_end) \ + return; \ + const uint32_t nbpr = (OW + owb - 1) / owb; \ + const uint32_t nrows_local = row_end - row_start; \ + const uint32_t total_blocks = nrows_local * nbpr; \ + for (uint32_t bi = 0; bi < total_blocks; bi++) { \ + const uint32_t buf = bi & 1u; \ + float * srcb = srcb2[buf]; \ + DST_CTYPE * dstb = dstb2[buf]; \ + const uint32_t r = row_start + bi / nbpr; \ + const uint32_t in = r / OH; \ + const uint32_t ioh = r % OH; \ + const uint32_t c0 = (bi % nbpr) * owb; \ + const uint32_t nb = MIN(owb, OW - c0); \ + const int32_t win0 = (int32_t) c0 * s0 - p0; \ + if (bi == 0) { \ + /* prologue: stage block 0 and wait - nothing to overlap with yet */ \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + const int32_t iih = (int32_t) ioh * s1 + (int32_t) ikh * d1 - p1; \ + if (iih < 0 || iih >= (int32_t) IH) \ + continue; \ + const int32_t lo = win0 < 0 ? -win0 : 0; \ + int32_t hi = (int32_t) IW - win0; \ + if (hi > (int32_t) Wb) \ + hi = (int32_t) Wb; \ + if (hi <= lo) \ + continue; \ + const uint32_t cpw = (uint32_t) (hi - lo); \ + float * vdst = srcb + (size_t) ikh * Wb + lo; \ + const dma_addr_t vsrc = src_data + (size_t) (((in * IC) * IH + iih) * IW + (win0 + lo)) * sizeof(float); \ + while (!dma_queue_push(dmaq, dma_make_data(vdst, vsrc), \ + (size_t) KH * Wb * sizeof(float), (size_t) IH * IW * sizeof(float), \ + cpw * sizeof(float), IC)) { \ + dma_queue_pop(dmaq); \ + } \ + } \ + dma_queue_flush(dmaq); \ + } \ + if (bi + 1 < total_blocks) { \ + /* prefetch: stage block bi+1 into the other slot; overlaps with this block's compute below */ \ + const uint32_t nbuf = 1u - buf; \ + float * nsrcb = srcb2[nbuf]; \ + const uint32_t nr = row_start + (bi + 1) / nbpr; \ + const uint32_t nin = nr / OH; \ + const uint32_t nioh = nr % OH; \ + const uint32_t nc0 = ((bi + 1) % nbpr) * owb; \ + const int32_t nwin0 = (int32_t) nc0 * s0 - p0; \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + const int32_t iih = (int32_t) nioh * s1 + (int32_t) ikh * d1 - p1; \ + if (iih < 0 || iih >= (int32_t) IH) \ + continue; \ + const int32_t lo = nwin0 < 0 ? -nwin0 : 0; \ + int32_t hi = (int32_t) IW - nwin0; \ + if (hi > (int32_t) Wb) \ + hi = (int32_t) Wb; \ + if (hi <= lo) \ + continue; \ + const uint32_t cpw = (uint32_t) (hi - lo); \ + float * vdst = nsrcb + (size_t) ikh * Wb + lo; \ + const dma_addr_t vsrc = src_data + (size_t) (((nin * IC) * IH + iih) * IW + (nwin0 + lo)) * sizeof(float); \ + while (!dma_queue_push(dmaq, dma_make_data(vdst, vsrc), \ + (size_t) KH * Wb * sizeof(float), (size_t) IH * IW * sizeof(float), \ + cpw * sizeof(float), IC)) { \ + dma_queue_pop(dmaq); \ + } \ + } \ + } \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); \ + for (uint32_t j = 0; j < nb; j++) { \ + const uint32_t iow = c0 + j; \ + DST_CTYPE * dst_patch = dstb + (uint64_t) j * patch_stride; \ + const int32_t iiw0 = (int32_t) iow * s0 - p0; \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + const int32_t iih = (int32_t) ioh * s1 + (int32_t) ikh * d1 - p1; \ + const int okh = (iih >= 0 && iih < (int32_t) IH); \ + for (uint32_t iic = 0; iic < IC; iic++) { \ + DST_CTYPE * out_run = dst_patch + iic * (KH * KW) + ikh * KW; \ + if (!okh) { \ + SPLAT_FN(out_run, 0.0f, KW); \ + continue; \ + } \ + const float * vrow = srcb + ((uint64_t) (iic * KH + ikh)) * Wb; /* col win0 at idx 0*/ \ + if (d0 == 1) { \ + /* contiguous run within the staged window: [lo,hi) in-bounds, tails zero pad */ \ + const int32_t lo = iiw0 < 0 ? -iiw0 : 0; \ + int32_t hi = (int32_t) IW - iiw0; \ + if (hi > (int32_t) KW) { \ + hi = (int32_t) KW; \ + } \ + if (hi <= lo) { \ + SPLAT_FN(out_run, 0.0f, KW); \ + } else { \ + if (lo > 0) { \ + SPLAT_FN(out_run, 0.0f, (uint32_t) lo); \ + } \ + COPY_FN((uint8_t *) (out_run + lo), (const uint8_t *) (vrow + (iiw0 + lo - win0)), \ + (uint32_t) (hi - lo)); \ + if (hi < (int32_t) KW) { \ + SPLAT_FN(out_run + hi, 0.0f, (KW - (uint32_t) hi)); \ + } \ + } \ + continue; \ + } \ + for (uint32_t ikw = 0; ikw < KW; ikw++) { \ + const int32_t iiw = iiw0 + (int32_t) ikw * d0; \ + out_run[ikw] = \ + (iiw < 0 || iiw >= (int32_t) IW) ? (DST_CTYPE) 0.0f : (DST_CTYPE) vrow[iiw - win0]; \ + } \ + } \ + } \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, r); \ + const dma_addr_t ddr = dst_data + (size_t) ((in * OH + ioh) * OW + c0) * patch_stride * (DST_ELEM); \ + dma_queue_push(dmaq, dma_make_data(ddr, dstb), \ + nb * patch_stride * (DST_ELEM), nb * patch_stride * (DST_ELEM), \ + nb * patch_stride * (DST_ELEM), 1); \ + dma_queue_flush(dmaq); \ + } \ + } +IM2COL_BLOCKED_DMA_BODY(im2col_blocked_dma_thread, __fp16, hvx_copy_f16_f32_uu, hvx_splat_f16_u, sizeof(__fp16), "blk-dma-f16") +IM2COL_BLOCKED_DMA_BODY(im2col_blocked_dma_f32_thread, float, hvx_copy_f32_uu, hvx_splat_f32_u, sizeof(float), "blk-dma-f32") + +// Exact-tiling patch-embed DMA fast path (s0==KW, p0=0, d0=1; and 2D s1==KH, +// p1=0, d1=1). Intentionally reads no stride/pad/dilation params so the inner +// copy stays tight and fully hoisted - do NOT graft the general gather in here. +#define IM2COL_PATCHEMBED_DMA_BODY(FNAME, DST_CTYPE, COPY_FN, SPLAT_FN, DST_ELEM, TAG) \ + static void FNAME(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_im2col_context * ictx = (struct htp_im2col_context *) data; \ + struct htp_ops_context * octx = ictx->octx; \ + struct htp_thread_trace * restrict tr = &octx->ctx->trace[ith]; \ + const struct htp_tensor * restrict src1 = octx->src[1]; \ + const struct htp_tensor * restrict dst = octx->dst; \ + const int32_t is_2D = octx->op_params[6] == 1; \ + const uint32_t N = is_2D ? src1->ne[3] : src1->ne[2]; \ + const uint32_t IC = is_2D ? src1->ne[2] : src1->ne[1]; \ + const uint32_t IH = is_2D ? src1->ne[1] : 1; \ + const uint32_t IW = src1->ne[0]; \ + const uint32_t KH = is_2D ? octx->src[0]->ne[1] : 1; \ + const uint32_t KW = octx->src[0]->ne[0]; \ + const uint32_t OH = is_2D ? dst->ne[2] : 1; \ + const uint32_t OW = dst->ne[1]; \ + const uint32_t patch_stride = IC * KH * KW; \ + const dma_addr_t src_data = src1->data; \ + const dma_addr_t dst_data = dst->data; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + uint8_t * src_base = ictx->pe_vtcm_src + ith * ictx->pe_src_size_per_thread; \ + uint8_t * dst_base = ictx->pe_vtcm_dst + ith * ictx->pe_dst_size_per_thread; \ + float * srcb = (float *) src_base; \ + DST_CTYPE * dstb = (DST_CTYPE *) dst_base; \ + const uint32_t row_end_max = ictx->pe_row_base + ictx->pe_nrows; \ + const uint32_t per_thread = ictx->pe_rows_per_thread; \ + const uint32_t row_start = ictx->pe_row_base + per_thread * ith; \ + const uint32_t row_end = MIN(row_start + per_thread, row_end_max); \ + if (row_start >= row_end) \ + return; \ + for (uint32_t r = row_start; r < row_end; r++) { \ + const uint32_t in = r / OH; \ + const uint32_t ioh = r % OH; \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + int32_t iih = (int32_t) ioh * (int32_t) KH + (int32_t) ikh; \ + int ok = (iih >= 0 && iih < (int32_t) IH); \ + for (uint32_t iic = 0; iic < IC; iic++) { \ + float * vdst = srcb + (size_t) (iic * KH + ikh) * IW; \ + const dma_addr_t vsrc = ok \ + ? (src_data + (size_t) ((in * IC + iic) * IH + iih) * IW * sizeof(float)) \ + : src_data; \ + dma_queue_push(dma_q, dma_make_data(vdst, vsrc), \ + IW * sizeof(float), IW * sizeof(float), IW * sizeof(float), ok ? 1 : 0); \ + } \ + } \ + for (uint32_t i = 0; i < IC * KH; i++) \ + dma_queue_pop(dma_q); \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); \ + for (uint32_t iow = 0; iow < OW; iow++) { \ + DST_CTYPE * dst_patch = dstb + (uint64_t) iow * patch_stride; \ + for (uint32_t ikh = 0; ikh < KH; ikh++) { \ + int32_t iih = (int32_t) ioh * (int32_t) KH + (int32_t) ikh; \ + for (uint32_t iic = 0; iic < IC; iic++) { \ + DST_CTYPE * out_run = dst_patch + iic * (KH * KW) + ikh * KW; \ + if (iih < 0 || iih >= (int32_t) IH) { \ + SPLAT_FN(out_run, 0.0f, KW); \ + continue; \ + } \ + const float * src_run = srcb + ((uint64_t) (iic * KH + ikh)) * IW + (uint64_t) iow * KW; \ + COPY_FN((uint8_t *) out_run, (const uint8_t *) src_run, KW); \ + } \ + } \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, r); \ + const dma_addr_t ddr_row = dst_data + (size_t) (in * OH + ioh) * OW * patch_stride * (DST_ELEM); \ + dma_queue_push(dma_q, dma_make_data(ddr_row, dstb), \ + OW * patch_stride * (DST_ELEM), OW * patch_stride * (DST_ELEM), \ + OW * patch_stride * (DST_ELEM), 1); \ + dma_queue_flush(dma_q); \ + } \ } IM2COL_PATCHEMBED_DMA_BODY(im2col_patchembed_dma_thread, __fp16, hvx_copy_f16_f32_uu, hvx_splat_f16_u, sizeof(__fp16), "pe-dma-f16") @@ -209,21 +390,30 @@ static bool im2col_use_patchembed_dma(const struct htp_ops_context * octx) { const int32_t p0 = octx->op_params[2], p1 = octx->op_params[3]; const int32_t d0 = octx->op_params[4], d1 = octx->op_params[5]; const int is_2D = octx->op_params[6] == 1; - if (!is_2D) { - return false; - } if (octx->dst->type != HTP_TYPE_F16 && octx->dst->type != HTP_TYPE_F32) { return false; } - const uint32_t KH = octx->src[0]->ne[1], KW = octx->src[0]->ne[0]; - if (s0 != (int32_t) KW || s1 != (int32_t) KH) { - return false; // non-overlapping + const uint32_t KH = is_2D ? octx->src[0]->ne[1] : 1; + const uint32_t KW = octx->src[0]->ne[0]; + if (s0 != (int32_t) KW) { + return false; // non-overlapping (width) + } + if (p0 != 0) { + return false; // no padding (width) } - if (p0 != 0 || p1 != 0) { - return false; // no padding + if (d0 != 1) { + return false; // no dilation (width) } - if (d0 != 1 || d1 != 1) { - return false; // no dilation + if (is_2D) { + if (s1 != (int32_t) KH) { + return false; // non-overlapping (height) + } + if (p1 != 0) { + return false; // no padding (height) + } + if (d1 != 1) { + return false; // no dilation (height) + } } return true; } @@ -233,8 +423,11 @@ static bool im2col_use_patchembed_dma(const struct htp_ops_context * octx) { static bool im2col_patchembed_dma_fits(struct htp_ops_context * octx, struct htp_im2col_context * ictx, uint32_t n_threads) { - const uint32_t IC = octx->src[1]->ne[2], IW = octx->src[1]->ne[0]; - const uint32_t KH = octx->src[0]->ne[1], KW = octx->src[0]->ne[0]; + const int32_t is_2D = octx->op_params[6] == 1; + const uint32_t IC = is_2D ? octx->src[1]->ne[2] : octx->src[1]->ne[1]; + const uint32_t IW = octx->src[1]->ne[0]; + const uint32_t KH = is_2D ? octx->src[0]->ne[1] : 1; + const uint32_t KW = octx->src[0]->ne[0]; const uint32_t OW = octx->dst->ne[1]; const uint32_t patch_stride = IC * KH * KW; @@ -257,6 +450,45 @@ static bool im2col_patchembed_dma_fits(struct htp_ops_context * octx, return true; } +// Sizes a per-thread 2x(src,dst) VTCM ping-pong for the blocked general kernel. +// Stages Wb=(owb-1)*s0+(KW-1)*d0+1 source cols per (iic,ikh) row and owb patches +// of dst. Picks the largest owb that fits; returns false if even owb=1 does not. +static bool im2col_blocked_dma_fits(struct htp_ops_context * octx, + struct htp_im2col_context * ictx, + uint32_t n_threads) { + const int32_t is_2D = octx->op_params[6] == 1; + const int32_t s0 = octx->op_params[0]; + const int32_t d0 = octx->op_params[4]; + const uint32_t IC = is_2D ? octx->src[1]->ne[2] : octx->src[1]->ne[1]; + const uint32_t KH = is_2D ? octx->src[0]->ne[1] : 1; + const uint32_t KW = octx->src[0]->ne[0]; + const uint32_t OW = octx->dst->ne[1]; + const uint32_t patch_stride = IC * KH * KW; + const uint32_t dst_elem = (octx->dst->type == HTP_TYPE_F16) ? sizeof(__fp16) : sizeof(float); + + for (uint32_t owb = (OW < 256 ? OW : 256); owb >= 1; owb--) { + const uint32_t Wb = (owb - 1) * (uint32_t) s0 + (KW - 1) * (uint32_t) d0 + 1; + const uint32_t src_row_bytes = hex_round_up(IC * KH * Wb * sizeof(float), 256); + const uint32_t dst_row_bytes = hex_round_up(owb * patch_stride * dst_elem, 256); + struct htp_im2col_vtcm_layout L; + htp_im2col_vtcm_layout_build(&L, src_row_bytes, dst_row_bytes, n_threads); + if (L.total_bytes <= octx->ctx->vtcm_size) { + uint8_t * const base = octx->ctx->vtcm_base; + ictx->pe_owb = owb; + ictx->pe_wb = Wb; + ictx->pe_src_row_bytes = src_row_bytes; + ictx->pe_dst_row_bytes = dst_row_bytes; + ictx->pe_vtcm_src = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src); + ictx->pe_vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, L.off_dst); + ictx->pe_src_size_per_thread = (uint32_t) L.src_bytes_per_thread; + ictx->pe_dst_size_per_thread = (uint32_t) L.dst_bytes_per_thread; + return true; + } + if (owb == 1) break; // avoid unsigned underflow + } + return false; +} + int op_im2col(struct htp_ops_context * octx) { const struct htp_tensor * src1 = octx->src[1]; const struct htp_tensor * dst = octx->dst; @@ -266,35 +498,83 @@ int op_im2col(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - const uint32_t N = src1->ne[3]; - const uint32_t OH = dst->ne[2]; - const uint32_t OW = dst->ne[1]; - const uint32_t npatches = N * OH * OW; - const uint32_t n_threads = MIN(octx->n_threads, npatches); + const int32_t is_2D = octx->op_params[6] == 1; + const uint32_t N = is_2D ? src1->ne[3] : src1->ne[2]; + const uint32_t OH = is_2D ? dst->ne[2] : 1; + const uint32_t OW = dst->ne[1]; + const uint32_t total_patches = N * OH * OW; + const uint32_t total_rows = N * OH; + + uint32_t patch_base = 0; + uint32_t npatches = total_patches; + if (octx->ctx->mdev.count > 1) { + const uint32_t patch_size = dst->nb[1]; + const uint32_t patches_per_chunk = + (patch_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(patch_size, HEX_L2_LINE_SIZE)) : 1; + const struct htp_tensor_mdev_range range = + htp_tensor_mdev_partition(total_patches, htp_tensor_mdev_data_aligned(dst) ? patches_per_chunk : 0, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + patch_base = range.start; + npatches = range.count; + } + + uint32_t row_base = 0; + uint32_t nrows = total_rows; + if (octx->ctx->mdev.count > 1) { + const uint32_t row_size = dst->nb[2]; + const uint32_t rows_per_chunk = + (row_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(row_size, HEX_L2_LINE_SIZE)) : 1; + const struct htp_tensor_mdev_range range = + htp_tensor_mdev_partition(total_rows, htp_tensor_mdev_data_aligned(dst) ? rows_per_chunk : 0, + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_base = range.start; + nrows = range.count; + } - if ((octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) || n_threads == 0) { + if (npatches == 0 && nrows == 0) { return HTP_STATUS_OK; } + const uint32_t n_threads = MIN(octx->n_threads, MAX(npatches, 1)); + struct htp_im2col_context ictx = { 0 }; - ictx.octx = octx; - ictx.npatches_per_thread = (npatches + n_threads - 1) / n_threads; + ictx.octx = octx; + ictx.patch_base = patch_base; + ictx.npatches = npatches; + ictx.npatches_per_thread = (npatches + n_threads - 1) / n_threads; // Clean non-overlapping patch-embed -> DMA kernel (if it fits VTCM); - // everything else (padding/dilation/stride edges) -> pure-DDR kernel. - if (im2col_use_patchembed_dma(octx)) { - const uint32_t nrows = N * OH; - const uint32_t pth = MIN(octx->n_threads, nrows); - if (pth > 0 && im2col_patchembed_dma_fits(octx, &ictx, pth)) { - ictx.pe_rows_per_thread = (nrows + pth - 1) / pth; - if (dst->type == HTP_TYPE_F16) { - work_queue_run(octx->ctx->work_queue, im2col_patchembed_dma_thread, &ictx, pth); - } else { - work_queue_run(octx->ctx->work_queue, im2col_patchembed_dma_f32_thread, &ictx, pth); + // everything else (padding/dilation/stride edges) -> blocked-staging DMA + // kernel; if neither fits VTCM -> pure-DDR kernel. + if (nrows > 0) { + const uint32_t pth = MIN(octx->n_threads, nrows); + if (pth > 0) { + ictx.pe_row_base = row_base; + ictx.pe_nrows = nrows; + const bool exact = im2col_use_patchembed_dma(octx); + if (exact && im2col_patchembed_dma_fits(octx, &ictx, pth)) { + ictx.pe_rows_per_thread = (nrows + pth - 1) / pth; + work_queue_run(octx->ctx->work_queue, + dst->type == HTP_TYPE_F16 ? im2col_patchembed_dma_thread + : im2col_patchembed_dma_f32_thread, &ictx, pth); + return HTP_STATUS_OK; + } + if (!exact && im2col_blocked_dma_fits(octx, &ictx, pth)) { + ictx.pe_rows_per_thread = (nrows + pth - 1) / pth; + work_queue_run(octx->ctx->work_queue, + dst->type == HTP_TYPE_F16 ? im2col_blocked_dma_thread + : im2col_blocked_dma_f32_thread, &ictx, pth); + return HTP_STATUS_OK; } - return HTP_STATUS_OK; } - // else: doesn't fit -> fall through to the pure-DDR kernel below. + } + // Fall through to pure-DDR. + if (npatches == 0) { + return HTP_STATUS_OK; + } + + if (htp_tensor_is_extended(src1) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; } if (dst->type == HTP_TYPE_F16) { diff --git a/ggml/src/ggml-hexagon/htp/main.c b/ggml/src/ggml-hexagon/htp/main.c index 880e20c9..b4b352b2 100644 --- a/ggml/src/ggml-hexagon/htp/main.c +++ b/ggml/src/ggml-hexagon/htp/main.c @@ -18,9 +18,10 @@ #include #include #include +#include #include "hex-utils.h" -#include "hex-dma.h" +#include "dma-queue.h" #include "hmx-queue.h" #define GGML_COMMON_DECL_C @@ -32,8 +33,10 @@ #include "htp_iface.h" #include "work-queue.h" #include "hex-profile.h" +#include "allreduce-ops.h" +#include "htp-fence.h" -#define HMX_QUEUE_CAPACITY 16 +#define HMX_QUEUE_CAPACITY 128 #define HMX_QUEUE_STACK_SIZE 16384 #define WORK_QUEUE_CAPACITY 16 #define WORK_QUEUE_STACK_SIZE 16384 @@ -46,6 +49,77 @@ struct htp_handle { struct htp_context * ctx; }; +static inline uint64_t htp_mmap(uint32_t fd, uint64_t size, uint32_t flags) { +#if __HVX_ARCH__ > 79 + if (flags & HTP_BUF_EXTENDED) { + HAP_mem_req_payload_t payload; + memset(&payload, 0, sizeof(payload)); + payload.request_id = HAP_MEM_MAP; + payload.mmap.len = size; + payload.mmap.prot = HAP_MEM_CACHE_NON_SHARED | HAP_PROT_READ; + payload.mmap.flags = HAP_MEM_FLAGS_EXTENDED_MAP; + payload.mmap.fd = fd; + + if (HAP_mem_request(&payload) != 0) { + FARF(ERROR, "extended mmap failed : fd %u size %llu", fd, (unsigned long long) size); + return 0; + } + + return payload.mmap.dsp_va; + } +#else + if (flags & HTP_BUF_EXTENDED) { + FARF(ERROR, "extended mmap is unsupported on v%d", __HVX_ARCH__); + return 0; + } +#endif + + if (size > UINT32_MAX) { + FARF(ERROR, "mmap failed : size %llu exceeds 32-bit limit", (unsigned long long) size); + return 0; + } + + void * va = (void *)-1; + for (int retry = 0; retry < 2; retry++) { +#if __HVX_ARCH__ > 73 + va = HAP_mmap2(NULL, (size_t) size, HAP_PROT_READ | HAP_PROT_WRITE, 0, fd, 0); +#else + if (size > HTP_MMAP_MAX_VMEM) { + FARF(ERROR, "mmap failed : size %llu exceeds 2GB limit for HAP_mmap", (unsigned long long) size); + abort(); + } + va = HAP_mmap(NULL, (int) size, HAP_PROT_READ | HAP_PROT_WRITE, 0, fd, 0); +#endif + if (va != (void *)-1 && va != NULL) { + return (uint64_t) (uintptr_t) va; + } + if (retry == 0) { + FARF(HIGH, "mmap failed first try (va %p fd %u size %llu), retrying...", va, fd, (unsigned long long) size); + } + } + return 0; +} + +static inline void htp_munmap(uint64_t va, uint64_t size, uint32_t flags) { +#if __HVX_ARCH__ > 79 + if (flags & HTP_BUF_EXTENDED) { + HAP_mem_req_payload_t payload; + memset(&payload, 0, sizeof(payload)); + payload.request_id = HAP_MEM_UNMAP; + payload.munmap.dsp_va = va; + payload.munmap.len = size; + HAP_mem_request(&payload); + return; + } +#endif + +#if __HVX_ARCH__ > 73 + HAP_munmap2((void *) (uintptr_t) va, (size_t) size); +#else + HAP_munmap((void *) (uintptr_t) va, (int) size); +#endif +} + AEEResult htp_iface_open(const char * uri, remote_handle64 * handle) { (void) uri; struct htp_handle * h = calloc(1, sizeof(*h)); @@ -127,14 +201,11 @@ AEEResult htp_iface_close(remote_handle64 handle) { // release the mmaps (if any) for (uint32_t i=0; immap[i].size) { -#if __HVX_ARCH__ > 73 - HAP_munmap2((void *) ctx->mmap[i].base, ctx->mmap[i].size); -#else - HAP_munmap((void *) ctx->mmap[i].base, ctx->mmap[i].size); -#endif + htp_munmap(ctx->mmap[i].base, ctx->mmap[i].size, ctx->mmap[i].flags); ctx->mmap[i].size = 0; - ctx->mmap[i].base = NULL; + ctx->mmap[i].base = 0; ctx->mmap[i].fd = -1; + ctx->mmap[i].flags = 0; } } @@ -155,7 +226,7 @@ AEEResult htp_iface_close(remote_handle64 handle) { return AEE_SUCCESS; } -AEEResult htp_iface_mmap(remote_handle64 handle, uint32_t fd, uint32_t size) { +AEEResult htp_iface_mmap(remote_handle64 handle, uint32_t fd, uint64_t size) { struct htp_handle * h = (struct htp_handle *) handle; if (!h || !h->ctx) { return AEE_EBADPARM; @@ -174,25 +245,17 @@ AEEResult htp_iface_mmap(remote_handle64 handle, uint32_t fd, uint32_t size) { for (uint32_t i=0; immap[i]; if (!m->size) { - FARF(HIGH, "mmap : fd %u size %u", fd, size); -#if __HVX_ARCH__ > 73 - void *va = HAP_mmap2(NULL, size, HAP_PROT_READ | HAP_PROT_WRITE, 0, fd, 0); -#else - if (size > HTP_MMAP_MAX_VMEM) { // HAP_mmap has a size limit of 2GB - FARF(ERROR, "mmap failed : size %u exceeds 2GB limit for HAP_mmap", (uint32_t) size); - abort(); // can't do much else at this point - } - - void *va = HAP_mmap(NULL, size, HAP_PROT_READ | HAP_PROT_WRITE, 0, fd, 0); -#endif - if (va == (void*)-1) { - FARF(ERROR, "mmap failed : va %p fd %u size %u", va, fd, (uint32_t) size); + FARF(HIGH, "mmap : fd %u size %llu", fd, (unsigned long long) size); + uint64_t va = htp_mmap(fd, size, 0); + if (va == 0) { + FARF(ERROR, "mmap failed : fd %u size %llu", fd, (unsigned long long) size); return AEE_EFAILED; } - m->base = (uint64_t) va; + m->base = va; m->fd = fd; m->size = size; + m->flags = 0; return AEE_SUCCESS; } @@ -211,15 +274,12 @@ AEEResult htp_iface_munmap(remote_handle64 handle, uint32 fd) { for (uint32_t i=0; immap[i]; if (fd < 0 || m->fd == fd) { - FARF(HIGH, "unmmap : base %p fd %u size %u", (void*) m->base, m->fd, (uint32_t) m->size); -#if __HVX_ARCH__ > 73 - HAP_munmap2((void *) m->base, m->size); -#else - HAP_munmap((void *) m->base, m->size); -#endif + FARF(HIGH, "unmmap : base 0x%llx fd %u size %llu", (unsigned long long) m->base, m->fd, (unsigned long long) m->size); + htp_munmap(m->base, m->size, m->flags); m->size = 0; m->base = NULL; m->fd = -1; + m->flags = 0; } } @@ -228,7 +288,7 @@ AEEResult htp_iface_munmap(remote_handle64 handle, uint32 fd) { static void vtcm_acquire(struct htp_context * ctx) { if (!ctx->vtcm_valid) { - int err = HAP_compute_res_acquire_cached(ctx->vtcm_rctx, 1000000u); + int err = HAP_compute_res_acquire_cached(ctx->vtcm_rctx, 10000000u); if (err != 0) { FARF(ERROR, "ggml-hex: failed to acquire VTCM: 0x%08x", (unsigned)err); abort(); @@ -378,8 +438,6 @@ AEEResult htp_iface_start(remote_handle64 handle, uint32_t sess_id, uint64_t dsp for (uint32_t i = 0; i < n_hvx; i++) { size_dma = hex_align_up(size_dma, dma_queue_alignof()); size_dma += dma_queue_sizeof(256); - size_dma = hex_align_up(size_dma, dma_queue_alignof()); - size_dma += dma_queue_alias_sizeof(); } offset = offset_dma + size_dma; @@ -522,16 +580,11 @@ AEEResult htp_iface_start(remote_handle64 handle, uint32_t sess_id, uint64_t dsp // Initialize DMA queues uint8_t * dma_ptr_curr = (uint8_t *) ((uintptr_t) block + offset_dma); size_t size_dma_q = dma_queue_sizeof(256); - size_t size_dma_alias = dma_queue_alias_sizeof(); for (int i = 0; i < ctx->n_threads; i++) { dma_ptr_curr = (uint8_t *) hex_align_up((uintptr_t) dma_ptr_curr, dma_queue_alignof()); - ctx->dma_cached[i] = dma_queue_init(dma_ptr_curr, 256, (uintptr_t) ctx->vtcm_base, ctx->vtcm_size, &ctx->trace[i]); + ctx->dma[i] = dma_queue_init(dma_ptr_curr, 256, &ctx->trace[i]); dma_ptr_curr += size_dma_q; - - dma_ptr_curr = (uint8_t *) hex_align_up((uintptr_t) dma_ptr_curr, dma_queue_alignof()); - ctx->dma[i] = dma_queue_alias_init(dma_ptr_curr, ctx->dma_cached[i], 1); - dma_ptr_curr += size_dma_alias; } ctx->ddr_spad_size = 512 * 1024; // 512 KB @@ -592,8 +645,7 @@ AEEResult htp_iface_stop(remote_handle64 handle) { work_queue_free(ctx->work_queue); for (int i = 0; i < ctx->n_threads; i++) { - dma_queue_alias_free(ctx->dma[i]); - dma_queue_free(ctx->dma_cached[i]); + dma_queue_free(ctx->dma[i]); } if (ctx->hmx_queue) { @@ -692,8 +744,81 @@ static inline void profile_stop(uint32_t mode, struct profile_data * d) { } } +static int op_fence(struct htp_ops_context * octx) { + struct htp_context *ctx = octx->ctx; + struct htp_thread_trace * tr = &ctx->trace[0]; + const uint32_t seq = (uint32_t) octx->op_params[0]; + const uint32_t mode = (uint32_t) octx->op_params[1]; + + htp_trace_event_start(tr, HTP_TRACE_EVT_FENCE, (uint16_t) seq); + + const struct htp_tensor * sync = octx->src[0]; + atomic_uint * sync_fence = (atomic_uint *) (uintptr_t) sync->data; + + if (mode == 1) { + htp_flush_dirty_ranges(ctx); + + htp_mdev_group_barrier(octx); + + if (ctx->mdev.idx == 0) { + htp_fence_write(sync_fence, seq, octx->status); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_FENCE, (uint16_t) seq); + FARF(HIGH, "ggml-hex: sync-signal : fence %p seq 0x%x status %d\n", sync_fence, seq, octx->status); + return octx->status; + } + + int status = HTP_STATUS_OK; + uint64_t spins = 0; + while (1) { + uint32_t sync_seq; + uint32_t sync_status; + htp_fence_read(sync_fence, &sync_seq, &sync_status); + if ((int32_t)(sync_seq - seq) >= 0) { + if (sync_status > HTP_STATUS_OK) { + FARF(ERROR, "ggml-hex: sync-wait peer failed with status %u : fence %p seq 0x%x\n", sync_status, sync_fence, seq); + status = sync_status; + } + break; + } + if (++spins > HTP_FENCE_TIMEOUT) { + FARF(ERROR, "ggml-hex: sync-wait TIMEOUT : fence %p spins %llu seq 0x%x\n", sync_fence, spins, seq); + status = HTP_STATUS_INTERNAL_ERR; + break; + } + hex_pause(); + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_FENCE, (uint16_t) seq); + + FARF(HIGH, "ggml-hex: sync-done : fence %p spins %llu seq 0x%x\n", sync_fence, spins, seq); + return status; +} + +static int op_mdev_group(struct htp_ops_context * octx) { + struct htp_context * ctx = octx->ctx; + const struct htp_tensor * sync = octx->src[0]; + ctx->mdev.idx = (uint16_t) octx->op_params[0]; + ctx->mdev.count = (uint16_t) sync->ne[1]; + if (ctx->mdev.count > 1) { + ctx->mdev.count_div = init_fastdiv_values(ctx->mdev.count); + ctx->mdev.fence_base = (uint8_t *) sync->data; + } + return HTP_STATUS_OK; +} + static int execute_op(struct htp_ops_context * octx) { switch (octx->op) { + case HTP_OP_MDEV_GROUP: + return op_mdev_group(octx); + + case HTP_OP_FENCE: + return op_fence(octx); + + case HTP_OP_ALLREDUCE: + case HTP_OP_ALLREDUCE_ADD: + return op_allreduce(octx); + case HTP_OP_MUL_MAT: case HTP_OP_MUL_MAT_ADD: return op_matmul(octx); @@ -701,11 +826,11 @@ static int execute_op(struct htp_ops_context * octx) { case HTP_OP_MUL_MAT_ID: return op_matmul_id(octx); - case HTP_OP_MUL_MAT_QKV: - return op_matmul_qkv(octx); + case HTP_OP_MUL_MAT_ID_NX: + return op_matmul_id_nx(octx); - case HTP_OP_MUL_MAT_FFN: - return op_matmul_ffn(octx); + case HTP_OP_MUL_MAT_NX: + return op_matmul_nx(octx); case HTP_OP_MUL: case HTP_OP_ADD: @@ -719,6 +844,7 @@ static int execute_op(struct htp_ops_context * octx) { case HTP_OP_RMS_NORM_MUL: case HTP_OP_SCALE: case HTP_OP_CLAMP: + case HTP_OP_LEAKY_RELU: case HTP_OP_SQR: case HTP_OP_SQRT: case HTP_OP_UNARY_SOFTPLUS: @@ -728,12 +854,17 @@ static int execute_op(struct htp_ops_context * octx) { case HTP_OP_UNARY_NEG: case HTP_OP_UNARY_EXP: case HTP_OP_UNARY_TANH: + case HTP_OP_UNARY_ABS: + case HTP_OP_UNARY_LOG: + case HTP_OP_UNARY_RELU: case HTP_OP_L2_NORM: return op_unary(octx); case HTP_OP_GLU_SWIGLU: case HTP_OP_GLU_SWIGLU_OAI: + case HTP_OP_GLU_SWIGLU_CLAMP: case HTP_OP_GLU_GEGLU: + case HTP_OP_GLU_GEGLU_QUICK: return op_activations(octx); case HTP_OP_SOFTMAX: @@ -755,6 +886,7 @@ static int execute_op(struct htp_ops_context * octx) { return op_sum_rows(octx); case HTP_OP_CPY: + case HTP_OP_CPY_FENCE: return op_cpy(octx); case HTP_OP_REPEAT: @@ -763,6 +895,9 @@ static int execute_op(struct htp_ops_context * octx) { case HTP_OP_ARGSORT: return op_argsort(octx); + case HTP_OP_TOP_K: + return op_top_k(octx); + case HTP_OP_SSM_CONV: return op_ssm_conv(octx); @@ -784,6 +919,9 @@ static int execute_op(struct htp_ops_context * octx) { case HTP_OP_IM2COL: return op_im2col(octx); + case HTP_OP_ROLL: + return op_roll(octx); + case HTP_OP_CONCAT: return op_concat(octx); @@ -798,7 +936,7 @@ static int execute_op(struct htp_ops_context * octx) { } FARF(ERROR, "Unknown Op %u", octx->op); - return -1; + return HTP_STATUS_NO_SUPPORT; } static inline bool reuse_buf(struct htp_context *ctx, uint32_t *m_reuse, struct htp_buf_desc *b) { @@ -806,7 +944,7 @@ static inline bool reuse_buf(struct htp_context *ctx, uint32_t *m_reuse, struct for (uint32_t i=0; immap + i; - if (m->size && m->fd == b->fd) { + if (m->size && m->fd == b->fd && m->flags == b->flags) { b->base = m->base; *m_reuse |= (1 << i); return true; @@ -818,48 +956,40 @@ static inline bool reuse_buf(struct htp_context *ctx, uint32_t *m_reuse, struct static inline void drop_mmap(struct htp_context *ctx, struct htp_mmap *m) { if (m->size) { - FARF(HIGH, "unmap : fd %u base %p size %u", m->fd, (void*) m->base, (uint32_t) m->size); -#if __HVX_ARCH__ > 73 - HAP_munmap2((void *) m->base, m->size); -#else - HAP_munmap((void *) m->base, m->size); -#endif + FARF(ALWAYS, "unmap : fd %u base 0x%llx size %llu", m->fd, (unsigned long long) m->base, (unsigned long long) m->size); + htp_munmap(m->base, m->size, m->flags); m->size = 0; m->base = 0; m->fd = -1; + m->flags = 0; } } -static inline void mmap_buf(struct htp_context *ctx, struct htp_buf_desc *b) { - if (b->base) return; // already mapped +static inline bool mmap_buf(struct htp_context *ctx, struct htp_buf_desc *b) { + if (b->base) return true; // already mapped // find unused mapping for (uint32_t i=0; i < HTP_MAX_MMAPS; i++) { struct htp_mmap *m = &ctx->mmap[i]; if (!m->size) { -#if __HVX_ARCH__ > 73 - void *va = HAP_mmap2(NULL, b->size, HAP_PROT_READ | HAP_PROT_WRITE, 0, b->fd, 0); -#else - if (b->size > HTP_MMAP_MAX_VMEM) { // HAP_mmap has a size limit of 2GB - FARF(ERROR, "mmap failed : size %u exceeds 2GB limit for HAP_mmap", (uint32_t) b->size); - abort(); // can't do much else at this point + uint64_t va = htp_mmap(b->fd, b->size, b->flags); + if (va == 0) { + FARF(HIGH, "mmap failed (will attempt defrag) : fd %u size %llu", b->fd, (unsigned long long) b->size); + return false; } - void *va = HAP_mmap(NULL, b->size, HAP_PROT_READ | HAP_PROT_WRITE, 0, b->fd, 0); -#endif - if (va == (void*)-1) { - FARF(ERROR, "mmap failed : va %p fd %u size %u", va, b->fd, (uint32_t) b->size); - abort(); // can't do much else at this point - } - - m->base = b->base = (uint64_t) va; + m->base = b->base = va; m->fd = b->fd; m->size = b->size; + m->flags = b->flags; - FARF(HIGH, "mmap : fd %u base %p size %u", m->fd, (void*) m->base, (uint32_t) m->size); - return; + FARF(ALWAYS, "mmap : fd %u base 0x%llx size %llu flags 0x%x", m->fd, (unsigned long long) m->base, (unsigned long long) m->size, m->flags); + return true; } } + + FARF(ERROR, "mmap failed : exceeded mapping capacity limit of %u", HTP_MAX_MMAPS); + return false; } static void prep_op_bufs(struct htp_context *ctx, struct htp_buf_desc *bufs, uint32_t n_bufs) { @@ -872,14 +1002,22 @@ static void prep_op_bufs(struct htp_context *ctx, struct htp_buf_desc *bufs, uin // See what we can reuse for (uint32_t i=0; i < n_bufs; i++) { struct htp_buf_desc *b = bufs + i; - if (reuse_buf(ctx, &m_reuse, b)) { b_reuse++; } else { e_vmem += b->size; } - FARF(HIGH, "prep-buf #%u : pass0 fd %u base %p size %u flags 0x%x", i, b->fd, (void*) b->base, (uint32_t) b->size, b->flags); + if (reuse_buf(ctx, &m_reuse, b)) { + b_reuse++; + } else if (!(b->flags & HTP_BUF_EXTENDED)) { + e_vmem += b->size; + } + FARF(HIGH, "prep-buf #%u : pass0 fd %u base 0x%llx size %llu flags 0x%x", i, b->fd, (unsigned long long) b->base, (unsigned long long) b->size, b->flags); } if (b_reuse == n_bufs) return; // all bufs reuse existing mappings // See how much vmem we have mmaped right now - for (uint32_t i=0; immap[i].size; } + for (uint32_t i=0; immap[i].flags & HTP_BUF_EXTENDED)) { + m_vmem += ctx->mmap[i].size; + } + } FARF(HIGH, "prep-bufs : pass1 mmap-vmem %zu extra-vmem %zu max-vmem %zu : n-bufs %u b-reuse %u", (size_t) m_vmem, (size_t) e_vmem, (size_t) ctx->max_vmem, n_bufs, b_reuse); @@ -888,27 +1026,54 @@ static void prep_op_bufs(struct htp_context *ctx, struct htp_buf_desc *bufs, uin // Drop unused mappings for (uint32_t i=0; i < HTP_MAX_MMAPS; i++) { bool used = m_reuse & (1<mmap + i); } + if (!used && !(ctx->mmap[i].flags & HTP_BUF_EXTENDED)) { + drop_mmap(ctx, ctx->mmap + i); + } } } - // Create missing mappings + // Create missing mappings (pass 1) + bool mmap_ok = true; for (uint32_t i=0; i < n_bufs; i++) { struct htp_buf_desc *b = bufs + i; - mmap_buf(ctx, b); - FARF(HIGH, "prep-buf #%u : pass1 fd %u base %p size %u flags 0x%x", i, b->fd, (void*) b->base, (uint32_t) b->size, b->flags); + if (!mmap_buf(ctx, b)) { + mmap_ok = false; + break; + } + FARF(HIGH, "prep-buf #%u : pass1 fd %u base 0x%llx size %llu flags 0x%x", i, b->fd, (unsigned long long) b->base, (unsigned long long) b->size, b->flags); + } + + if (!mmap_ok) { + // Attempt defragmentation: drop 32-bit mappings and remap (pass 2) + FARF(HIGH, "prep-bufs : dropping 32-bit mappings to defragment address space"); + for (uint32_t i=0; i < HTP_MAX_MMAPS; i++) { + if (!(ctx->mmap[i].flags & HTP_BUF_EXTENDED)) { + drop_mmap(ctx, ctx->mmap + i); + } + } + + for (uint32_t i=0; i < n_bufs; i++) { + struct htp_buf_desc *b = bufs + i; + if (!(b->flags & HTP_BUF_EXTENDED)) { + b->base = 0; + } + if (!mmap_buf(ctx, b)) { + FARF(ERROR, "prep-bufs : mmap failed after defragmentation (fd %u size %llu)", b->fd, (unsigned long long) b->size); + abort(); + } + FARF(HIGH, "prep-buf #%u : pass2 fd %u base 0x%llx size %llu flags 0x%x", i, b->fd, (unsigned long long) b->base, (unsigned long long) b->size, b->flags); + } } } static void prep_tensor(struct htp_context *ctx, struct htp_buf_desc *bufs, struct htp_tensor *tens, uint32_t idx, struct htp_tensor *t) { - uint32_t offset = t->data; - uint32_t size = t->size; + uint64_t offset = t->data; uint32_t bi = t->bi; - t->data = (uint32_t) (bufs[bi].base + offset); // update data to the actual pointer + t->data = bufs[bi].base + offset; // update data to the actual pointer - FARF(HIGH, "prep-tensor #%u: bi %u offset %u size %u data %p : %u:%u:%u:%u", idx, t->bi, offset, t->size, (void*) t->data, - t->ne[0], t->ne[1], t->ne[3], t->ne[3]); + FARF(HIGH, "prep-tensor #%u: bi %u offset %llu size %u data 0x%llx : %u:%u:%u:%u", idx, t->bi, (unsigned long long) offset, t->size, (unsigned long long) t->data, + t->ne[0], t->ne[1], t->ne[2], t->ne[3]); } static void prep_tensors(struct htp_context *ctx, struct htp_buf_desc *bufs, struct htp_tensor *tens, uint32_t n_tens) { @@ -917,11 +1082,19 @@ static void prep_tensors(struct htp_context *ctx, struct htp_buf_desc *bufs, str } } -static int proc_op_req(struct htp_ops_context * octx, struct htp_tensor *tens, uint32_t idx, struct htp_op_desc * op) { - memcpy(octx->op_params, op->params, sizeof(octx->op_params)); +static void mdev_group_init(struct htp_context * ctx, const struct htp_opbatch_req * req) { + memset(&ctx->mdev, 0, sizeof(ctx->mdev)); + ctx->mdev.fence_seq = (uint32_t)((req->seq & 0xfffff) << 12); +} + +static int proc_op_req(struct htp_ops_context * octx, struct htp_buf_desc * bufs, uint32_t n_bufs, + struct htp_tensor * tens, uint32_t idx, struct htp_op_desc * op) { + memcpy(octx->op_params, op->params, sizeof(octx->op_params)); memcpy(octx->kernel_params, op->kernel_params, sizeof(octx->kernel_params)); - octx->flags = op->flags; - octx->op = op->opcode; + octx->flags = op->flags; + octx->op = op->opcode; + octx->n_threads = octx->ctx->n_threads; + octx->n_threads_div = octx->ctx->n_threads_div; FARF(HIGH, "proc-op #%u: opcode %u flags 0x%x", idx, octx->op, octx->flags); @@ -930,16 +1103,14 @@ static int proc_op_req(struct htp_ops_context * octx, struct htp_tensor *tens, u uint16_t src_idx = op->src[i]; if (src_idx == 0xffff) { octx->src[i] = NULL; - octx->src_dma[i] = NULL; continue; } struct htp_tensor *src = tens + src_idx; octx->src[i] = src; - octx->src_dma[i] = octx->ctx->dma; // FIXME: ? octx->ctx->dma_cached : octx->ctx->dma; - FARF(HIGH, "prep-src #%u: data %p size %u : %u:%u:%u:%u", op->src[i], (void*) src->data, src->size, - src->ne[0], src->ne[1], src->ne[3], src->ne[3]); + FARF(HIGH, "prep-src #%u: data 0x%llx size %u : %u:%u:%u:%u", op->src[i], (unsigned long long) src->data, src->size, + src->ne[0], src->ne[1], src->ne[2], src->ne[3]); } htp_tensor_flush_all(octx->ctx, octx->src, HTP_OP_MAX_INPUTS); @@ -949,20 +1120,22 @@ static int proc_op_req(struct htp_ops_context * octx, struct htp_tensor *tens, u uint16_t dst_idx = op->dst[i]; if (dst_idx == 0xffff) { octx->dsts[i] = NULL; - octx->dst_dma[i] = NULL; continue; } struct htp_tensor *dst = tens + dst_idx; octx->dsts[i] = dst; - octx->dst_dma[i] = octx->ctx->dma; // FIXME: ? octx->ctx->dma_cached : octx->ctx->dma; - FARF(HIGH, "prep-dst[%u] #%u: data %p size %u : %u:%u:%u:%u", i, dst_idx, (void*) dst->data, dst->size, + FARF(HIGH, "prep-dst[%u] #%u: data 0x%llx size %u : %u:%u:%u:%u", i, dst_idx, (unsigned long long) dst->data, dst->size, dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3]); } + htp_tensor_dirty_all(octx->ctx, octx->dsts, HTP_OP_MAX_OUTPUTS); + + htp_mdev_group_barrier(octx); + int status = execute_op(octx); - htp_tensor_dirty_all(octx->ctx, octx->dsts, HTP_OP_MAX_OUTPUTS); + htp_ops_context_set_status(octx, status); octx->src0_spad.src = NULL; octx->src1_spad.src = NULL; @@ -970,7 +1143,7 @@ static int proc_op_req(struct htp_ops_context * octx, struct htp_tensor *tens, u octx->src3_spad.src = NULL; octx->dst_spad.src = NULL; - return status; + return octx->status; } static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_req * req, const struct dspqueue_buffer * dbuf) { @@ -992,7 +1165,7 @@ static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_r return; } - FARF(HIGH, "processing opbatch #%u: n-bufs %u n-tensors %u n-ops %u n-traces %u : m-size %u b-size %u t-size %u o-size %u", req->id, + FARF(HIGH, "processing opbatch #%llu: n-bufs %u n-tensors %u n-ops %u n-traces %u : m-size %u b-size %u t-size %u o-size %u", (unsigned long long) req->seq, n_bufs, n_tens, n_ops, req->n_traces, dbuf->size, b_size, t_size, o_size); // Setup descriptor pointers @@ -1029,8 +1202,11 @@ static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_r struct htp_ops_context *octx = &ctx->octx; memset(octx, 0, sizeof(*octx)); - octx->n_threads = ctx->n_threads; - octx->ctx = ctx; + octx->n_threads = ctx->n_threads; + octx->n_threads_div = ctx->n_threads_div; + octx->ctx = ctx; + + mdev_group_init(ctx, req); work_queue_wakeup(ctx->work_queue); if (ctx->hmx_queue) { @@ -1038,15 +1214,18 @@ static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_r } int op_status = HTP_STATUS_OK; - for (uint32_t i = 0; i < n_ops && op_status == HTP_STATUS_OK; i++) { + octx->status = HTP_STATUS_OK; + for (uint32_t i = 0; i < n_ops; i++) { struct profile_data prof; profile_start(ctx->profiler, &prof); - op_status = proc_op_req(octx, tens, i, &ops[i]); + op_status = proc_op_req(octx, bufs, n_bufs, tens, i, &ops[i]); profile_stop(ctx->profiler, &prof); + htp_ops_context_set_status(octx, op_status); + if (ctx->profiler) { pds[i].opcode = ops[i].opcode; pds[i].usecs = prof.usecs; @@ -1069,12 +1248,14 @@ static void process_opbatch(struct htp_context * ctx, const struct htp_opbatch_r qurt_mem_cache_clean((qurt_addr_t) 0, 0, QURT_MEM_CACHE_FLUSH_INVALIDATE_ALL, QURT_MEM_DCACHE); htp_trace_event_stop(&ctx->trace[0], HTP_TRACE_EVT_L2FLUSH, 0); + htp_mdev_group_barrier(octx); + profile_stop(HTP_PROF_BASIC, &batch_prof); struct htp_opbatch_rsp rsp; memset(&rsp, 0, sizeof(rsp)); - rsp.id = req->id; - rsp.status = op_status; + rsp.seq = req->seq; + rsp.status = octx->status; rsp.n_bufs = n_bufs; rsp.n_tensors = n_tens; rsp.n_ops = n_ops; diff --git a/ggml/src/ggml-hexagon/htp/matmul-ops.c b/ggml/src/ggml-hexagon/htp/matmul-ops.c index 9d385469..a09bc7a2 100644 --- a/ggml/src/ggml-hexagon/htp/matmul-ops.c +++ b/ggml/src/ggml-hexagon/htp/matmul-ops.c @@ -11,7 +11,7 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #include "hvx-dump.h" #include "hvx-arith.h" @@ -21,25 +21,17 @@ #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" #include "matmul-ops.h" #include "htp-vtcm.h" -static void hvx_tensor_add_f32_grid( - const struct htp_tensor * restrict dst, - const struct htp_tensor * restrict src2, - uint32_t start_row, - uint32_t end_row, - uint32_t start_col, - uint32_t end_col, - const struct fastdiv_values * div_ne11_12, - const struct fastdiv_values * div_ne11 -); - typedef struct { float *dst; - const float *src2; + dma_addr_t src2_addr; + size_t src2_bytes; const float *activation; - const __fp16 *weight; + dma_addr_t weight; + dma_queue * weight_dma; int m; int k; int n; @@ -55,15 +47,29 @@ typedef struct { size_t src0_nb3; size_t src1_nb2; size_t src1_nb3; - size_t dst_nb2; - size_t dst_nb3; size_t src2_nb2; size_t src2_nb3; + size_t dst_nb2; + size_t dst_nb3; + int r2; + int r3; + struct fastdiv_values div_r2; + struct fastdiv_values div_r3; } hmx_mm_f16_f32_batched_params_t; +static bool htp_matmul_has_extended_weight(const struct htp_ops_context * octx, uint32_t n_weights) { + for (uint32_t i = 0; i < n_weights; ++i) { + if (htp_tensor_is_extended(octx->src[i])) { + return true; + } + } + return false; +} + struct htp_mm_context { const char * type; struct htp_ops_context * octx; + const struct htp_tensor * act; void (*vec_dot_1x1)(const uint32_t n, float * restrict s0, const void * restrict vx0, @@ -84,8 +90,12 @@ struct htp_mm_context { // Precomputed values uint32_t src0_nrows_per_thread; + uint32_t src0_row_start; + uint32_t src0_row_end; uint32_t src0_row_size_padded; uint32_t src1_nrows; + uint32_t cur_m_start; + uint32_t cur_m_rows; struct fastdiv_values mm_div_ne12_ne1; struct fastdiv_values mm_div_ne1; @@ -130,6 +140,23 @@ struct htp_mm_context { uint32_t vtcm_dst_size_per_thread; }; +static int htp_mm_init_context( + struct htp_ops_context * octx, + const struct htp_mm_kernel_params * kparams +) { + if (!htp_ops_context_set_n_threads(octx, (uint32_t) kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } + + if (kparams->n_hmx) { + if (kparams->n_act_threads <= 0 || kparams->n_act_threads > (int32_t) octx->n_threads) { + return HTP_STATUS_INVAL_PARAMS; + } + } + + return HTP_STATUS_OK; +} + // vdelta control to expand first 32 e8m0 values into 32 uint32 elements static const uint8_t __attribute__((aligned(128))) expand_x32_e8m0[128] = { 0x00, 0x00, 0x00, 0x00, 0x01, 0x04, 0x00, 0x00, 0x02, 0x00, 0x08, 0x08, 0x01, 0x02, 0x00, 0x04, 0x04, 0x00, 0x00, @@ -203,7 +230,7 @@ static const uint8_t __attribute__((aligned(VLEN))) kvalues_mxfp4_lut[] = { #define htp_matmul_preamble \ struct htp_mm_context * mmctx = data; \ struct htp_ops_context * octx = mmctx->octx; \ - dma_queue *dma_queue = octx->ctx->dma[ith]; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ uint32_t src0_nrows_per_thread = mmctx->src0_nrows_per_thread; \ htp_matmul_tensors_preamble; @@ -219,433 +246,285 @@ static inline void hvx_mm_run_quant_task(struct htp_mm_context * mmctx, unsigned } } -// *** matmul with support for 4d tensors and full broadcasting - -static void hvx_mm_4d(unsigned int nth, unsigned int ith, void * data) { - htp_matmul_preamble; - - assert(ne12 % ne02 == 0); - assert(ne13 % ne03 == 0); - - // This is the size of the first dimension of the result, so we can iterate that way. (see the ASSERT above, these are the same numbers) - const uint32_t nr0 = ne0; - - // This is the size of the rest of the dimensions of the result - const uint32_t nr1 = ne1 * ne2 * ne3; - - // distribute the thread work across the inner or outer loop based on which one is larger - uint32_t nchunk0 = nr0 > nr1 ? nth : 1; // parallelize by src0 rows - uint32_t nchunk1 = nr0 > nr1 ? 1 : nth; // parallelize by src1 rows - - // The number of elements in each chunk - const uint32_t dr0 = (nr0 + nchunk0 - 1) / nchunk0; - const uint32_t dr1 = (nr1 + nchunk1 - 1) / nchunk1; - - uint32_t current_chunk = ith; - - const uint32_t ith0 = current_chunk % nchunk0; - const uint32_t ith1 = current_chunk / nchunk0; - - const uint32_t ir0_start = dr0 * ith0; - const uint32_t ir0_end = MIN(ir0_start + dr0, nr0); - const uint32_t ir1_start = dr1 * ith1; - const uint32_t ir1_end = MIN(ir1_start + dr1, nr1); - // no work for this thread - if (ir0_start >= ir0_end || ir1_start >= ir1_end) { - return; - } - - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0_start); - - const uint32_t blck_0 = 64; - const uint32_t blck_1 = 64; - - for (uint32_t iir1 = ir1_start; iir1 < ir1_end; iir1 += blck_1) { - for (uint32_t iir0 = ir0_start; iir0 < ir0_end; iir0 += blck_0) { - for (uint32_t ir1 = iir1; ir1 < MIN(iir1 + blck_1, ir1_end); ir1++) { - const uint32_t i13 = fastdiv(ir1, &mmctx->mm_div_ne12_ne1); - const uint32_t i12 = fastdiv(ir1 - i13 * ne12 * ne1, &mmctx->mm_div_ne1); - const uint32_t i11 = (ir1 - i13 * ne12 * ne1 - i12 * ne1); - - // broadcast src0 into src1 - const uint32_t i03 = fastdiv(i13, &mmctx->mm_div_r3); - const uint32_t i02 = fastdiv(i12, &mmctx->mm_div_r2); - - const uint32_t i1 = i11; - const uint32_t i2 = i12; - const uint32_t i3 = i13; - - const uint8_t * restrict src0_base = (const uint8_t *) src0->data + (0 + i02 * nb02 + i03 * nb03); - const uint8_t * restrict src1_col = (const uint8_t *) src1->data + (i11 * nb11 + i12 * nb12 + i13 * nb13); - float * dst_col = (float *) ((uint8_t * restrict) dst->data + (i1 * nb1 + i2 * nb2 + i3 * nb3)); - - const uint32_t ir0_block_end = MIN(iir0 + blck_0, ir0_end); - for (uint32_t ir0 = iir0; ir0 < ir0_block_end; ir0++) { - const uint8_t * restrict src0_row = src0_base + ir0 * nb01; - mmctx->vec_dot_1x1(ne00, &dst_col[ir0], src0_row, src1_col); - } - } - } - } - - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0_start); - if (src2) { - hvx_tensor_add_f32_grid(dst, src2, ir1_start, ir1_end, ir0_start, ir0_end, &mmctx->mm_div_ne12_ne1, &mmctx->mm_div_ne1); - } -} - -#include "hmx-mm-kernels-tiled.h" +// hvx kernels first: the HMX Q6_K dequantizer reuses unpack_q6_k_group from there #include "hvx-mm-kernels-tiled.h" -#include "hvx-mm-kernels-flat.h" +#include "hvx-mm-kernels-float.h" +#include "hmx-mm-kernels-tiled.h" // Specialized repacked matmul macros -#define MATMUL_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ -static void hvx_mm_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ - htp_matmul_preamble; \ - \ - const uint32_t src0_nrows = ne01 * ne02 * ne03; \ - const uint32_t src1_nrows = ne11 * ne12 * ne13; \ - \ - const uint32_t src0_start_row = src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \ - \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ - const uint32_t n_prefetch = kparams->n_prefetch; \ - assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ - \ - const size_t dst_row_size = nb1; \ - const size_t src1_row_size = nb11; \ - const size_t src1_stride = mmctx->vtcm_src1_stride; \ - const size_t src2_stride = src2 ? ((src2->ne[1] == 1) ? 0 : src2->nb[1]) : 0; \ - \ - uint8_t * restrict vtcm_dst_ptr = mmctx->vtcm_dst + mmctx->vtcm_dst_size_per_thread * ith; \ - uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ - uint8_t * restrict src1_data = mmctx->vtcm_src1; \ - \ - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; \ - \ - const uint32_t tile_size = TILE_SIZE; \ - const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ - \ - uint32_t n_k_tiles_w = ne00 / 32; \ - uint32_t n_k_tiles_a = ne10 / 32; \ - uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ - uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ - \ - uint32_t ct_start = src0_start_row / 32; \ - uint32_t ct_end = (src0_end_row + 31) / 32; \ - \ - uint32_t push_ct = ct_start; \ - if (src0_start_row < src0_end_row) { \ - for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ - src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - } \ - } \ - \ - hvx_mm_run_quant_task(mmctx, ith); \ - \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ - const uint8_t * w_tile = dma_queue_pop(dma_queue).dst; \ - \ - int valid_rows = (int)ne0 - (int)(ct * 32); \ - valid_rows = MIN(32, MAX(0, valid_rows)); \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - uint32_t ir1 = 0; \ - for (; ir1 + 1 < src1_nrows; ir1 += 2) { \ - const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); \ - const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); \ - float * restrict dst_row0 = (float *) (dst->data + ((ir1+0) * dst_row_size)); \ - float * restrict dst_row1 = (float *) (dst->data + ((ir1+1) * dst_row_size)); \ - \ - float * dst_ptr0 = &dst_row0[ct * 32]; \ - float * dst_ptr1 = &dst_row1[ct * 32]; \ - \ - const float * src2_ptr0 = NULL; \ - const float * src2_ptr1 = NULL; \ - if (src2) { \ - const float * restrict src2_row0 = (const float *) ((const uint8_t *) src2->data + ((ir1+0) * src2_stride)); \ - const float * restrict src2_row1 = (const float *) ((const uint8_t *) src2->data + ((ir1+1) * src2_stride)); \ - src2_ptr0 = &src2_row0[ct * 32]; \ - src2_ptr1 = &src2_row1[ct * 32]; \ - } \ - DOT_2X2(ne10, dst_ptr0, dst_ptr1, w_tile, src1_col0, src1_col1, valid_rows, src2_ptr0, src2_ptr1); \ - } \ - \ - for (; ir1 < src1_nrows; ++ir1) { \ - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); \ - float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); \ - float * dst_ptr = &dst_row[ct * 32]; \ - \ - const float * src2_ptr = NULL; \ - if (src2) { \ - const float * restrict src2_row = (const float *) ((const uint8_t *) src2->data + (ir1 * src2_stride)); \ - src2_ptr = &src2_row[ct * 32]; \ - } \ - DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, src2_ptr); \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - \ - if (push_ct < ct_end) { \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile, src0_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - push_ct++; \ - } \ - } \ +#define MATMUL_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ +static void hvx_mm_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ + htp_matmul_preamble; \ + \ + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; \ + const uint32_t src1_nrows = mmctx->cur_m_rows ? mmctx->cur_m_rows : (ne11 * ne12 * ne13); \ + const uint32_t cur_m_start = mmctx->cur_m_start; \ + \ + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; \ + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + \ + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ + const uint32_t n_prefetch = kparams->n_prefetch; \ + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ + \ + const size_t dst_row_size = nb1; \ + const size_t src1_row_size = nb11; \ + const size_t src1_stride = mmctx->vtcm_src1_stride; \ + const size_t src2_stride = src2 ? ((src2->ne[1] == 1) ? 0 : src2->nb[1]) : 0; \ + \ + uint8_t * restrict vtcm_dst_ptr = mmctx->vtcm_dst + mmctx->vtcm_dst_size_per_thread * ith; \ + uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ + uint8_t * restrict src1_data = mmctx->vtcm_src1; \ + \ + const dma_addr_t src0_row = src0->data; \ + \ + const uint32_t tile_size = TILE_SIZE; \ + const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ + \ + uint32_t n_k_tiles_w = ne00 / 32; \ + uint32_t n_k_tiles_a = ne10 / 32; \ + uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ + uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ + \ + uint32_t ct_start = src0_start_row / 32; \ + uint32_t ct_end = (src0_end_row + 31) / 32; \ + \ + uint32_t push_ct = ct_start; \ + if (src0_start_row < src0_end_row) { \ + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ + src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + } \ + } \ + \ + hvx_mm_run_quant_task(mmctx, ith); \ + \ + if (src0_start_row >= src0_end_row) { \ + return; \ + } \ + \ + for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; \ + \ + int valid_rows = (int)ne0 - (int)(ct * 32); \ + valid_rows = MIN(32, MAX(0, valid_rows)); \ + \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + uint32_t ir1 = 0; \ + for (; ir1 + 1 < src1_nrows; ir1 += 2) { \ + const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); \ + const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); \ + float * restrict dst_row0 = (float *) (dst->data + ((cur_m_start + ir1+0) * dst_row_size)); \ + float * restrict dst_row1 = (float *) (dst->data + ((cur_m_start + ir1+1) * dst_row_size)); \ + \ + float * dst_ptr0 = &dst_row0[ct * 32]; \ + float * dst_ptr1 = &dst_row1[ct * 32]; \ + \ + const float * src2_ptr0 = NULL; \ + const float * src2_ptr1 = NULL; \ + if (src2) { \ + const float * restrict src2_row0 = (const float *) ((const uint8_t *) src2->data + ((cur_m_start + ir1+0) * src2_stride)); \ + const float * restrict src2_row1 = (const float *) ((const uint8_t *) src2->data + ((cur_m_start + ir1+1) * src2_stride)); \ + src2_ptr0 = &src2_row0[ct * 32]; \ + src2_ptr1 = &src2_row1[ct * 32]; \ + } \ + DOT_2X2(ne10, dst_ptr0, dst_ptr1, w_tile, src1_col0, src1_col1, valid_rows, src2_ptr0, src2_ptr1); \ + } \ + \ + for (; ir1 < src1_nrows; ++ir1) { \ + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); \ + float * restrict dst_row = (float *) (dst->data + ((cur_m_start + ir1) * dst_row_size)); \ + float * dst_ptr = &dst_row[ct * 32]; \ + \ + const float * src2_ptr = NULL; \ + if (src2) { \ + const float * restrict src2_row = (const float *) ((const uint8_t *) src2->data + ((cur_m_start + ir1) * src2_stride)); \ + src2_ptr = &src2_row[ct * 32]; \ + } \ + DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, src2_ptr); \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + \ + if (push_ct < ct_end) { \ + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), \ + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + push_ct++; \ + } \ + } \ } -#define MATVEC_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X1) \ -static void hvx_mv_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ - htp_matmul_preamble; \ - \ - const uint32_t src0_nrows = ne01; \ - \ - const uint32_t src0_start_row = src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \ - \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ - const uint32_t n_prefetch = kparams->n_prefetch; \ - assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ - \ - const size_t dst_row_size = nb1; \ - const size_t src1_row_size = nb11; \ - const size_t src1_stride = mmctx->vtcm_src1_stride; \ - \ - uint8_t * vtcm_dst_ptr = mmctx->vtcm_dst + mmctx->vtcm_dst_size_per_thread * ith; \ - uint8_t * vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ - uint8_t * src1_data = mmctx->vtcm_src1; \ - \ - float * tmp = (float *) vtcm_dst_ptr; \ - \ - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; \ - \ - const uint8_t * restrict src1_col = (const uint8_t *) src1_data; \ - float * restrict dst_col = (float *) dst->data; \ - \ - const uint32_t tile_size = TILE_SIZE; \ - const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ - \ - uint32_t n_k_tiles_w = ne00 / 32; \ - uint32_t n_k_tiles_a = ne10 / 32; \ - uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ - uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ - \ - uint32_t ct_start = src0_start_row / 32; \ - uint32_t ct_end = (src0_end_row + 31) / 32; \ - \ - uint32_t push_ct = ct_start; \ - if (src0_start_row < src0_end_row) { \ - if (src2) { \ - float * vtcm_src2_ptr = (float *) mmctx->vtcm_src2 + src0_start_row; \ - const float * src2_ptr = (const float *) src2->data + src0_start_row; \ - int slice_size = (int)MIN(src0_end_row, ne0) - (int)src0_start_row; \ - if (slice_size > 0) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr, src2_ptr), \ - slice_size * sizeof(float), slice_size * sizeof(float), slice_size * sizeof(float), 1); \ - dma_queue_pop_nowait(dma_queue); \ - } \ - } \ - for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ - src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - } \ - } \ - \ - hvx_mm_run_quant_task(mmctx, ith); \ - \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ - const uint8_t * w_tile = dma_queue_pop(dma_queue).dst; \ - \ - float * dst_ptr = &tmp[ct * 32 - src0_start_row]; \ - int valid_rows = (int)ne0 - (int)(ct * 32); \ - valid_rows = MIN(32, MAX(0, valid_rows)); \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, NULL); \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - \ - if (push_ct < ct_end) { \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile, src0_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - push_ct++; \ - } \ - } \ - \ - int copy_cnt = (int)MIN(src0_end_row, ne0) - (int)src0_start_row; \ - if (copy_cnt > 0) { \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct_end); \ - if (src2) { \ - hvx_add_f32_uaa((uint8_t *) &dst_col[src0_start_row], \ - (const uint8_t *) tmp, \ - (const uint8_t *) ((const float *) mmctx->vtcm_src2 + src0_start_row), \ - copy_cnt); \ - } else { \ - hvx_copy_f32_ua((uint8_t *) &dst_col[src0_start_row], (uint8_t *) tmp, copy_cnt); \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct_end); \ - } \ +#define MATVEC_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X1) \ +static void hvx_mv_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ + htp_matmul_preamble; \ + \ + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; \ + \ + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; \ + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + \ + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ + const uint32_t n_prefetch = kparams->n_prefetch; \ + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ + \ + const size_t dst_row_size = nb1; \ + const size_t src1_row_size = nb11; \ + const size_t src1_stride = mmctx->vtcm_src1_stride; \ + \ + uint8_t * vtcm_dst_ptr = mmctx->vtcm_dst + mmctx->vtcm_dst_size_per_thread * ith; \ + uint8_t * vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ + uint8_t * src1_data = mmctx->vtcm_src1; \ + \ + float * tmp = (float *) vtcm_dst_ptr; \ + \ + const dma_addr_t src0_row = src0->data; \ + \ + const uint8_t * restrict src1_col = (const uint8_t *) src1_data; \ + float * restrict dst_col = (float *) dst->data; \ + \ + const uint32_t tile_size = TILE_SIZE; \ + const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ + \ + uint32_t n_k_tiles_w = ne00 / 32; \ + uint32_t n_k_tiles_a = ne10 / 32; \ + uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ + uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ + \ + uint32_t ct_start = src0_start_row / 32; \ + uint32_t ct_end = (src0_end_row + 31) / 32; \ + \ + uint32_t push_ct = ct_start; \ + if (src0_start_row < src0_end_row) { \ + if (src2) { \ + float * vtcm_src2_ptr = (float *) mmctx->vtcm_src2 + src0_start_row; \ + const dma_addr_t src2_addr = src2->data + src0_start_row * sizeof(float); \ + int slice_size = (int)MIN(src0_end_row, ne0) - (int)src0_start_row; \ + if (slice_size > 0) { \ + dma_queue_push(dma_q, dma_make_data(vtcm_src2_ptr, src2_addr), \ + slice_size * sizeof(float), slice_size * sizeof(float), slice_size * sizeof(float), 1); \ + dma_queue_pop_nowait(dma_q); \ + } \ + } \ + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ + src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + } \ + } \ + \ + hvx_mm_run_quant_task(mmctx, ith); \ + \ + if (src0_start_row >= src0_end_row) { \ + return; \ + } \ + \ + for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; \ + \ + float * dst_ptr = &tmp[ct * 32 - src0_start_row]; \ + int valid_rows = (int)ne0 - (int)(ct * 32); \ + valid_rows = MIN(32, MAX(0, valid_rows)); \ + \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, NULL); \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + \ + if (push_ct < ct_end) { \ + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), \ + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + push_ct++; \ + } \ + } \ + \ + int copy_cnt = (int)MIN(src0_end_row, ne0) - (int)src0_start_row; \ + if (copy_cnt > 0) { \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct_end); \ + if (src2) { \ + hvx_add_f32_uaa((uint8_t *) &dst_col[src0_start_row], \ + (const uint8_t *) tmp, \ + (const uint8_t *) ((const float *) mmctx->vtcm_src2 + src0_start_row), \ + copy_cnt); \ + } else { \ + hvx_copy_f32_ua((uint8_t *) &dst_col[src0_start_row], (uint8_t *) tmp, copy_cnt); \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct_end); \ + } \ } -#define MATMUL_QKV_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ -static void hvx_mm_qkv_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ +#define MATMUL_NX_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ +static void hvx_mm_nx_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ struct htp_mm_context * mmctx = data; \ struct htp_ops_context * octx = mmctx->octx; \ + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ + const uint32_t n_weights = kparams->n_weights; \ \ - const struct htp_tensor * restrict src0 = octx->src[0]; /* Wk */ \ - const struct htp_tensor * restrict src1 = octx->src[1]; /* x */ \ - const struct htp_tensor * restrict src2 = octx->src[2]; /* Wv */ \ - const struct htp_tensor * restrict src3 = octx->src[3]; /* Wq */ \ - const struct htp_tensor * restrict dst_k = octx->dsts[0]; \ - const struct htp_tensor * restrict dst_v = octx->dsts[1]; \ - const struct htp_tensor * restrict dst_q = octx->dsts[2]; \ - \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne10 = src1->ne[0]; \ - const uint32_t src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; \ - \ - const size_t dst_k_row_size = dst_k->nb[1]; /* K and V share output width */ \ - const size_t dst_q_row_size = dst_q->nb[1]; /* Q may be wider (GQA) */ \ + const struct htp_tensor * restrict act = octx->src[n_weights]; /* x */ \ + const uint32_t ne10 = act->ne[0]; \ + const uint32_t src1_nrows = act->ne[1] * act->ne[2] * act->ne[3]; \ const size_t src1_stride = mmctx->vtcm_src1_stride; \ \ - uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ - uint8_t * restrict vtcm_src2_ptr = mmctx->vtcm_src2 + mmctx->vtcm_src2_size_per_thread * ith; \ - uint8_t * restrict vtcm_src3_ptr = mmctx->vtcm_src3 + mmctx->vtcm_src3_size_per_thread * ith; \ - uint8_t * restrict src1_data = mmctx->vtcm_src1; \ + uint8_t * restrict vtcm_weight_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ + uint8_t * restrict src1_data = mmctx->vtcm_src1; \ \ struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ const uint32_t n_prefetch = kparams->n_prefetch; \ assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ \ - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; \ - const uint8_t * restrict src2_row = (const uint8_t *) src2->data; \ - const uint8_t * restrict src3_row = (const uint8_t *) src3->data; \ - \ const uint32_t tile_size = TILE_SIZE; \ const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ - \ - uint32_t n_k_tiles_w = ne00 / 32; \ uint32_t n_k_tiles_a = ne10 / 32; \ - uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ \ - dma_queue * dma_queue = octx->ctx->dma[ith]; \ - \ - /* 1. Process K and V together */ \ - const uint32_t src0_nrows_kv = src0->ne[1] * src0->ne[2] * src0->ne[3]; /* src0 is Wk */ \ - uint32_t src0_nrows_per_thread_kv = (src0_nrows_kv + nth - 1) / nth; \ - src0_nrows_per_thread_kv = hex_round_up(src0_nrows_per_thread_kv, 32); \ - \ - const uint32_t start_row_kv = src0_nrows_per_thread_kv * ith; \ - const uint32_t end_row_kv = MIN(start_row_kv + src0_nrows_per_thread_kv, src0_nrows_kv); \ - \ - uint32_t ct_start_kv = start_row_kv / 32; \ - uint32_t ct_end_kv = (end_row_kv + 31) / 32; \ - \ - uint32_t push_ct = ct_start_kv; \ - if (start_row_kv < end_row_kv) { \ - for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end_kv; d++, push_ct++) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ - src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + d * tile_row_transfer_size_aligned, \ - src2_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - } \ - } \ - \ hvx_mm_run_quant_task(mmctx, ith); \ \ - if (start_row_kv < end_row_kv) { \ - \ - for (uint32_t ct = ct_start_kv; ct < ct_end_kv; ct++) { \ - const uint8_t * w_tile_k = dma_queue_pop(dma_queue).dst; \ - const uint8_t * w_tile_v = dma_queue_pop(dma_queue).dst; \ - \ - int valid_rows = (int)src0->ne[1] - (int)(ct * 32); \ - valid_rows = MIN(32, MAX(0, valid_rows)); \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ith); \ - uint32_t ir1 = 0; \ - for (; ir1 + 1 < src1_nrows; ir1 += 2) { \ - const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); \ - const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); \ - \ - float * restrict dst_row0_k = (float *) (dst_k->data + ((ir1+0) * dst_k_row_size)); \ - float * restrict dst_row1_k = (float *) (dst_k->data + ((ir1+1) * dst_k_row_size)); \ - float * dst_ptr0_k = &dst_row0_k[ct * 32]; \ - float * dst_ptr1_k = &dst_row1_k[ct * 32]; \ - \ - float * restrict dst_row0_v = (float *) (dst_v->data + ((ir1+0) * dst_k_row_size)); \ - float * restrict dst_row1_v = (float *) (dst_v->data + ((ir1+1) * dst_k_row_size)); \ - float * dst_ptr0_v = &dst_row0_v[ct * 32]; \ - float * dst_ptr1_v = &dst_row1_v[ct * 32]; \ + for (uint32_t widx = 0; widx < n_weights; widx++) { \ + const struct htp_tensor * restrict src_w = octx->src[widx]; \ + const struct htp_tensor * restrict dst = octx->dsts[widx]; \ + if (!src_w || !dst) continue; \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ \ - DOT_2X2(ne10, dst_ptr0_k, dst_ptr1_k, w_tile_k, src1_col0, src1_col1, valid_rows, NULL, NULL); \ - DOT_2X2(ne10, dst_ptr0_v, dst_ptr1_v, w_tile_v, src1_col0, src1_col1, valid_rows, NULL, NULL); \ - } \ - \ - for (; ir1 < src1_nrows; ++ir1) { \ - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); \ + const uint32_t ne00 = src_w->ne[0]; \ + const uint32_t ne01 = src_w->ne[1]; \ + const size_t dst_row_size = dst->nb[1]; \ + const dma_addr_t src_w_row = src_w->data; \ \ - float * restrict dst_row_k = (float *) (dst_k->data + (ir1 * dst_k_row_size)); \ - float * dst_ptr_k = &dst_row_k[ct * 32]; \ + uint32_t n_k_tiles_w = ne00 / 32; \ + uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ \ - float * restrict dst_row_v = (float *) (dst_v->data + (ir1 * dst_k_row_size)); \ - float * dst_ptr_v = &dst_row_v[ct * 32]; \ - \ - DOT_2X1(ne10, dst_ptr_k, w_tile_k, src1_col, valid_rows, NULL); \ - DOT_2X1(ne10, dst_ptr_v, w_tile_v, src1_col, valid_rows, NULL); \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ith); \ - \ - if (push_ct < ct_end_kv) { \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile_k, src0_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile_v, src2_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - push_ct++; \ - } \ + uint32_t src0_start_row = 0; \ + uint32_t src0_end_row = ne01; \ + if (octx->ctx->mdev.count > 1) { \ + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); \ + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(ne01, can_split ? 32 : 0, \ + octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); \ + src0_start_row = range.start; \ + src0_end_row = range.start + range.count; \ } \ - } \ \ - /* 2. Process Q separately */ \ - const uint32_t src0_nrows_q = src3->ne[1] * src3->ne[2] * src3->ne[3]; /* src3 is Wq */ \ - uint32_t src0_nrows_per_thread_q = (src0_nrows_q + nth - 1) / nth; \ - src0_nrows_per_thread_q = hex_round_up(src0_nrows_per_thread_q, 32); \ + const uint32_t nrows = src0_end_row - src0_start_row; \ + uint32_t src0_nrows_per_thread = fastdiv(nrows + nth - 1, &octx->n_threads_div); \ + src0_nrows_per_thread = hex_round_up(src0_nrows_per_thread, 32); \ \ - const uint32_t start_row_q = src0_nrows_per_thread_q * ith; \ - const uint32_t end_row_q = MIN(start_row_q + src0_nrows_per_thread_q, src0_nrows_q); \ + const uint32_t start_row = src0_start_row + src0_nrows_per_thread * ith; \ + const uint32_t end_row = MIN(start_row + src0_nrows_per_thread, src0_end_row); \ + if (start_row >= end_row) continue; \ \ - if (start_row_q < end_row_q) { \ - uint32_t ct_start_q = start_row_q / 32; \ - uint32_t ct_end_q = (end_row_q + 31) / 32; \ + uint32_t ct_start = start_row / 32; \ + uint32_t ct_end = (end_row + 31) / 32; \ \ - uint32_t push_ct = ct_start_q; \ - for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end_q; d++, push_ct++) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src3_ptr + d * tile_row_transfer_size_aligned, \ - src3_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + uint32_t push_ct = ct_start; \ + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ + dma_queue_push(dma_q, dma_make_data(vtcm_weight_ptr + d * tile_row_transfer_size_aligned, \ + src_w_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ } \ \ - for (uint32_t ct = ct_start_q; ct < ct_end_q; ct++) { \ - const uint8_t * w_tile_q = dma_queue_pop(dma_queue).dst; \ - \ - int valid_rows = (int)src3->ne[1] - (int)(ct * 32); \ + for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; \ + int valid_rows = (int)ne01 - (int)(ct * 32); \ valid_rows = MIN(32, MAX(0, valid_rows)); \ \ htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ @@ -654,26 +533,24 @@ static void hvx_mm_qkv_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); \ const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); \ \ - float * restrict dst_row0_q = (float *) (dst_q->data + ((ir1+0) * dst_q_row_size)); \ - float * restrict dst_row1_q = (float *) (dst_q->data + ((ir1+1) * dst_q_row_size)); \ - float * dst_ptr0_q = &dst_row0_q[ct * 32]; \ - float * dst_ptr1_q = &dst_row1_q[ct * 32]; \ + float * restrict dst_row0 = (float *) (dst->data + ((ir1+0) * dst_row_size)); \ + float * restrict dst_row1 = (float *) (dst->data + ((ir1+1) * dst_row_size)); \ + float * dst_ptr0 = &dst_row0[ct * 32]; \ + float * dst_ptr1 = &dst_row1[ct * 32]; \ \ - DOT_2X2(ne10, dst_ptr0_q, dst_ptr1_q, w_tile_q, src1_col0, src1_col1, valid_rows, NULL, NULL); \ + DOT_2X2(ne10, dst_ptr0, dst_ptr1, w_tile, src1_col0, src1_col1, valid_rows, NULL, NULL); \ } \ \ for (; ir1 < src1_nrows; ++ir1) { \ const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); \ - \ - float * restrict dst_row_q = (float *) (dst_q->data + (ir1 * dst_q_row_size)); \ - float * dst_ptr_q = &dst_row_q[ct * 32]; \ - \ - DOT_2X1(ne10, dst_ptr_q, w_tile_q, src1_col, valid_rows, NULL); \ + float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); \ + float * dst_ptr = &dst_row[ct * 32]; \ + DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, NULL); \ } \ htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ \ - if (push_ct < ct_end_q) { \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile_q, src3_row + push_ct * tile_row_stride), \ + if (push_ct < ct_end) { \ + dma_queue_push(dma_q, dma_make_data(w_tile, src_w_row + push_ct * tile_row_stride), \ aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ push_ct++; \ } \ @@ -681,172 +558,64 @@ static void hvx_mm_qkv_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, } \ } -#define MATMUL_FFN_2D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ -static void hvx_mm_ffn_2d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_mm_context * mmctx = data; \ - struct htp_ops_context * octx = mmctx->octx; \ - \ - const struct htp_tensor * restrict src0 = octx->src[0]; /* Wgate */ \ - const struct htp_tensor * restrict src1 = octx->src[1]; /* y */ \ - const struct htp_tensor * restrict src2 = octx->src[2]; /* Wup */ \ - const struct htp_tensor * restrict dst_gate = octx->dsts[0]; \ - const struct htp_tensor * restrict dst_up = octx->dsts[1]; \ - \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne10 = src1->ne[0]; \ - const uint32_t src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; \ - \ - const size_t dst_row_size = dst_gate->nb[1]; \ - const size_t src1_stride = mmctx->vtcm_src1_stride; \ - \ - uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ - uint8_t * restrict vtcm_src2_ptr = mmctx->vtcm_src2 + mmctx->vtcm_src2_size_per_thread * ith; \ - uint8_t * restrict src1_data = mmctx->vtcm_src1; \ - \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; \ - const uint8_t * restrict src2_row = (const uint8_t *) src2->data; \ - \ - const uint32_t tile_size = TILE_SIZE; \ - const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ - \ - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ - const uint32_t n_prefetch = kparams->n_prefetch; \ - assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ - \ - uint32_t n_k_tiles_w = ne00 / 32; \ - uint32_t n_k_tiles_a = ne10 / 32; \ - uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ - uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ - dma_queue * dma_queue = octx->ctx->dma[ith]; \ - \ - const uint32_t src0_nrows = ne01 * src0->ne[2] * src0->ne[3]; \ - const uint32_t src0_start_row = mmctx->src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + mmctx->src0_nrows_per_thread, src0_nrows); \ - \ - uint32_t ct_start = src0_start_row / 32; \ - uint32_t ct_end = (src0_end_row + 31) / 32; \ - \ - uint32_t push_ct = ct_start; \ - if (src0_start_row < src0_end_row) { \ - for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ - src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + d * tile_row_transfer_size_aligned, \ - src2_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - } \ - } \ - \ - hvx_mm_run_quant_task(mmctx, ith); \ - \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ - const uint8_t * w_tile_gate = dma_queue_pop(dma_queue).dst; \ - const uint8_t * w_tile_up = dma_queue_pop(dma_queue).dst; \ - \ - int valid_rows = (int)ne01 - (int)(ct * 32); \ - valid_rows = MIN(32, MAX(0, valid_rows)); \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - uint32_t ir1 = 0; \ - for (; ir1 + 1 < src1_nrows; ir1 += 2) { \ - const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); \ - const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); \ - \ - float * restrict dst_row0_gate = (float *) (dst_gate->data + ((ir1+0) * dst_row_size)); \ - float * restrict dst_row1_gate = (float *) (dst_gate->data + ((ir1+1) * dst_row_size)); \ - float * dst_ptr0_gate = &dst_row0_gate[ct * 32]; \ - float * dst_ptr1_gate = &dst_row1_gate[ct * 32]; \ - \ - float * restrict dst_row0_up = (float *) (dst_up->data + ((ir1+0) * dst_row_size)); \ - float * restrict dst_row1_up = (float *) (dst_up->data + ((ir1+1) * dst_row_size)); \ - float * dst_ptr0_up = &dst_row0_up[ct * 32]; \ - float * dst_ptr1_up = &dst_row1_up[ct * 32]; \ - \ - DOT_2X2(ne10, dst_ptr0_gate, dst_ptr1_gate, w_tile_gate, src1_col0, src1_col1, valid_rows, NULL, NULL); \ - DOT_2X2(ne10, dst_ptr0_up, dst_ptr1_up, w_tile_up, src1_col0, src1_col1, valid_rows, NULL, NULL); \ - } \ - \ - for (; ir1 < src1_nrows; ++ir1) { \ - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); \ - \ - float * restrict dst_row_gate = (float *) (dst_gate->data + (ir1 * dst_row_size)); \ - float * dst_ptr_gate = &dst_row_gate[ct * 32]; \ - \ - float * restrict dst_row_up = (float *) (dst_up->data + (ir1 * dst_row_size)); \ - float * dst_ptr_up = &dst_row_up[ct * 32]; \ - \ - DOT_2X1(ne10, dst_ptr_gate, w_tile_gate, src1_col, valid_rows, NULL); \ - DOT_2X1(ne10, dst_ptr_up, w_tile_up, src1_col, valid_rows, NULL); \ - } \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ - \ - if (push_ct < ct_end) { \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile_gate, src0_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile_up, src2_row + push_ct * tile_row_stride), \ - aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ - push_ct++; \ - } \ - } \ -} - MATMUL_2D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x2, tiled_vec_dot_q4_0_32x1) MATMUL_2D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x2, tiled_vec_dot_q4_1_32x1) MATMUL_2D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x2, tiled_vec_dot_q8_0_32x1) +MATMUL_2D_REPACKED_IMPL(q6_k, 896, tiled_vec_dot_q6_k_32x2, tiled_vec_dot_q6_k_32x1) MATMUL_2D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x2, tiled_vec_dot_iq4nl_32x1) MATMUL_2D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x2, tiled_vec_dot_mxfp4_32x1) -MATMUL_2D_REPACKED_IMPL(q4_0_flat, 576, flat_vec_dot_q4_0_32x2, flat_vec_dot_q4_0_32x1) -MATMUL_2D_REPACKED_IMPL(q4_1_flat, 640, flat_vec_dot_q4_1_32x2, flat_vec_dot_q4_1_32x1) -MATMUL_2D_REPACKED_IMPL(q8_0_flat, 1088, flat_vec_dot_q8_0_32x2, flat_vec_dot_q8_0_32x1) -MATMUL_2D_REPACKED_IMPL(iq4nl_flat, 576, flat_vec_dot_iq4nl_32x2, flat_vec_dot_iq4nl_32x1) -MATMUL_2D_REPACKED_IMPL(mxfp4_flat, 544, flat_vec_dot_mxfp4_32x2, flat_vec_dot_mxfp4_32x1) - -#define QUANTIZE_IMPL(name, log_name, kernel_fn, dst_row_size_expr) \ -static void name(unsigned int nth, unsigned int ith, void * data) { \ - struct htp_mm_context * mmctx = data; \ - struct htp_ops_context * octx = mmctx->octx; \ - const struct htp_tensor * src = octx->src[1]; \ - const uint32_t ne0 = src->ne[0]; \ - const uint32_t ne1 = src->ne[1]; \ - const uint32_t ne2 = src->ne[2]; \ - const uint32_t ne3 = src->ne[3]; \ - const uint32_t nrows = ne1 * ne2 * ne3; \ - const uint32_t nrows_per_thread = mmctx->n_quant_rows_per_thread; \ - \ - const uint32_t ir_first = nrows_per_thread * ith; \ - if (ir_first >= nrows) { \ - return; \ - } \ - \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, ir_first); \ - \ - uint8_t * restrict dst = mmctx->vtcm_src1; \ - const uint32_t ir_last = MIN(ir_first + nrows_per_thread, nrows); \ - const size_t src_row_size = src->nb[1]; \ - const size_t dst_row_size = (dst_row_size_expr); \ - const uint8_t * restrict src_data = (const uint8_t *) src->data + (src_row_size * ir_first); \ - uint8_t * restrict dst_data = (uint8_t *) dst + (dst_row_size * ir_first); \ - uint8_t * restrict tmp_data = (uint8_t *) mmctx->vtcm_dst + (mmctx->vtcm_dst_size_per_thread * ith); \ - kernel_fn(src_data, dst_data, tmp_data, ne0, ir_last - ir_first, src_row_size, dst_row_size); \ - \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_QUANT, ir_first); \ +#define QUANTIZE_IMPL(name, log_name, kernel_fn, dst_row_size_expr) \ +static void name(unsigned int nth, unsigned int ith, void * data) { \ + struct htp_mm_context * mmctx = data; \ + struct htp_ops_context * octx = mmctx->octx; \ + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ + const struct htp_tensor * src = mmctx->act; \ + const uint32_t ne0 = src->ne[0]; \ + const uint32_t nrows = mmctx->cur_m_rows ? mmctx->cur_m_rows : mmctx->src1_nrows; \ + const uint32_t nrows_per_thread = mmctx->n_quant_rows_per_thread; \ + \ + const uint32_t ir_first = nrows_per_thread * ith; \ + if (ir_first >= nrows) { \ + return; \ + } \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, ir_first); \ + \ + uint8_t * restrict dst = mmctx->vtcm_src1; \ + const uint32_t ir_last = MIN(ir_first + nrows_per_thread, nrows); \ + const size_t src_row_size = src->nb[1]; \ + const size_t dst_row_size = (dst_row_size_expr); \ + uint8_t * restrict tmp_data = (uint8_t *) mmctx->vtcm_dst + (mmctx->vtcm_dst_size_per_thread * ith); \ + \ + const bool is_contiguous = (src->nb[2] == src->ne[1] * src->nb[1]) && (src->nb[3] == src->ne[2] * src->nb[2]); \ + if (is_contiguous) { \ + const uint8_t * restrict src_data = (const uint8_t *) src->data + (src_row_size * (mmctx->cur_m_start + ir_first)); \ + uint8_t * restrict dst_data = (uint8_t *) dst + (dst_row_size * ir_first); \ + kernel_fn(src_data, dst_data, tmp_data, ne0, ir_last - ir_first, src_row_size, dst_row_size); \ + } else { \ + const uint32_t ne12_ne1 = src->ne[2] * src->ne[1]; \ + for (uint32_t ir = ir_first; ir < ir_last; ++ir) { \ + const uint32_t ir1 = mmctx->cur_m_start + ir; \ + const uint32_t i13 = fastdiv(ir1, &kparams->div_ne12_ne1); \ + const uint32_t rem = ir1 - i13 * ne12_ne1; \ + const uint32_t i12 = fastdiv(rem, &kparams->div_ne1); \ + const uint32_t i11 = rem - i12 * src->ne[1]; \ + const uint8_t * restrict row_src = (const uint8_t *) src->data + ((size_t) i11 * src->nb[1] + (size_t) i12 * src->nb[2] + (size_t) i13 * src->nb[3]); \ + uint8_t * restrict row_dst = dst + (dst_row_size * ir); \ + kernel_fn(row_src, row_dst, tmp_data, ne0, 1, src_row_size, dst_row_size); \ + } \ + } \ + \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_QUANT, ir_first); \ } QUANTIZE_IMPL(quantize_f32_q8_0_tiled, "quantize-f32-q8_0_tiled", quantize_f32_q8_0_tiled_kernel, htp_mm_q8_0_tiled_row_size(ne0)) QUANTIZE_IMPL(quantize_f32_q8_1_tiled, "quantize-f32-q8_1_tiled", quantize_f32_q8_1_tiled_kernel, htp_mm_q8_1_tiled_row_size(ne0)) -QUANTIZE_IMPL(quantize_f32_q8_0_flat, "quantize-f32-q8_0_flat", quantize_f32_q8_0_flat_kernel, htp_mm_q8_0_flat_row_size(ne0)) -QUANTIZE_IMPL(quantize_f32_q8_1_flat, "quantize-f32-q8_1_flat", quantize_f32_q8_1_flat_kernel, htp_mm_q8_1_flat_row_size(ne0)) -QUANTIZE_IMPL(quantize_f32_f32_flat, "quantize-f32-f32", quantize_f32_f32_flat_kernel, mmctx->vtcm_src1_stride) -QUANTIZE_IMPL(quantize_f32_f16_flat, "quantize-f32-f16", quantize_f32_f16_flat_kernel, mmctx->vtcm_src1_stride) -QUANTIZE_IMPL(quantize_f16_f16_flat, "quantize-f16-f16", quantize_f16_f16_flat_kernel, mmctx->vtcm_src1_stride) +QUANTIZE_IMPL(quantize_f32_f32, "quantize-f32-f32", quantize_f32_f32_kernel, mmctx->vtcm_src1_stride) +QUANTIZE_IMPL(quantize_f32_f16, "quantize-f32-f16", quantize_f32_f16_kernel, mmctx->vtcm_src1_stride) +QUANTIZE_IMPL(quantize_f16_f16, "quantize-f16-f16", quantize_f16_f16_kernel, mmctx->vtcm_src1_stride) static void quantize_f32_q8_0_tiled_block(unsigned int nth, unsigned int ith, void * data) { struct htp_mm_context * mmctx = data; @@ -854,7 +623,7 @@ static void quantize_f32_q8_0_tiled_block(unsigned int nth, unsigned int ith, vo struct htp_thread_trace * tr = &octx->ctx->trace[ith]; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, mmctx->quant_ib_first[ith]); - const struct htp_tensor * src = octx->src[1]; + const struct htp_tensor * src = mmctx->act; quantize_f32_q8_0_tiled_block_kernel( (const float *) src->data, @@ -878,7 +647,7 @@ static void quantize_f32_q8_1_tiled_block(unsigned int nth, unsigned int ith, vo struct htp_thread_trace * tr = &octx->ctx->trace[ith]; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_QUANT, mmctx->quant_ib_first[ith]); - const struct htp_tensor * src = octx->src[1]; + const struct htp_tensor * src = mmctx->act; quantize_f32_q8_1_tiled_block_kernel( (const float *) src->data, @@ -899,40 +668,150 @@ static void quantize_f32_q8_1_tiled_block(unsigned int nth, unsigned int ith, vo MATVEC_2D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x1) MATVEC_2D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x1) MATVEC_2D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x1) +MATVEC_2D_REPACKED_IMPL(q6_k, 896, tiled_vec_dot_q6_k_32x1) MATVEC_2D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x1) MATVEC_2D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x1) -MATVEC_2D_REPACKED_IMPL(q4_0_flat, 576, flat_vec_dot_q4_0_32x1) -MATVEC_2D_REPACKED_IMPL(q4_1_flat, 640, flat_vec_dot_q4_1_32x1) -MATVEC_2D_REPACKED_IMPL(q8_0_flat, 1088, flat_vec_dot_q8_0_32x1) -MATVEC_2D_REPACKED_IMPL(iq4nl_flat, 576, flat_vec_dot_iq4nl_32x1) -MATVEC_2D_REPACKED_IMPL(mxfp4_flat, 544, flat_vec_dot_mxfp4_32x1) - - -MATMUL_QKV_2D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x2, tiled_vec_dot_q4_0_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x2, tiled_vec_dot_q4_1_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x2, tiled_vec_dot_q8_0_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x2, tiled_vec_dot_iq4nl_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x2, tiled_vec_dot_mxfp4_32x1) - -MATMUL_QKV_2D_REPACKED_IMPL(q4_0_flat, 576, flat_vec_dot_q4_0_32x2, flat_vec_dot_q4_0_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(q4_1_flat, 640, flat_vec_dot_q4_1_32x2, flat_vec_dot_q4_1_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(q8_0_flat, 1088, flat_vec_dot_q8_0_32x2, flat_vec_dot_q8_0_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(iq4nl_flat, 576, flat_vec_dot_iq4nl_32x2, flat_vec_dot_iq4nl_32x1) -MATMUL_QKV_2D_REPACKED_IMPL(mxfp4_flat, 544, flat_vec_dot_mxfp4_32x2, flat_vec_dot_mxfp4_32x1) - - -MATMUL_FFN_2D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x2, tiled_vec_dot_q4_0_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x2, tiled_vec_dot_q4_1_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x2, tiled_vec_dot_q8_0_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x2, tiled_vec_dot_iq4nl_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x2, tiled_vec_dot_mxfp4_32x1) +MATMUL_NX_2D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x2, tiled_vec_dot_q4_0_32x1) +MATMUL_NX_2D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x2, tiled_vec_dot_q4_1_32x1) +MATMUL_NX_2D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x2, tiled_vec_dot_q8_0_32x1) +MATMUL_NX_2D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x2, tiled_vec_dot_iq4nl_32x1) +MATMUL_NX_2D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x2, tiled_vec_dot_mxfp4_32x1) + +#define MATMUL_4D_REPACKED_IMPL(SUFFIX, TILE_SIZE, DOT_2X2, DOT_2X1) \ +static void hvx_mm_4d_repacked_##SUFFIX(unsigned int nth, unsigned int ith, void * data) { \ + htp_matmul_preamble; \ + \ + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; \ + const uint32_t cur_m_rows = mmctx->cur_m_rows ? mmctx->cur_m_rows : (ne11 * ne12 * ne13); \ + const uint32_t cur_m_start = mmctx->cur_m_start; \ + \ + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; \ + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); \ + \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + \ + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; \ + const uint32_t n_prefetch = kparams->n_prefetch; \ + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); \ + \ + const size_t dst_row_size = nb1; \ + const size_t src1_stride = mmctx->vtcm_src1_stride; \ + \ + uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; \ + uint8_t * restrict src1_data = mmctx->vtcm_src1; \ + \ + const uint32_t tile_size = TILE_SIZE; \ + const uint32_t aligned_tile_size = hex_align_up(tile_size, 128); \ + \ + const uint32_t n_k_tiles_w = ne00 / 32; \ + const uint32_t n_k_tiles_a = ne10 / 32; \ + const uint32_t tile_row_stride = n_k_tiles_w * tile_size; \ + const uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; \ + const uint32_t src0_slice_stride = ((ne01 + 31) / 32) * tile_row_stride; \ + \ + const uint32_t ct_start = src0_start_row / 32; \ + const uint32_t ct_end = (src0_end_row + 31) / 32; \ + \ + hvx_mm_run_quant_task(mmctx, ith); \ + \ + if (src0_start_row >= src0_end_row || cur_m_rows == 0) { \ + return; \ + } \ + \ + const uint32_t total_batches = ne12 * ne13; \ + const uint32_t b_start = fastdiv(cur_m_start, &kparams->div_ne1); \ + uint32_t b_end = fastdiv(cur_m_start + cur_m_rows + ne11 - 1, &kparams->div_ne1); \ + b_end = MIN(b_end, total_batches); \ + \ + uint32_t b_grp_start = b_start; \ + while (b_grp_start < b_end) { \ + const uint32_t b3 = fastdiv(b_grp_start, &kparams->div_ne12); \ + const uint32_t b2 = b_grp_start - b3 * ne12; \ + const uint32_t i02 = fastdiv(b2, &kparams->div_r2); \ + const uint32_t i03 = fastdiv(b3, &kparams->div_r3); \ + \ + uint32_t b_grp_end = b_grp_start + 1; \ + while (b_grp_end < b_end) { \ + const uint32_t cur_b3 = fastdiv(b_grp_end, &kparams->div_ne12); \ + const uint32_t cur_b2 = b_grp_end - cur_b3 * ne12; \ + const uint32_t cur_i02 = fastdiv(cur_b2, &kparams->div_r2); \ + const uint32_t cur_i03 = fastdiv(cur_b3, &kparams->div_r3); \ + if (cur_i02 != i02 || cur_i03 != i03) { \ + break; \ + } \ + b_grp_end++; \ + } \ + \ + const uint32_t slice_idx = i03 * ne02 + i02; \ + const dma_addr_t src0_slice = src0->data + (size_t) slice_idx * src0_slice_stride; \ + \ + uint32_t push_ct = ct_start; \ + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { \ + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, \ + src0_slice + (size_t) push_ct * tile_row_stride), \ + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + } \ + \ + for (uint32_t ct = ct_start; ct < ct_end; ct++) { \ + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; \ + \ + int valid_rows = (int)ne0 - (int)(ct * 32); \ + valid_rows = MIN(32, MAX(0, valid_rows)); \ + \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + for (uint32_t b = b_grp_start; b < b_grp_end; b++) { \ + const uint32_t b_m_start = b * ne11; \ + const uint32_t m_first = MAX(cur_m_start, b_m_start); \ + const uint32_t m_last = MIN(cur_m_start + cur_m_rows, b_m_start + ne11); \ + if (m_first >= m_last) continue; \ + \ + const uint32_t cur_b3 = fastdiv(b, &kparams->div_ne12); \ + const uint32_t cur_b2 = b - cur_b3 * ne12; \ + uint8_t * dst_batch_base = (uint8_t *) dst->data + (size_t) cur_b2 * nb2 + (size_t) cur_b3 * nb3; \ + \ + const uint32_t chunk_m_offset = m_first - cur_m_start; \ + const uint32_t dst_m_offset = m_first - b_m_start; \ + const uint32_t batch_nrows = m_last - m_first; \ + \ + uint32_t ir1 = 0; \ + for (; ir1 + 1 < batch_nrows; ir1 += 2) { \ + const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (chunk_m_offset + ir1 + 0) * src1_stride); \ + const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (chunk_m_offset + ir1 + 1) * src1_stride); \ + float * restrict dst_row0 = (float *) (dst_batch_base + (dst_m_offset + ir1 + 0) * dst_row_size); \ + float * restrict dst_row1 = (float *) (dst_batch_base + (dst_m_offset + ir1 + 1) * dst_row_size); \ + float * dst_ptr0 = &dst_row0[ct * 32]; \ + float * dst_ptr1 = &dst_row1[ct * 32]; \ + DOT_2X2(ne10, dst_ptr0, dst_ptr1, w_tile, src1_col0, src1_col1, valid_rows, NULL, NULL); \ + } \ + for (; ir1 < batch_nrows; ++ir1) { \ + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + (chunk_m_offset + ir1) * src1_stride); \ + float * restrict dst_row = (float *) (dst_batch_base + (dst_m_offset + ir1) * dst_row_size); \ + float * dst_ptr = &dst_row[ct * 32]; \ + DOT_2X1(ne10, dst_ptr, w_tile, src1_col, valid_rows, NULL); \ + } \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); \ + \ + if (push_ct < ct_end) { \ + dma_queue_push(dma_q, dma_make_data(w_tile, src0_slice + (size_t) push_ct * tile_row_stride), \ + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); \ + push_ct++; \ + } \ + } \ + b_grp_start = b_grp_end; \ + } \ + if (src2) { \ + hvx_tensor_add_f32_grid(dst, src2, cur_m_start, cur_m_start + cur_m_rows, src0_start_row, src0_end_row, &kparams->div_ne12_ne1, &kparams->div_ne1); \ + } \ +} -MATMUL_FFN_2D_REPACKED_IMPL(q4_0_flat, 576, flat_vec_dot_q4_0_32x2, flat_vec_dot_q4_0_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(q4_1_flat, 640, flat_vec_dot_q4_1_32x2, flat_vec_dot_q4_1_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(q8_0_flat, 1088, flat_vec_dot_q8_0_32x2, flat_vec_dot_q8_0_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(iq4nl_flat, 576, flat_vec_dot_iq4nl_32x2, flat_vec_dot_iq4nl_32x1) -MATMUL_FFN_2D_REPACKED_IMPL(mxfp4_flat, 544, flat_vec_dot_mxfp4_32x2, flat_vec_dot_mxfp4_32x1) +MATMUL_4D_REPACKED_IMPL(q4_0, 576, tiled_vec_dot_q4_0_32x2, tiled_vec_dot_q4_0_32x1) +MATMUL_4D_REPACKED_IMPL(q4_1, 640, tiled_vec_dot_q4_1_32x2, tiled_vec_dot_q4_1_32x1) +MATMUL_4D_REPACKED_IMPL(q8_0, 1088, tiled_vec_dot_q8_0_32x2, tiled_vec_dot_q8_0_32x1) +MATMUL_4D_REPACKED_IMPL(q6_k, 896, tiled_vec_dot_q6_k_32x2, tiled_vec_dot_q6_k_32x1) +MATMUL_4D_REPACKED_IMPL(iq4nl, 576, tiled_vec_dot_iq4nl_32x2, tiled_vec_dot_iq4nl_32x1) +MATMUL_4D_REPACKED_IMPL(mxfp4, 544, tiled_vec_dot_mxfp4_32x2, tiled_vec_dot_mxfp4_32x1) static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { htp_matmul_preamble; @@ -942,11 +821,12 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); const uint32_t prefetch_mask = n_prefetch - 1; - const uint32_t src0_nrows = ne01 * ne02 * ne03; // src0 rows - const uint32_t src1_nrows = ne11 * ne12 * ne13; // src1 rows + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; // src0 rows + const uint32_t src1_nrows = mmctx->cur_m_rows ? mmctx->cur_m_rows : mmctx->src1_nrows; // src1 rows + const uint32_t cur_m_start = mmctx->cur_m_start; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U); struct htp_thread_trace * tr = &octx->ctx->trace[ith]; @@ -963,7 +843,7 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; uint8_t * restrict src1_data = mmctx->vtcm_src1; - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; + const dma_addr_t src0_row = src0->data; // Prefill vtcm with src0 rows if (src0_start_row < src0_end_row) { @@ -972,7 +852,7 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { if (is0 >= (int)n_prefetch) { break; } - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 2); } } @@ -985,23 +865,23 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { // Process src0 rows for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); - // Process src1 columns in pairs (2×2 tiling) + // Process src1 columns in pairs (2x2 tiling) uint32_t ir1 = 0; for (; ir1 + 1 < src1_nrows; ir1 += 2) { const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); - float * restrict dst_row0 = (float *) (dst->data + ((ir1+0) * dst_row_size)); - float * restrict dst_row1 = (float *) (dst->data + ((ir1+1) * dst_row_size)); + float * restrict dst_row0 = (float *) (dst->data + ((cur_m_start + ir1+0) * dst_row_size)); + float * restrict dst_row1 = (float *) (dst->data + ((cur_m_start + ir1+1) * dst_row_size)); mmctx->vec_dot_2x2(ne00, &dst_row0[ir0], &dst_row1[ir0], ss0, ss0 + src0_stride, src1_col0, src1_col1); } - // Handle remaining src1 rows (fallback to 2×1) + // Handle remaining src1 rows (fallback to 2x1) for (; ir1 < src1_nrows; ++ir1) { const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); + float * restrict dst_row = (float *) (dst->data + ((cur_m_start + ir1) * dst_row_size)); mmctx->vec_dot_2x1(ne00, &dst_row[ir0], ss0, ss0 + src0_stride, src1_col); } htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); @@ -1010,7 +890,7 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { const int pr0 = (ir0 + n_prefetch); const int is0 = (pr0 - src0_start_row) & prefetch_mask; if (pr0 < src0_end_row_x2) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 2); } } @@ -1019,31 +899,31 @@ static void hvx_mm_2d(unsigned int nth, unsigned int ith, void * data) { if (src0_end_row != src0_end_row_x2) { uint32_t ir0 = src0_end_row_x2; const int is0 = (ir0 - src0_start_row) & prefetch_mask; - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 1); - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); #pragma unroll(2) for (uint32_t ir1 = 0; ir1 < src1_nrows; ++ir1) { const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); + float * restrict dst_row = (float *) (dst->data + ((cur_m_start + ir1) * dst_row_size)); mmctx->vec_dot_1x1(ne00, &dst_row[ir0], ss0, src1_col); } htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); } if (src2) { - hvx_tensor_add_f32_grid(dst, src2, 0, src1_nrows, src0_start_row, src0_end_row, &kparams->div_ne12_ne1, &kparams->div_ne1); + hvx_tensor_add_f32_grid(dst, src2, cur_m_start, cur_m_start + src1_nrows, src0_start_row, src0_end_row, &kparams->div_ne12_ne1, &kparams->div_ne1); } } static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { htp_matmul_preamble; - const uint32_t src0_nrows = ne01; + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); struct htp_thread_trace * tr = &octx->ctx->trace[ith]; @@ -1061,7 +941,7 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { float * tmp = (float *) vtcm_dst_ptr; - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; + const dma_addr_t src0_row = src0->data; const uint8_t * restrict src1_col = (const uint8_t *) src1_data; float * restrict dst_col = (float *) dst->data; @@ -1076,12 +956,12 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { if (src0_start_row < src0_end_row) { if (src2) { float * vtcm_src2_ptr = (float *) mmctx->vtcm_src2 + src0_start_row; - const float * src2_ptr = (const float *) src2->data + src0_start_row; + const dma_addr_t src2_addr = src2->data + src0_start_row * sizeof(float); int slice_size = (int)src0_end_row - (int)src0_start_row; if (slice_size > 0) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr, src2_ptr), + dma_queue_push(dma_q, dma_make_data(vtcm_src2_ptr, src2_addr), slice_size * sizeof(float), slice_size * sizeof(float), slice_size * sizeof(float), 1); - dma_queue_pop_nowait(dma_queue); + dma_queue_pop_nowait(dma_q); } } for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { @@ -1089,7 +969,7 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { if (is0 >= n_prefetch) { break; } - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 2); } } @@ -1102,7 +982,7 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { // Process src0 rows for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); mmctx->vec_dot_2x1(ne00, &tmp[ir0 - src0_start_row], ss0, ss0 + src0_stride, src1_col); htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); @@ -1111,7 +991,7 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { const uint32_t pr0 = (ir0 + n_prefetch); const uint32_t is0 = (pr0 - src0_start_row) & prefetch_mask; if (pr0 < src0_end_row_x2) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 2); } } @@ -1120,9 +1000,9 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { if (src0_end_row != src0_end_row_x2) { const uint32_t ir0 = src0_end_row_x2; const uint32_t is0 = (ir0 - src0_start_row) & prefetch_mask; - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 1); - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); mmctx->vec_dot_1x1(ne00, &tmp[ir0 - src0_start_row], ss0, src1_col); htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); @@ -1143,55 +1023,200 @@ static void hvx_mv_2d(unsigned int nth, unsigned int ith, void * data) { } } -#define MMID_MATRIX_ROW(row_id, i1) matrix_rows[(row_id) * mmctx->mapping_stride + (i1)] - -static void hvx_mm_id(unsigned int nth, unsigned int ith, void * data) { +static void hvx_mm_4d(unsigned int nth, unsigned int ith, void * data) { htp_matmul_preamble; - const struct htp_tensor * restrict ids = octx->src[2]; - - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - - const uint32_t src0_nrows = ne01; // src0 rows per expert - const uint32_t src1_nrows = ne11; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); - - hvx_mm_run_quant_task(mmctx, ith); - - if (src0_start_row >= src0_end_row) { - return; - } - - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; const uint32_t n_prefetch = kparams->n_prefetch; assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); + const uint32_t prefetch_mask = n_prefetch - 1; - const uint32_t n_ids = ids->ne[0]; // n_expert_used - const uint32_t n_as = ne02; // n_expert + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; + const uint32_t cur_m_rows = mmctx->cur_m_rows ? mmctx->cur_m_rows : (ne11 * ne12 * ne13); + const uint32_t cur_m_start = mmctx->cur_m_start; - const uint32_t * matrix_row_counts = mmctx->matrix_row_counts; - const struct mmid_row_mapping * matrix_rows = mmctx->matrix_rows; + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); + const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U); - const size_t dst_row_size = nb1; - const size_t src1_row_size = htp_mm_q8_0_tiled_row_size(ne10); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + const size_t dst_row_size = nb1; + const size_t src0_row_size = nb01; + const size_t src0_stride = mmctx->vtcm_src0_stride; const size_t src1_stride = mmctx->vtcm_src1_stride; - // Per-thread VTCMs for all tensors uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; - uint8_t * restrict src1_data = mmctx->vtcm_src1; + uint8_t * restrict src1_data = mmctx->vtcm_src1; - for (uint32_t cur_a = 0; cur_a < n_as; ++cur_a) { - const int32_t cne1 = matrix_row_counts[cur_a]; + hvx_mm_run_quant_task(mmctx, ith); + + if (src0_start_row >= src0_end_row || cur_m_rows == 0) { + return; + } + + const uint32_t total_batches = ne12 * ne13; + const uint32_t b_start = fastdiv(cur_m_start, &kparams->div_ne1); + uint32_t b_end = fastdiv(cur_m_start + cur_m_rows + ne11 - 1, &kparams->div_ne1); + b_end = MIN(b_end, total_batches); + + uint32_t b_grp_start = b_start; + while (b_grp_start < b_end) { + const uint32_t b3 = fastdiv(b_grp_start, &kparams->div_ne12); + const uint32_t b2 = b_grp_start - b3 * ne12; + const uint32_t i02 = fastdiv(b2, &kparams->div_r2); + const uint32_t i03 = fastdiv(b3, &kparams->div_r3); + + uint32_t b_grp_end = b_grp_start + 1; + while (b_grp_end < b_end) { + const uint32_t cur_b3 = fastdiv(b_grp_end, &kparams->div_ne12); + const uint32_t cur_b2 = b_grp_end - cur_b3 * ne12; + const uint32_t cur_i02 = fastdiv(cur_b2, &kparams->div_r2); + const uint32_t cur_i03 = fastdiv(cur_b3, &kparams->div_r3); + if (cur_i02 != i02 || cur_i03 != i03) { + break; + } + b_grp_end++; + } + + const dma_addr_t src0_row = src0->data + ((size_t) i02 * nb02 + (size_t) i03 * nb03); + + for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { + const int is0 = (ir0 - src0_start_row); + if (is0 >= (int)n_prefetch) { + break; + } + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + (size_t) ir0 * src0_row_size), + src0_stride, src0_row_size, src0_row_size, 2); + } + + for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + for (uint32_t b = b_grp_start; b < b_grp_end; b++) { + const uint32_t b_m_start = b * ne11; + const uint32_t m_first = MAX(cur_m_start, b_m_start); + const uint32_t m_last = MIN(cur_m_start + cur_m_rows, b_m_start + ne11); + if (m_first >= m_last) continue; + + const uint32_t cur_b3 = fastdiv(b, &kparams->div_ne12); + const uint32_t cur_b2 = b - cur_b3 * ne12; + uint8_t * dst_batch_base = (uint8_t *) dst->data + (size_t) cur_b2 * nb2 + (size_t) cur_b3 * nb3; + + const uint32_t chunk_m_offset = m_first - cur_m_start; + const uint32_t dst_m_offset = m_first - b_m_start; + const uint32_t batch_nrows = m_last - m_first; + + uint32_t ir1 = 0; + for (; ir1 + 1 < batch_nrows; ir1 += 2) { + const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (chunk_m_offset + ir1 + 0) * src1_stride); + const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (chunk_m_offset + ir1 + 1) * src1_stride); + float * restrict dst_row0 = (float *) (dst_batch_base + (dst_m_offset + ir1 + 0) * dst_row_size); + float * restrict dst_row1 = (float *) (dst_batch_base + (dst_m_offset + ir1 + 1) * dst_row_size); + mmctx->vec_dot_2x2(ne00, &dst_row0[ir0], &dst_row1[ir0], ss0, ss0 + src0_stride, src1_col0, src1_col1); + } + for (; ir1 < batch_nrows; ++ir1) { + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + (chunk_m_offset + ir1) * src1_stride); + float * restrict dst_row = (float *) (dst_batch_base + (dst_m_offset + ir1) * dst_row_size); + mmctx->vec_dot_2x1(ne00, &dst_row[ir0], ss0, ss0 + src0_stride, src1_col); + } + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + + const int pr0 = (ir0 + n_prefetch); + const int is0 = (pr0 - src0_start_row) & prefetch_mask; + if (pr0 < src0_end_row_x2) { + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + (size_t) pr0 * src0_row_size), + src0_stride, src0_row_size, src0_row_size, 2); + } + } + + if (src0_end_row != src0_end_row_x2) { + uint32_t ir0 = src0_end_row_x2; + const int is0 = (ir0 - src0_start_row) & prefetch_mask; + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + (size_t) ir0 * src0_row_size), + src0_stride, src0_row_size, src0_row_size, 1); + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + for (uint32_t b = b_grp_start; b < b_grp_end; b++) { + const uint32_t b_m_start = b * ne11; + const uint32_t m_first = MAX(cur_m_start, b_m_start); + const uint32_t m_last = MIN(cur_m_start + cur_m_rows, b_m_start + ne11); + if (m_first >= m_last) continue; + + const uint32_t cur_b3 = fastdiv(b, &kparams->div_ne12); + const uint32_t cur_b2 = b - cur_b3 * ne12; + uint8_t * dst_batch_base = (uint8_t *) dst->data + (size_t) cur_b2 * nb2 + (size_t) cur_b3 * nb3; + + const uint32_t chunk_m_offset = m_first - cur_m_start; + const uint32_t dst_m_offset = m_first - b_m_start; + const uint32_t batch_nrows = m_last - m_first; + + for (uint32_t ir1 = 0; ir1 < batch_nrows; ++ir1) { + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + (chunk_m_offset + ir1) * src1_stride); + float * restrict dst_row = (float *) (dst_batch_base + (dst_m_offset + ir1) * dst_row_size); + mmctx->vec_dot_1x1(ne00, &dst_row[ir0], ss0, src1_col); + } + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + } + + b_grp_start = b_grp_end; + } + + if (src2) { + hvx_tensor_add_f32_grid(dst, src2, cur_m_start, cur_m_start + cur_m_rows, src0_start_row, src0_end_row, &kparams->div_ne12_ne1, &kparams->div_ne1); + } +} + +#define MMID_MATRIX_ROW(row_id, i1) matrix_rows[(row_id) * mmctx->mapping_stride + (i1)] + +static void hvx_mm_id(unsigned int nth, unsigned int ith, void * data) { + htp_matmul_preamble; + + const struct htp_tensor * restrict ids = octx->src[2]; + + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; // src0 rows per expert + const uint32_t src1_nrows = ne11; + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); + + hvx_mm_run_quant_task(mmctx, ith); + + if (src0_start_row >= src0_end_row) { + return; + } + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const uint32_t n_prefetch = kparams->n_prefetch; + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); + + const uint32_t n_ids = ids->ne[0]; // n_expert_used + const uint32_t n_as = ne02; // n_expert + + const uint32_t * matrix_row_counts = mmctx->matrix_row_counts; + const struct mmid_row_mapping * matrix_rows = mmctx->matrix_rows; + + const size_t dst_row_size = nb1; + const size_t src1_row_size = htp_mm_q8_0_tiled_row_size(ne10); + + const size_t src1_stride = mmctx->vtcm_src1_stride; + + // Per-thread VTCMs for all tensors + uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; + uint8_t * restrict src1_data = mmctx->vtcm_src1; + + for (uint32_t cur_a = 0; cur_a < n_as; ++cur_a) { + const int32_t cne1 = matrix_row_counts[cur_a]; if (cne1 == 0) { continue; } - const uint8_t * src0_row = (const uint8_t *) src0->data + cur_a * nb02; + const dma_addr_t src0_row = src0->data + cur_a * nb02; const uint32_t tile_size = htp_mm_get_weight_tile_size(src0->type); const uint32_t aligned_tile_size = htp_mm_get_weight_aligned_tile_size(src0->type); @@ -1205,12 +1230,12 @@ static void hvx_mm_id(unsigned int nth, unsigned int ith, void * data) { uint32_t push_ct = ct_start; for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); } for (uint32_t ct = ct_start; ct < ct_end; ct++) { - const uint8_t * w_tile = dma_queue_pop(dma_queue).dst; + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; int valid_rows = (int)ne01 - (int)(ct * 32); valid_rows = MIN(32, MAX(0, valid_rows)); @@ -1230,7 +1255,7 @@ static void hvx_mm_id(unsigned int nth, unsigned int ith, void * data) { htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); if (push_ct < ct_end) { - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile, src0_row + push_ct * tile_row_stride), + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); push_ct++; } @@ -1243,9 +1268,9 @@ static void hvx_mv_id(unsigned int nth, unsigned int ith, void * data) { const struct htp_tensor * restrict ids = octx->src[2]; - const uint32_t src0_nrows = ne01; // src0 rows per expert - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); + const uint32_t src0_nrows = mmctx->src0_row_end - mmctx->src0_row_start; // src0 rows per expert + const uint32_t src0_start_row = mmctx->src0_row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, mmctx->src0_row_end); hvx_mm_run_quant_task(mmctx, ith); @@ -1278,7 +1303,7 @@ static void hvx_mv_id(unsigned int nth, unsigned int ith, void * data) { } assert(eid < (int32_t) n_ids); - const uint8_t * restrict src0_row = (const uint8_t *) src0->data + eid * nb02; + const dma_addr_t src0_row = src0->data + eid * nb02; const uint8_t * restrict src1_col = (const uint8_t *) src1_data; float * restrict dst_row = (float *) (dst->data + ie1 * nb1); @@ -1294,12 +1319,12 @@ static void hvx_mv_id(unsigned int nth, unsigned int ith, void * data) { uint32_t push_ct = ct_start; for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); } for (uint32_t ct = ct_start; ct < ct_end; ct++) { - const uint8_t * w_tile = dma_queue_pop(dma_queue).dst; + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; int valid_rows = (int)ne01 - (int)(ct * 32); valid_rows = MIN(32, MAX(0, valid_rows)); @@ -1309,7 +1334,7 @@ static void hvx_mv_id(unsigned int nth, unsigned int ith, void * data) { htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); if (push_ct < ct_end) { - dma_queue_push(dma_queue, dma_make_ptr((uint8_t *)w_tile, src0_row + push_ct * tile_row_stride), + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), aligned_tile_size, tile_size, tile_size, n_k_tiles_a); push_ct++; } @@ -1317,6 +1342,199 @@ static void hvx_mv_id(unsigned int nth, unsigned int ith, void * data) { } } +static void hvx_mv_id_nx(unsigned int nth, unsigned int ith, void * data) { + struct htp_mm_context * mmctx = (struct htp_mm_context *) data; + struct htp_ops_context * octx = mmctx->octx; + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const uint32_t n_weights = kparams->n_weights; + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const struct htp_tensor * restrict ids = octx->src[n_weights + 1]; + + hvx_mm_run_quant_task(mmctx, ith); + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + const uint32_t n_prefetch = kparams->n_prefetch; + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); + + const uint32_t n_aids = ids->ne[0]; + const uint32_t n_ids = src0->ne[2]; + + uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; + uint8_t * restrict src1_data = mmctx->vtcm_src1; + + for (uint32_t ie1 = 0; ie1 < n_aids; ++ie1) { + const int32_t eid = *(const int32_t *) ((const uint8_t *) ids->data + ie1 * ids->nb[0]); + if (eid < 0) continue; + assert(eid < (int32_t) n_ids); + + for (uint32_t p = 0; p < n_weights; ++p) { + const struct htp_tensor * restrict src_w = octx->src[p]; + const struct htp_tensor * restrict dst = octx->dsts[p]; + if (!src_w || !dst) continue; + dma_queue * dma_q = octx->ctx->dma[ith]; + + const uint32_t ne01 = src_w->ne[1]; + uint32_t start_row = 0; + uint32_t end_row = ne01; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(ne01, can_split ? 32 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + start_row = range.start; + end_row = range.start + range.count; + } + + const uint32_t nrows = end_row - start_row; + uint32_t src0_nrows_per_thread = fastdiv(nrows + nth - 1, &octx->n_threads_div); + src0_nrows_per_thread = hex_round_up(src0_nrows_per_thread, 32); + + const uint32_t src0_start_row = start_row + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, end_row); + if (src0_start_row >= src0_end_row) continue; + + const dma_addr_t src0_row = src_w->data + eid * src_w->nb[2]; + const uint8_t * restrict src1_col = (const uint8_t *) src1_data; + float * restrict dst_row = (float *) (dst->data + ie1 * dst->nb[1]); + + const uint32_t tile_size = htp_mm_get_weight_tile_size(src_w->type); + const uint32_t aligned_tile_size = htp_mm_get_weight_aligned_tile_size(src_w->type); + const uint32_t n_k_tiles_w = src_w->ne[0] / 32; + const uint32_t n_k_tiles_a = act->ne[0] / 32; + const uint32_t tile_row_stride = n_k_tiles_w * tile_size; + const uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; + + const uint32_t ct_start = src0_start_row / 32; + const uint32_t ct_end = (src0_end_row + 31) / 32; + + uint32_t push_ct = ct_start; + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); + } + + for (uint32_t ct = ct_start; ct < ct_end; ct++) { + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; + + int valid_rows = (int)src_w->ne[1] - (int)(ct * 32); + valid_rows = MIN(32, MAX(0, valid_rows)); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); + mmctx->vec_dot_32x1(act->ne[0], &dst_row[ct * 32], w_tile, src1_col, valid_rows, NULL); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); + + if (push_ct < ct_end) { + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); + push_ct++; + } + } + } + } +} + +static void hvx_mm_id_nx(unsigned int nth, unsigned int ith, void * data) { + struct htp_mm_context * mmctx = (struct htp_mm_context *) data; + struct htp_ops_context * octx = mmctx->octx; + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const uint32_t n_weights = kparams->n_weights; + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const struct htp_tensor * restrict ids = octx->src[n_weights + 1]; + + hvx_mm_run_quant_task(mmctx, ith); + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + const uint32_t n_prefetch = kparams->n_prefetch; + assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); + + const uint32_t n_as = src0->ne[2]; + + const uint32_t * matrix_row_counts = mmctx->matrix_row_counts; + const struct mmid_row_mapping * matrix_rows = mmctx->matrix_rows; + + const size_t src1_stride = mmctx->vtcm_src1_stride; + + uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; + uint8_t * restrict src1_data = mmctx->vtcm_src1; + + for (uint32_t cur_a = 0; cur_a < n_as; ++cur_a) { + const int32_t cne1 = matrix_row_counts[cur_a]; + if (cne1 == 0) continue; + + for (uint32_t p = 0; p < n_weights; ++p) { + const struct htp_tensor * restrict src_w = octx->src[p]; + const struct htp_tensor * restrict dst = octx->dsts[p]; + if (!src_w || !dst) continue; + dma_queue * dma_q = octx->ctx->dma[ith]; + + const uint32_t ne01 = src_w->ne[1]; + uint32_t start_row = 0; + uint32_t end_row = ne01; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(ne01, can_split ? 32 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + start_row = range.start; + end_row = range.start + range.count; + } + + const uint32_t nrows = end_row - start_row; + uint32_t src0_nrows_per_thread = fastdiv(nrows + nth - 1, &octx->n_threads_div); + src0_nrows_per_thread = hex_round_up(src0_nrows_per_thread, 32); + + const uint32_t src0_start_row = start_row + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, end_row); + if (src0_start_row >= src0_end_row) continue; + + const dma_addr_t src0_row = src_w->data + cur_a * src_w->nb[2]; + + const uint32_t tile_size = htp_mm_get_weight_tile_size(src_w->type); + const uint32_t aligned_tile_size = htp_mm_get_weight_aligned_tile_size(src_w->type); + const uint32_t n_k_tiles_w = src_w->ne[0] / 32; + const uint32_t n_k_tiles_a = act->ne[0] / 32; + const uint32_t tile_row_stride = n_k_tiles_w * tile_size; + const uint32_t tile_row_transfer_size_aligned = n_k_tiles_a * aligned_tile_size; + + const uint32_t ct_start = src0_start_row / 32; + const uint32_t ct_end = (src0_end_row + 31) / 32; + + uint32_t push_ct = ct_start; + for (uint32_t d = 0; d < n_prefetch && push_ct < ct_end; d++, push_ct++) { + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + d * tile_row_transfer_size_aligned, src0_row + push_ct * tile_row_stride), + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); + } + + for (uint32_t ct = ct_start; ct < ct_end; ct++) { + const uint8_t * w_tile = (void *) dma_queue_pop(dma_q).dst; + + int valid_rows = (int)src_w->ne[1] - (int)(ct * 32); + valid_rows = MIN(32, MAX(0, valid_rows)); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ct); + for (uint32_t cid = 0; cid < (uint32_t) cne1; ++cid) { + struct mmid_row_mapping row_mapping = MMID_MATRIX_ROW(cur_a, cid); + const int rm1 = row_mapping.i1; + const int rm2 = row_mapping.i2; + + const uint32_t ir1 = fastmodulo(rm1, act->ne[1], &mmctx->mm_div_ne11); + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + (ir1 + rm2 * act->ne[1]) * src1_stride); + float * restrict dst_row = (float *) (dst->data + (rm1 * dst->nb[1] + rm2 * dst->nb[2])); + + mmctx->vec_dot_32x1(act->ne[0], &dst_row[ct * 32], w_tile, src1_col, valid_rows, NULL); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ct); + + if (push_ct < ct_end) { + dma_queue_push(dma_q, dma_make_data(w_tile, src0_row + push_ct * tile_row_stride), + aligned_tile_size, tile_size, tile_size, n_k_tiles_a); + push_ct++; + } + } + } + } +} + static int hvx_mm_init_vec_dot(struct htp_mm_context * mmctx, enum htp_data_type type) { switch (type) { case HTP_TYPE_Q4_0: @@ -1324,6 +1542,7 @@ static int hvx_mm_init_vec_dot(struct htp_mm_context * mmctx, enum htp_data_type mmctx->vec_dot_32x1 = tiled_vec_dot_q4_0_32x1; return 0; case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: mmctx->type = "q4_1_tiled-f32"; mmctx->vec_dot_32x1 = tiled_vec_dot_q4_1_32x1; return 0; @@ -1331,6 +1550,10 @@ static int hvx_mm_init_vec_dot(struct htp_mm_context * mmctx, enum htp_data_type mmctx->type = "q8_0_tiled-f32"; mmctx->vec_dot_32x1 = tiled_vec_dot_q8_0_32x1; return 0; + case HTP_TYPE_Q6_K: + mmctx->type = "q6_k_tiled-f32"; + mmctx->vec_dot_32x1 = tiled_vec_dot_q6_k_32x1; + return 0; case HTP_TYPE_IQ4_NL: mmctx->type = "iq4nl_tiled-f32"; mmctx->vec_dot_32x1 = tiled_vec_dot_iq4nl_32x1; @@ -1353,18 +1576,39 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { struct htp_mm_context mmctx_struct = {0}; struct htp_mm_context * mmctx = &mmctx_struct; mmctx->octx = octx; + mmctx->act = src1; const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - const uint32_t src0_nrows = ne01 * ne02 * ne03; + const uint32_t src0_nrows = ne01; const uint32_t src1_nrows = ne11 * ne12 * ne13; + mmctx->src1_nrows = src1_nrows; + + uint32_t src0_row_start = 0; + uint32_t src0_row_end = src0_nrows; + + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, can_split ? 32 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + src0_row_start = range.start; + src0_row_end = range.start + range.count; + } + + if (src0_row_start >= src0_row_end) { + return HTP_STATUS_OK; + } + + const uint32_t nrows = src0_row_end - src0_row_start; + mmctx->src0_row_start = src0_row_start; + mmctx->src0_row_end = src0_row_end; bool is_repacked = (src0->type == HTP_TYPE_Q4_0 || src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q8_0 || src0->type == HTP_TYPE_IQ4_NL || - src0->type == HTP_TYPE_MXFP4); + src0->type == HTP_TYPE_MXFP4 || src0->type == HTP_TYPE_Q6_K || + src0->type == HTP_TYPE_Q4_K); // Compute src0_nrows_per_thread - mmctx->src0_nrows_per_thread = (src0_nrows + octx->n_threads - 1) / octx->n_threads; + mmctx->src0_nrows_per_thread = fastdiv(nrows + octx->n_threads - 1, &octx->n_threads_div); if (is_repacked) { mmctx->src0_nrows_per_thread = hex_round_up(mmctx->src0_nrows_per_thread, 32); } else { @@ -1381,12 +1625,30 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { worker_callback_t quant_task_func; worker_callback_t matmul_job_func; uint32_t n_quant_tasks = 1; - if (src1_nrows > 1) { + const bool is_batched = (ne12 > 1 || ne13 > 1 || ne02 > 1 || ne03 > 1); + if (is_batched) { + if (is_repacked) { + switch (src0->type) { + case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_4d_repacked_q4_0; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: matmul_job_func = hvx_mm_4d_repacked_q4_1; break; + case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_4d_repacked_q8_0; break; + case HTP_TYPE_Q6_K: matmul_job_func = hvx_mm_4d_repacked_q6_k; break; + case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_4d_repacked_iq4nl; break; + case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_4d_repacked_mxfp4; break; + default: return HTP_STATUS_NO_SUPPORT; + } + } else { + matmul_job_func = hvx_mm_4d; + } + } else if (src1_nrows > 1) { if (is_repacked) { switch (src0->type) { case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_2d_repacked_q4_0; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_2d_repacked_q4_1; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: matmul_job_func = hvx_mm_2d_repacked_q4_1; break; case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_2d_repacked_q8_0; break; + case HTP_TYPE_Q6_K: matmul_job_func = hvx_mm_2d_repacked_q6_k; break; case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_2d_repacked_iq4nl; break; case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_2d_repacked_mxfp4; break; default: return HTP_STATUS_NO_SUPPORT; @@ -1398,8 +1660,10 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { if (is_repacked) { switch (src0->type) { case HTP_TYPE_Q4_0: matmul_job_func = hvx_mv_2d_repacked_q4_0; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mv_2d_repacked_q4_1; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: matmul_job_func = hvx_mv_2d_repacked_q4_1; break; case HTP_TYPE_Q8_0: matmul_job_func = hvx_mv_2d_repacked_q8_0; break; + case HTP_TYPE_Q6_K: matmul_job_func = hvx_mv_2d_repacked_q6_k; break; case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mv_2d_repacked_iq4nl; break; case HTP_TYPE_MXFP4: matmul_job_func = hvx_mv_2d_repacked_mxfp4; break; default: return HTP_STATUS_NO_SUPPORT; @@ -1413,7 +1677,7 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { switch (kparams->kernel_type) { case HTP_MM_KERNEL_HVX_F16_F16_VTCM: - quant_task_func = (src1->type == HTP_TYPE_F32) ? quantize_f32_f16_flat : quantize_f16_f16_flat; + quant_task_func = (src1->type == HTP_TYPE_F32) ? quantize_f32_f16 : quantize_f16_f16; mmctx->type = "f16-f16"; mmctx->vec_dot_1x1 = vec_dot_f16_f16_aa_1x1; mmctx->vec_dot_2x1 = vec_dot_f16_f16_aa_2x1; @@ -1421,34 +1685,8 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { src1_row_size = hex_round_up(ne10 * 2, 128); break; - case HTP_MM_KERNEL_HVX_F16_F32_DDR: - mmctx->type = "f16-f32"; - mmctx->vec_dot_1x1 = vec_dot_f16_f32_uu_1x1; - matmul_job_func = hvx_mm_4d; - mmctx->mm_div_ne12_ne1 = kparams->div_ne12_ne1; - mmctx->mm_div_ne1 = kparams->div_ne1; - mmctx->mm_div_r2 = kparams->div_r2; - mmctx->mm_div_r3 = kparams->div_r3; - need_quant = false; - quant_task_func = NULL; - src1_row_size = nb11; - break; - - case HTP_MM_KERNEL_HVX_F16_F16_DDR: - mmctx->type = "f16-f16"; - mmctx->vec_dot_1x1 = vec_dot_f16_f16_uu_1x1; - matmul_job_func = hvx_mm_4d; - mmctx->mm_div_ne12_ne1 = kparams->div_ne12_ne1; - mmctx->mm_div_ne1 = kparams->div_ne1; - mmctx->mm_div_r2 = kparams->div_r2; - mmctx->mm_div_r3 = kparams->div_r3; - src1_row_size = nb11; - need_quant = false; - quant_task_func = NULL; - break; - case HTP_MM_KERNEL_HVX_F32_F32_VTCM: - quant_task_func = quantize_f32_f32_flat; + quant_task_func = quantize_f32_f32; mmctx->type = "f32-f32"; mmctx->vec_dot_1x1 = vec_dot_f32_f32_aa_1x1; mmctx->vec_dot_2x1 = vec_dot_f32_f32_aa_2x1; @@ -1456,46 +1694,6 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { src1_row_size = hex_round_up(ne10 * 4, 128); break; - case HTP_MM_KERNEL_HVX_F32_F32_DDR: - quant_task_func = NULL; - mmctx->type = "f32-f32"; - mmctx->vec_dot_1x1 = vec_dot_f32_f32_uu_1x1; - mmctx->mm_div_ne12_ne1 = kparams->div_ne12_ne1; - mmctx->mm_div_ne1 = kparams->div_ne1; - mmctx->mm_div_r2 = kparams->div_r2; - mmctx->mm_div_r3 = kparams->div_r3; - src1_row_size = nb11; - need_quant = false; - matmul_job_func = hvx_mm_4d; - break; - - case HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT: { - n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_flat : quantize_f32_q8_0_flat; - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); - - if (src1_nrows > 1) { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_2d_repacked_q4_0_flat; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_2d_repacked_q4_1_flat; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_2d_repacked_q8_0_flat; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_2d_repacked_iq4nl_flat; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_2d_repacked_mxfp4_flat; break; - default: return HTP_STATUS_NO_SUPPORT; - } - } else { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mv_2d_repacked_q4_0_flat; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mv_2d_repacked_q4_1_flat; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mv_2d_repacked_q8_0_flat; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mv_2d_repacked_iq4nl_flat; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mv_2d_repacked_mxfp4_flat; break; - default: return HTP_STATUS_NO_SUPPORT; - } - } - break; - } - case HTP_MM_KERNEL_HVX_QUANT_BLOCK: case HTP_MM_KERNEL_HVX_QUANT_ROW: default: @@ -1507,9 +1705,9 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { const uint32_t nb = (ne10 + qk - 1) / qk; const uint32_t total_nb = src1_nrows * nb; - if (src1_nrows < octx->n_threads) { + if (src1_nrows < octx->n_threads && !is_batched) { n_quant_tasks = MIN(total_nb, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; for (uint32_t ith = 0; ith < n_quant_tasks; ++ith) { uint32_t ib_first = (total_nb * ith) / n_quant_tasks; uint32_t ib_last = (total_nb * (ith + 1)) / n_quant_tasks; @@ -1520,15 +1718,19 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { } } else { n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; } - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + src1_row_size = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); break; } + const uint32_t m_chunk = (kparams->m_chunk > 0 && (uint32_t) kparams->m_chunk < src1_nrows) + ? (uint32_t) kparams->m_chunk : src1_nrows; + const uint32_t m_layout_rows = m_chunk; + struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, ne10, src1_nrows, octx->n_threads, - dst_row_size, src0_row_size, src1_row_size, src2 ? src2->nb[1] : 0, kparams->n_prefetch, false, false, false); + htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, ne10, m_layout_rows, octx->n_threads, + dst_row_size, src0_row_size, src1_row_size, src2 ? src2->nb[1] : 0, kparams->n_prefetch, false, false); if (kparams->kernel_type == HTP_MM_KERNEL_HVX_F16_F16_VTCM || kparams->kernel_type == HTP_MM_KERNEL_HVX_F32_F32_VTCM || @@ -1536,13 +1738,13 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_BLOCK) { mmctx->vtcm_src1_size_per_thread = L.src1_bytes; } else { - mmctx->vtcm_src1_size_per_thread = L.src1_bytes / octx->n_threads; + mmctx->vtcm_src1_size_per_thread = fastdiv(L.src1_bytes, &octx->n_threads_div); } - mmctx->vtcm_src0_size_per_thread = L.src0_bytes / octx->n_threads; - mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads; + mmctx->vtcm_src0_size_per_thread = fastdiv(L.src0_bytes, &octx->n_threads_div); + mmctx->vtcm_dst_size_per_thread = fastdiv(L.dst_bytes, &octx->n_threads_div); - size_t vtcm_size = kparams->vtcm_size > 0 ? (size_t)kparams->vtcm_size : L.total_bytes; + const size_t vtcm_size = L.total_bytes; FARF(HIGH, "matmul-%s : src0-vtcm-size %zu src1-vtcm-size %zu dst-vtcm-size %zu (%zu)\n", mmctx->type, L.src0_bytes, L.src1_bytes, L.dst_bytes, vtcm_size); @@ -1570,314 +1772,149 @@ static int hvx_mm_matmul(struct htp_ops_context * octx) { mmctx->vtcm_src0_stride = src0_row_size_padded; mmctx->vtcm_src1_stride = src1_row_size; - if (need_quant) { - mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; - mmctx->quant_task_func = quant_task_func; - mmctx->n_quant_tasks = n_quant_tasks; - atomic_init(&mmctx->quant_barrier, n_quant_tasks); + if (kparams->m_chunk > 0 && (uint32_t) kparams->m_chunk < src1_nrows) { + atomic_init(&mmctx->quant_barrier, 0); + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); + + for (uint32_t m_start = 0; m_start < src1_nrows; m_start += m_chunk) { + const uint32_t cur_m_rows = MIN(src1_nrows - m_start, m_chunk); + mmctx->cur_m_start = m_start; + mmctx->cur_m_rows = cur_m_rows; + + if (need_quant) { + const uint32_t quant_tasks = MIN(cur_m_rows, octx->n_threads); + mmctx->n_quant_rows_per_thread = (cur_m_rows + quant_tasks - 1) / quant_tasks; + mmctx->n_quant_tasks = quant_tasks; + atomic_store(&mmctx->quant_barrier, quant_tasks); + mmctx->quant_task_func = quant_task_func; + } else { + mmctx->quant_task_func = NULL; + mmctx->n_quant_tasks = 0; + } + + worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, octx->n_threads); + } } else { - mmctx->quant_task_func = NULL; - mmctx->n_quant_tasks = 0; - } + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; + + if (need_quant) { + mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; + mmctx->quant_task_func = quant_task_func; + mmctx->n_quant_tasks = n_quant_tasks; + atomic_init(&mmctx->quant_barrier, n_quant_tasks); + } else { + mmctx->quant_task_func = NULL; + mmctx->n_quant_tasks = 0; + } - htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); - worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, octx->n_threads); + worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, octx->n_threads); + } return HTP_STATUS_OK; } -static void hvx_mm_qkv_2d(unsigned int nth, unsigned int ith, void * data) { +static void hvx_mm_nx_2d(unsigned int nth, unsigned int ith, void * data) { struct htp_mm_context * mmctx = data; struct htp_ops_context * octx = mmctx->octx; + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const uint32_t n_weights = kparams->n_weights; - const struct htp_tensor * restrict src0 = octx->src[0]; // Wk - const struct htp_tensor * restrict src1 = octx->src[1]; // x - const struct htp_tensor * restrict src2 = octx->src[2]; // Wv - const struct htp_tensor * restrict src3 = octx->src[3]; // Wq - const struct htp_tensor * restrict dst_k = octx->dsts[0]; - const struct htp_tensor * restrict dst_v = octx->dsts[1]; - const struct htp_tensor * restrict dst_q = octx->dsts[2]; - - const uint32_t ne00 = src0->ne[0]; - const uint32_t ne01 = src0->ne[1]; - const uint32_t ne02 = src0->ne[2]; - const uint32_t ne03 = src0->ne[3]; - - const uint32_t ne11 = src1->ne[1]; - const uint32_t ne12 = src1->ne[2]; - const uint32_t ne13 = src1->ne[3]; - - const uint32_t src0_nrows = ne01 * ne02 * ne03; - const uint32_t src1_nrows = ne11 * ne12 * ne13; - - const uint32_t src0_nrows_per_thread = mmctx->src0_nrows_per_thread; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); - const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U); - - const size_t dst_k_row_size = dst_k->nb[1]; // K and V share output width - const size_t dst_q_row_size = dst_q->nb[1]; // Q may be wider (GQA) - const size_t src0_row_size = src0->nb[1]; - const size_t src2_row_size = src2->nb[1]; - const size_t src3_row_size = src3->nb[1]; - - const size_t src0_stride = mmctx->vtcm_src0_stride; - const size_t src2_stride = mmctx->vtcm_src2_stride; - const size_t src3_stride = mmctx->vtcm_src3_stride; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const uint32_t src1_nrows = act->ne[1] * act->ne[2] * act->ne[3]; const size_t src1_stride = mmctx->vtcm_src1_stride; uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; - uint8_t * restrict vtcm_src2_ptr = mmctx->vtcm_src2 + mmctx->vtcm_src2_size_per_thread * ith; - uint8_t * restrict vtcm_src3_ptr = mmctx->vtcm_src3 + mmctx->vtcm_src3_size_per_thread * ith; - uint8_t * restrict src1_data = mmctx->vtcm_src1; - - dma_queue * dma_queue = octx->ctx->dma[ith]; + uint8_t * restrict src1_data = mmctx->vtcm_src1; - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; const uint32_t n_prefetch = kparams->n_prefetch; assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); const uint32_t prefetch_mask = n_prefetch - 1; - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; - const uint8_t * restrict src2_row = (const uint8_t *) src2->data; - const uint8_t * restrict src3_row = (const uint8_t *) src3->data; - - // Prefill spad with src0, src2, src3 rows - if (src0_start_row < src0_end_row) { - for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { - const int is0 = (ir0 - src0_start_row); - if (is0 >= (int)n_prefetch) { - break; - } - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), - src0_stride, src0_row_size, src0_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + ir0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src3_ptr + is0 * src3_stride, src3_row + ir0 * src3_row_size), - src3_stride, src3_row_size, src3_row_size, 2); - } - } + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; hvx_mm_run_quant_task(mmctx, ith); - if (src0_start_row >= src0_end_row) { - return; - } - - // Process rows - for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss2 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss3 = dma_queue_pop(dma_queue).dst; - - // Process src1 columns in pairs (2×2 tiling) - uint32_t ir1 = 0; - for (; ir1 + 1 < src1_nrows; ir1 += 2) { - const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); - const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); - - float * restrict dst_row0_k = (float *) (dst_k->data + ((ir1+0) * dst_k_row_size)); - float * restrict dst_row1_k = (float *) (dst_k->data + ((ir1+1) * dst_k_row_size)); - mmctx->vec_dot_2x2(ne00, &dst_row0_k[ir0], &dst_row1_k[ir0], ss0, ss0 + src0_stride, src1_col0, src1_col1); - - float * restrict dst_row0_v = (float *) (dst_v->data + ((ir1+0) * dst_k_row_size)); - float * restrict dst_row1_v = (float *) (dst_v->data + ((ir1+1) * dst_k_row_size)); - mmctx->vec_dot_2x2(ne00, &dst_row0_v[ir0], &dst_row1_v[ir0], ss2, ss2 + src2_stride, src1_col0, src1_col1); - - float * restrict dst_row0_q = (float *) (dst_q->data + ((ir1+0) * dst_q_row_size)); - float * restrict dst_row1_q = (float *) (dst_q->data + ((ir1+1) * dst_q_row_size)); - mmctx->vec_dot_2x2(ne00, &dst_row0_q[ir0], &dst_row1_q[ir0], ss3, ss3 + src3_stride, src1_col0, src1_col1); + for (uint32_t widx = 0; widx < n_weights; widx++) { + const struct htp_tensor * restrict src_w = octx->src[widx]; + const struct htp_tensor * restrict dst = octx->dsts[widx]; + if (!src_w || !dst) continue; + dma_queue * dma_q = octx->ctx->dma[ith]; + + const uint32_t ne00 = src_w->ne[0]; + const uint32_t ne01 = src_w->ne[1]; + uint32_t start_row = 0; + uint32_t end_row = ne01; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(ne01, can_split ? 32 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + start_row = range.start; + end_row = range.start + range.count; } - // Handle remaining src1 rows (fallback to 2×1) - for (; ir1 < src1_nrows; ++ir1) { - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - - float * restrict dst_row_k = (float *) (dst_k->data + (ir1 * dst_k_row_size)); - mmctx->vec_dot_2x1(ne00, &dst_row_k[ir0], ss0, ss0 + src0_stride, src1_col); + const uint32_t nrows = end_row - start_row; + uint32_t src0_nrows_per_thread = fastdiv(nrows + nth - 1, &octx->n_threads_div); + src0_nrows_per_thread += (src0_nrows_per_thread & 1); - float * restrict dst_row_v = (float *) (dst_v->data + (ir1 * dst_k_row_size)); - mmctx->vec_dot_2x1(ne00, &dst_row_v[ir0], ss2, ss2 + src2_stride, src1_col); + const uint32_t src0_start_row = start_row + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, end_row); + const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U); + if (src0_start_row >= src0_end_row) continue; - float * restrict dst_row_q = (float *) (dst_q->data + (ir1 * dst_q_row_size)); - mmctx->vec_dot_2x1(ne00, &dst_row_q[ir0], ss3, ss3 + src3_stride, src1_col); - } + const size_t dst_row_size = dst->nb[1]; + const size_t src0_row_size = src_w->nb[1]; + const size_t src0_stride = hex_round_up(src0_row_size, 128); - // Prefetch next (n + vtcm_nrows) rows - const int pr0 = (ir0 + n_prefetch); - const int is0 = (pr0 - src0_start_row) & prefetch_mask; - if (pr0 < src0_end_row_x2) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), - src0_stride, src0_row_size, src0_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + pr0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src3_ptr + is0 * src3_stride, src3_row + pr0 * src3_row_size), - src3_stride, src3_row_size, src3_row_size, 2); - } - } + const dma_addr_t src0_row = src_w->data; - // Process last row (if any) - if (src0_end_row != src0_end_row_x2) { - uint32_t ir0 = src0_end_row_x2; - const int is0 = (ir0 - src0_start_row) & prefetch_mask; - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), - src0_stride, src0_row_size, src0_row_size, 1); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + ir0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 1); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src3_ptr + is0 * src3_stride, src3_row + ir0 * src3_row_size), - src3_stride, src3_row_size, src3_row_size, 1); - - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss2 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss3 = dma_queue_pop(dma_queue).dst; - - for (uint32_t ir1 = 0; ir1 < src1_nrows; ++ir1) { - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - - float * restrict dst_row_k = (float *) (dst_k->data + (ir1 * dst_k_row_size)); - mmctx->vec_dot_1x1(ne00, &dst_row_k[ir0], ss0, src1_col); - - float * restrict dst_row_v = (float *) (dst_v->data + (ir1 * dst_k_row_size)); - mmctx->vec_dot_1x1(ne00, &dst_row_v[ir0], ss2, src1_col); - - float * restrict dst_row_q = (float *) (dst_q->data + (ir1 * dst_q_row_size)); - mmctx->vec_dot_1x1(ne00, &dst_row_q[ir0], ss3, src1_col); - } - } -} - -static void hvx_mm_ffn_2d(unsigned int nth, unsigned int ith, void * data) { - struct htp_mm_context * mmctx = data; - struct htp_ops_context * octx = mmctx->octx; - - const struct htp_tensor * restrict src0 = octx->src[0]; // Wgate - const struct htp_tensor * restrict src1 = octx->src[1]; // y - const struct htp_tensor * restrict src2 = octx->src[2]; // Wup - const struct htp_tensor * restrict dst_gate = octx->dsts[0]; - const struct htp_tensor * restrict dst_up = octx->dsts[1]; - - const uint32_t ne00 = src0->ne[0]; - const uint32_t ne01 = src0->ne[1]; - const uint32_t ne02 = src0->ne[2]; - const uint32_t ne03 = src0->ne[3]; - - const uint32_t ne11 = src1->ne[1]; - const uint32_t ne12 = src1->ne[2]; - const uint32_t ne13 = src1->ne[3]; - - const uint32_t src0_nrows = ne01 * ne02 * ne03; - const uint32_t src1_nrows = ne11 * ne12 * ne13; - - const uint32_t src0_nrows_per_thread = mmctx->src0_nrows_per_thread; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); - const uint32_t src0_end_row_x2 = src0_start_row + ((src0_end_row - src0_start_row) & ~1U); - - const size_t dst_row_size = dst_gate->nb[1]; - const size_t src0_row_size = src0->nb[1]; - const size_t src2_row_size = src2->nb[1]; - - const size_t src0_stride = mmctx->vtcm_src0_stride; - const size_t src2_stride = mmctx->vtcm_src2_stride; - const size_t src1_stride = mmctx->vtcm_src1_stride; - - uint8_t * restrict vtcm_src0_ptr = mmctx->vtcm_src0 + mmctx->vtcm_src0_size_per_thread * ith; - uint8_t * restrict vtcm_src2_ptr = mmctx->vtcm_src2 + mmctx->vtcm_src2_size_per_thread * ith; - uint8_t * restrict src1_data = mmctx->vtcm_src1; - - dma_queue * dma_queue = octx->ctx->dma[ith]; - - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - const uint32_t n_prefetch = kparams->n_prefetch; - assert(n_prefetch >= 2 && n_prefetch <= HTP_MM_MAX_PREFETCH && (n_prefetch & (n_prefetch - 1)) == 0); - const uint32_t prefetch_mask = n_prefetch - 1; - - const uint8_t * restrict src0_row = (const uint8_t *) src0->data; - const uint8_t * restrict src2_row = (const uint8_t *) src2->data; - - // Prefill spad with src0, src2 rows - if (src0_start_row < src0_end_row) { for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { const int is0 = (ir0 - src0_start_row); - if (is0 >= (int)n_prefetch) { - break; - } - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + if (is0 >= (int)n_prefetch) break; + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), src0_stride, src0_row_size, src0_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + ir0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 2); } - } - - hvx_mm_run_quant_task(mmctx, ith); - - if (src0_start_row >= src0_end_row) { - return; - } - - // Process rows - for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss2 = dma_queue_pop(dma_queue).dst; - - // Process src1 columns in pairs (2×2 tiling) - uint32_t ir1 = 0; - for (; ir1 + 1 < src1_nrows; ir1 += 2) { - const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); - const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); - - float * restrict dst_row0_gate = (float *) (dst_gate->data + ((ir1+0) * dst_row_size)); - float * restrict dst_row1_gate = (float *) (dst_gate->data + ((ir1+1) * dst_row_size)); - mmctx->vec_dot_2x2(ne00, &dst_row0_gate[ir0], &dst_row1_gate[ir0], ss0, ss0 + src0_stride, src1_col0, src1_col1); - - float * restrict dst_row0_up = (float *) (dst_up->data + ((ir1+0) * dst_row_size)); - float * restrict dst_row1_up = (float *) (dst_up->data + ((ir1+1) * dst_row_size)); - mmctx->vec_dot_2x2(ne00, &dst_row0_up[ir0], &dst_row1_up[ir0], ss2, ss2 + src2_stride, src1_col0, src1_col1); - } - - // Handle remaining src1 rows (fallback to 2×1) - for (; ir1 < src1_nrows; ++ir1) { - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - - float * restrict dst_row_gate = (float *) (dst_gate->data + (ir1 * dst_row_size)); - mmctx->vec_dot_2x1(ne00, &dst_row_gate[ir0], ss0, ss0 + src0_stride, src1_col); - float * restrict dst_row_up = (float *) (dst_up->data + (ir1 * dst_row_size)); - mmctx->vec_dot_2x1(ne00, &dst_row_up[ir0], ss2, ss2 + src2_stride, src1_col); - } + for (uint32_t ir0 = src0_start_row; ir0 < src0_end_row_x2; ir0 += 2) { + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + uint32_t ir1 = 0; + for (; ir1 + 1 < src1_nrows; ir1 += 2) { + const uint8_t * restrict src1_col0 = (const uint8_t *) (src1_data + (ir1+0) * src1_stride); + const uint8_t * restrict src1_col1 = (const uint8_t *) (src1_data + (ir1+1) * src1_stride); + float * restrict dst_row0 = (float *) (dst->data + ((ir1+0) * dst_row_size)); + float * restrict dst_row1 = (float *) (dst->data + ((ir1+1) * dst_row_size)); + mmctx->vec_dot_2x2(ne00, &dst_row0[ir0], &dst_row1[ir0], ss0, ss0 + src0_stride, src1_col0, src1_col1); + } + for (; ir1 < src1_nrows; ++ir1) { + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); + float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); + mmctx->vec_dot_2x1(ne00, &dst_row[ir0], ss0, ss0 + src0_stride, src1_col); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); - // Prefetch next rows - const int pr0 = (ir0 + n_prefetch); - const int is0 = (pr0 - src0_start_row) & prefetch_mask; - if (pr0 < src0_end_row_x2) { - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), - src0_stride, src0_row_size, src0_row_size, 2); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + pr0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 2); + const int pr0 = (ir0 + n_prefetch); + const int is0 = (pr0 - src0_start_row) & prefetch_mask; + if (pr0 < src0_end_row_x2) { + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + pr0 * src0_row_size), + src0_stride, src0_row_size, src0_row_size, 2); + } } - } - - // Process last row (if any) - if (src0_end_row != src0_end_row_x2) { - uint32_t ir0 = src0_end_row_x2; - const int is0 = (ir0 - src0_start_row) & prefetch_mask; - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), - src0_stride, src0_row_size, src0_row_size, 1); - dma_queue_push(dma_queue, dma_make_ptr(vtcm_src2_ptr + is0 * src2_stride, src2_row + ir0 * src2_row_size), - src2_stride, src2_row_size, src2_row_size, 1); - - const uint8_t * ss0 = dma_queue_pop(dma_queue).dst; - const uint8_t * ss2 = dma_queue_pop(dma_queue).dst; - - for (uint32_t ir1 = 0; ir1 < src1_nrows; ++ir1) { - const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); - - float * restrict dst_row_gate = (float *) (dst_gate->data + (ir1 * dst_row_size)); - mmctx->vec_dot_1x1(ne00, &dst_row_gate[ir0], ss0, src1_col); - float * restrict dst_row_up = (float *) (dst_up->data + (ir1 * dst_row_size)); - mmctx->vec_dot_1x1(ne00, &dst_row_up[ir0], ss2, src1_col); + if (src0_end_row != src0_end_row_x2) { + uint32_t ir0 = src0_end_row_x2; + const int is0 = (ir0 - src0_start_row) & prefetch_mask; + dma_queue_push(dma_q, dma_make_data(vtcm_src0_ptr + is0 * src0_stride, src0_row + ir0 * src0_row_size), + src0_stride, src0_row_size, src0_row_size, 1); + const uint8_t * ss0 = (void *) dma_queue_pop(dma_q).dst; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir0); + for (uint32_t ir1 = 0; ir1 < src1_nrows; ++ir1) { + const uint8_t * restrict src1_col = (const uint8_t *) (src1_data + ir1 * src1_stride); + float * restrict dst_row = (float *) (dst->data + (ir1 * dst_row_size)); + mmctx->vec_dot_1x1(ne00, &dst_row[ir0], ss0, src1_col); + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir0); } } } @@ -1900,6 +1937,7 @@ DEQUANTIZE_WORKER_LOOP_IMPL(q4_1) DEQUANTIZE_WORKER_LOOP_IMPL(iq4_nl) DEQUANTIZE_WORKER_LOOP_IMPL(mxfp4) DEQUANTIZE_WORKER_LOOP_IMPL(q8_0) +DEQUANTIZE_WORKER_LOOP_IMPL(q6_k) static void convert_f16_worker_loop(unsigned int n, unsigned int i, void *data) { tiled_dequantize_state_t *state = (tiled_dequantize_state_t *)data; @@ -1949,36 +1987,36 @@ static void transfer_output_chunk_worker_fn(unsigned int n, unsigned int i, void } typedef struct { - const struct mmid_row_mapping *matrix_rows; - __fp16 *dst; - const float *src; - uint32_t n_tasks; - uint32_t n_tot_chunks; - uint32_t n_chunks_per_task; - uint32_t k_block; - uint32_t k_stride; - uint32_t k_valid; - struct htp_thread_trace * traces; - struct htp_context * ctx; - float * vtcm_f32_act; - size_t vtcm_f32_act_bytes_per_thread; - uint32_t dma_step_rows; - uint32_t dma_step_rows_shift; + struct htp_context * ctx; + struct htp_thread_trace * traces; + __fp16 * dst; + const float * src; + const struct mmid_row_mapping * matrix_rows; + float * vtcm_f32_act; + uint32_t n_tasks; + uint32_t n_tot_chunks; + uint32_t n_chunks_per_task; + uint32_t k_block; + uint32_t k_stride; + uint32_t k_valid; + size_t vtcm_f32_act_bytes_per_thread; + uint32_t dma_step_rows; + uint32_t dma_step_rows_shift; } activation_transfer_task_state_t; typedef struct { - __fp16 *dst; - const float *src; + struct htp_context * ctx; + struct htp_thread_trace * traces; + __fp16 * dst; + const float * src; + float * vtcm_f32_act; uint32_t n_rows; uint32_t k_block; uint32_t k_stride; uint32_t k_valid; uint32_t n_col_chunks; struct fastdiv_values n_threads_div; - float *vtcm_f32_act; size_t vtcm_f32_act_bytes; - struct htp_thread_trace *traces; - struct htp_context *ctx; uint32_t dma_step_rows; uint32_t dma_step_rows_shift; } activation_transfer_col_chunk_state_t; @@ -2006,7 +2044,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined_col_chunk( // Push step 0 if (n_steps > 0 && n_rows > 0) { uint32_t nrows_to_fetch = hex_smin(n_rows, R); - dma_queue_push(dma_q, dma_make_ptr(thread_f32_act, src + c_first), + dma_queue_push(dma_q, dma_make_data(thread_f32_act, src + c_first), c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch); } // Push step 1 @@ -2016,7 +2054,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined_col_chunk( uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R); const float *next_src = src + next_r * k_stride + c_first; float *next_buf = thread_f32_act + 1 * R * c_len; - dma_queue_push(dma_q, dma_make_ptr(next_buf, next_src), + dma_queue_push(dma_q, dma_make_data(next_buf, next_src), c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch); } } @@ -2047,7 +2085,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined_col_chunk( if (next_r < n_rows) { uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R); const float *next_src = src + next_r * k_stride + c_first; - dma_queue_push(dma_q, dma_make_ptr(curr_buf, next_src), + dma_queue_push(dma_q, dma_make_data(curr_buf, next_src), c_len * sizeof(float), k_stride * sizeof(float), k_chunk_valid * sizeof(float), nrows_to_fetch); } } @@ -2151,7 +2189,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined( // Push step 0 if (n_steps > 0 && n_rows > 0) { uint32_t nrows_to_fetch = hex_smin(n_rows, R); - dma_queue_push(dma_q, dma_make_ptr(thread_f32_act, src), + dma_queue_push(dma_q, dma_make_data(thread_f32_act, src), k_block * sizeof(float), k_stride * sizeof(float), k_valid * sizeof(float), nrows_to_fetch); } // Push step 1 (if valid) @@ -2161,7 +2199,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined( uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R); const float *next_src = src + next_r * k_stride; float *next_buf = thread_f32_act + 1 * R * k_block; - dma_queue_push(dma_q, dma_make_ptr(next_buf, next_src), + dma_queue_push(dma_q, dma_make_data(next_buf, next_src), k_block * sizeof(float), k_stride * sizeof(float), k_valid * sizeof(float), nrows_to_fetch); } } @@ -2190,7 +2228,7 @@ static void transfer_activation_chunk_fp32_to_fp16_dma_pipelined( if (next_r < n_rows) { uint32_t nrows_to_fetch = hex_smin(n_rows - next_r, R); const float *next_src = src + next_r * k_stride; - dma_queue_push(dma_q, dma_make_ptr(curr_buf, next_src), + dma_queue_push(dma_q, dma_make_data(curr_buf, next_src), k_block * sizeof(float), k_stride * sizeof(float), k_valid * sizeof(float), nrows_to_fetch); } } @@ -2222,9 +2260,10 @@ static void transfer_activation_chunk_worker_fn(unsigned int n, unsigned int i, } typedef struct { - const struct mmid_row_mapping *matrix_rows; - __fp16 *dst; - const float *src; + struct htp_thread_trace * traces; + const struct mmid_row_mapping * matrix_rows; + __fp16 * dst; + const float * src; uint32_t n_tasks; uint32_t n_tot_chunks; uint32_t n_chunks_per_task; @@ -2238,13 +2277,13 @@ typedef struct { uint32_t start_row; uint32_t cne1; uint32_t k_valid; - struct htp_thread_trace *traces; } activation_transfer_gathered_task_state_t; typedef struct { - const struct mmid_row_mapping *matrix_rows; - const __fp16 *vtcm_src; - float *dst; + struct htp_thread_trace * traces; + const struct mmid_row_mapping * matrix_rows; + const __fp16 * vtcm_src; + float * dst; uint32_t n_tasks; uint32_t n_tot_chunks; uint32_t n_chunks_per_task; @@ -2255,17 +2294,16 @@ typedef struct { size_t dst_nb2; uint32_t start_row; uint32_t cne1; - struct htp_thread_trace *traces; } output_transfer_scattered_task_state_t; static void transfer_activation_chunk_gathered_worker_fn(unsigned int n, unsigned int i, void *data) { activation_transfer_gathered_task_state_t *st = data; struct htp_thread_trace * tr = &st->traces[i]; - int chunk_idx = i; - int chunk_size = st->n_chunks_per_task; + int chunk_idx = i; + int chunk_size = st->n_chunks_per_task; int vtcm_start_row = chunk_idx * chunk_size; - int start_row = st->start_row + vtcm_start_row; - int n_rows = hex_smin(st->cne1 - start_row, chunk_size); + int start_row = st->start_row + vtcm_start_row; + int n_rows = hex_smin(st->cne1 - start_row, chunk_size); if (n_rows > 0) { htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, chunk_idx); transfer_activation_chunk_fp32_to_fp16_gathered( @@ -2357,17 +2395,17 @@ static void dequantize_tiled_weight_chunk_to_fp16_tiles( } typedef struct { - float *dst; - const float *src2; - const __fp16 *vtcm_src; - uint32_t n_rows; - uint32_t n_cols; - uint32_t dst_stride; - uint32_t src2_stride; - uint32_t dst_cols; - struct fastdiv_values n_threads_div; - struct htp_thread_trace *traces; - struct htp_context *ctx; + struct htp_context * ctx; + struct htp_thread_trace * traces; + float * dst; + const __fp16 * vtcm_src; + const float * src2; + uint32_t n_rows; + uint32_t n_cols; + uint32_t dst_stride; + uint32_t src2_stride; + uint32_t dst_cols; + struct fastdiv_values n_threads_div; } output_transfer_col_chunk_state_t; static void transfer_output_chunk_col_chunk_worker_fn(unsigned int n, unsigned int i, void *data) { @@ -2376,19 +2414,19 @@ static void transfer_output_chunk_col_chunk_worker_fn(unsigned int n, unsigned i struct htp_thread_trace * tr = &st->traces[i]; uint32_t n_blocks = st->n_cols / 32; - uint32_t b_first = fastdiv(n_blocks * i, &st->n_threads_div); - uint32_t b_last = fastdiv(n_blocks * (i + 1), &st->n_threads_div); - uint32_t c_first = b_first * 32; - uint32_t c_last = b_last * 32; - uint32_t c_len = c_last - c_first; + uint32_t b_first = fastdiv(n_blocks * i, &st->n_threads_div); + uint32_t b_last = fastdiv(n_blocks * (i + 1), &st->n_threads_div); + uint32_t c_first = b_first * 32; + uint32_t c_last = b_last * 32; + uint32_t c_len = c_last - c_first; if (c_len == 0) return; htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_O_PROC, c_first); - float *dst = st->dst + c_first; - const float *src2 = st->src2 ? (st->src2 + c_first) : NULL; const __fp16 *vtcm_src = st->vtcm_src + b_first * HTP_MM_HMX_TILE_N_ELMS; + const float *src2 = st->src2 ? (st->src2 + c_first) : NULL; + float *dst = st->dst + c_first; int chunk_dst_cols = (int)st->dst_cols - (int)c_first; if (chunk_dst_cols > 0) { @@ -2409,7 +2447,7 @@ static void transfer_output_chunk_threaded(struct htp_context *ctx, float *dst, uint32_t n_blocks = (uint32_t)n_cols / 32; if (n_threads > 1 && n_blocks >= (uint32_t)n_threads) { - struct fastdiv_values n_threads_div = init_fastdiv_values(n_threads); + struct fastdiv_values n_threads_div = (n_threads == (int)ctx->n_threads) ? ctx->n_threads_div : init_fastdiv_values(n_threads); output_transfer_col_chunk_state_t col_state; col_state.dst = dst; col_state.src2 = src2; @@ -2539,8 +2577,7 @@ static void transfer_activation_chunk_threaded(const struct activation_transfer_ state.ctx = ctx; state.vtcm_f32_act = vtcm_f32_act; - int active_threads = hex_smin(n_threads, (int)state.n_tasks); - state.vtcm_f32_act_bytes_per_thread = hex_align_down(vtcm_f32_act_bytes / active_threads, 128); + state.vtcm_f32_act_bytes_per_thread = hex_align_down(fastdiv(vtcm_f32_act_bytes, act_threads_div), 128); uint32_t dma_step_rows = 2; uint32_t dma_step_rows_shift = 1; @@ -2555,6 +2592,7 @@ static void transfer_activation_chunk_threaded(const struct activation_transfer_ state.dma_step_rows = dma_step_rows; state.dma_step_rows_shift = dma_step_rows_shift; + int active_threads = hex_smin(n_threads, (int)state.n_tasks); if (state.n_tasks == 1 || n_threads == 1) { transfer_activation_chunk_worker_fn(1, 0, &state); } else { @@ -2597,10 +2635,12 @@ static inline void hmx_matmul_job_init(hmx_matmul_job_t * job, } static int hmx_mm_2d_f32(struct htp_context *ctx, + dma_queue *weight_dma, float *restrict dst, - const float *restrict src2, + dma_addr_t src2_addr, + size_t src2_bytes, const float *activation, - const uint8_t *weight, + dma_addr_t weight, int m, int k, int n, int act_stride, int weight_stride, @@ -2634,40 +2674,308 @@ static int hmx_mm_2d_f32(struct htp_context *ctx, switch (weight_type) { case HTP_TYPE_Q4_0: dequant_worker_fn = dequantize_tiled_worker_loop_q4_0; break; case HTP_TYPE_IQ4_NL: dequant_worker_fn = dequantize_tiled_worker_loop_iq4_nl; break; - case HTP_TYPE_Q4_1: dequant_worker_fn = dequantize_tiled_worker_loop_q4_1; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: dequant_worker_fn = dequantize_tiled_worker_loop_q4_1; break; + case HTP_TYPE_MXFP4: dequant_worker_fn = dequantize_tiled_worker_loop_mxfp4; break; + case HTP_TYPE_Q8_0: dequant_worker_fn = dequantize_tiled_worker_loop_q8_0; break; + case HTP_TYPE_Q6_K: dequant_worker_fn = dequantize_tiled_worker_loop_q6_k; break; + case HTP_TYPE_F16: dequant_worker_fn = convert_f16_worker_loop; break; + case HTP_TYPE_F32: dequant_worker_fn = quantize_f32_worker_loop; break; + default: + return -1; + } + + const int n_k_tiles = k / HTP_MM_HMX_TILE_N_COLS; + const struct fastdiv_values n_k_tiles_div = init_fastdiv_values(n_k_tiles); + + const bool is_quant = (weight_type != HTP_TYPE_F16 && weight_type != HTP_TYPE_F32); + const size_t vec_dot_size = k * sizeof(__fp16); + const size_t vtcm_budget = ctx->vtcm_size; + + const uint32_t dma_dst_stride = is_quant ? aligned_tile_size : row_stride; + const uint32_t dma_src_stride = is_quant ? tile_size : weight_stride; + const uint32_t dma_width_bytes = is_quant ? tile_size : row_stride; + + size_t m_chunk_n_rows = m_chunk; + size_t n_chunk_n_cols = n_chunk; + size_t vtcm_used = vtcm_size; + + const size_t qweight_row_stride = is_quant ? (size_t)(n_k_tiles * aligned_tile_size) / 32 : 0; + + struct htp_mm_hmx_vtcm_layout L; + htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_2D, weight_type, k, m_chunk_n_rows, n_chunk_n_cols, 1, false, pipeline, act_threads, aligned_tile_size, src2_bytes); + + vtcm_used = L.total_bytes; + if (vtcm_used > vtcm_budget) { + FARF(ERROR, "hmx-mm-2d-precomputed: VTCM overflow: used %zu budget %zu, m %d k %d n %d mc %zu nc %zu", + vtcm_used, vtcm_budget, m, k, n, m_chunk_n_rows, n_chunk_n_cols); + return -1; + } + + uint8_t * const base = (uint8_t *) ctx->vtcm_base; + __fp16 *vtcm_weight_raw[2] = { + VTCM_LAYOUT_PTR(__fp16, base, L.off_weight[0]), + VTCM_LAYOUT_PTR_OPTIONAL(__fp16, base, L.off_weight[1], pipeline) + }; + + __fp16 *vtcm_f16_act = VTCM_LAYOUT_PTR(__fp16, base, L.off_act); + float *vtcm_f32_act = VTCM_LAYOUT_PTR(float, base, L.off_act_f32); + __fp16 *vtcm_output = VTCM_LAYOUT_PTR(__fp16, base, L.off_dst[0]); + void *vtcm_scratch0 = VTCM_LAYOUT_PTR(void, base, L.off_scratch[0]); + void *vtcm_scratch1 = VTCM_LAYOUT_PTR_OPTIONAL(void, base, L.off_scratch[1], pipeline); + void *vtcm_scratch2 = VTCM_LAYOUT_PTR_OPTIONAL(void, base, L.off_dst[1], pipeline); + __fp16 *vtcm_scales = VTCM_LAYOUT_PTR(__fp16, base, L.off_scales); + + hmx_init_column_scales(vtcm_scales, Q6_V_vsplat_R(0x3c00)); // scale: 1.0, bias: 0.0 in FP16 + + const bool has_src2 = (src2_bytes > 0 && src2_addr != 0); + float *vtcm_src2 = VTCM_LAYOUT_PTR_OPTIONAL(float, base, L.off_src2, has_src2); + if (has_src2) { + dma_queue_push(weight_dma, dma_make_data(vtcm_src2, src2_addr), hex_align_up(src2_bytes, 128), 0, src2_bytes, 1); + dma_queue_pop(weight_dma); + } + + FARF(HIGH, "hmx-mm-2d: m %d k %d n %d wtype %d mc %zu nc %zu vtcm %zu/%zu", + m, k, n, weight_type, m_chunk_n_rows, n_chunk_n_cols, vtcm_used, vtcm_budget); + + int n_chunk_cnt = hmx_ceil_div(n, n_chunk_n_cols); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); + + if (pipeline) { + // --- Asynchronous Pipelined Loop --- + hmx_matmul_job_t job_slots[2]; // persistent double-buffered job descriptors + + for (size_t mr = 0; mr < m; mr += m_chunk_n_rows) { + const size_t n_rows = hex_smin(m - mr, m_chunk_n_rows); + + void *vtcm_weight_bufs[2] = { vtcm_scratch0, vtcm_scratch1 }; + void *vtcm_output_bufs[2] = { vtcm_output, vtcm_scratch2 }; + + struct activation_transfer_params act_params = { + .ctx = ctx, + .dst = vtcm_f16_act, + .src = activation + mr * act_stride, + .n_rows = (int) n_rows, + .k_block = k, + .k_stride = act_stride, + .n_threads = act_threads, + .act_threads_div = act_threads_div, + .k_div = k_div, + .k_valid = k_valid, + .vtcm_f32_act = vtcm_f32_act, + .vtcm_f32_act_bytes = L.act_f32_bytes, + }; + transfer_activation_chunk_threaded(&act_params); + + // Prologue: push A0 and optionally A1 (if n_chunk_cnt > 1) + const size_t n_cols_A0 = hex_smin(n - 0 * n_chunk_n_cols, n_chunk_n_cols); + const uint32_t height_A0 = is_quant ? (n_cols_A0 / 32) * n_k_tiles : n_cols_A0; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[0], weight), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_A0); + + if (1 < n_chunk_cnt) { + const size_t n_cols_A1 = hex_smin(n - 1 * n_chunk_n_cols, n_chunk_n_cols); + const uint32_t height_A1 = is_quant ? (n_cols_A1 / 32) * n_k_tiles : n_cols_A1; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[1], weight + n_chunk_n_cols * weight_stride), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_A1); + } + + // Main loop: pop A_i -> dequantize A_i -> push A_{i+2} -> submit C_i -> wait C_{i-1} and store D_{i-1} + for (int i = 0; i < n_chunk_cnt; ++i) { + const size_t nc = i * n_chunk_n_cols; + const size_t nc_p2 = nc + 2 * n_chunk_n_cols; + + const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); + const size_t n_cols_p2 = hex_smin(n - nc_p2, n_chunk_n_cols); + + // 1. pop A_i + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; + + // 2. dequantize A_i + dequantize_tiled_weight_chunk_to_fp16_tiles( + ctx, vtcm_weight_bufs[i % 2], curr_raw, + n_cols, k, row_stride, weight_type, + n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); + + // 3. push A_{i+2} (if i+2 < n_chunk_cnt) + if (i + 2 < n_chunk_cnt) { + const uint32_t height_p2 = is_quant ? (n_cols_p2 / 32) * n_k_tiles : n_cols_p2; + dma_queue_push(weight_dma, dma_make_data(curr_raw, weight + nc_p2 * weight_stride), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_p2); + } + + // 4. submit C_i + hmx_matmul_job_init(&job_slots[i % 2], (__fp16 *) vtcm_output_bufs[i % 2], + (__fp16 *) vtcm_f16_act, (__fp16 *) vtcm_weight_bufs[i % 2], + vtcm_scales, hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS), + hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS), k / HTP_MM_HMX_TILE_N_ROWS); + hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job_slots[i % 2])); + + // 5. wait C_{i-1} and store D_{i-1} (multi-thread HVX, parallel with C_i) + if (i > 0) { + hmx_queue_pop(ctx->hmx_queue); + const size_t nc_prev = (i - 1) * n_chunk_n_cols; + const size_t n_cols_prev = hex_smin(n - nc_prev, n_chunk_n_cols); + float *output_chunk = dst + (mr * dst_stride + nc_prev); + const float *src2_chunk = has_src2 ? (vtcm_src2 + mr * src2_stride + nc_prev) : NULL; + int chunk_dst_cols = dst_cols - (int)nc_prev; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output_bufs[(i - 1) % 2], n_rows, n_cols_prev, dst_stride, src2_stride, chunk_dst_cols, n_threads); + } + } + } + + // Epilogue: wait C_{last} and store D_{last} + hmx_queue_pop(ctx->hmx_queue); + const size_t nc_last = (n_chunk_cnt - 1) * n_chunk_n_cols; + const size_t n_cols_last = hex_smin(n - nc_last, n_chunk_n_cols); + float *output_chunk = dst + (mr * dst_stride + nc_last); + const float *src2_chunk = has_src2 ? (vtcm_src2 + mr * src2_stride + nc_last) : NULL; + int chunk_dst_cols = dst_cols - (int)nc_last; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output_bufs[(n_chunk_cnt - 1) % 2], n_rows, n_cols_last, dst_stride, src2_stride, chunk_dst_cols, n_threads); + } + } + } else { + // --- Synchronous loop (m <= 32 or fallback) --- + hmx_matmul_job_t job; + for (size_t mr = 0; mr < m; mr += m_chunk_n_rows) { + const size_t n_rows = hex_smin(m - mr, m_chunk_n_rows); + + struct activation_transfer_params act_params = { + .ctx = ctx, + .dst = vtcm_f16_act, + .src = activation + mr * act_stride, + .n_rows = (int) n_rows, + .k_block = k, + .k_stride = act_stride, + .n_threads = act_threads, + .act_threads_div = act_threads_div, + .k_div = k_div, + .k_valid = k_valid, + .vtcm_f32_act = vtcm_f32_act, + .vtcm_f32_act_bytes = L.act_f32_bytes, + }; + transfer_activation_chunk_threaded(&act_params); + + // A0: Pre-fetch the first weight chunk (nc = 0) + if (n > 0) { + const size_t n_cols = hex_smin(n, n_chunk_n_cols); + const uint32_t height = is_quant ? (n_cols / 32) * n_k_tiles : n_cols; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[0], weight), dma_dst_stride, dma_src_stride, dma_width_bytes, height); + } + + for (size_t nc = 0; nc < n; nc += n_chunk_n_cols) { + const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); + const size_t n_row_tiles = hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS); + const size_t n_col_tiles = hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS); + + // A: Wait for weight DMA + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; + + // B: Weight Dequantize (Threaded) + dequantize_tiled_weight_chunk_to_fp16_tiles( + ctx, vtcm_scratch0, curr_raw, + n_cols, k, row_stride, weight_type, + n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); + + // Start weight DMA for the next chunk early + const size_t nc_next = nc + n_chunk_n_cols; + if (nc_next < n) { + const size_t n_cols_next = hex_smin(n - nc_next, n_chunk_n_cols); + const uint32_t height_next = is_quant ? (n_cols_next / 32) * n_k_tiles : n_cols_next; + dma_queue_push(weight_dma, dma_make_data(curr_raw, weight + nc_next * weight_stride), dma_dst_stride, dma_src_stride, dma_width_bytes, height_next); + } + + // C: HMX Compute (Queue-based) + hmx_matmul_job_init(&job, vtcm_output, vtcm_f16_act, vtcm_scratch0, vtcm_scales, n_row_tiles, n_col_tiles, k / HTP_MM_HMX_TILE_N_ROWS); + hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job)); + hmx_queue_pop(ctx->hmx_queue); + + // D: Output Store + float *output_chunk = dst + (mr * dst_stride + nc); + const float *src2_chunk = has_src2 ? (vtcm_src2 + mr * src2_stride + nc) : NULL; + int chunk_dst_cols = dst_cols - (int)nc; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output, n_rows, n_cols, dst_stride, src2_stride, chunk_dst_cols, n_threads); + } + } + } + } + + return 0; +} + +static int hmx_mm_nx_2d_f32(struct htp_ops_context * octx, const struct htp_mm_kernel_params * kparams) { + struct htp_context * ctx = octx->ctx; + struct htp_thread_trace * tr = &ctx->trace[0]; + htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); + + const uint32_t n_weights = kparams->n_weights; + if (n_weights == 0 || n_weights > HTP_OP_MAX_OUTPUTS) { + return HTP_STATUS_INVAL_PARAMS; + } + + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + + const int weight_type = (int) src0->type; + const int k = (int) act->ne[0]; + const int k_valid = (int) act->ne[0]; + const int m = (int) (act->ne[1] * act->ne[2] * act->ne[3]); + const int act_stride = (int) (act->nb[1] / sizeof(float)); + const float * activation = (const float *) act->data; + + if (k % 32 != 0) { return HTP_STATUS_NO_SUPPORT; } + if (!hex_is_aligned(activation, VLEN)) { return HTP_STATUS_NO_SUPPORT; } + + size_t row_stride = htp_mm_get_tiled_row_stride(weight_type, k); + if (row_stride == 0) { + return HTP_STATUS_NO_SUPPORT; + } + + worker_callback_t dequant_worker_fn = NULL; + switch (weight_type) { + case HTP_TYPE_Q4_0: dequant_worker_fn = dequantize_tiled_worker_loop_q4_0; break; + case HTP_TYPE_IQ4_NL: dequant_worker_fn = dequantize_tiled_worker_loop_iq4_nl; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: dequant_worker_fn = dequantize_tiled_worker_loop_q4_1; break; case HTP_TYPE_MXFP4: dequant_worker_fn = dequantize_tiled_worker_loop_mxfp4; break; case HTP_TYPE_Q8_0: dequant_worker_fn = dequantize_tiled_worker_loop_q8_0; break; + case HTP_TYPE_Q6_K: dequant_worker_fn = dequantize_tiled_worker_loop_q6_k; break; case HTP_TYPE_F16: dequant_worker_fn = convert_f16_worker_loop; break; case HTP_TYPE_F32: dequant_worker_fn = quantize_f32_worker_loop; break; default: - return -1; + return HTP_STATUS_NO_SUPPORT; } const int n_k_tiles = k / HTP_MM_HMX_TILE_N_COLS; const struct fastdiv_values n_k_tiles_div = init_fastdiv_values(n_k_tiles); const bool is_quant = (weight_type != HTP_TYPE_F16 && weight_type != HTP_TYPE_F32); - const size_t vec_dot_size = k * sizeof(__fp16); const size_t vtcm_budget = ctx->vtcm_size; + const int m_chunk_n_rows = kparams->m_chunk; + const int n_chunk_n_cols = kparams->n_chunk; + const int pipeline = kparams->pipeline; + const int n_threads = octx->n_threads; + const int act_threads = kparams->n_act_threads; + const struct fastdiv_values * act_threads_div = &kparams->div_n_act_threads; + const struct fastdiv_values * k_div = &kparams->div_ne00_padded; + const int tile_size = kparams->tile_size; + const int aligned_tile_size = kparams->aligned_tile_size; + const uint32_t dma_dst_stride = is_quant ? aligned_tile_size : row_stride; - const uint32_t dma_src_stride = is_quant ? tile_size : weight_stride; const uint32_t dma_width_bytes = is_quant ? tile_size : row_stride; - size_t m_chunk_n_rows = m_chunk; - size_t n_chunk_n_cols = n_chunk; - size_t vtcm_used = vtcm_size; - - const size_t qweight_row_stride = is_quant ? (size_t)(n_k_tiles * aligned_tile_size) / 32 : 0; - struct htp_mm_hmx_vtcm_layout L; - htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_2D, weight_type, k, m_chunk_n_rows, n_chunk_n_cols, 1, false, pipeline, act_threads, aligned_tile_size); + htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_2D, weight_type, k, m_chunk_n_rows, n_chunk_n_cols, 1, false, pipeline, act_threads, aligned_tile_size, 0); - vtcm_used = L.total_bytes; - if (vtcm_used > vtcm_budget) { - FARF(ERROR, "hmx-mm-2d-precomputed: VTCM overflow: used %zu budget %zu, m %d k %d n %d mc %zu nc %zu", - vtcm_used, vtcm_budget, m, k, n, m_chunk_n_rows, n_chunk_n_cols); - return -1; + if (L.total_bytes > vtcm_budget) { + FARF(ERROR, "hmx-mm-nx-2d: VTCM overflow: used %zu budget %zu, m %d k %d mc %d nc %d", + L.total_bytes, vtcm_budget, m, k, m_chunk_n_rows, n_chunk_n_cols); + return HTP_STATUS_VTCM_TOO_SMALL; } uint8_t * const base = (uint8_t *) ctx->vtcm_base; @@ -2686,19 +2994,31 @@ static int hmx_mm_2d_f32(struct htp_context *ctx, hmx_init_column_scales(vtcm_scales, Q6_V_vsplat_R(0x3c00)); // scale: 1.0, bias: 0.0 in FP16 - FARF(HIGH, "hmx-mm-2d: m %d k %d n %d wtype %d mc %zu nc %zu vtcm %zu/%zu", - m, k, n, weight_type, m_chunk_n_rows, n_chunk_n_cols, vtcm_used, vtcm_budget); + int m_start = 0; + int m_rows = m; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(octx->dsts[0], sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition((uint32_t) m, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + m_start = (int) range.start; + m_rows = (int) range.count; + } - int n_chunk_cnt = hmx_ceil_div(n, n_chunk_n_cols); + if (m_rows == 0) { + return HTP_STATUS_OK; + } + + FARF(HIGH, "hmx-mm-nx-2d: n_weights %u m %d (%d..%d) k %d wtype %d mc %d nc %d vtcm %zu/%zu", + n_weights, m, m_start, m_start + m_rows, k, weight_type, m_chunk_n_rows, n_chunk_n_cols, L.total_bytes, vtcm_budget); htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); + const size_t mr_end = (size_t)(m_start + m_rows); + if (pipeline) { - // --- Asynchronous Pipelined Loop --- - hmx_matmul_job_t job_slots[2]; // persistent double-buffered job descriptors + hmx_matmul_job_t job_slots[2]; - for (size_t mr = 0; mr < m; mr += m_chunk_n_rows) { - const size_t n_rows = hex_smin(m - mr, m_chunk_n_rows); + for (size_t mr = (size_t) m_start; mr < mr_end; mr += m_chunk_n_rows) { + const size_t n_rows = hex_smin(mr_end - mr, m_chunk_n_rows); void *vtcm_weight_bufs[2] = { vtcm_scratch0, vtcm_scratch1 }; void *vtcm_output_bufs[2] = { vtcm_output, vtcm_scratch2 }; @@ -2719,80 +3039,87 @@ static int hmx_mm_2d_f32(struct htp_context *ctx, }; transfer_activation_chunk_threaded(&act_params); - // Prologue: push A0 and optionally A1 (if n_chunk_cnt > 1) - const size_t n_cols_A0 = hex_smin(n - 0 * n_chunk_n_cols, n_chunk_n_cols); - const uint32_t height_A0 = is_quant ? (n_cols_A0 / 32) * n_k_tiles : n_cols_A0; - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_weight_raw[0], weight), - dma_dst_stride, dma_src_stride, dma_width_bytes, height_A0); - - if (1 < n_chunk_cnt) { - const size_t n_cols_A1 = hex_smin(n - 1 * n_chunk_n_cols, n_chunk_n_cols); - const uint32_t height_A1 = is_quant ? (n_cols_A1 / 32) * n_k_tiles : n_cols_A1; - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_weight_raw[1], weight + n_chunk_n_cols * weight_stride), - dma_dst_stride, dma_src_stride, dma_width_bytes, height_A1); - } - - // Main loop: pop A_i -> dequantize A_i -> push A_{i+2} -> submit C_i -> wait C_{i-1} and store D_{i-1} - for (int i = 0; i < n_chunk_cnt; ++i) { - const size_t nc = i * n_chunk_n_cols; - const size_t nc_p2 = nc + 2 * n_chunk_n_cols; + for (uint32_t p = 0; p < n_weights; p++) { + const struct htp_tensor * restrict src_w = octx->src[p]; + const struct htp_tensor * restrict dst = octx->dsts[p]; + if (!src_w || !dst) continue; + + const dma_addr_t weight = src_w->data; + dma_queue * weight_dma = octx->ctx->dma[0]; + float * dst_ptr = (float *) dst->data; + const size_t n = src_w->ne[1]; + if (n == 0) continue; + const size_t weight_stride = src_w->nb[1]; + const size_t dst_stride = dst->nb[1] / sizeof(float); + const int dst_cols = (int) dst->ne[0]; + const int n_chunk_cnt = hmx_ceil_div(n, n_chunk_n_cols); + + const uint32_t dma_src_stride = is_quant ? tile_size : weight_stride; + + const size_t n_cols_A0 = hex_smin(n - 0 * n_chunk_n_cols, n_chunk_n_cols); + const uint32_t height_A0 = is_quant ? (n_cols_A0 / 32) * n_k_tiles : n_cols_A0; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[0], weight), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_A0); + + if (1 < n_chunk_cnt) { + const size_t n_cols_A1 = hex_smin(n - 1 * n_chunk_n_cols, n_chunk_n_cols); + const uint32_t height_A1 = is_quant ? (n_cols_A1 / 32) * n_k_tiles : n_cols_A1; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[1], weight + n_chunk_n_cols * weight_stride), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_A1); + } - const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); - const size_t n_cols_p2 = hex_smin(n - nc_p2, n_chunk_n_cols); + for (int i = 0; i < n_chunk_cnt; ++i) { + const size_t nc = i * n_chunk_n_cols; + const size_t nc_p2 = nc + 2 * n_chunk_n_cols; - // 1. pop A_i - void * curr_raw = dma_queue_pop(ctx->dma[0]).dst; + const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); + const size_t n_cols_p2 = hex_smin(n - nc_p2, n_chunk_n_cols); - // 2. dequantize A_i - dequantize_tiled_weight_chunk_to_fp16_tiles( - ctx, vtcm_weight_bufs[i % 2], curr_raw, - n_cols, k, row_stride, weight_type, - n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; - // 3. push A_{i+2} (if i+2 < n_chunk_cnt) - if (i + 2 < n_chunk_cnt) { - const uint32_t height_p2 = is_quant ? (n_cols_p2 / 32) * n_k_tiles : n_cols_p2; - dma_queue_push(ctx->dma[0], dma_make_ptr(curr_raw, weight + nc_p2 * weight_stride), - dma_dst_stride, dma_src_stride, dma_width_bytes, height_p2); - } + dequantize_tiled_weight_chunk_to_fp16_tiles( + ctx, vtcm_weight_bufs[i % 2], curr_raw, + n_cols, k, row_stride, weight_type, + n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); - // 4. submit C_i - hmx_matmul_job_init(&job_slots[i % 2], (__fp16 *) vtcm_output_bufs[i % 2], - (__fp16 *) vtcm_f16_act, (__fp16 *) vtcm_weight_bufs[i % 2], - vtcm_scales, hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS), - hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS), k / HTP_MM_HMX_TILE_N_ROWS); - hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job_slots[i % 2])); + if (i + 2 < n_chunk_cnt) { + const uint32_t height_p2 = is_quant ? (n_cols_p2 / 32) * n_k_tiles : n_cols_p2; + dma_queue_push(weight_dma, dma_make_data(curr_raw, weight + nc_p2 * weight_stride), + dma_dst_stride, dma_src_stride, dma_width_bytes, height_p2); + } - // 5. wait C_{i-1} and store D_{i-1} (multi-thread HVX, parallel with C_i) - if (i > 0) { - hmx_queue_pop(ctx->hmx_queue); - const size_t nc_prev = (i - 1) * n_chunk_n_cols; - const size_t n_cols_prev = hex_smin(n - nc_prev, n_chunk_n_cols); - float *output_chunk = dst + (mr * dst_stride + nc_prev); - const float *src2_chunk = src2 ? (src2 + mr * src2_stride + nc_prev) : NULL; - int chunk_dst_cols = dst_cols - (int)nc_prev; - if (chunk_dst_cols > 0) { - transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output_bufs[(i - 1) % 2], n_rows, n_cols_prev, dst_stride, src2_stride, chunk_dst_cols, n_threads); + hmx_matmul_job_init(&job_slots[i % 2], (__fp16 *) vtcm_output_bufs[i % 2], + (__fp16 *) vtcm_f16_act, (__fp16 *) vtcm_weight_bufs[i % 2], + vtcm_scales, hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS), + hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS), k / HTP_MM_HMX_TILE_N_ROWS); + hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job_slots[i % 2])); + + if (i > 0) { + hmx_queue_pop(ctx->hmx_queue); + const size_t nc_prev = (i - 1) * n_chunk_n_cols; + const size_t n_cols_prev = hex_smin(n - nc_prev, n_chunk_n_cols); + float *output_chunk = dst_ptr + (mr * dst_stride + nc_prev); + int chunk_dst_cols = dst_cols - (int)nc_prev; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, NULL, vtcm_output_bufs[(i - 1) % 2], n_rows, n_cols_prev, dst_stride, 0, chunk_dst_cols, n_threads); + } } } - } - // Epilogue: wait C_{last} and store D_{last} - hmx_queue_pop(ctx->hmx_queue); - const size_t nc_last = (n_chunk_cnt - 1) * n_chunk_n_cols; - const size_t n_cols_last = hex_smin(n - nc_last, n_chunk_n_cols); - float *output_chunk = dst + (mr * dst_stride + nc_last); - const float *src2_chunk = src2 ? (src2 + mr * src2_stride + nc_last) : NULL; - int chunk_dst_cols = dst_cols - (int)nc_last; - if (chunk_dst_cols > 0) { - transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output_bufs[(n_chunk_cnt - 1) % 2], n_rows, n_cols_last, dst_stride, src2_stride, chunk_dst_cols, n_threads); + hmx_queue_pop(ctx->hmx_queue); + const size_t nc_last = (n_chunk_cnt - 1) * n_chunk_n_cols; + const size_t n_cols_last = hex_smin(n - nc_last, n_chunk_n_cols); + float *output_chunk = dst_ptr + (mr * dst_stride + nc_last); + int chunk_dst_cols = dst_cols - (int)nc_last; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, NULL, vtcm_output_bufs[(n_chunk_cnt - 1) % 2], n_rows, n_cols_last, dst_stride, 0, chunk_dst_cols, n_threads); + } } } } else { - // --- Synchronous loop (m <= 32 or fallback) --- hmx_matmul_job_t job; - for (size_t mr = 0; mr < m; mr += m_chunk_n_rows) { - const size_t n_rows = hex_smin(m - mr, m_chunk_n_rows); + for (size_t mr = (size_t) m_start; mr < mr_end; mr += m_chunk_n_rows) { + const size_t n_rows = hex_smin(mr_end - mr, m_chunk_n_rows); struct activation_transfer_params act_params = { .ctx = ctx, @@ -2810,69 +3137,69 @@ static int hmx_mm_2d_f32(struct htp_context *ctx, }; transfer_activation_chunk_threaded(&act_params); - // A0: Pre-fetch the first weight chunk (nc = 0) - if (n > 0) { - const size_t n_cols = hex_smin(n, n_chunk_n_cols); - const uint32_t height = is_quant ? (n_cols / 32) * n_k_tiles : n_cols; - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_weight_raw[0], weight), dma_dst_stride, dma_src_stride, dma_width_bytes, height); - } + for (uint32_t p = 0; p < n_weights; p++) { + const struct htp_tensor * restrict src_w = octx->src[p]; + const struct htp_tensor * restrict dst = octx->dsts[p]; + if (!src_w || !dst) continue; + + const dma_addr_t weight = src_w->data; + dma_queue * weight_dma = octx->ctx->dma[0]; + float * dst_ptr = (float *) dst->data; + const size_t n = src_w->ne[1]; + if (n == 0) continue; + const size_t weight_stride = src_w->nb[1]; + const size_t dst_stride = dst->nb[1] / sizeof(float); + const int dst_cols = (int) dst->ne[0]; + + const uint32_t dma_src_stride = is_quant ? tile_size : weight_stride; + + if (n > 0) { + const size_t n_cols = hex_smin(n, n_chunk_n_cols); + const uint32_t height = is_quant ? (n_cols / 32) * n_k_tiles : n_cols; + dma_queue_push(weight_dma, dma_make_data(vtcm_weight_raw[0], weight), dma_dst_stride, dma_src_stride, dma_width_bytes, height); + } - for (size_t nc = 0; nc < n; nc += n_chunk_n_cols) { - const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); - const size_t n_row_tiles = hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS); - const size_t n_col_tiles = hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS); + for (size_t nc = 0; nc < n; nc += n_chunk_n_cols) { + const size_t n_cols = hex_smin(n - nc, n_chunk_n_cols); + const size_t n_row_tiles = hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS); + const size_t n_col_tiles = hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS); - // A: Wait for weight DMA - void * curr_raw = dma_queue_pop(ctx->dma[0]).dst; + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; - // B: Weight Dequantize (Threaded) - dequantize_tiled_weight_chunk_to_fp16_tiles( - ctx, vtcm_scratch0, curr_raw, - n_cols, k, row_stride, weight_type, - n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); + dequantize_tiled_weight_chunk_to_fp16_tiles( + ctx, vtcm_scratch0, curr_raw, + n_cols, k, row_stride, weight_type, + n_k_tiles, n_k_tiles_div, dequant_worker_fn, n_threads); - // Start weight DMA for the next chunk early - const size_t nc_next = nc + n_chunk_n_cols; - if (nc_next < n) { - const size_t n_cols_next = hex_smin(n - nc_next, n_chunk_n_cols); - const uint32_t height_next = is_quant ? (n_cols_next / 32) * n_k_tiles : n_cols_next; - dma_queue_push(ctx->dma[0], dma_make_ptr(curr_raw, weight + nc_next * weight_stride), dma_dst_stride, dma_src_stride, dma_width_bytes, height_next); - } + const size_t nc_next = nc + n_chunk_n_cols; + if (nc_next < n) { + const size_t n_cols_next = hex_smin(n - nc_next, n_chunk_n_cols); + const uint32_t height_next = is_quant ? (n_cols_next / 32) * n_k_tiles : n_cols_next; + dma_queue_push(weight_dma, dma_make_data(curr_raw, weight + nc_next * weight_stride), dma_dst_stride, dma_src_stride, dma_width_bytes, height_next); + } - // C: HMX Compute (Queue-based) - hmx_matmul_job_init(&job, vtcm_output, vtcm_f16_act, vtcm_scratch0, vtcm_scales, n_row_tiles, n_col_tiles, k / HTP_MM_HMX_TILE_N_ROWS); - hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job)); - hmx_queue_pop(ctx->hmx_queue); + hmx_matmul_job_init(&job, vtcm_output, vtcm_f16_act, vtcm_scratch0, vtcm_scales, n_row_tiles, n_col_tiles, k / HTP_MM_HMX_TILE_N_ROWS); + hmx_queue_push(ctx->hmx_queue, hmx_queue_make_desc(hmx_matmul_worker_fn, &job)); + hmx_queue_pop(ctx->hmx_queue); - // D: Output Store - float *output_chunk = dst + (mr * dst_stride + nc); - const float *src2_chunk = src2 ? (src2 + mr * src2_stride + nc) : NULL; - int chunk_dst_cols = dst_cols - (int)nc; - if (chunk_dst_cols > 0) { - transfer_output_chunk_threaded(ctx, output_chunk, src2_chunk, vtcm_output, n_rows, n_cols, dst_stride, src2_stride, chunk_dst_cols, n_threads); + float *output_chunk = dst_ptr + (mr * dst_stride + nc); + int chunk_dst_cols = dst_cols - (int)nc; + if (chunk_dst_cols > 0) { + transfer_output_chunk_threaded(ctx, output_chunk, NULL, vtcm_output, n_rows, n_cols, dst_stride, 0, chunk_dst_cols, n_threads); + } } } } } - return 0; -} - -static inline int hmx_mm_batch_r2(const hmx_mm_f16_f32_batched_params_t *params) { - return params->ne02 > 0 ? params->ne12 / params->ne02 : 1; -} - -static inline int hmx_mm_batch_r3(const hmx_mm_f16_f32_batched_params_t *params) { - return params->ne03 > 0 ? params->ne13 / params->ne03 : 1; + return HTP_STATUS_OK; } -static inline const __fp16 *hmx_mm_weight_batch_ptr(const hmx_mm_f16_f32_batched_params_t *params, - int dst_b2, int dst_b3) { - const int r2 = hmx_mm_batch_r2(params); - const int r3 = hmx_mm_batch_r3(params); - return (const __fp16 *) ((const uint8_t *) params->weight + - (size_t) (dst_b2 / r2) * params->src0_nb2 + - (size_t) (dst_b3 / r3) * params->src0_nb3); +static inline dma_addr_t hmx_mm_weight_batch_data(const hmx_mm_f16_f32_batched_params_t *params, + int dst_b2, int dst_b3) { + const size_t b2_idx = (params->r2 <= 1) ? (size_t) dst_b2 : (size_t) fastdiv((uint32_t) dst_b2, ¶ms->div_r2); + const size_t b3_idx = (params->r3 <= 1) ? (size_t) dst_b3 : (size_t) fastdiv((uint32_t) dst_b3, ¶ms->div_r3); + return params->weight + b2_idx * params->src0_nb2 + b3_idx * params->src0_nb3; } static inline const float *hmx_mm_activation_batch_ptr(const hmx_mm_f16_f32_batched_params_t *params, @@ -2889,13 +3216,6 @@ static inline float *hmx_mm_dst_batch_ptr(const hmx_mm_f16_f32_batched_params_t (size_t) dst_b3 * params->dst_nb3); } -static inline const float *hmx_mm_src2_batch_ptr(const hmx_mm_f16_f32_batched_params_t *params, - int src2_b2, int src2_b3) { - return params->src2 ? (const float *) ((const uint8_t *) params->src2 + - (size_t) src2_b2 * params->src2_nb2 + - (size_t) src2_b3 * params->src2_nb3) : NULL; -} - static int hmx_mm_f16_f32_batched_simple(struct htp_context *ctx, const hmx_mm_f16_f32_batched_params_t *params, int m_chunk, int n_chunk, int pipeline, int n_threads, int act_threads, int vtcm_size, @@ -2903,15 +3223,18 @@ static int hmx_mm_f16_f32_batched_simple(struct htp_context *ctx, int ret = 0; for (int b3 = 0; b3 < params->ne13 && ret == 0; ++b3) { for (int b2 = 0; b2 < params->ne12 && ret == 0; ++b2) { - ret = hmx_mm_2d_f32(ctx, hmx_mm_dst_batch_ptr(params, b2, b3), - hmx_mm_src2_batch_ptr(params, b2, b3), - hmx_mm_activation_batch_ptr(params, b2, b3), - (const uint8_t *)hmx_mm_weight_batch_ptr(params, b2, b3), - params->m, params->k, params->n, - params->act_stride, params->weight_stride * (int)sizeof(__fp16), - HTP_TYPE_F16, params->k, params->dst_stride, params->src2_stride, params->n, - m_chunk, n_chunk, pipeline, n_threads, act_threads, - act_threads_div, k_div, 0, 0, vtcm_size); + dma_addr_t cur_src2_addr = params->src2_addr ? (params->src2_addr + + (dma_addr_t) b2 * params->src2_nb2 + + (dma_addr_t) b3 * params->src2_nb3) : 0; + ret = hmx_mm_2d_f32(ctx, params->weight_dma, hmx_mm_dst_batch_ptr(params, b2, b3), + cur_src2_addr, params->src2_bytes, + hmx_mm_activation_batch_ptr(params, b2, b3), + hmx_mm_weight_batch_data(params, b2, b3), + params->m, params->k, params->n, + params->act_stride, params->weight_stride * (int)sizeof(__fp16), + HTP_TYPE_F16, params->k, params->dst_stride, params->src2_stride, params->n, + m_chunk, n_chunk, pipeline, n_threads, act_threads, + act_threads_div, k_div, 0, 0, vtcm_size); } } return ret; @@ -2928,7 +3251,7 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ if (params->k % 32 != 0 || params->n % 32 != 0) { return -1; } if (!hex_is_aligned(params->dst, VLEN) || !hex_is_aligned(params->activation, VLEN)) { return -1; } - const int group_size = hmx_mm_batch_r2(params); + const int group_size = params->r2; const size_t vtcm_budget = ctx->vtcm_size; // Check if the precomputed parameters are grouped or simple. @@ -2953,7 +3276,7 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ size_t vtcm_used = vtcm_size; struct htp_mm_hmx_vtcm_layout L; - htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_F16_BATCHED, HTP_TYPE_F16, params->k, m_chunk_n_rows, n_chunk_n_cols, group_size, use_dma_activation, false, act_threads, 0); + htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_F16_BATCHED, HTP_TYPE_F16, params->k, m_chunk_n_rows, n_chunk_n_cols, group_size, use_dma_activation, false, act_threads, 0, params->src2_bytes); if (L.total_bytes > vtcm_budget) { FARF(HIGH, "%s: grouped layout overflowed VTCM, falling back to simple batched loop", __func__); @@ -2970,6 +3293,13 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ __fp16 *vtcm_scales = VTCM_LAYOUT_PTR(__fp16, base, L.off_scales); float *vtcm_f32_act = VTCM_LAYOUT_PTR_OPTIONAL(float, base, L.off_act_f32, use_dma_activation); + const bool has_src2 = (params->src2_bytes > 0 && params->src2_addr != 0); + float *vtcm_src2 = VTCM_LAYOUT_PTR_OPTIONAL(float, base, L.off_src2, has_src2); + if (has_src2) { + dma_queue_push(params->weight_dma, dma_make_data(vtcm_src2, params->src2_addr), hex_align_up(params->src2_bytes, 128), 0, params->src2_bytes, 1); + dma_queue_pop(params->weight_dma); + } + hmx_init_column_scales(vtcm_scales, Q6_V_vsplat_R(0x3c00)); // scale: 1.0, bias: 0.0 in FP16 FARF(HIGH, "%s: grouped path m=%d k=%d n=%d group=%d streams=%d mc=%zu nc=%zu vtcm=%zu/%zu", @@ -2986,7 +3316,8 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ for (int b3 = 0; b3 < params->ne13; ++b3) { for (int b2_base = 0; b2_base < params->ne12; b2_base += group_size) { - const __fp16 *weight_group = hmx_mm_weight_batch_ptr(params, b2_base, b3); + const dma_addr_t weight_group = hmx_mm_weight_batch_data(params, b2_base, b3); + dma_queue * weight_dma = params->weight_dma; for (size_t mr = 0; mr < (size_t) params->m; mr += m_chunk_n_rows) { const size_t n_rows = hex_smin((size_t) params->m - mr, m_chunk_n_rows); @@ -3020,12 +3351,12 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ // Prologue: Push A0 and A1 (if exists) { const size_t n_cols_first = hex_smin((size_t) params->n, n_chunk_n_cols); - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_scratch0, weight_group), + dma_queue_push(weight_dma, dma_make_data(vtcm_scratch0, weight_group), fp16_row_bytes, weight_row_bytes, fp16_row_bytes, n_cols_first); } if (n_chunk_n_cols < (size_t) params->n) { const size_t n_cols_second = hex_smin((size_t) params->n - n_chunk_n_cols, n_chunk_n_cols); - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_scratch1, weight_group + params->weight_stride), + dma_queue_push(weight_dma, dma_make_data(vtcm_scratch1, weight_group + params->weight_stride * sizeof(__fp16)), fp16_row_bytes, weight_row_bytes, fp16_row_bytes, n_cols_second); } @@ -3034,16 +3365,16 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ const size_t n_col_tiles = hmx_ceil_div((int) n_cols, HTP_MM_HMX_TILE_N_COLS); { - void * curr_raw = dma_queue_pop(ctx->dma[0]).dst; + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; hmx_interleave_rows_to_tiles(vtcm_weight, (const __fp16 *) curr_raw, n_cols, params->k, params->k, 0, n_cols); const size_t nc_next = nc + n_chunk_n_cols * 2; if (nc_next < (size_t) params->n) { const size_t n_cols_next = hex_smin((size_t) params->n - nc_next, n_chunk_n_cols); - const __fp16 *next_weight_chunk = weight_group + nc_next * params->weight_stride; + const dma_addr_t next_weight_chunk = weight_group + nc_next * params->weight_stride * sizeof(__fp16); - dma_queue_push(ctx->dma[0], dma_make_ptr(curr_raw, next_weight_chunk), + dma_queue_push(weight_dma, dma_make_data(curr_raw, next_weight_chunk), fp16_row_bytes, weight_row_bytes, fp16_row_bytes, n_cols_next); } } @@ -3059,11 +3390,11 @@ static int hmx_mm_f16_f32_batched(struct htp_context *ctx, const hmx_mm_f16_f32_ { float *output = hmx_mm_dst_batch_ptr(params, b2_base + g, b3) + mr * params->dst_stride + nc; - const float *src2_chunk = params->src2 ? (hmx_mm_src2_batch_ptr(params, b2_base + g, b3) + mr * params->src2_stride + nc) : NULL; + const float *src2_chunk = has_src2 ? (vtcm_src2 + mr * params->src2_stride + nc) : NULL; int chunk_dst_cols = params->n - (int)nc; if (chunk_dst_cols > 0) { transfer_output_chunk_threaded(ctx, output, src2_chunk, vtcm_output, (int) n_rows, (int) n_cols, - params->dst_stride, params->src2_stride, chunk_dst_cols, ctx->n_threads); + params->dst_stride, params->src2_stride, chunk_dst_cols, n_threads); } } } @@ -3172,9 +3503,10 @@ static void transfer_output_chunk_scattered_threaded( } static int hmx_mm_id_2d_f32(struct htp_context *ctx, + dma_queue *weight_dma, float *restrict dst, const float *activation, - const uint8_t *weight, + dma_addr_t weight, int m, int k, int n, int k_valid, int ne11, @@ -3184,7 +3516,10 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, int weight_type, const struct mmid_row_mapping *matrix_rows, int cur_a, - int mapping_stride) { + int mapping_stride, + int m_start, + int m_end, + int n_threads) { struct htp_thread_trace * tr = &ctx->trace[0]; htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); @@ -3203,9 +3538,11 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, switch (weight_type) { case HTP_TYPE_Q4_0: dequant_worker_fn = dequantize_tiled_worker_loop_q4_0; break; case HTP_TYPE_IQ4_NL: dequant_worker_fn = dequantize_tiled_worker_loop_iq4_nl; break; - case HTP_TYPE_Q4_1: dequant_worker_fn = dequantize_tiled_worker_loop_q4_1; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: dequant_worker_fn = dequantize_tiled_worker_loop_q4_1; break; case HTP_TYPE_MXFP4: dequant_worker_fn = dequantize_tiled_worker_loop_mxfp4; break; case HTP_TYPE_Q8_0: dequant_worker_fn = dequantize_tiled_worker_loop_q8_0; break; + case HTP_TYPE_Q6_K: dequant_worker_fn = dequantize_tiled_worker_loop_q6_k; break; case HTP_TYPE_F16: dequant_worker_fn = convert_f16_worker_loop; break; case HTP_TYPE_F32: dequant_worker_fn = quantize_f32_worker_loop; break; default: @@ -3215,7 +3552,6 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, const int n_k_tiles = k / HTP_MM_HMX_TILE_N_COLS; const struct fastdiv_values n_k_tiles_div = init_fastdiv_values(n_k_tiles); - const int n_threads = ctx->n_threads; const bool is_quant = (weight_type != HTP_TYPE_F16 && weight_type != HTP_TYPE_F32); const size_t vec_dot_size = k * sizeof(__fp16); @@ -3236,8 +3572,9 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, htp_mm_hmx_get_2d_chunk_costs(weight_type, k, /*pipeline=*/false, aligned_tile_size, &size_per_n, &size_per_m, &size_per_mn); + const size_t overhead = htp_mm_hmx_get_2d_overhead(/*pipeline=*/false, /*is_matmul_id=*/true); size_t m_chunk_n_rows = 0, n_chunk_n_cols = 0; - if (htp_mm_hmx_compute_chunks(vtcm_budget, /*overhead=*/256, size_per_n, size_per_m, size_per_mn, + if (htp_mm_hmx_compute_chunks(vtcm_budget, overhead, size_per_n, size_per_m, size_per_mn, m_padded, n, /*m_block_cost=*/(size_t) n * HTP_MM_HMX_COST_W_DEQUANT, /*n_block_cost=*/(size_t) m_padded * HTP_MM_HMX_COST_A_CONVERT, &m_chunk_n_rows, &n_chunk_n_cols, &vtcm_used)) { @@ -3270,8 +3607,8 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, hmx_matmul_job_t job; - for (size_t mr = 0; mr < (size_t) m_padded; mr += m_chunk_n_rows) { - const size_t n_rows = hex_smin(m_padded - mr, m_chunk_n_rows); + for (size_t mr = (size_t) m_start; mr < (size_t) m_end; mr += m_chunk_n_rows) { + const size_t n_rows = hex_smin((size_t) m_end - mr, m_chunk_n_rows); const size_t n_row_tiles = hmx_ceil_div(n_rows, HTP_MM_HMX_TILE_N_ROWS); transfer_activation_chunk_gathered_threaded( @@ -3282,7 +3619,7 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, if (n > 0) { const size_t n_cols = hex_smin((size_t) n, n_chunk_n_cols); const uint32_t height = is_quant ? (n_cols / 32) * n_k_tiles : n_cols; - dma_queue_push(ctx->dma[0], dma_make_ptr(vtcm_weight, weight), + dma_queue_push(weight_dma, dma_make_data(vtcm_weight, weight), dma_dst_stride, dma_src_stride, dma_width_bytes, height); } @@ -3291,7 +3628,7 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, const size_t n_col_tiles = hmx_ceil_div(n_cols, HTP_MM_HMX_TILE_N_COLS); // A: Wait for weight DMA - void * curr_raw = dma_queue_pop(ctx->dma[0]).dst; + void * curr_raw = (void *) dma_queue_pop(weight_dma).dst; // B: Weight Dequantize (Threaded) dequantize_tiled_weight_chunk_to_fp16_tiles( @@ -3305,7 +3642,7 @@ static int hmx_mm_id_2d_f32(struct htp_context *ctx, if (nc_next < (size_t) n) { const size_t n_cols_next = hex_smin((size_t) n - nc_next, n_chunk_n_cols); const uint32_t height_next = is_quant ? (n_cols_next / 32) * n_k_tiles : n_cols_next; - dma_queue_push(ctx->dma[0], dma_make_ptr(curr_raw, weight + nc_next * weight_stride), + dma_queue_push(weight_dma, dma_make_data(curr_raw, weight + nc_next * weight_stride), dma_dst_stride, dma_src_stride, dma_width_bytes, height_next); } @@ -3335,31 +3672,52 @@ static int hmx_mm_op_matmul(struct htp_ops_context * octx, const struct htp_mm_k const int act_stride = (int)(src1->nb[1] / sizeof(float)); const int wgt_stride = (int)(src0->nb[1] / sizeof(__fp16)); - const float * src2_ptr = NULL; + int m_start = 0; + int m_rows = m_total; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition((uint32_t) m_total, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + m_start = (int) range.start; + m_rows = (int) range.count; + } + + if (m_rows == 0) { + return HTP_STATUS_OK; + } + + dma_addr_t src2_addr = 0; + size_t src2_bytes = 0; uint32_t src2_stride = 0; size_t src2_nb2 = 0; size_t src2_nb3 = 0; if (src2) { - src2_ptr = (const float *) src2->data; src2_stride = (src2->ne[1] == 1) ? 0 : (uint32_t) (src2->nb[1] / sizeof(float)); + src2_addr = src2->data + (dma_addr_t) m_start * src2_stride * sizeof(float); + src2_bytes = (size_t) kparams->vtcm_src2_size; src2_nb2 = (src2->ne[2] == 1) ? 0 : src2->nb[2]; src2_nb3 = (src2->ne[3] == 1) ? 0 : src2->nb[3]; } + const int dst_stride = (int)(dst->nb[1] / sizeof(float)); + float * dst_ptr = (float *) dst->data + m_start * dst_stride; + const float * act_ptr = (const float *) src1->data + m_start * act_stride; + int ret = -1; - const int n_threads = MIN(kparams->n_threads, (int) octx->n_threads); + const int n_threads = kparams->n_threads; if (kparams->kernel_type == HTP_MM_KERNEL_HMX_F16_BATCHED) { hmx_mm_f16_f32_batched_params_t batch_params = { - .dst = (float *) dst->data, - .src2 = src2_ptr, - .activation = (float *) src1->data, - .weight = (const __fp16 *) src0->data, - .m = m_total, + .dst = dst_ptr, + .src2_addr = src2_addr, + .src2_bytes = src2_bytes, + .activation = act_ptr, + .weight = src0->data, + .weight_dma = octx->ctx->dma[0], + .m = m_rows, .k = k, .n = n, .act_stride = act_stride, .weight_stride = wgt_stride, - .dst_stride = (int) (dst->nb[1] / sizeof(float)), + .dst_stride = dst_stride, .src2_stride = src2_stride, .ne02 = ne02, .ne03 = ne03, @@ -3373,6 +3731,10 @@ static int hmx_mm_op_matmul(struct htp_ops_context * octx, const struct htp_mm_k .dst_nb3 = dst->nb[3], .src2_nb2 = src2_nb2, .src2_nb3 = src2_nb3, + .r2 = (ne02 > 0) ? (ne12 / ne02) : 1, + .r3 = (ne03 > 0) ? (ne13 / ne03) : 1, + .div_r2 = kparams->div_r2, + .div_r3 = kparams->div_r3, }; ret = hmx_mm_f16_f32_batched(octx->ctx, &batch_params, kparams->m_chunk, kparams->n_chunk, @@ -3383,9 +3745,9 @@ static int hmx_mm_op_matmul(struct htp_ops_context * octx, const struct htp_mm_k kparams->vtcm_size); } else { ret = hmx_mm_2d_f32( - octx->ctx, (float*) dst->data, src2_ptr, (float*) src1->data, (const uint8_t *) src0->data, - m_total, k, n, act_stride, (int) src0->nb[1], (int) src0->type, (int) src1->ne[0], - (int)(dst->nb[1] / sizeof(float)), src2_stride, (int)dst->ne[0], + octx->ctx, octx->ctx->dma[0], dst_ptr, src2_addr, src2_bytes, act_ptr, src0->data, + m_rows, k, n, act_stride, (int) src0->nb[1], (int) src0->type, (int) src1->ne[0], + dst_stride, src2_stride, (int)dst->ne[0], kparams->m_chunk, kparams->n_chunk, kparams->pipeline, n_threads, kparams->n_act_threads, &kparams->div_n_act_threads, @@ -3394,62 +3756,225 @@ static int hmx_mm_op_matmul(struct htp_ops_context * octx, const struct htp_mm_k ); } - if (ret != 0) { - FARF(ERROR, "HMX matmul failed (ret=%d)\n", ret); - return HTP_STATUS_INTERNAL_ERR; - } - return HTP_STATUS_OK; -} + if (ret != 0) { + FARF(ERROR, "HMX matmul failed (ret=%d)\n", ret); + return HTP_STATUS_INTERNAL_ERR; + } + return HTP_STATUS_OK; +} + +int op_matmul(struct htp_ops_context * octx) { + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + + const int status = htp_mm_init_context(octx, kparams); + if (status != HTP_STATUS_OK) { + return status; + } + + if (kparams->n_hmx) { + return hmx_mm_op_matmul(octx, kparams); + } + + return hvx_mm_matmul(octx); +} + +static int hmx_mm_op_matmul_id( + struct htp_ops_context * octx, + struct htp_mm_context * mmctx +) { + const uint32_t * matrix_row_counts = mmctx->matrix_row_counts; + const struct mmid_row_mapping * matrix_rows = mmctx->matrix_rows; + htp_matmul_tensors_preamble; + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const int n_ids = octx->src[2]->ne[0]; + const int n_as = ne02; + + for (uint32_t cur_a = 0; cur_a < n_as; ++cur_a) { + const int32_t cne1 = matrix_row_counts[cur_a]; + if (cne1 == 0) continue; + + const int m_padded = hex_align_up(cne1, 32); + int m_start = 0, m_end = m_padded; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && (uint32_t) cne1 >= octx->ctx->mdev.count; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition((uint32_t) m_padded, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + m_start = (int) range.start; + m_end = (int) (range.start + range.count); + } + if (m_start >= m_end) continue; + + int ret = hmx_mm_id_2d_f32(octx->ctx, octx->ctx->dma[0], (float*) dst->data, (float*) src1->data, + src0->data + cur_a * nb02, + cne1, ne00, ne01, + ne10, + ne11, + nb11, nb12, + nb1, nb2, + (int) src0->nb[1], (int) src0->type, + matrix_rows, cur_a, mmctx->mapping_stride, + m_start, m_end, (int) octx->n_threads); + if (ret != 0) { + FARF(ERROR, "HMX matmul failed for expert %u, error %d\n", cur_a, ret); + return HTP_STATUS_NO_SUPPORT; + } + } + + return HTP_STATUS_OK; +} + +static int hvx_mm_matmul_id( + struct htp_ops_context * octx, + struct htp_mm_context * mmctx, + work_queue_func_t hvx_mmid_task_func +) { + htp_matmul_tensors_preamble; + const uint32_t src0_row_size_padded = mmctx->src0_row_size_padded; + const uint32_t src1_nrows = mmctx->src1_nrows; + + struct htp_thread_trace * tr = &octx->ctx->trace[0]; + htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); + + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + const struct htp_tensor * restrict ids = octx->src[2]; + const size_t src0_row_size = nb01; + + const uint32_t qk = QK_Q8_0_TILED; + const uint32_t nb = (ne10 + qk - 1) / qk; + const uint32_t total_nb = src1_nrows * nb; + + work_queue_func_t quant_task_func; + uint32_t n_quant_tasks = 1; + if (src1_nrows < octx->n_threads) { + n_quant_tasks = MIN(total_nb, octx->n_threads); + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; + for (uint32_t ith = 0; ith < n_quant_tasks; ++ith) { + uint32_t ib_first = (total_nb * ith) / n_quant_tasks; + uint32_t ib_last = (total_nb * (ith + 1)) / n_quant_tasks; + mmctx->quant_ib_first[ith] = ib_first; + mmctx->quant_ib_last[ith] = ib_last; + mmctx->quant_r[ith] = ib_first / nb; + mmctx->quant_c[ith] = ib_first % nb; + } + } else { + n_quant_tasks = MIN(src1_nrows, octx->n_threads); + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; + } + size_t src1_row_size = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + + struct htp_mm_hvx_vtcm_layout L; + htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, ne10, src1_nrows, octx->n_threads, + 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, true, false); + + const size_t vtcm_size = L.total_bytes; + + FARF(HIGH, "matmul-id-%s : src0-spad-size %zu src1-spad-size %zu src2-spad-size 0 dst-spad-size %zu (%zu)\n", mmctx->type, + L.src0_bytes, L.src1_bytes, L.dst_bytes, vtcm_size); + + FARF(HIGH, "matmul-id-%s : %ux%ux%ux%u * %ux%ux%ux%u (%ux%ux%ux%u) -> %ux%ux%ux%u (0x%p, 0x%p, 0x%p)\n", mmctx->type, + src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], + ids->ne[0], ids->ne[1], ids->ne[2], ids->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], src0->data, + src1->data, dst->data); + + // Make sure the reserved vtcm size is sufficient + if (octx->ctx->vtcm_size < vtcm_size) { + FARF(ERROR, "matmul-id-%s : current VTCM reservation %zu is too small, needed %zu\n", mmctx->type, octx->ctx->vtcm_size, vtcm_size); + return HTP_STATUS_VTCM_TOO_SMALL; + } + + uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; + mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); + mmctx->vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src0); + mmctx->vtcm_src2 = NULL; + mmctx->vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, L.off_dst); + + octx->src1_spad.src = NULL; + octx->src0_spad.src = NULL; + octx->src2_spad.src = NULL; + octx->dst_spad.src = NULL; + + mmctx->vtcm_src0_stride = src0_row_size_padded; + mmctx->vtcm_src1_stride = src1_row_size; -int op_matmul(struct htp_ops_context * octx) { - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + mmctx->vtcm_src0_size_per_thread = fastdiv(L.src0_bytes, &octx->n_threads_div); + mmctx->vtcm_src1_size_per_thread = L.src1_bytes; + mmctx->vtcm_src2_size_per_thread = 0; + mmctx->vtcm_dst_size_per_thread = fastdiv(L.dst_bytes, &octx->n_threads_div); - if (kparams->n_hmx) { - return hmx_mm_op_matmul(octx, kparams); - } + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; - return hvx_mm_matmul(octx); + mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; + mmctx->quant_task_func = quant_task_func; + mmctx->n_quant_tasks = n_quant_tasks; + atomic_init(&mmctx->quant_barrier, n_quant_tasks); + + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); + + worker_pool_run_func(octx->ctx->worker_pool, hvx_mmid_task_func, mmctx, octx->n_threads); + + return HTP_STATUS_OK; } -static int hmx_mm_op_matmul_id( +static int hmx_mm_op_matmul_id_nx( struct htp_ops_context * octx, struct htp_mm_context * mmctx ) { const uint32_t * matrix_row_counts = mmctx->matrix_row_counts; const struct mmid_row_mapping * matrix_rows = mmctx->matrix_rows; - htp_matmul_tensors_preamble; const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - const int n_ids = octx->src[2]->ne[0]; - const int n_as = ne02; + const uint32_t n_weights = kparams->n_weights; + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const int n_as = src0->ne[2]; - for (uint32_t cur_a = 0; cur_a < n_as; ++cur_a) { + for (uint32_t cur_a = 0; cur_a < (uint32_t) n_as; ++cur_a) { const int32_t cne1 = matrix_row_counts[cur_a]; if (cne1 == 0) continue; - int ret = hmx_mm_id_2d_f32(octx->ctx, (float*) dst->data, (float*) src1->data, - (const uint8_t *) src0->data + cur_a * nb02, - cne1, ne00, ne01, - ne10, - ne11, - nb11, nb12, - nb1, nb2, - (int) src0->nb[1], (int) src0->type, - matrix_rows, cur_a, mmctx->mapping_stride); - if (ret != 0) { - FARF(ERROR, "HMX matmul failed for expert %u, error %d\n", cur_a, ret); - return HTP_STATUS_NO_SUPPORT; + const int m_padded = hex_align_up(cne1, 32); + int m_start = 0, m_end = m_padded; + if (octx->ctx->mdev.count > 1) { + bool can_split = (uint32_t) cne1 >= octx->ctx->mdev.count; + for (uint32_t p = 0; p < n_weights && can_split; ++p) { + const struct htp_tensor * restrict dst = octx->dsts[p]; + can_split = !dst || htp_tensor_mdev_data_aligned(dst); + } + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition((uint32_t) m_padded, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + m_start = (int) range.start; + m_end = (int) (range.start + range.count); + } + if (m_start >= m_end) continue; + + for (uint32_t p = 0; p < n_weights; ++p) { + const struct htp_tensor * restrict src_w = octx->src[p]; + const struct htp_tensor * restrict dst = octx->dsts[p]; + if (!src_w || !dst) continue; + + int ret = hmx_mm_id_2d_f32(octx->ctx, octx->ctx->dma[0], (float*) dst->data, (float*) act->data, + src_w->data + cur_a * src_w->nb[2], + cne1, src_w->ne[0], src_w->ne[1], + act->ne[0], + act->ne[1], + act->nb[1], act->nb[2], + dst->nb[1], dst->nb[2], + (int) src_w->nb[1], (int) src_w->type, + matrix_rows, cur_a, mmctx->mapping_stride, + m_start, m_end, (int) octx->n_threads); + if (ret != 0) { + FARF(ERROR, "HMX matmul ID NX failed for expert %u weight %u, error %d\n", cur_a, p, ret); + return HTP_STATUS_NO_SUPPORT; + } } } return HTP_STATUS_OK; } -static int hvx_mm_matmul_id( +static int hvx_mm_matmul_id_nx( struct htp_ops_context * octx, struct htp_mm_context * mmctx, work_queue_func_t hvx_mmid_task_func ) { - htp_matmul_tensors_preamble; const uint32_t src0_row_size_padded = mmctx->src0_row_size_padded; const uint32_t src1_nrows = mmctx->src1_nrows; @@ -3457,18 +3982,21 @@ static int hvx_mm_matmul_id( htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - const struct htp_tensor * restrict ids = octx->src[2]; - const size_t src0_row_size = nb01; + const uint32_t n_weights = kparams->n_weights; + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const struct htp_tensor * restrict ids = octx->src[n_weights + 1]; + const size_t src0_row_size = src0->nb[1]; const uint32_t qk = QK_Q8_0_TILED; - const uint32_t nb = (ne10 + qk - 1) / qk; + const uint32_t nb = (act->ne[0] + qk - 1) / qk; const uint32_t total_nb = src1_nrows * nb; work_queue_func_t quant_task_func; uint32_t n_quant_tasks = 1; if (src1_nrows < octx->n_threads) { n_quant_tasks = MIN(total_nb, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; for (uint32_t ith = 0; ith < n_quant_tasks; ++ith) { uint32_t ib_first = (total_nb * ith) / n_quant_tasks; uint32_t ib_last = (total_nb * (ith + 1)) / n_quant_tasks; @@ -3479,54 +4007,53 @@ static int hvx_mm_matmul_id( } } else { n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; } - size_t src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + size_t src1_row_size = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(act->ne[0]) : htp_mm_q8_0_tiled_row_size(act->ne[0]); struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, ne10, src1_nrows, octx->n_threads, - 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, true, false, false); - - size_t vtcm_size = kparams->vtcm_size > 0 ? (size_t)kparams->vtcm_size : L.total_bytes; - - FARF(HIGH, "matmul-id-%s : src0-spad-size %zu src1-spad-size %zu src2-spad-size 0 dst-spad-size %zu (%zu)\n", mmctx->type, - L.src0_bytes, L.src1_bytes, L.dst_bytes, vtcm_size); + htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, act->ne[0], src1_nrows, octx->n_threads, + 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, true, false); - FARF(HIGH, "matmul-id-%s : %ux%ux%ux%u * %ux%ux%ux%u (%ux%ux%ux%u) -> %ux%ux%ux%u (0x%p, 0x%p, 0x%p)\n", mmctx->type, - src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], - ids->ne[0], ids->ne[1], ids->ne[2], ids->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], src0->data, - src1->data, dst->data); + const size_t vtcm_size = L.total_bytes; - // Make sure the reserved vtcm size is sufficient if (octx->ctx->vtcm_size < vtcm_size) { - FARF(ERROR, "matmul-id-%s : current VTCM reservation %zu is too small, needed %zu\n", mmctx->type, octx->ctx->vtcm_size, vtcm_size); + FARF(ERROR, "matmul-id-nx: current VTCM reservation %zu is too small, needed %zu\n", + octx->ctx->vtcm_size, vtcm_size); return HTP_STATUS_VTCM_TOO_SMALL; } uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; - mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); mmctx->vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src0); - mmctx->vtcm_src2 = NULL; + mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); mmctx->vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, L.off_dst); - octx->src1_spad.src = NULL; - octx->src0_spad.src = NULL; - octx->src2_spad.src = NULL; - octx->dst_spad.src = NULL; + octx->src0_spad.src = NULL; + octx->src1_spad.src = NULL; + octx->src2_spad.src = NULL; + octx->src3_spad.src = NULL; + octx->dst_spad.src = NULL; - mmctx->vtcm_src0_stride = src0_row_size_padded; + mmctx->vtcm_src0_stride = 0; mmctx->vtcm_src1_stride = src1_row_size; - mmctx->vtcm_src0_size_per_thread = L.src0_bytes / octx->n_threads; + mmctx->vtcm_src0_size_per_thread = fastdiv(L.src0_bytes, &octx->n_threads_div); mmctx->vtcm_src1_size_per_thread = L.src1_bytes; - mmctx->vtcm_src2_size_per_thread = 0; - mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads; + mmctx->vtcm_dst_size_per_thread = fastdiv(L.dst_bytes, &octx->n_threads_div); + + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; - mmctx->quant_task_func = quant_task_func; - mmctx->n_quant_tasks = n_quant_tasks; + mmctx->quant_task_func = quant_task_func; + mmctx->n_quant_tasks = n_quant_tasks; atomic_init(&mmctx->quant_barrier, n_quant_tasks); + FARF(HIGH, "matmul-id-nx: src0 %d:%d:%d type %s nrows %u, src1 %d:%d:%d nrows %u, vtcm %zu/%zu, threads %d\n", + src0->ne[0], src0->ne[1], src0->ne[2], mmctx->type, src0->ne[1], + act->ne[0], act->ne[1], act->ne[2], src1_nrows, + L.total_bytes, octx->ctx->vtcm_size, octx->n_threads); + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); worker_pool_run_func(octx->ctx->worker_pool, hvx_mmid_task_func, mmctx, octx->n_threads); @@ -3604,16 +4131,25 @@ static inline void scan_expert_ids( int op_matmul_id(struct htp_ops_context * octx) { htp_matmul_tensors_preamble; + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + struct htp_mm_context mmctx_struct = {0}; + struct htp_mm_context * mmctx = &mmctx_struct; + + const int status = htp_mm_init_context(octx, kparams); + if (status != HTP_STATUS_OK) { + return status; + } + struct htp_thread_trace * tr = &octx->ctx->trace[0]; htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); - struct htp_mm_context mmctx_struct = {0}; - struct htp_mm_context * mmctx = &mmctx_struct; mmctx->octx = octx; - - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + mmctx->act = src1; const struct htp_tensor * restrict ids = octx->src[2]; + if (htp_tensor_is_extended(ids) || htp_tensor_is_extended(src1) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } const size_t src0_row_size = nb01; const size_t dst_row_size = nb1; @@ -3623,9 +4159,6 @@ int op_matmul_id(struct htp_ops_context * octx) { const uint32_t src0_nrows = ne01; // per expert const uint32_t src1_nrows = ne11 * ne12 * ne13; - mmctx->src0_nrows_per_thread = (src0_nrows + octx->n_threads - 1) / octx->n_threads; - mmctx->src0_nrows_per_thread = hex_round_up(mmctx->src0_nrows_per_thread, 32); - // row groups const int n_ids = ids->ne[0]; // n_expert_used const int n_as = ne02; // n_expert @@ -3667,9 +4200,11 @@ int op_matmul_id(struct htp_ops_context * octx) { mmctx->matrix_row_counts = matrix_row_counts; mmctx->matrix_rows = matrix_rows; mmctx->mapping_stride = mapping_stride; - mmctx->mm_div_ne11 = kparams->div_ne11; + mmctx->mm_div_ne11 = kparams->div_ne1; mmctx->src0_row_size_padded = src0_row_size_padded; mmctx->src1_nrows = src1_nrows; + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); @@ -3677,6 +4212,29 @@ int op_matmul_id(struct htp_ops_context * octx) { if (kparams->n_hmx) { s = hmx_mm_op_matmul_id(octx, mmctx); } else { + uint32_t src0_row_start = 0; + uint32_t src0_row_end = src0_nrows; + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_can_row_partition(dst, sizeof(float)); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, can_split ? 32 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + src0_row_start = range.start; + src0_row_end = range.start + range.count; + } + + if (src0_row_start >= src0_row_end) { + if (mapping_buf != octx->ctx->ddr_spad_base) { + free(mapping_buf); + } + return HTP_STATUS_OK; + } + + const uint32_t nrows = src0_row_end - src0_row_start; + mmctx->src0_row_start = src0_row_start; + mmctx->src0_row_end = src0_row_end; + + mmctx->src0_nrows_per_thread = fastdiv(nrows + octx->n_threads - 1, &octx->n_threads_div); + mmctx->src0_nrows_per_thread = hex_round_up(mmctx->src0_nrows_per_thread, 32); + if (hvx_mm_init_vec_dot(mmctx, src0->type) == 0) { s = hvx_mm_matmul_id(octx, mmctx, src1_nrows > 1 ? hvx_mm_id : hvx_mv_id); } else { @@ -3691,183 +4249,138 @@ int op_matmul_id(struct htp_ops_context * octx) { return s; } -int op_matmul_qkv(struct htp_ops_context * octx) { - struct htp_thread_trace * tr = &octx->ctx->trace[0]; - htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); - - const struct htp_tensor * restrict src0 = octx->src[0]; // Wk - const struct htp_tensor * restrict src1 = octx->src[1]; // x - const struct htp_tensor * restrict src2 = octx->src[2]; // Wv - const struct htp_tensor * restrict src3 = octx->src[3]; // Wq - const struct htp_tensor * restrict dst_k = octx->dsts[0]; - const struct htp_tensor * restrict dst_v = octx->dsts[1]; - const struct htp_tensor * restrict dst_q = octx->dsts[2]; - - bool is_repacked = (src0->type == HTP_TYPE_Q4_0 || src0->type == HTP_TYPE_Q4_1 || - src0->type == HTP_TYPE_Q8_0 || src0->type == HTP_TYPE_IQ4_NL || - src0->type == HTP_TYPE_MXFP4); - +int op_matmul_id_nx(struct htp_ops_context * octx) { + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; struct htp_mm_context mmctx_struct = {0}; struct htp_mm_context * mmctx = &mmctx_struct; - mmctx->octx = octx; - - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; - - // Compute src0_nrows_per_thread - mmctx->src0_nrows_per_thread = (src0_nrows + octx->n_threads - 1) / octx->n_threads; - if (is_repacked) { - mmctx->src0_nrows_per_thread = hex_round_up(mmctx->src0_nrows_per_thread, 32); - } else { - mmctx->src0_nrows_per_thread += (mmctx->src0_nrows_per_thread & 1); // round up to even + const int status = htp_mm_init_context(octx, kparams); + if (status != HTP_STATUS_OK) { + return status; } - const size_t src0_row_size = src0->nb[1]; - const size_t src0_row_size_padded = hex_round_up(src0_row_size, 128); + struct htp_thread_trace * tr = &octx->ctx->trace[0]; + htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); - if (hvx_mm_init_vec_dot(mmctx, src0->type) != 0) { + mmctx->octx = octx; + const uint32_t n_weights = kparams->n_weights; + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict act = octx->src[n_weights]; + const struct htp_tensor * restrict ids = octx->src[n_weights + 1]; + if (htp_tensor_is_extended(ids) || htp_tensor_is_extended(act)) { return HTP_STATUS_NO_SUPPORT; } - - const uint32_t qk = QK_Q8_0_TILED; - const uint32_t nb = (src1->ne[0] + qk - 1) / qk; - const uint32_t total_nb = src1_nrows * nb; - - worker_callback_t quant_task_func; - uint32_t n_quant_tasks = 1; - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_flat : quantize_f32_q8_0_flat; - } else if (src1_nrows < octx->n_threads) { - n_quant_tasks = MIN(total_nb, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; - for (uint32_t ith = 0; ith < n_quant_tasks; ++ith) { - uint32_t ib_first = (total_nb * ith) / n_quant_tasks; - uint32_t ib_last = (total_nb * (ith + 1)) / n_quant_tasks; - mmctx->quant_ib_first[ith] = ib_first; - mmctx->quant_ib_last[ith] = ib_last; - mmctx->quant_r[ith] = ib_first / nb; - mmctx->quant_c[ith] = ib_first % nb; + for (uint32_t p = 0; p < n_weights; p++) { + if (octx->dsts[p] && htp_tensor_is_extended(octx->dsts[p])) { + return HTP_STATUS_NO_SUPPORT; } - } else { - n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; } - size_t src1_row_size; - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(src1->ne[0]) : htp_mm_q8_0_flat_row_size(src1->ne[0]); - } else { - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(src1->ne[0]) : htp_mm_q8_0_tiled_row_size(src1->ne[0]); - } + mmctx->act = act; - struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, src1->ne[0], src1_nrows, octx->n_threads, - 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, false, true, false); + const size_t src0_row_size = src0->nb[1]; + const size_t src0_row_size_padded = hex_round_up(src0_row_size, 128); - size_t vtcm_size = kparams->vtcm_size > 0 ? (size_t)kparams->vtcm_size : L.total_bytes; + const uint32_t src1_nrows = act->ne[1] * act->ne[2] * act->ne[3]; - if (octx->ctx->vtcm_size < vtcm_size) { - FARF(ERROR, "matmul-qkv: current VTCM reservation %zu is too small, needed %zu\n", - octx->ctx->vtcm_size, vtcm_size); - return HTP_STATUS_VTCM_TOO_SMALL; - } + const int n_ids = ids->ne[0]; + const int n_as = src0->ne[2]; - uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; - mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); - mmctx->vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src0); - mmctx->vtcm_src2 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src2); - mmctx->vtcm_src3 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src3); - mmctx->vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, L.off_dst); + uint8_t * mapping_buf = octx->ctx->ddr_spad_base; + uint32_t mapping_stride = 1; + uint32_t * matrix_row_counts = (uint32_t *) mapping_buf; + struct mmid_row_mapping * matrix_rows = NULL; - octx->src1_spad.src = NULL; - octx->src0_spad.src = NULL; - octx->src2_spad.src = NULL; - octx->src3_spad.src = NULL; - octx->dst_spad.src = NULL; + if (src1_nrows > 1) { + const size_t matrix_row_counts_size = n_as * sizeof(uint32_t); + assert(octx->ctx->ddr_spad_size >= matrix_row_counts_size); - mmctx->vtcm_src0_stride = is_repacked ? 0 : src0_row_size_padded; - mmctx->vtcm_src2_stride = is_repacked ? 0 : src0_row_size_padded; - mmctx->vtcm_src3_stride = is_repacked ? 0 : src0_row_size_padded; - mmctx->vtcm_src1_stride = src1_row_size; + hex_l2fetch_block((const void *) ids->data, ids->ne[1] * ids->nb[1]); - mmctx->vtcm_src0_size_per_thread = L.src0_bytes / octx->n_threads; - mmctx->vtcm_src1_size_per_thread = L.src1_bytes; - mmctx->vtcm_src2_size_per_thread = L.src2_bytes / octx->n_threads; - mmctx->vtcm_src3_size_per_thread = L.src3_bytes / octx->n_threads; - mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads; + memset(matrix_row_counts, 0, matrix_row_counts_size); + scan_expert_ids(ids, n_ids, n_as, matrix_row_counts, NULL, 0); - mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; - mmctx->quant_task_func = quant_task_func; - mmctx->n_quant_tasks = n_quant_tasks; - atomic_init(&mmctx->quant_barrier, n_quant_tasks); + uint32_t max_count = hvx_reduce_max_i32((const uint8_t *) matrix_row_counts, n_as); + mapping_stride = max_count > 0 ? max_count : 1; - // Run fused matmul - const uint32_t n_matmul_jobs = octx->n_threads; - worker_callback_t matmul_job_func; - if (is_repacked) { - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_qkv_2d_repacked_q4_0_flat; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_qkv_2d_repacked_q4_1_flat; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_qkv_2d_repacked_q8_0_flat; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_qkv_2d_repacked_iq4nl_flat; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_qkv_2d_repacked_mxfp4_flat; break; - default: return HTP_STATUS_NO_SUPPORT; - } - } else { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_qkv_2d_repacked_q4_0; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_qkv_2d_repacked_q4_1; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_qkv_2d_repacked_q8_0; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_qkv_2d_repacked_iq4nl; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_qkv_2d_repacked_mxfp4; break; - default: return HTP_STATUS_NO_SUPPORT; + size_t matrix_row_map_size = n_as * mapping_stride * sizeof(struct mmid_row_mapping); + const size_t total_map_size = matrix_row_counts_size + matrix_row_map_size; + + if (total_map_size > octx->ctx->ddr_spad_size) { + mapping_buf = memalign(128, total_map_size); + if (!mapping_buf) { + return HTP_STATUS_INTERNAL_ERR; } } - } else { - matmul_job_func = hvx_mm_qkv_2d; + + matrix_row_counts = (uint32_t *) mapping_buf; + matrix_rows = (struct mmid_row_mapping *) (mapping_buf + matrix_row_counts_size); + + memset(matrix_row_counts, 0, n_as * sizeof(uint32_t)); + scan_expert_ids(ids, n_ids, n_as, matrix_row_counts, matrix_rows, mapping_stride); } + mmctx->matrix_row_counts = matrix_row_counts; + mmctx->matrix_rows = matrix_rows; + mmctx->mapping_stride = mapping_stride; + mmctx->mm_div_ne11 = kparams->div_ne1; + mmctx->src0_row_size_padded = src0_row_size_padded; + mmctx->src1_nrows = src1_nrows; + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; + htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); - worker_pool_run_func(octx->ctx->worker_pool, matmul_job_func, mmctx, n_matmul_jobs); + int s; + if (kparams->n_hmx) { + s = hmx_mm_op_matmul_id_nx(octx, mmctx); + } else { + if (hvx_mm_init_vec_dot(mmctx, src0->type) == 0) { + s = hvx_mm_matmul_id_nx(octx, mmctx, src1_nrows > 1 ? hvx_mm_id_nx : hvx_mv_id_nx); + } else { + s = HTP_STATUS_NO_SUPPORT; + } + } - return HTP_STATUS_OK; + if (mapping_buf != octx->ctx->ddr_spad_base) { + free(mapping_buf); + } + + return s; } +int op_matmul_nx(struct htp_ops_context * octx) { + const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; + + const int status = htp_mm_init_context(octx, kparams); + if (status != HTP_STATUS_OK) { + return status; + } + + if (kparams->n_hmx) { + return hmx_mm_nx_2d_f32(octx, kparams); + } -int op_matmul_ffn(struct htp_ops_context * octx) { struct htp_thread_trace * tr = &octx->ctx->trace[0]; htp_trace_event_start(tr, HTP_TRACE_EVT_INIT, 0); - const struct htp_tensor * restrict src0 = octx->src[0]; // Wgate - const struct htp_tensor * restrict src1 = octx->src[1]; // y - const struct htp_tensor * restrict src2 = octx->src[2]; // Wup - const struct htp_tensor * restrict dst_gate = octx->dsts[0]; - const struct htp_tensor * restrict dst_up = octx->dsts[1]; + const uint32_t n_weights = kparams->n_weights; + + const struct htp_tensor * restrict src0 = octx->src[0]; // first weight + const struct htp_tensor * restrict act = octx->src[n_weights]; // activation x bool is_repacked = (src0->type == HTP_TYPE_Q4_0 || src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q8_0 || src0->type == HTP_TYPE_IQ4_NL || - src0->type == HTP_TYPE_MXFP4); + src0->type == HTP_TYPE_MXFP4 || src0->type == HTP_TYPE_Q4_K); struct htp_mm_context mmctx_struct = {0}; struct htp_mm_context * mmctx = &mmctx_struct; mmctx->octx = octx; + mmctx->act = act; - const struct htp_mm_kernel_params * kparams = (const struct htp_mm_kernel_params *) octx->kernel_params; - - const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t src1_nrows = src1->ne[1] * src1->ne[2] * src1->ne[3]; - - // Compute src0_nrows_per_thread - mmctx->src0_nrows_per_thread = (src0_nrows + octx->n_threads - 1) / octx->n_threads; - if (is_repacked) { - mmctx->src0_nrows_per_thread = hex_round_up(mmctx->src0_nrows_per_thread, 32); - } else { - mmctx->src0_nrows_per_thread += (mmctx->src0_nrows_per_thread & 1); // round up to even - } + const uint32_t src1_nrows = act->ne[1] * act->ne[2] * act->ne[3]; + mmctx->src1_nrows = src1_nrows; + mmctx->cur_m_start = 0; + mmctx->cur_m_rows = src1_nrows; const size_t src0_row_size = src0->nb[1]; const size_t src0_row_size_padded = hex_round_up(src0_row_size, 128); @@ -3877,19 +4390,16 @@ int op_matmul_ffn(struct htp_ops_context * octx) { } const uint32_t qk = QK_Q8_0_TILED; - const uint32_t nb = (src1->ne[0] + qk - 1) / qk; + const uint32_t nb = (act->ne[0] + qk - 1) / qk; const uint32_t total_nb = src1_nrows * nb; worker_callback_t quant_task_func; uint32_t n_quant_tasks = 1; - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_flat : quantize_f32_q8_0_flat; - } else if (src1_nrows < octx->n_threads) { + if (src1_nrows < octx->n_threads) { n_quant_tasks = MIN(total_nb, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled_block : quantize_f32_q8_0_tiled_block; for (uint32_t ith = 0; ith < n_quant_tasks; ++ith) { - uint32_t ib_first = (total_nb * (ith + 0)) / n_quant_tasks; + uint32_t ib_first = (total_nb * ith) / n_quant_tasks; uint32_t ib_last = (total_nb * (ith + 1)) / n_quant_tasks; mmctx->quant_ib_first[ith] = ib_first; mmctx->quant_ib_last[ith] = ib_last; @@ -3898,46 +4408,42 @@ int op_matmul_ffn(struct htp_ops_context * octx) { } } else { n_quant_tasks = MIN(src1_nrows, octx->n_threads); - quant_task_func = (src0->type == HTP_TYPE_Q4_1) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; + quant_task_func = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) ? quantize_f32_q8_1_tiled : quantize_f32_q8_0_tiled; } - size_t src1_row_size; - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(src1->ne[0]) : htp_mm_q8_0_flat_row_size(src1->ne[0]); - } else { - src1_row_size = (src0->type == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(src1->ne[0]) : htp_mm_q8_0_tiled_row_size(src1->ne[0]); - } + const size_t src1_row_size = (src0->type == HTP_TYPE_Q4_1 || src0->type == HTP_TYPE_Q4_K) + ? htp_mm_q8_1_tiled_row_size(act->ne[0]) + : htp_mm_q8_0_tiled_row_size(act->ne[0]); struct htp_mm_hvx_vtcm_layout L; - htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, src1->ne[0], src1_nrows, octx->n_threads, - 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, false, false, true); + htp_mm_hvx_vtcm_layout_build(&L, kparams->kernel_type, src0->type, act->ne[0], src1_nrows, octx->n_threads, + 0, src0_row_size, src1_row_size, 0, kparams->n_prefetch, false, true); - size_t vtcm_size = kparams->vtcm_size > 0 ? (size_t)kparams->vtcm_size : L.total_bytes; + const size_t vtcm_size = L.total_bytes; if (octx->ctx->vtcm_size < vtcm_size) { - FARF(ERROR, "matmul-ffn: current VTCM reservation %zu is too small, needed %zu\n", octx->ctx->vtcm_size, vtcm_size); + FARF(ERROR, "matmul-nx: current VTCM reservation %zu is too small, needed %zu\n", + octx->ctx->vtcm_size, vtcm_size); return HTP_STATUS_VTCM_TOO_SMALL; } uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; - mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); mmctx->vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src0); - mmctx->vtcm_src2 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src2); + mmctx->vtcm_src1 = VTCM_LAYOUT_PTR(uint8_t, base, L.off_src1); mmctx->vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, L.off_dst); - octx->src1_spad.src = NULL; octx->src0_spad.src = NULL; + octx->src1_spad.src = NULL; octx->src2_spad.src = NULL; + octx->src3_spad.src = NULL; octx->dst_spad.src = NULL; mmctx->vtcm_src0_stride = is_repacked ? 0 : src0_row_size_padded; - mmctx->vtcm_src2_stride = is_repacked ? 0 : src0_row_size_padded; mmctx->vtcm_src1_stride = src1_row_size; - mmctx->vtcm_src0_size_per_thread = L.src0_bytes / octx->n_threads; + mmctx->vtcm_src0_size_per_thread = fastdiv(L.src0_bytes, &octx->n_threads_div); mmctx->vtcm_src1_size_per_thread = L.src1_bytes; - mmctx->vtcm_src2_size_per_thread = L.src2_bytes / octx->n_threads; - mmctx->vtcm_dst_size_per_thread = L.dst_bytes / octx->n_threads; + mmctx->vtcm_dst_size_per_thread = fastdiv(L.dst_bytes, &octx->n_threads_div); mmctx->n_quant_rows_per_thread = (src1_nrows + n_quant_tasks - 1) / n_quant_tasks; mmctx->quant_task_func = quant_task_func; @@ -3948,27 +4454,17 @@ int op_matmul_ffn(struct htp_ops_context * octx) { const uint32_t n_matmul_jobs = octx->n_threads; worker_callback_t matmul_job_func; if (is_repacked) { - if (kparams->kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_ffn_2d_repacked_q4_0_flat; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_ffn_2d_repacked_q4_1_flat; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_ffn_2d_repacked_q8_0_flat; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_ffn_2d_repacked_iq4nl_flat; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_ffn_2d_repacked_mxfp4_flat; break; - default: return HTP_STATUS_NO_SUPPORT; - } - } else { - switch (src0->type) { - case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_ffn_2d_repacked_q4_0; break; - case HTP_TYPE_Q4_1: matmul_job_func = hvx_mm_ffn_2d_repacked_q4_1; break; - case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_ffn_2d_repacked_q8_0; break; - case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_ffn_2d_repacked_iq4nl; break; - case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_ffn_2d_repacked_mxfp4; break; - default: return HTP_STATUS_NO_SUPPORT; - } + switch (src0->type) { + case HTP_TYPE_Q4_0: matmul_job_func = hvx_mm_nx_2d_repacked_q4_0; break; + case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: matmul_job_func = hvx_mm_nx_2d_repacked_q4_1; break; + case HTP_TYPE_Q8_0: matmul_job_func = hvx_mm_nx_2d_repacked_q8_0; break; + case HTP_TYPE_IQ4_NL: matmul_job_func = hvx_mm_nx_2d_repacked_iq4nl; break; + case HTP_TYPE_MXFP4: matmul_job_func = hvx_mm_nx_2d_repacked_mxfp4; break; + default: return HTP_STATUS_NO_SUPPORT; } } else { - matmul_job_func = hvx_mm_ffn_2d; + matmul_job_func = hvx_mm_nx_2d; } htp_trace_event_stop(tr, HTP_TRACE_EVT_INIT, 0); diff --git a/ggml/src/ggml-hexagon/htp/matmul-ops.h b/ggml/src/ggml-hexagon/htp/matmul-ops.h index 6c393664..fe9dbb61 100644 --- a/ggml/src/ggml-hexagon/htp/matmul-ops.h +++ b/ggml/src/ggml-hexagon/htp/matmul-ops.h @@ -25,6 +25,11 @@ extern "C" { #define HTP_MM_WEIGHT_TILE_SIZE_Q8_0 1088 #define HTP_MM_WEIGHT_TILE_SIZE_IQ4_NL 576 #define HTP_MM_WEIGHT_TILE_SIZE_MXFP4 544 +// Q6_K native 6-bit tile (32 rows x 32 k), vrmpy-ready: byte 4*row+b of a vector holds k = 4*group+b +// vectors 0..3: low nibbles, vector i holds group 2i (low nibble) and group 2i+1 (high nibble) +// vectors 4..5: high 2 bits, vector m holds groups 4m..4m+3 at bit offsets 0,2,4,6 +// vector 6: fp16 scales per row, d * scales[]: k 0..15 in lanes 0..31, k 16..31 in lanes 32..63 +#define HTP_MM_WEIGHT_TILE_SIZE_Q6_K 896 // --- Weight Repacked Aligned Tile Sizes --- #define HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q4_0 640 @@ -32,6 +37,7 @@ extern "C" { #define HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q8_0 1152 #define HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_IQ4_NL 640 #define HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_MXFP4 640 +#define HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q6_K 896 // --- Activation Tiled Block Sizes (including padding) --- #define HTP_MM_ACT_TILE_SIZE_Q8_0 1152 @@ -56,17 +62,11 @@ enum htp_mm_kernel_type { // HVX floating-point paths HTP_MM_KERNEL_HVX_F16_F16_VTCM, - HTP_MM_KERNEL_HVX_F16_F16_DDR, - HTP_MM_KERNEL_HVX_F16_F32_DDR, - HTP_MM_KERNEL_HVX_F32_F32_VTCM, - HTP_MM_KERNEL_HVX_F32_F32_DDR, - HTP_MM_KERNEL_HVX_F32_F16_DDR, // HVX quantized paths HTP_MM_KERNEL_HVX_QUANT_ROW, // standard row-wise parallel quantization HTP_MM_KERNEL_HVX_QUANT_BLOCK, // parallel block-wise quantization - HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT, // row-wise fallback flat quantization }; // Op-specific struct for precomputed matmul params @@ -88,13 +88,14 @@ struct htp_mm_kernel_params { int32_t vtcm_src2_size; // src2 scratchpad size in VTCM (fused only) int32_t vtcm_src3_size; // src3 scratchpad size in VTCM (fused only) int32_t vtcm_dst_size; // dst scratchpad size in VTCM + int32_t n_weights; // Number of weights for fused NX // Precomputed division values struct fastdiv_values div_ne12_ne1; struct fastdiv_values div_ne1; struct fastdiv_values div_r2; struct fastdiv_values div_r3; - struct fastdiv_values div_ne11; + struct fastdiv_values div_ne12; struct fastdiv_values div_n_act_threads; struct fastdiv_values div_ne00_padded; }; @@ -133,7 +134,8 @@ static inline int htp_mm_hmx_compute_chunks(size_t vtcm_total, size_t best_mn = 0; size_t best_m = 0, best_n = 0; - const size_t n_max = hex_align_down((size_t)n, HTP_MM_HMX_TILE_N_COLS); + const size_t max_nc_budget = (usable / per_n_cost); + const size_t n_max = hex_align_down(hex_smin((size_t)n, max_nc_budget), HTP_MM_HMX_TILE_N_COLS); for (size_t nc = n_max; nc >= HTP_MM_HMX_TILE_N_COLS; nc -= HTP_MM_HMX_TILE_N_COLS) { size_t n_fixed = 0, ncmn = 0, mc_denom = 0; if (hex_mul_overflow(nc, per_n_cost, &n_fixed)) continue; @@ -193,9 +195,12 @@ static inline uint32_t htp_mm_get_weight_tile_size(int weight_type) { case HTP_TYPE_IQ4_NL: return HTP_MM_WEIGHT_TILE_SIZE_Q4_0; case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: return HTP_MM_WEIGHT_TILE_SIZE_Q4_1; case HTP_TYPE_Q8_0: return HTP_MM_WEIGHT_TILE_SIZE_Q8_0; + case HTP_TYPE_Q6_K: + return HTP_MM_WEIGHT_TILE_SIZE_Q6_K; case HTP_TYPE_MXFP4: return HTP_MM_WEIGHT_TILE_SIZE_MXFP4; default: @@ -209,9 +214,12 @@ static inline uint32_t htp_mm_get_weight_aligned_tile_size(int weight_type) { case HTP_TYPE_IQ4_NL: return HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q4_0; case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: return HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q4_1; case HTP_TYPE_Q8_0: return HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q8_0; + case HTP_TYPE_Q6_K: + return HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_Q6_K; case HTP_TYPE_MXFP4: return HTP_MM_WEIGHT_ALIGNED_TILE_SIZE_MXFP4; default: @@ -232,27 +240,15 @@ static inline size_t htp_mm_q8_1_tiled_row_size(uint32_t ne) { return nb_32 * HTP_MM_ACT_TILE_SIZE_Q8_1; } -static inline size_t htp_mm_q8_0_flat_row_size(uint32_t ne) { - const uint32_t quants_size = hex_align_up(ne, 128); - const uint32_t num_scales = (ne + 31) / 32; - const uint32_t scales_size = hex_align_up(num_scales * 2, 128); - return quants_size + scales_size; -} - -static inline size_t htp_mm_q8_1_flat_row_size(uint32_t ne) { - const uint32_t quants_size = hex_align_up(ne, 128); - const uint32_t num_scales = (ne + 31) / 32; - const uint32_t scales_size = hex_align_up(num_scales * 4, 128); - return quants_size + scales_size; -} - static inline size_t htp_mm_get_tiled_row_stride(int weight_type, uint32_t k) { uint32_t nb = (k + QK_Q4_0_TILED - 1) / QK_Q4_0_TILED; switch (weight_type) { case HTP_TYPE_Q4_0: case HTP_TYPE_IQ4_NL: case HTP_TYPE_Q4_1: + case HTP_TYPE_Q4_K: case HTP_TYPE_Q8_0: + case HTP_TYPE_Q6_K: case HTP_TYPE_MXFP4: return (size_t) nb * htp_mm_get_weight_tile_size(weight_type); case HTP_TYPE_F16: @@ -298,6 +294,15 @@ static inline void htp_mm_hmx_get_batched_chunk_costs( *size_per_mn_out = sizeof(uint16_t); } +static inline size_t htp_mm_hmx_get_2d_overhead(bool pipeline, bool is_matmul_id) { + size_t num_regions = pipeline ? 7 : (is_matmul_id ? 4 : 5); + return num_regions * HTP_MM_HMX_TILE_SIZE + 256; +} + +static inline size_t htp_mm_hmx_get_batched_overhead(void) { + return 5 * HTP_MM_HMX_TILE_SIZE + 256; +} + struct htp_mm_hmx_vtcm_layout { // Byte offsets from vtcm_base for each region size_t off_weight[2]; // [1] is only used when pipelined @@ -306,6 +311,7 @@ struct htp_mm_hmx_vtcm_layout { size_t off_dst[2]; // [1] is only used when pipelined size_t off_scratch[2]; // dequantization scratch pads size_t off_scales; // HMX scales (256 bytes) + size_t off_src2; // src2 bias in VTCM // Cached sizes of regions for HMX kernel use size_t weight_area_bytes; @@ -314,6 +320,7 @@ struct htp_mm_hmx_vtcm_layout { size_t output_area_bytes; size_t scratch_bytes[2]; size_t act_head_stride; + size_t src2_bytes; size_t total_bytes; }; @@ -347,7 +354,8 @@ static inline void htp_mm_hmx_vtcm_layout_build( bool use_dma_activation, bool pipeline, uint32_t act_threads, - uint32_t aligned_tile_size + uint32_t aligned_tile_size, + size_t src2_size ) { size_t off = 0; @@ -365,6 +373,7 @@ static inline void htp_mm_hmx_vtcm_layout_build( size_t off_group_a = 0; VTCM_LAYOUT_ALLOC(off_group_a, off_act, activation_area_size); VTCM_LAYOUT_ALLOC(off_group_a, off_scales, HTP_MM_HMX_TILE_SIZE); // Padded to 2K for alignment and future persistent data + VTCM_LAYOUT_ALLOC_OPTIONAL(off_group_a, off_src2, hex_align_up(src2_size, HTP_MM_HMX_TILE_SIZE), src2_size > 0); // Group B: Compute-only buffers (starts at off_group_a) size_t off_group_b = off_group_a; @@ -393,6 +402,7 @@ static inline void htp_mm_hmx_vtcm_layout_build( L->scratch_bytes[0] = scratch_area_size; L->scratch_bytes[1] = scratch_area_size; L->act_head_stride = act_head_stride; + L->src2_bytes = src2_size; off = off_group_a + hex_smax(group_b_size, group_c_size); } else { @@ -416,6 +426,7 @@ static inline void htp_mm_hmx_vtcm_layout_build( size_t off_group_a = 0; VTCM_LAYOUT_ALLOC(off_group_a, off_scales, HTP_MM_HMX_TILE_SIZE); // Padded to 2K for alignment and future persistent data VTCM_LAYOUT_ALLOC(off_group_a, off_act, act_area_size); + VTCM_LAYOUT_ALLOC_OPTIONAL(off_group_a, off_src2, hex_align_up(src2_size, HTP_MM_HMX_TILE_SIZE), src2_size > 0); // Group B: Compute-only buffers (starts at off_group_a) size_t off_group_b = off_group_a; @@ -443,6 +454,7 @@ static inline void htp_mm_hmx_vtcm_layout_build( L->scratch_bytes[0] = scratch0_size; L->scratch_bytes[1] = scratch1_size; L->act_head_stride = 0; + L->src2_bytes = src2_size; off = off_group_a + hex_smax(group_b_size, group_c_size); } @@ -463,9 +475,9 @@ static inline void htp_mm_hvx_vtcm_layout_build( size_t src2_row_size, uint32_t n_prefetch, bool is_matmul_id, - bool is_fused_qkv, - bool is_fused_ffn + bool is_fused_nx ) { + (void)src1_row_size; size_t src0_sz = 0; size_t src1_sz = 0; size_t src2_sz = src2_row_size > 0 ? htp_mm_round_up(src2_row_size, 128) : 0; @@ -474,51 +486,37 @@ static inline void htp_mm_hvx_vtcm_layout_build( const bool is_repack = (wtype == HTP_TYPE_Q4_0 || wtype == HTP_TYPE_Q4_1 || wtype == HTP_TYPE_Q8_0 || wtype == HTP_TYPE_IQ4_NL || - wtype == HTP_TYPE_MXFP4); + wtype == HTP_TYPE_MXFP4 || wtype == HTP_TYPE_Q6_K || + wtype == HTP_TYPE_Q4_K); - if (is_fused_qkv || is_fused_ffn) { + if (is_fused_nx) { const size_t src0_row_size_padded = hex_round_up(src0_row_size, 128); const size_t quant_scratch_size = hex_round_up(ne10 * sizeof(float), QK_Q8_0_TILED * sizeof(float)) * n_threads; - size_t src0_sz_per_thread = 0; - size_t src2_sz_per_thread = 0; - size_t src3_sz_per_thread = 0; + size_t weight_sz_per_thread = 0; if (is_repack) { uint32_t aligned_tile_size = htp_mm_get_weight_aligned_tile_size(wtype); uint32_t n_k_tiles = hex_round_up(ne10, 32) / 32; uint32_t tile_row_size = n_k_tiles * aligned_tile_size; - src0_sz_per_thread = hex_round_up(n_prefetch * tile_row_size, 128); - src2_sz_per_thread = hex_round_up(n_prefetch * tile_row_size, 128); - if (is_fused_qkv) { - src3_sz_per_thread = hex_round_up(n_prefetch * tile_row_size, 128); - } + weight_sz_per_thread = hex_round_up(n_prefetch * tile_row_size, 128); } else { - src0_sz_per_thread = hex_round_up(n_prefetch * src0_row_size_padded, 128); - src2_sz_per_thread = hex_round_up(n_prefetch * src0_row_size_padded, 128); - if (is_fused_qkv) { - src3_sz_per_thread = hex_round_up(n_prefetch * src0_row_size_padded, 128); - } + weight_sz_per_thread = hex_round_up(n_prefetch * src0_row_size_padded, 128); } - size_t flat_src1_row_size = (wtype == HTP_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); - size_t tiled_src1_row_size = (wtype == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + size_t tiled_act_row_size = (wtype == HTP_TYPE_Q4_1 || wtype == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + size_t act_sz = hex_round_up(tiled_act_row_size * src1_nrows, 128); - if (kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT) { - src1_sz = hex_round_up(flat_src1_row_size * src1_nrows, 128); - } else { - src1_sz = hex_round_up(tiled_src1_row_size * src1_nrows, 128); - } - - src0_sz = src0_sz_per_thread * n_threads; - src2_sz = src2_sz_per_thread * n_threads; - src3_sz = src3_sz_per_thread * n_threads; + src0_sz = weight_sz_per_thread * n_threads; // shared single-weight prefetch buffer + src1_sz = act_sz; // quantized activation buffer + src2_sz = 0; + src3_sz = 0; dst_sz = quant_scratch_size; } else if (is_matmul_id) { const size_t src0_row_size_padded = htp_mm_round_up(src0_row_size, 128); - const size_t src1_row_size_tiled = (wtype == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) - : htp_mm_q8_0_tiled_row_size(ne10); + const size_t src1_row_size_tiled = (wtype == HTP_TYPE_Q4_1 || wtype == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) + : htp_mm_q8_0_tiled_row_size(ne10); size_t src0_sz_per_thread = htp_mm_round_up(n_prefetch * src0_row_size_padded, 256); src1_sz = htp_mm_round_up(src1_row_size_tiled * src1_nrows, 256); @@ -533,6 +531,8 @@ static inline void htp_mm_hvx_vtcm_layout_build( src0_sz = src0_sz_per_thread * n_threads; dst_sz = htp_mm_round_up(ne10 * sizeof(float), QK_Q8_0_TILED * sizeof(float)) * n_threads; + src2_sz = 0; + src3_sz = 0; } else { const size_t src0_row_size_padded = htp_mm_round_up(src0_row_size, 128); const size_t dst_nrows = (src1_nrows > 1) ? 0 : 1; @@ -545,15 +545,6 @@ static inline void htp_mm_hvx_vtcm_layout_build( dst_sz = dst_nrows > 0 ? htp_mm_round_up(dst_row_size, 128) * n_threads : 0; break; } - case HTP_MM_KERNEL_HVX_F16_F32_DDR: - case HTP_MM_KERNEL_HVX_F16_F16_DDR: - case HTP_MM_KERNEL_HVX_F32_F32_DDR: - case HTP_MM_KERNEL_HVX_F32_F16_DDR: { - src0_sz = htp_mm_round_up(n_prefetch * src0_row_size, 256) * n_threads; - src1_sz = htp_mm_round_up(n_prefetch * src1_row_size, 256) * n_threads; - dst_sz = dst_nrows > 0 ? htp_mm_round_up(dst_row_size, 128) * n_threads : 0; - break; - } case HTP_MM_KERNEL_HVX_F32_F32_VTCM: { size_t f32_src1_row_size = htp_mm_round_up(ne10 * 4, 128); src1_sz = htp_mm_round_up(f32_src1_row_size * src1_nrows, 256); @@ -563,7 +554,7 @@ static inline void htp_mm_hvx_vtcm_layout_build( } case HTP_MM_KERNEL_HVX_QUANT_BLOCK: case HTP_MM_KERNEL_HVX_QUANT_ROW: { - size_t q_src1_row_size = (wtype == HTP_TYPE_Q4_1) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); + size_t q_src1_row_size = (wtype == HTP_TYPE_Q4_1 || wtype == HTP_TYPE_Q4_K) ? htp_mm_q8_1_tiled_row_size(ne10) : htp_mm_q8_0_tiled_row_size(ne10); src0_sz = htp_mm_round_up(n_prefetch * src0_row_size_padded, 256); src1_sz = htp_mm_round_up(q_src1_row_size * src1_nrows, 256); @@ -579,34 +570,8 @@ static inline void htp_mm_hvx_vtcm_layout_build( } size_t quant_scratch_size_per_thread = htp_mm_round_up(ne10 * sizeof(float), QK_Q8_0_TILED * sizeof(float)); - size_t dst_size_per_thread = dst_nrows > 0 ? htp_mm_round_up(dst_row_size, 128) : 0; - if (dst_size_per_thread < quant_scratch_size_per_thread) { - dst_size_per_thread = quant_scratch_size_per_thread; - } - dst_sz = dst_size_per_thread * n_threads; - break; - } - case HTP_MM_KERNEL_HVX_QUANT_ROW_FLAT: { - size_t q_src1_row_size = (wtype == HTP_TYPE_Q4_1) ? htp_mm_q8_1_flat_row_size(ne10) : htp_mm_q8_0_flat_row_size(ne10); - - src0_sz = htp_mm_round_up(n_prefetch * src0_row_size_padded, 256); - src1_sz = htp_mm_round_up(q_src1_row_size * src1_nrows, 256); - - src0_sz = src0_sz * n_threads; - - if (is_repack) { - uint32_t aligned_tile_size = htp_mm_get_weight_aligned_tile_size(wtype); - uint32_t n_k_tiles = ne10 / 32; - uint32_t tile_row_size = n_k_tiles * aligned_tile_size; - size_t repacked_vtcm_size = htp_mm_round_up(n_prefetch * tile_row_size, 256); - src0_sz = repacked_vtcm_size * n_threads; - } - - size_t quant_scratch_size_per_thread = htp_mm_round_up(ne10 * sizeof(float), QK_Q8_0_TILED * sizeof(float)); - size_t dst_size_per_thread = dst_nrows > 0 ? htp_mm_round_up(dst_row_size, 128) : 0; - if (dst_size_per_thread < quant_scratch_size_per_thread) { - dst_size_per_thread = quant_scratch_size_per_thread; - } + size_t dst_slice_per_thread = (dst_nrows > 0 && src1_nrows == 1) ? htp_mm_round_up((dst_row_size + n_threads - 1) / n_threads, 128) : 0; + size_t dst_size_per_thread = (dst_slice_per_thread > quant_scratch_size_per_thread) ? dst_slice_per_thread : quant_scratch_size_per_thread; dst_sz = dst_size_per_thread * n_threads; break; } @@ -616,8 +581,8 @@ static inline void htp_mm_hvx_vtcm_layout_build( } size_t off = 0; - VTCM_LAYOUT_ALLOC(off, off_src1, src1_sz); VTCM_LAYOUT_ALLOC(off, off_src0, src0_sz); + VTCM_LAYOUT_ALLOC(off, off_src1, src1_sz); VTCM_LAYOUT_ALLOC(off, off_src2, src2_sz); VTCM_LAYOUT_ALLOC(off, off_src3, src3_sz); VTCM_LAYOUT_ALLOC(off, off_dst, dst_sz); @@ -630,19 +595,99 @@ static inline void htp_mm_hvx_vtcm_layout_build( L->total_bytes = off; } +static inline bool htp_mm_hvx_solve_vtcm_params( + int kernel_type, + int wtype, + uint32_t ne10, + uint32_t src1_nrows, + uint32_t n_threads, + size_t dst_row_size, + size_t src0_row_size, + size_t src1_row_size, + size_t src2_row_size, + uint32_t n_prefetch, + size_t vtcm_budget, + struct htp_mm_hvx_vtcm_layout * L_out, + uint32_t * m_chunk_out +) { + struct htp_mm_hvx_vtcm_layout L; + htp_mm_hvx_vtcm_layout_build( + &L, kernel_type, wtype, ne10, src1_nrows, n_threads, + dst_row_size, src0_row_size, src1_row_size, src2_row_size, n_prefetch, false, false + ); + + if (L.total_bytes <= vtcm_budget) { + *L_out = L; + *m_chunk_out = src1_nrows; + return true; + } + + const size_t fixed_bytes = L.src0_bytes + L.src2_bytes + L.dst_bytes; + if (vtcm_budget <= fixed_bytes) { + return false; + } + + const size_t avail_act = vtcm_budget - fixed_bytes; + size_t row_size = 0; + if (kernel_type == HTP_MM_KERNEL_HVX_QUANT_ROW || kernel_type == HTP_MM_KERNEL_HVX_QUANT_BLOCK) { + row_size = (wtype == HTP_TYPE_Q4_1 || wtype == HTP_TYPE_Q4_K) + ? htp_mm_q8_1_tiled_row_size(ne10) + : htp_mm_q8_0_tiled_row_size(ne10); + } else if (kernel_type == HTP_MM_KERNEL_HVX_F16_F16_VTCM) { + row_size = hex_round_up(ne10 * 2, 128); + } else { + row_size = hex_round_up(ne10 * 4, 128); + } + if (row_size == 0) { + return false; + } + + uint32_t m_chunk = (uint32_t) (avail_act / row_size); + if (m_chunk > 1) { + m_chunk &= ~1U; + } + if (m_chunk > src1_nrows) { + m_chunk = src1_nrows; + } + if (m_chunk < 1) { + return false; + } + + htp_mm_hvx_vtcm_layout_build( + &L, kernel_type, wtype, ne10, m_chunk, n_threads, + dst_row_size, src0_row_size, src1_row_size, src2_row_size, n_prefetch, false, false + ); + + while (m_chunk > 2 && L.total_bytes > vtcm_budget) { + m_chunk -= 2; + htp_mm_hvx_vtcm_layout_build( + &L, kernel_type, wtype, ne10, m_chunk, n_threads, + dst_row_size, src0_row_size, src1_row_size, src2_row_size, n_prefetch, false, false + ); + } + + if (L.total_bytes <= vtcm_budget) { + *L_out = L; + *m_chunk_out = m_chunk; + return true; + } + + return false; +} + static inline size_t htp_mm_hmx_get_2d_vtcm_size( - int wtype, uint32_t k, size_t mc, size_t nc, bool pipeline, uint32_t act_threads, uint32_t aligned_tile_size + int wtype, uint32_t k, size_t mc, size_t nc, bool pipeline, uint32_t act_threads, uint32_t aligned_tile_size, size_t src2_size ) { struct htp_mm_hmx_vtcm_layout L; - htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_2D, wtype, k, mc, nc, 1, false, pipeline, act_threads, aligned_tile_size); + htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_2D, wtype, k, mc, nc, 1, false, pipeline, act_threads, aligned_tile_size, src2_size); return L.total_bytes; } static inline size_t htp_mm_hmx_get_batched_vtcm_size( - int wtype, uint32_t k, size_t mc, size_t nc, uint32_t group_size, bool use_dma_activation, bool pipeline, uint32_t act_threads) { + int wtype, uint32_t k, size_t mc, size_t nc, uint32_t group_size, bool use_dma_activation, bool pipeline, uint32_t act_threads, size_t src2_size) { (void)pipeline; struct htp_mm_hmx_vtcm_layout L; - htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_F16_BATCHED, wtype, k, mc, nc, group_size, use_dma_activation, false, act_threads, 0); + htp_mm_hmx_vtcm_layout_build(&L, HTP_MM_KERNEL_HMX_F16_BATCHED, wtype, k, mc, nc, group_size, use_dma_activation, false, act_threads, 0, src2_size); return L.total_bytes; } @@ -655,6 +700,7 @@ static inline bool htp_mm_hmx_solve_batched_params( bool use_dma_activation, int n_threads, bool pipeline, + size_t src2_size, size_t vtcm_budget, size_t * m_chunk_out, size_t * n_chunk_out, @@ -669,7 +715,7 @@ static inline bool htp_mm_hmx_solve_batched_params( int act_threads = n_threads; while (act_threads >= 1) { - size_t group_overhead = 256; + size_t group_overhead = htp_mm_hmx_get_batched_overhead() + (src2_size > 0 ? hex_align_up(src2_size, HTP_MM_HMX_TILE_SIZE) : 0); size_t group_size_per_n, group_size_per_m, group_size_per_mn; htp_mm_hmx_get_batched_chunk_costs(k, group_size, &group_size_per_n, &group_size_per_m, &group_size_per_mn); @@ -680,7 +726,7 @@ static inline bool htp_mm_hmx_solve_batched_params( if (htp_mm_hmx_compute_chunks(vtcm_budget, group_overhead, group_size_per_n, group_size_per_m, group_size_per_mn, hex_align_up(ne11, 32), ne01_padded, (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) ne11 * HTP_MM_HMX_COST_A_CONVERT, &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) { - size_t exact_size = htp_mm_hmx_get_batched_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, group_size, use_dma_activation, pipeline, act_threads); + size_t exact_size = htp_mm_hmx_get_batched_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, group_size, use_dma_activation, pipeline, act_threads, src2_size); if (exact_size <= vtcm_budget) { size_t mblocks = ((size_t) ne11 + m_chunk_candidate - 1) / m_chunk_candidate; if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) { @@ -720,6 +766,7 @@ static inline bool htp_mm_hmx_solve_2d_params( bool pipeline, bool is_matmul_id, uint32_t aligned_tile_size, + size_t src2_size, size_t vtcm_budget, size_t * m_chunk_out, size_t * n_chunk_out, @@ -736,7 +783,7 @@ static inline bool htp_mm_hmx_solve_2d_params( int act_threads = n_threads; while (act_threads >= 1) { - size_t simple_2d_overhead = 256; + size_t simple_2d_overhead = htp_mm_hmx_get_2d_overhead(pipeline, is_matmul_id) + (src2_size > 0 ? hex_align_up(src2_size, HTP_MM_HMX_TILE_SIZE) : 0); size_t simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn; htp_mm_hmx_get_2d_chunk_costs(wtype, k, pipeline, aligned_tile_size, &simple_2d_size_per_n, &simple_2d_size_per_m, &simple_2d_size_per_mn); @@ -747,7 +794,7 @@ static inline bool htp_mm_hmx_solve_2d_params( if (htp_mm_hmx_compute_chunks(vtcm_budget, simple_2d_overhead, simple_2d_size_per_n, simple_2d_size_per_m, simple_2d_size_per_mn, m_for_chunks, ne01_padded, (size_t) ne01_padded * HTP_MM_HMX_COST_W_DEQUANT, (size_t) m_for_cost * HTP_MM_HMX_COST_A_CONVERT, &m_chunk_candidate, &n_chunk_candidate, &vtcm_size_candidate) == 0) { - size_t exact_size = htp_mm_hmx_get_2d_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, pipeline, is_matmul_id ? 0 : act_threads, aligned_tile_size); + size_t exact_size = htp_mm_hmx_get_2d_vtcm_size(wtype, k, m_chunk_candidate, n_chunk_candidate, pipeline, is_matmul_id ? 0 : act_threads, aligned_tile_size, src2_size); if (exact_size <= vtcm_budget) { size_t mblocks = ((size_t) m_for_cost + m_chunk_candidate - 1) / m_chunk_candidate; if (mblocks < best_mblocks || (mblocks == best_mblocks && act_threads > best_act_threads)) { diff --git a/ggml/src/ggml-hexagon/htp/pad-ops.c b/ggml/src/ggml-hexagon/htp/pad-ops.c index aaa72b31..85f25a8e 100644 --- a/ggml/src/ggml-hexagon/htp/pad-ops.c +++ b/ggml/src/ggml-hexagon/htp/pad-ops.c @@ -7,13 +7,16 @@ #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" /* Circular wrap: maps any integer x into [0, n) */ static inline uint32_t wrap_around(int32_t x, uint32_t n) { @@ -48,6 +51,15 @@ static inline const uint8_t * pad_src_row_ptr(const struct htp_tensor * src, + (i3 - (uint32_t)lp3) * src->nb[3]; } +static inline dma_addr_t pad_src_row_data(const struct htp_tensor * src, + uint32_t i1, uint32_t i2, uint32_t i3, + int32_t lp1, int32_t lp2, int32_t lp3) { + return src->data + + (i1 - (uint32_t)lp1) * src->nb[1] + + (i2 - (uint32_t)lp2) * src->nb[2] + + (i3 - (uint32_t)lp3) * src->nb[3]; +} + /* Compute the DDR src row pointer for a circular row (wrap-around indexing) */ static inline const uint8_t * pad_circ_src_row_ptr(const struct htp_tensor * src, uint32_t i1, uint32_t i2, uint32_t i3, @@ -58,6 +70,15 @@ static inline const uint8_t * pad_circ_src_row_ptr(const struct htp_tensor * src + wrap_around((int32_t)i3 - lp3, src->ne[3]) * src->nb[3]; } +static inline dma_addr_t pad_circ_src_row_data(const struct htp_tensor * src, + uint32_t i1, uint32_t i2, uint32_t i3, + int32_t lp1, int32_t lp2, int32_t lp3) { + return src->data + + wrap_around((int32_t)i1 - lp1, src->ne[1]) * src->nb[1] + + wrap_around((int32_t)i2 - lp2, src->ne[2]) * src->nb[2] + + wrap_around((int32_t)i3 - lp3, src->ne[3]) * src->nb[3]; +} + struct htp_pad_context { struct htp_ops_context * octx; @@ -68,6 +89,7 @@ struct htp_pad_context { uint32_t nrows_per_thread; uint32_t total_dst_rows; + uint32_t row_start; size_t type_size; @@ -78,43 +100,43 @@ struct htp_pad_context { size_t dst_row_size_aligned; }; -#define htp_pad_preamble \ - const struct htp_tensor * src = octx->src[0]; \ - const struct htp_tensor * dst = octx->dst; \ - \ - const uint32_t ne00 = src->ne[0]; \ - const uint32_t nb00 = src->nb[0]; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; \ - \ - const int32_t lp0 = pctx->lp0, rp0 = pctx->rp0; \ - const int32_t lp1 = pctx->lp1, rp1 = pctx->rp1; \ - const int32_t lp2 = pctx->lp2, rp2 = pctx->rp2; \ - const int32_t lp3 = pctx->lp3, rp3 = pctx->rp3; \ - \ - const size_t type_size = pctx->type_size; \ - \ - const uint32_t row_start = pctx->nrows_per_thread * ith; \ - const uint32_t row_end = MIN(row_start + pctx->nrows_per_thread, pctx->total_dst_rows); - - -#define htp_pad_dma_preamble \ - const size_t src_row_size = pctx->src_row_size; \ - const size_t src_row_size_aligned = pctx->src_row_size_aligned; \ - const size_t dst_row_size = pctx->dst_row_size; \ - const size_t dst_row_size_aligned = pctx->dst_row_size_aligned; \ - \ +#define htp_pad_preamble \ + const struct htp_tensor * src = octx->src[0]; \ + const struct htp_tensor * dst = octx->dst; \ + \ + const uint32_t ne00 = src->ne[0]; \ + const uint32_t nb00 = src->nb[0]; \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + const uint32_t ne2 = dst->ne[2]; \ + const uint32_t ne3 = dst->ne[3]; \ + \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ + const uint32_t nb3 = dst->nb[3]; \ + \ + const int32_t lp0 = pctx->lp0, rp0 = pctx->rp0; \ + const int32_t lp1 = pctx->lp1, rp1 = pctx->rp1; \ + const int32_t lp2 = pctx->lp2, rp2 = pctx->rp2; \ + const int32_t lp3 = pctx->lp3, rp3 = pctx->rp3; \ + \ + const size_t type_size = pctx->type_size; \ + \ + const uint32_t row_start = pctx->row_start + pctx->nrows_per_thread * ith; \ + const uint32_t row_end = MIN(row_start + pctx->nrows_per_thread, pctx->row_start + pctx->total_dst_rows); + + +#define htp_pad_dma_preamble \ + const size_t src_row_size = pctx->src_row_size; \ + const size_t src_row_size_aligned = pctx->src_row_size_aligned; \ + const size_t dst_row_size = pctx->dst_row_size; \ + const size_t dst_row_size_aligned = pctx->dst_row_size_aligned; \ + \ uint8_t * src_spad_base = octx->src0_spad.data + ith * octx->src0_spad.size_per_thread; \ uint8_t * dst_spad_base = octx->dst_spad.data + ith * octx->dst_spad.size_per_thread; \ \ - dma_queue * dma = octx->ctx->dma[ith]; + dma_queue * dma_q = octx->ctx->dma[ith]; // --------------------------------------------------------------------------- // HVX vectorized PAD kernel @@ -125,8 +147,8 @@ static void pad_job_per_thread_hvx(unsigned int nth, unsigned int ith, void * da struct htp_ops_context * octx = pctx->octx; htp_pad_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, row_start); for (uint32_t dst_row = row_start; dst_row < row_end; dst_row++) { uint32_t i1, i2, i3; @@ -165,18 +187,17 @@ static void pad_job_per_thread_hvx(unsigned int nth, unsigned int ith, void * da } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, row_start); - FARF(HIGH, "pad-hvx %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u usec %u\n", + FARF(HIGH, "pad-hvx %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u\n", ith, nth, src->ne[0], src->ne[1], src->ne[2], src->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - row_start, row_end, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + row_start, row_end); } // --------------------------------------------------------------------------- -// HVX + DMA PAD kernel — aligned, double-buffered +// HVX + DMA PAD kernel - aligned, double-buffered // --------------------------------------------------------------------------- static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void * data) { @@ -185,9 +206,6 @@ static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void htp_pad_preamble; htp_pad_dma_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - // ----------------------------------------------------------------------- // Priming phase: push 2 pairs of (dummy_dst_DMA, src_DMA) to seed the // double-buffer pipeline before the main loop begins. @@ -196,9 +214,9 @@ static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void uint8_t * src_spad_cur = src_spad_base + spad_idx * src_row_size_aligned; uint8_t * dst_spad_cur = dst_spad_base + spad_idx * dst_row_size_aligned; - dma_queue_push_vtcm_to_ddr(dma, - dma_make_ptr((uint8_t *)dst->data, dst_spad_cur), - dst_row_size, dst_row_size_aligned, 0); + dma_queue_push(dma_q, + dma_make_data(dst->data, dst_spad_cur), + dst_row_size, dst_row_size_aligned, dst_row_size, 0); uint32_t i1, i2, i3; pad_decompose_row(ir, ne1, ne2, &i1, &i2, &i3); @@ -207,35 +225,37 @@ static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void lp2, rp2, ne2, lp3, rp3, ne3); - const uint8_t * src_ptr = interior - ? pad_src_row_ptr(src, i1, i2, i3, lp1, lp2, lp3) : NULL; + const dma_addr_t src_data = interior + ? pad_src_row_data(src, i1, i2, i3, lp1, lp2, lp3) : src->data; // Interior row: real DMA (1 row) from DDR to VTCM. // Border row: null DMA (nrows=0) - dma_queue_push_ddr_to_vtcm(dma, - dma_make_ptr(src_spad_cur, - src_ptr ? src_ptr : (const uint8_t *)src_spad_cur), - src_row_size_aligned, src_row_size, src_ptr ? 1 : 0); + dma_queue_push(dma_q, + dma_make_data(src_spad_cur, src_data), + src_row_size_aligned, src_row_size, src_row_size, interior ? 1 : 0); } // ----------------------------------------------------------------------- // Main loop: pop completed DMAs, compute in VTCM with aligned HVX ops, // push dst DMA and prefetch src for the next+1 row. // ----------------------------------------------------------------------- + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t ir = row_start; ir < row_end; ir++) { - uint8_t * dst_spad_cur = (uint8_t *) dma_queue_pop(dma).src; - uint8_t * src_spad_cur = (uint8_t *) dma_queue_pop(dma).dst; + uint8_t * dst_spad_cur = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * src_spad_cur = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i1, i2, i3; pad_decompose_row(ir, ne1, ne2, &i1, &i2, &i3); - uint8_t * dst_ptr = (uint8_t *) dst->data + i1 * nb1 + i2 * nb2 + i3 * nb3; + const dma_addr_t dst_data = dst->data + i1 * nb1 + i2 * nb2 + i3 * nb3; const int interior = pad_is_interior(i1, i2, i3, lp1, rp1, ne1, lp2, rp2, ne2, lp3, rp3, ne3); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); if (!interior) { hvx_splat_f32_a(dst_spad_cur, 0.0f, ne0); } else { @@ -249,10 +269,11 @@ static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void hvx_copy_f32_ua(dst_interior, src_spad_cur, ne00); } } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - dma_queue_push_vtcm_to_ddr(dma, - dma_make_ptr(dst_ptr, dst_spad_cur), - dst_row_size, dst_row_size_aligned, 1); + dma_queue_push(dma_q, + dma_make_data(dst_data, dst_spad_cur), + dst_row_size, dst_row_size_aligned, dst_row_size, 1); const uint32_t next_row = ir + 2; if (next_row < row_end) { @@ -262,26 +283,22 @@ static void pad_job_per_thread_hvx_dma(unsigned int nth, unsigned int ith, void lp1, rp1, ne1, lp2, rp2, ne2, lp3, rp3, ne3); - const uint8_t * next_src_ptr = next_interior - ? pad_src_row_ptr(src, ni1, ni2, ni3, lp1, lp2, lp3) : NULL; + const dma_addr_t next_src_data = next_interior + ? pad_src_row_data(src, ni1, ni2, ni3, lp1, lp2, lp3) : src->data; - dma_queue_push_ddr_to_vtcm(dma, - dma_make_ptr(src_spad_cur, - next_src_ptr ? next_src_ptr : (const uint8_t *)src_spad_cur), - src_row_size_aligned, src_row_size, next_src_ptr ? 1 : 0); + dma_queue_push(dma_q, + dma_make_data(src_spad_cur, next_src_data), + src_row_size_aligned, src_row_size, src_row_size, next_interior ? 1 : 0); } } - dma_queue_flush(dma); + dma_queue_flush(dma_q); - t2 = HAP_perf_get_qtimer_count(); - - FARF(HIGH, "pad-hvx-dma %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u usec %u\n", + FARF(HIGH, "pad-hvx-dma %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u\n", ith, nth, src->ne[0], src->ne[1], src->ne[2], src->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - row_start, row_end, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + row_start, row_end); } // --------------------------------------------------------------------------- @@ -293,8 +310,8 @@ static void pad_job_per_thread_hvx_circular(unsigned int nth, unsigned int ith, struct htp_ops_context * octx = pctx->octx; htp_pad_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, row_start); for (uint32_t dst_row = row_start; dst_row < row_end; dst_row++) { uint32_t i1, i2, i3; @@ -344,18 +361,17 @@ static void pad_job_per_thread_hvx_circular(unsigned int nth, unsigned int ith, } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, row_start); - FARF(HIGH, "pad-hvx-circ %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u usec %u\n", + FARF(HIGH, "pad-hvx-circ %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u\n", ith, nth, src->ne[0], src->ne[1], src->ne[2], src->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - row_start, row_end, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + row_start, row_end); } // --------------------------------------------------------------------------- -// HVX + DMA circular PAD kernel — aligned, double-buffered +// HVX + DMA circular PAD kernel - aligned, double-buffered // --------------------------------------------------------------------------- static void pad_job_per_thread_hvx_circular_dma(unsigned int nth, unsigned int ith, void * data) { @@ -364,9 +380,6 @@ static void pad_job_per_thread_hvx_circular_dma(unsigned int nth, unsigned int i htp_pad_preamble; htp_pad_dma_preamble; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - // ----------------------------------------------------------------------- // Priming phase: push 2 pairs of (dummy_dst_DMA, src_DMA) to seed the // double-buffer pipeline. Every row is a real src DMA (no null DMAs). @@ -375,30 +388,33 @@ static void pad_job_per_thread_hvx_circular_dma(unsigned int nth, unsigned int i uint8_t * src_spad_cur = src_spad_base + spad_idx * src_row_size_aligned; uint8_t * dst_spad_cur = dst_spad_base + spad_idx * dst_row_size_aligned; - dma_queue_push_vtcm_to_ddr(dma, - dma_make_ptr((uint8_t *)dst->data, dst_spad_cur), - dst_row_size, dst_row_size_aligned, 0); + dma_queue_push(dma_q, + dma_make_data(dst->data, dst_spad_cur), + dst_row_size, dst_row_size_aligned, dst_row_size, 0); uint32_t pi1, pi2, pi3; pad_decompose_row(ir, ne1, ne2, &pi1, &pi2, &pi3); - dma_queue_push_ddr_to_vtcm(dma, - dma_make_ptr(src_spad_cur, pad_circ_src_row_ptr(src, pi1, pi2, pi3, lp1, lp2, lp3)), - src_row_size_aligned, src_row_size, 1); + const dma_addr_t src_data = pad_circ_src_row_data(src, pi1, pi2, pi3, lp1, lp2, lp3); + dma_queue_push(dma_q, + dma_make_data(src_spad_cur, src_data), + src_row_size_aligned, src_row_size, src_row_size, 1); } // ----------------------------------------------------------------------- // Main loop: pop completed DMAs, assemble circular row in VTCM with // aligned HVX ops, push dst DMA and prefetch src for the next+1 row. // ----------------------------------------------------------------------- + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + for (uint32_t ir = row_start; ir < row_end; ir++) { - uint8_t * dst_spad_cur = (uint8_t *) dma_queue_pop(dma).src; - uint8_t * src_spad_cur = (uint8_t *) dma_queue_pop(dma).dst; + uint8_t * dst_spad_cur = (uint8_t *) dma_queue_pop(dma_q).src; + uint8_t * src_spad_cur = (uint8_t *) dma_queue_pop(dma_q).dst; uint32_t i1, i2, i3; pad_decompose_row(ir, ne1, ne2, &i1, &i2, &i3); - uint8_t * dst_ptr = (uint8_t *) dst->data + i1 * nb1 + i2 * nb2 + i3 * nb3; - + const dma_addr_t dst_data = dst->data + i1 * nb1 + i2 * nb2 + i3 * nb3; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); if (lp0 > 0) { uint8_t * dst_left = dst_spad_cur; const uint8_t * src_left = src_spad_cur + (size_t)(ne00 - (uint32_t)lp0) * type_size; @@ -430,32 +446,30 @@ static void pad_job_per_thread_hvx_circular_dma(unsigned int nth, unsigned int i } } } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir); - dma_queue_push_vtcm_to_ddr(dma, - dma_make_ptr(dst_ptr, dst_spad_cur), - dst_row_size, dst_row_size_aligned, 1); + dma_queue_push(dma_q, + dma_make_data(dst_data, dst_spad_cur), + dst_row_size, dst_row_size_aligned, dst_row_size, 1); const uint32_t next_row = ir + 2; if (next_row < row_end) { uint32_t nri1, nri2, nri3; pad_decompose_row(next_row, ne1, ne2, &nri1, &nri2, &nri3); - dma_queue_push_ddr_to_vtcm(dma, - dma_make_ptr(src_spad_cur, - pad_circ_src_row_ptr(src, nri1, nri2, nri3, lp1, lp2, lp3)), - src_row_size_aligned, src_row_size, 1); + const dma_addr_t next_src_data = pad_circ_src_row_data(src, nri1, nri2, nri3, lp1, lp2, lp3); + dma_queue_push(dma_q, + dma_make_data(src_spad_cur, next_src_data), + src_row_size_aligned, src_row_size, src_row_size, 1); } } - dma_queue_flush(dma); + dma_queue_flush(dma_q); - t2 = HAP_perf_get_qtimer_count(); - - FARF(HIGH, "pad-hvx-circ-dma %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u usec %u\n", + FARF(HIGH, "pad-hvx-circ-dma %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u\n", ith, nth, src->ne[0], src->ne[1], src->ne[2], src->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - row_start, row_end, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + row_start, row_end); } int op_pad(struct htp_ops_context * octx) { @@ -471,10 +485,6 @@ int op_pad(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { - return HTP_STATUS_OK; - } - const int32_t lp0 = octx->op_params[0]; const int32_t rp0 = octx->op_params[1]; const int32_t lp1 = octx->op_params[2]; @@ -489,21 +499,39 @@ int op_pad(struct htp_ops_context * octx) { const uint32_t ne00 = src0->ne[0]; const uint32_t total_dst_rows = dst->ne[1] * dst->ne[2] * dst->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, total_dst_rows > 0 ? total_dst_rows : 1); + const size_t dst_row_size = (size_t)ne0 * type_size; + + uint32_t row_start = 0; + uint32_t nrows = total_dst_rows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, type_size, (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_dst_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } + + const uint32_t n_threads = octx->n_threads; const size_t src_row_size = (size_t)ne00 * type_size; - const size_t dst_row_size = (size_t)ne0 * type_size; const size_t src_row_size_aligned = hex_round_up(src_row_size, VLEN); const size_t dst_row_size_aligned = hex_round_up(dst_row_size, VLEN); // Total VTCM needed: 2 buffers (ping+pong) for src and dst, per thread const size_t vtcm_needed = (size_t)n_threads * 2 * (src_row_size_aligned + dst_row_size_aligned); - const int use_dma = (src0->nb[0] == (uint32_t)type_size) && - (ne00 >= 512) && - (octx->ctx->vtcm_base != NULL) && + const int use_dma = (src0->nb[0] == (uint32_t)type_size) && (ne00 >= 512) && (octx->ctx->vtcm_size >= vtcm_needed); + if (!use_dma && (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst))) { + return HTP_STATUS_NO_SUPPORT; + } + if (use_dma) { octx->src0_spad.size_per_thread = 2 * src_row_size_aligned; octx->dst_spad.size_per_thread = 2 * dst_row_size_aligned; @@ -521,8 +549,9 @@ int op_pad(struct htp_ops_context * octx) { .lp1 = lp1, .rp1 = rp1, .lp2 = lp2, .rp2 = rp2, .lp3 = lp3, .rp3 = rp3, - .nrows_per_thread = (total_dst_rows + n_threads - 1) / n_threads, - .total_dst_rows = total_dst_rows, + .nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .total_dst_rows = nrows, + .row_start = row_start, .type_size = type_size, .src_row_size = src_row_size, .src_row_size_aligned = src_row_size_aligned, @@ -537,11 +566,10 @@ int op_pad(struct htp_ops_context * octx) { dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], lp0, rp0, lp1, rp1, lp2, rp2, lp3, rp3); - if (circular && use_dma) { worker_pool_run_func(octx->ctx->worker_pool, pad_job_per_thread_hvx_circular_dma, &pctx, n_threads); } - else if (circular) { worker_pool_run_func(octx->ctx->worker_pool, pad_job_per_thread_hvx_circular, &pctx, n_threads); } - else if (use_dma) { worker_pool_run_func(octx->ctx->worker_pool, pad_job_per_thread_hvx_dma, &pctx, n_threads); } - else { worker_pool_run_func(octx->ctx->worker_pool, pad_job_per_thread_hvx, &pctx, n_threads); } + if (circular && use_dma) { work_queue_run(octx->ctx->work_queue, pad_job_per_thread_hvx_circular_dma, &pctx, n_threads); } + else if (circular) { work_queue_run(octx->ctx->work_queue, pad_job_per_thread_hvx_circular, &pctx, n_threads); } + else if (use_dma) { work_queue_run(octx->ctx->work_queue, pad_job_per_thread_hvx_dma, &pctx, n_threads); } + else { work_queue_run(octx->ctx->work_queue, pad_job_per_thread_hvx, &pctx, n_threads); } return HTP_STATUS_OK; } - diff --git a/ggml/src/ggml-hexagon/htp/repeat-ops.c b/ggml/src/ggml-hexagon/htp/repeat-ops.c index a6f2f0ed..2551be22 100644 --- a/ggml/src/ggml-hexagon/htp/repeat-ops.c +++ b/ggml/src/ggml-hexagon/htp/repeat-ops.c @@ -12,8 +12,10 @@ #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "htp-tensor.h" struct htp_repeat_context { struct htp_ops_context * octx; @@ -25,6 +27,7 @@ struct htp_repeat_context { uint32_t nrows_per_thread; uint32_t total_dst_rows; // ne1 * ne2 * ne3 + uint32_t row_start; size_t type_size; }; @@ -62,11 +65,11 @@ static void repeat_job_per_thread(unsigned int nth, unsigned int ith, void * dat const size_t row_bytes = ne00 * rctx->type_size; - const uint32_t row_start = rctx->nrows_per_thread * ith; - const uint32_t row_end = MIN(row_start + rctx->nrows_per_thread, rctx->total_dst_rows); + const uint32_t row_start = rctx->row_start + rctx->nrows_per_thread * ith; + const uint32_t row_end = MIN(row_start + rctx->nrows_per_thread, rctx->row_start + rctx->total_dst_rows); - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, row_start); for (uint32_t dst_row = row_start; dst_row < row_end; dst_row++) { // Decompose flat dst row index into (i1, i2, i3) @@ -89,12 +92,12 @@ static void repeat_job_per_thread(unsigned int nth, unsigned int ith, void * dat } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, row_start); - FARF(HIGH, "repeat %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u usec %u\n", + FARF(HIGH, "repeat %d/%d: (%ux%ux%ux%u) -> (%ux%ux%ux%u) rows %u:%u\n", ith, nth, src->ne[0], src->ne[1], src->ne[2], src->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - row_start, row_end, (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + row_start, row_end); } int op_repeat(struct htp_ops_context * octx) { @@ -119,21 +122,39 @@ int op_repeat(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - const uint32_t total_dst_rows = dst->ne[1] * dst->ne[2] * dst->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, total_dst_rows); + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + const uint32_t total_dst_rows = dst->ne[1] * dst->ne[2] * dst->ne[3]; + const size_t dst_row_size = dst->ne[0] * type_size; + + uint32_t row_start = 0; + uint32_t nrows = total_dst_rows; - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, type_size, (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_dst_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { return HTP_STATUS_OK; } + const uint32_t n_threads = octx->n_threads; + struct htp_repeat_context rctx = { .octx = octx, .nr0 = dst->ne[0] / src0->ne[0], .nr1 = dst->ne[1] / src0->ne[1], .nr2 = dst->ne[2] / src0->ne[2], .nr3 = dst->ne[3] / src0->ne[3], - .nrows_per_thread = (total_dst_rows + n_threads - 1) / n_threads, - .total_dst_rows = total_dst_rows, + .nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .total_dst_rows = nrows, + .row_start = row_start, .type_size = type_size, }; @@ -142,7 +163,7 @@ int op_repeat(struct htp_ops_context * octx) { dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], rctx.nr0, rctx.nr1, rctx.nr2, rctx.nr3); - worker_pool_run_func(octx->ctx->worker_pool, repeat_job_per_thread, &rctx, n_threads); + work_queue_run(octx->ctx->work_queue, repeat_job_per_thread, &rctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/roll-ops.c b/ggml/src/ggml-hexagon/htp/roll-ops.c new file mode 100644 index 00000000..9c373f56 --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/roll-ops.c @@ -0,0 +1,316 @@ +#pragma clang diagnostic ignored "-Wunused-variable" +#pragma clang diagnostic ignored "-Wunused-function" +#pragma clang diagnostic ignored "-Wunused-but-set-variable" + +#include +#include + +#include + +#include "dma-queue.h" +#include "hvx-utils.h" + +#define GGML_COMMON_DECL_C +#include "ggml-common.h" +#include "htp-ctx.h" +#include "hex-common.h" +#include "hex-profile.h" +#include "htp-ops.h" +#include "htp-tensor.h" + +struct htp_roll_context { + struct htp_ops_context * octx; + + uint32_t row_start; + uint32_t nrows; + uint32_t nrows_per_thread; + + struct fastdiv_values div_ne1; + struct fastdiv_values div_ne2_ne1; +}; + +static inline uint32_t htp_roll_wrap(int32_t i, uint32_t ne) { + if (i < 0) { + return (uint32_t) (i + (int32_t) ne); + } + if ((uint32_t) i >= ne) { + return (uint32_t) i - ne; + } + return (uint32_t) i; +} + +#define htp_roll_preamble \ + const struct htp_tensor * src0 = octx->src[0]; \ + const struct htp_tensor * dst = octx->dst; \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + const uint32_t ne2 = dst->ne[2]; \ + const uint32_t ne3 = dst->ne[3]; \ + \ + const uint32_t nb01 = src0->nb[1]; \ + const uint32_t nb02 = src0->nb[2]; \ + const uint32_t nb03 = src0->nb[3]; \ + \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ + const uint32_t nb3 = dst->nb[3]; \ + \ + const int32_t s0 = octx->op_params[0]; \ + const int32_t s1 = octx->op_params[1]; \ + const int32_t s2 = octx->op_params[2]; \ + const int32_t s3 = octx->op_params[3]; \ + \ + const uint32_t i0_src0 = htp_roll_wrap(-s0, ne0); \ + const uint32_t n0 = ne0 - i0_src0; + +#define htp_roll_dma_preamble dma_queue * q = octx->ctx->dma[0]; + +static inline void roll_dma_push(dma_queue * q, + dma_addr_t dst, + dma_addr_t src, + uint32_t dst_stride, + uint32_t src_stride, + uint32_t bytes, + uint32_t nrows) { + if (bytes == 0 || nrows == 0) { + return; + } + + if (!dma_queue_push(q, dma_make_data(dst, src), dst_stride, src_stride, bytes, nrows)) { + dma_queue_flush(q); + dma_queue_push(q, dma_make_data(dst, src), + dst_stride, src_stride, bytes, nrows); + } +} + +static inline void roll_dma_push_rows(dma_queue * q, + const struct htp_tensor * dst, + const struct htp_tensor * src0, + uint32_t dst_row, + uint32_t src_row, + uint32_t nrows, + uint32_t row_size, + uint32_t i0_src0) { + const dma_addr_t dst_base = dst->data + (size_t) dst_row * row_size; + const dma_addr_t src_base = src0->data + (size_t) src_row * row_size; + const uint32_t n0 = src0->ne[0] - i0_src0; + + roll_dma_push(q, dst_base, src_base + (size_t) i0_src0 * sizeof(float), + row_size, row_size, n0 * sizeof(float), nrows); + roll_dma_push(q, dst_base + (size_t) n0 * sizeof(float), src_base, + row_size, row_size, i0_src0 * sizeof(float), nrows); +} + +// Same row-wrap split as roll_dma_push_rows, but addressed with explicit byte strides so it +// also works for a src0 that is row-contiguous only (e.g. a permuted view) rather than fully packed. +static inline void roll_dma_push_range(dma_queue * q, + dma_addr_t dst_row, + dma_addr_t src_row, + uint32_t dst_stride, + uint32_t src_stride, + uint32_t nrows, + uint32_t i0_src0, + uint32_t n0) { + roll_dma_push(q, dst_row, src_row + (size_t) i0_src0 * sizeof(float), + dst_stride, src_stride, n0 * sizeof(float), nrows); + roll_dma_push(q, dst_row + (size_t) n0 * sizeof(float), src_row, + dst_stride, src_stride, i0_src0 * sizeof(float), nrows); +} + +static int roll_dma_f32_contiguous(struct htp_ops_context * octx) { + htp_roll_preamble; + htp_roll_dma_preamble; + + const uint32_t row_size = ne0 * sizeof(float); + + if (s1 == 0 && s2 == 0 && s3 == 0) { + roll_dma_push_rows(q, dst, src0, 0, 0, ne1 * ne2 * ne3, row_size, i0_src0); + dma_queue_flush(q); + return HTP_STATUS_OK; + } + + if (s1 == 0) { + const uint32_t i2_src0 = htp_roll_wrap(-s2, ne2); + for (uint32_t i3 = 0; i3 < ne3; i3++) { + const uint32_t i03 = htp_roll_wrap((int32_t) i3 - s3, ne3); + const uint32_t dst_row0 = i3 * ne2 * ne1; + const uint32_t src_row0 = (i03 * ne2 + i2_src0) * ne1; + const uint32_t n2_first = ne2 - i2_src0; + + roll_dma_push_rows(q, dst, src0, dst_row0, src_row0, n2_first * ne1, + row_size, i0_src0); + roll_dma_push_rows(q, dst, src0, dst_row0 + n2_first * ne1, i03 * ne2 * ne1, + i2_src0 * ne1, row_size, i0_src0); + } + + dma_queue_flush(q); + return HTP_STATUS_OK; + } + + const uint32_t i1_src0 = htp_roll_wrap(-s1, ne1); + const uint32_t n1_first = ne1 - i1_src0; + + for (uint32_t i3 = 0; i3 < ne3; i3++) { + const uint32_t i03 = htp_roll_wrap((int32_t) i3 - s3, ne3); + for (uint32_t i2 = 0; i2 < ne2; i2++) { + const uint32_t i02 = htp_roll_wrap((int32_t) i2 - s2, ne2); + const uint32_t dst_row0 = (i3 * ne2 + i2) * ne1; + const uint32_t src_row0 = (i03 * ne2 + i02) * ne1; + + roll_dma_push_rows(q, dst, src0, dst_row0, src_row0 + i1_src0, + n1_first, row_size, i0_src0); + roll_dma_push_rows(q, dst, src0, dst_row0 + n1_first, src_row0, + i1_src0, row_size, i0_src0); + } + } + + dma_queue_flush(q); + return HTP_STATUS_OK; +} + +// DMA path for a row-contiguous but otherwise arbitrarily strided src0 (e.g. a permuted view). +// Same row-wrap split as above, one DMA push per (i2,i3), addressed via the real nb01/nb02/nb03 +// instead of assuming a packed layout. +static int roll_dma_f32_strided(struct htp_ops_context * octx) { + htp_roll_preamble; + htp_roll_dma_preamble; + + const uint32_t i1_src0 = htp_roll_wrap(-s1, ne1); + const uint32_t n1_first = ne1 - i1_src0; + + for (uint32_t i3 = 0; i3 < ne3; i3++) { + const uint32_t i03 = htp_roll_wrap((int32_t) i3 - s3, ne3); + for (uint32_t i2 = 0; i2 < ne2; i2++) { + const uint32_t i02 = htp_roll_wrap((int32_t) i2 - s2, ne2); + + const dma_addr_t dst_row0 = dst->data + (size_t) i2 * nb2 + (size_t) i3 * nb3; + const dma_addr_t src_row0 = src0->data + (size_t) i02 * nb02 + (size_t) i03 * nb03; + + roll_dma_push_range(q, dst_row0, src_row0 + (size_t) i1_src0 * nb01, + nb1, nb01, n1_first, i0_src0, n0); + roll_dma_push_range(q, dst_row0 + (size_t) n1_first * nb1, src_row0, + nb1, nb01, i1_src0, i0_src0, n0); + } + } + + dma_queue_flush(q); + return HTP_STATUS_OK; +} + +static void roll_thread_f32(unsigned int nth, unsigned int ith, void * data) { + struct htp_roll_context * rctx = (struct htp_roll_context *) data; + struct htp_ops_context * octx = rctx->octx; + + htp_roll_preamble; + + const uint32_t row_start = rctx->row_start + rctx->nrows_per_thread * ith; + const uint32_t row_end = MIN(row_start + rctx->nrows_per_thread, rctx->row_start + rctx->nrows); + if (row_start >= row_end) { + return; + } + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, row_start); + + for (uint32_t row = row_start; row < row_end; row++) { + const uint32_t i3 = fastdiv(row, &rctx->div_ne2_ne1); + const uint32_t rem = row - i3 * ne2 * ne1; + const uint32_t i2 = fastdiv(rem, &rctx->div_ne1); + const uint32_t i1 = rem - i2 * ne1; + + const uint32_t i01 = htp_roll_wrap((int32_t) i1 - s1, ne1); + const uint32_t i02 = htp_roll_wrap((int32_t) i2 - s2, ne2); + const uint32_t i03 = htp_roll_wrap((int32_t) i3 - s3, ne3); + + const uint8_t * src_row = (const uint8_t *) (uintptr_t) src0->data + i01*nb01 + i02*nb02 + i03*nb03; + uint8_t * dst_row = (uint8_t *) (uintptr_t) dst->data + i1*nb1 + i2*nb2 + i3*nb3; + + hex_l2fetch(src_row + i0_src0 * sizeof(float), n0 * sizeof(float), ne0 * sizeof(float), 1); + hvx_copy_uu(dst_row, src_row + i0_src0 * sizeof(float), n0, sizeof(float)); + + if (i0_src0 != 0) { + hex_l2fetch(src_row, i0_src0 * sizeof(float), ne0 * sizeof(float), 1); + hvx_copy_uu(dst_row + n0 * sizeof(float), src_row, i0_src0, sizeof(float)); + } + } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, row_start); + + FARF(HIGH, "roll %d/%d: (%ux%ux%ux%u) rows %u:%u shift=(%d,%d,%d,%d)\n", + ith, nth, ne0, ne1, ne2, ne3, + row_start, row_end, s0, s1, s2, s3); +} + +int execute_op_roll_f32(struct htp_ops_context * octx) { + htp_roll_preamble; + + if (src0->type != HTP_TYPE_F32 || dst->type != HTP_TYPE_F32) { + FARF(ERROR, "roll: unsupported type %u -> %u\n", src0->type, dst->type); + return HTP_STATUS_NO_SUPPORT; + } + + if (src0->nb[0] != sizeof(float) || dst->nb[0] != sizeof(float)) { + FARF(ERROR, "roll: unsupported nb0 %u -> %u\n", src0->nb[0], dst->nb[0]); + return HTP_STATUS_NO_SUPPORT; + } + + if (src0->ne[0] != ne0 || src0->ne[1] != ne1 || + src0->ne[2] != ne2 || src0->ne[3] != ne3) { + FARF(ERROR, "roll: shape mismatch\n"); + return HTP_STATUS_INVAL_PARAMS; + } + + const uint32_t total_rows = ne1 * ne2 * ne3; + const size_t dst_row_size = ne0 * sizeof(float); + + uint32_t row_start = 0; + uint32_t nrows = total_rows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(float), (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } + + if (octx->ctx->mdev.count <= 1) { + if (htp_tensor_is_contiguous(src0, sizeof(float)) && htp_tensor_is_contiguous(dst, sizeof(float))) { + return roll_dma_f32_contiguous(octx); + } + return roll_dma_f32_strided(octx); + } + + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + const uint32_t n_threads = octx->n_threads; + struct htp_roll_context rctx = { + .octx = octx, + .row_start = row_start, + .nrows = nrows, + .nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .div_ne1 = init_fastdiv_values(dst->ne[1]), + .div_ne2_ne1 = init_fastdiv_values(dst->ne[2] * dst->ne[1]), + }; + + work_queue_run(octx->ctx->work_queue, roll_thread_f32, &rctx, n_threads); + + return HTP_STATUS_OK; +} + +int op_roll(struct htp_ops_context * octx) { + switch (octx->src[0]->type) { + case HTP_TYPE_F32: + return execute_op_roll_f32(octx); + + default: + return HTP_STATUS_NO_SUPPORT; + } +} diff --git a/ggml/src/ggml-hexagon/htp/rope-ops.c b/ggml/src/ggml-hexagon/htp/rope-ops.c index 5bc7d74f..f6b4d383 100644 --- a/ggml/src/ggml-hexagon/htp/rope-ops.c +++ b/ggml/src/ggml-hexagon/htp/rope-ops.c @@ -9,7 +9,7 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #include "hex-fastdiv.h" @@ -17,8 +17,8 @@ #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" #include "htp-tensor.h" +#include "rope-ops.h" // Redefined the rope type constants as we can't include ggml.h #define HTP_ROPE_TYPE_NORMAL 0 @@ -27,9 +27,6 @@ #define HTP_ROPE_TYPE_VISION 24 #define HTP_ROPE_TYPE_IMROPE 40 -#define HTP_ROPE_SPAD_NROWS 16 -#define HTP_ROPE_SPAD_BLOCK (HTP_ROPE_SPAD_NROWS/2) - #define htp_rope_preamble \ const uint32_t ne00 = src0->ne[0]; \ const uint32_t ne01 = src0->ne[1]; \ @@ -53,6 +50,7 @@ struct htp_rope_context { int32_t n_dims; + int32_t n_offs; int32_t mode; int32_t n_ctx_orig; int32_t sections[4]; @@ -64,26 +62,31 @@ struct htp_rope_context { float beta_fast; float beta_slow; float theta_scale; + float theta_scale_32; + float theta_powers[32]; float corr_dims[2]; uint32_t src0_nrows_per_thread; - size_t spad_stride; struct htp_ops_context * octx; + uint8_t * vtcm_base; + size_t spad_per_thread; + size_t theta_cache_offset; + size_t src0_row_size; size_t src0_row_stride; size_t dst_row_size; size_t dst_row_stride; size_t src0_row_size_aligned; - size_t dst_row_size_aligned; - size_t theta_cache_offset; uint32_t src0_nrows; + uint32_t row_start; + uint32_t nrows; struct fastdiv_values div_ne2_ne1; struct fastdiv_values div_ne1; - uint64_t t_start; + const float * freq_factors; }; static float rope_yarn_ramp(const float low, const float high, const int i0) { @@ -111,94 +114,80 @@ static inline void rope_yarn_one(float theta, float freq_scale, float * corr_dim mscale_final *= 1.0f + 0.1f * logf(1.0f / freq_scale); } - cache[i0 + 0] = cosf(theta_final) * mscale_final; - cache[i0 + 1] = sinf(theta_final) * mscale_final; + const uint32_t b = i0 / 64; + const uint32_t k = (i0 % 64) / 2; + cache[b * 64 + k] = cosf(theta_final) * mscale_final; + cache[b * 64 + 32 + k] = sinf(theta_final) * mscale_final; +} + +// 32 thetas -> 32 deinterleaved pairs [cos[32] | sin[32]] at cache[i0]. +static inline void rope_cache_hvx_32(float * cache, uint32_t i0, + HVX_Vector v_theta, + const float * freq_factors, + HVX_Vector v_freq_scale, + HVX_Vector v_mscale) { + if (freq_factors) { + HVX_Vector v_ff = hvx_vmemu(freq_factors + i0 / 2); + v_theta = hvx_vec_mul_f32_f32(v_theta, hvx_vec_inverse_f32(v_ff)); + } + + HVX_Vector v_theta_final = hvx_vec_mul_f32_f32(v_theta, v_freq_scale); + HVX_Vector vcos; + HVX_Vector vsin; + hvx_vec_sincos_f32(v_theta_final, &vcos, &vsin); + vcos = hvx_vec_mul_f32_f32(vcos, v_mscale); + vsin = hvx_vec_mul_f32_f32(vsin, v_mscale); + + if (((uintptr_t) (cache + i0)) % 128 == 0) { + hvx_vmem(cache + i0 + 0) = vcos; + hvx_vmem(cache + i0 + 32) = vsin; + } else { + hvx_vec_store_u(cache + i0 + 0, 32 * sizeof(float), vcos); + hvx_vec_store_u(cache + i0 + 32, 32 * sizeof(float), vsin); + } } static __attribute__((noinline)) void rope_cache_init(const float theta_base, const float freq_scale, const float * freq_factors, float * corr_dims, - const uint32_t ne0, + const uint32_t n_cache, const float ext_factor, const float mscale, float * cache, - const float theta_scale) { + const float theta_scale, + const float * theta_powers, + const float theta_scale_32) { // ref: https://github.com/jquesnelle/yarn/blob/master/scaled_rope/LlamaYaRNScaledRotaryEmbedding.py -#if __HVX_ARCH__ >= 79 - const bool is_v79_or_newer = true; -#else - const bool is_v79_or_newer = false; -#endif - - if (is_v79_or_newer && ext_factor == 0.0f) { + if (ext_factor == 0.0f) { // Fast path: fully vectorized // We process 32 pairs (64 elements) per iteration. - const uint32_t n_blocks = ne0 / 64; - - // Initialize theta scale powers: [1.0f, theta_scale, theta_scale^2, ..., theta_scale^31] - float __attribute__((aligned(128))) theta_powers[32]; - theta_powers[0] = 1.0f; - for (int j = 1; j < 32; j++) { - theta_powers[j] = theta_powers[j - 1] * theta_scale; - } - HVX_Vector v_theta_powers = hvx_vmem(theta_powers); + const uint32_t n_blocks = n_cache / 64; + HVX_Vector v_theta_powers = hvx_vmemu(theta_powers); HVX_Vector v_freq_scale = hvx_vec_splat_f32(freq_scale); HVX_Vector v_mscale = hvx_vec_splat_f32(mscale); - // Base theta starts at theta_base float theta_block = theta_base; - // The scale factor for the next block is theta_scale^32 - float theta_scale_32 = 1.0f; - for (int j = 0; j < 32; j++) { - theta_scale_32 *= theta_scale; - } for (uint32_t b = 0; b < n_blocks; b++) { uint32_t i0 = b * 64; HVX_Vector v_theta_base = hvx_vec_splat_f32(theta_block); HVX_Vector v_theta = hvx_vec_mul_f32_f32(v_theta_base, v_theta_powers); - - if (freq_factors) { - // Load 32 elements of freq_factors - HVX_Vector v_ff = hvx_vmemu(freq_factors + i0 / 2); - HVX_Vector v_inv_ff = hvx_vec_inverse_f32(v_ff); - v_theta = hvx_vec_mul_f32_f32(v_theta, v_inv_ff); - } - - HVX_Vector v_theta_final = hvx_vec_mul_f32_f32(v_theta, v_freq_scale); - - HVX_Vector vcos = hvx_vec_cos_f32(v_theta_final); - HVX_Vector vsin = hvx_vec_sin_f32(v_theta_final); - - vcos = hvx_vec_mul_f32_f32(vcos, v_mscale); - vsin = hvx_vec_mul_f32_f32(vsin, v_mscale); - - HVX_VectorPair vstore = Q6_W_vshuff_VVR(vsin, vcos, -4); - - if (((uintptr_t)cache) % 128 == 0) { - hvx_vmem(cache + i0 + 0) = Q6_V_lo_W(vstore); - hvx_vmem(cache + i0 + 32) = Q6_V_hi_W(vstore); - } else { - hvx_vec_store_u(cache + i0 + 0, 32 * sizeof(float), Q6_V_lo_W(vstore)); - hvx_vec_store_u(cache + i0 + 32, 32 * sizeof(float), Q6_V_hi_W(vstore)); - } - + rope_cache_hvx_32(cache, i0, v_theta, freq_factors, v_freq_scale, v_mscale); theta_block *= theta_scale_32; } // Leftovers float theta = theta_block; - for (uint32_t i0 = n_blocks * 64; i0 < ne0; i0 += 2) { + for (uint32_t i0 = n_blocks * 64; i0 < n_cache; i0 += 2) { const float ff = freq_factors ? freq_factors[i0 / 2] : 1.0f; rope_yarn_one(theta / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); theta *= theta_scale; } } else { - // Fallback to original scalar loop float theta = theta_base; - for (uint32_t i0 = 0; i0 < ne0; i0 += 2) { + for (uint32_t i0 = 0; i0 < n_cache; i0 += 2) { const float ff = freq_factors ? freq_factors[i0 / 2] : 1.0f; rope_yarn_one(theta / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); theta *= theta_scale; @@ -206,6 +195,72 @@ static __attribute__((noinline)) void rope_cache_init(const float theta_base, } } +static inline float mrope_pick_theta(float theta_t, float theta_h, float theta_w, float theta_e, + int sector, const int32_t sections[4], int sec_w, int sec_e, + bool is_imrope) { + if (is_imrope) { + if (sector % 3 == 0 && sector < 3 * sections[0]) { return theta_t; } + else if (sector % 3 == 1 && sector < 3 * sections[1]) { return theta_h; } + else if (sector % 3 == 2 && sector < 3 * sections[2]) { return theta_w; } + else { return theta_e; } + } + if (sector < sections[0]) { return theta_t; } + else if (sector < sec_w) { return theta_h; } + else if (sector < sec_e) { return theta_w; } + else { return theta_e; } +} + +// lane j is 1 when (j % 3) == rem +static const float __attribute__((aligned(128))) mrope_mod3_eq0[32] = { + 1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0 +}; +static const float __attribute__((aligned(128))) mrope_mod3_eq1[32] = { + 0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1 +}; +static const float __attribute__((aligned(128))) mrope_mod3_eq2[32] = { + 0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0,1,0,0 +}; + +static const float __attribute__((aligned(128))) mrope_k_ramp[32] = { + 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15, + 16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31 +}; + +static inline HVX_VectorPred mrope_mask_eq1(const float * m) { + return Q6_Q_vcmp_gt_VsfVsf(hvx_vmemu(m), Q6_V_vzero()); +} + +// IMROPE without wrap: theta[k] = pos[k % 3] * scale^k +static inline HVX_Vector mrope_thetas_imrope_mod3(float pos_t, float pos_h, float pos_w, + uint32_t k0, HVX_Vector v_powers, float scale_block) { + const int r = (int) (k0 % 3); + const float * mt = (r == 0) ? mrope_mod3_eq0 : (r == 1) ? mrope_mod3_eq2 : mrope_mod3_eq1; + const float * mh = (r == 0) ? mrope_mod3_eq1 : (r == 1) ? mrope_mod3_eq0 : mrope_mod3_eq2; + + HVX_Vector v = hvx_vec_splat_f32(pos_w); + v = Q6_V_vmux_QVV(mrope_mask_eq1(mh), hvx_vec_splat_f32(pos_h), v); + v = Q6_V_vmux_QVV(mrope_mask_eq1(mt), hvx_vec_splat_f32(pos_t), v); + v = hvx_vec_mul_f32_f32(v, v_powers); + return hvx_vec_mul_f32_f32(v, hvx_vec_splat_f32(scale_block)); +} + +// Contiguous MROPE without wrap: theta[k] = pos[section(k)] * scale^k +static inline HVX_Vector mrope_thetas_contig(float pos_t, float pos_h, float pos_w, float pos_e, + uint32_t k0, int s0, int sec_w, int sec_e, + HVX_Vector v_powers, float scale_block) { + HVX_Vector v_k = hvx_vec_add_f32_f32(hvx_vec_splat_f32((float) k0), hvx_vmemu(mrope_k_ramp)); + HVX_VectorPred lt_s0 = Q6_Q_vcmp_gt_VsfVsf(hvx_vec_splat_f32((float) s0), v_k); + HVX_VectorPred lt_sw = Q6_Q_vcmp_gt_VsfVsf(hvx_vec_splat_f32((float) sec_w), v_k); + HVX_VectorPred lt_se = Q6_Q_vcmp_gt_VsfVsf(hvx_vec_splat_f32((float) sec_e), v_k); + + HVX_Vector v = hvx_vec_splat_f32(pos_e); + v = Q6_V_vmux_QVV(lt_se, hvx_vec_splat_f32(pos_w), v); + v = Q6_V_vmux_QVV(lt_sw, hvx_vec_splat_f32(pos_h), v); + v = Q6_V_vmux_QVV(lt_s0, hvx_vec_splat_f32(pos_t), v); + v = hvx_vec_mul_f32_f32(v, v_powers); + return hvx_vec_mul_f32_f32(v, hvx_vec_splat_f32(scale_block)); +} + // pos_t/h/w/e: the four position ids for this sequence step (t=time, h=height, w=width, e=extra). // sections[4]: number of head dims assigned to each position component. static __attribute__((noinline)) void mrope_cache_init(const float pos_t, @@ -218,23 +273,71 @@ static __attribute__((noinline)) void mrope_cache_init(const float pos_t, const float freq_scale, const float * freq_factors, float * corr_dims, - const uint32_t ne0, + const uint32_t n_cache, const float ext_factor, const float mscale, float * cache, - const float theta_scale) { + const float theta_scale, + const float * theta_powers, + const float theta_scale_32) { const int sect_dims = sections[0] + sections[1] + sections[2] + sections[3]; const int sec_w = sections[0] + sections[1]; const int sec_e = sec_w + sections[2]; + const uint32_t n_pairs = n_cache / 2; + + const bool no_wrap = (sect_dims > 0) && (n_pairs <= (uint32_t) sect_dims); + const bool imrope_mod3 = is_imrope && !indep_sects && no_wrap + && sections[0] > 0 && sections[1] > 0 && sections[2] > 0 + && n_pairs <= (uint32_t) (3 * sections[0]) + && n_pairs <= (uint32_t) (3 * sections[1]) + && n_pairs <= (uint32_t) (3 * sections[2]); + const bool contig = !is_imrope && !indep_sects && no_wrap; + + if (ext_factor == 0.0f && (imrope_mod3 || contig)) { + HVX_Vector v_powers = hvx_vmemu(theta_powers); + HVX_Vector v_freq_scale = hvx_vec_splat_f32(freq_scale); + HVX_Vector v_mscale = hvx_vec_splat_f32(mscale); + float scale_block = 1.0f; + const uint32_t n_blocks = n_cache / 64; + + for (uint32_t b = 0; b < n_blocks; b++) { + const uint32_t i0 = b * 64; + const uint32_t k0 = b * 32; + HVX_Vector v_theta = imrope_mod3 + ? mrope_thetas_imrope_mod3(pos_t, pos_h, pos_w, k0, v_powers, scale_block) + : mrope_thetas_contig(pos_t, pos_h, pos_w, pos_e, k0, sections[0], sec_w, sec_e, + v_powers, scale_block); + rope_cache_hvx_32(cache, i0, v_theta, freq_factors, v_freq_scale, v_mscale); + scale_block *= theta_scale_32; + } + + float theta_k = scale_block; + for (uint32_t k = n_blocks * 32; k < n_pairs; k++) { + const uint32_t i0 = 2 * k; + const float pos = mrope_pick_theta(pos_t, pos_h, pos_w, pos_e, + (int) k, sections, sec_w, sec_e, is_imrope); + const float ff = freq_factors ? freq_factors[k] : 1.0f; + rope_yarn_one(pos * theta_k / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); + theta_k *= theta_scale; + } + return; + } float theta_t = pos_t; float theta_h = pos_h; float theta_w = pos_w; float theta_e = pos_e; - for (uint32_t i0 = 0; i0 < ne0; i0 += 2) { - const float ff = freq_factors ? freq_factors[i0 / 2] : 1.0f; - const int sector = (i0 / 2) % sect_dims; + const bool use_hvx = (ext_factor == 0.0f); + float __attribute__((aligned(128))) thetas[32]; + uint32_t n_thetas = 0; + uint32_t block_i0 = 0; + + HVX_Vector v_freq_scale = hvx_vec_splat_f32(freq_scale); + HVX_Vector v_mscale = hvx_vec_splat_f32(mscale); + + for (uint32_t i0 = 0; i0 < n_cache; i0 += 2) { + const int sector = (i0 / 2) % sect_dims; if (indep_sects) { // Reset theta when crossing into a new section. @@ -244,28 +347,34 @@ static __attribute__((noinline)) void mrope_cache_init(const float pos_t, else if (sector == sec_e) { theta_e = pos_e; } } - float theta; - if (is_imrope) { - // Interleaved: sector mod 3 selects component - if (sector % 3 == 0 && sector < 3 * sections[0]) { theta = theta_t; } - else if (sector % 3 == 1 && sector < 3 * sections[1]) { theta = theta_h; } - else if (sector % 3 == 2 && sector < 3 * sections[2]) { theta = theta_w; } - else { theta = theta_e; } + const float theta = mrope_pick_theta(theta_t, theta_h, theta_w, theta_e, + sector, sections, sec_w, sec_e, is_imrope); + + if (use_hvx) { + if (n_thetas == 0) { + block_i0 = i0; + } + thetas[n_thetas++] = theta; + if (n_thetas == 32) { + rope_cache_hvx_32(cache, block_i0, hvx_vmemu(thetas), freq_factors, v_freq_scale, v_mscale); + n_thetas = 0; + } } else { - // Contiguous sections - if (sector < sections[0]) { theta = theta_t; } - else if (sector < sec_w) { theta = theta_h; } - else if (sector < sec_e) { theta = theta_w; } - else { theta = theta_e; } + const float ff = freq_factors ? freq_factors[i0 / 2] : 1.0f; + rope_yarn_one(theta / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); } - rope_yarn_one(theta / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); - theta_t *= theta_scale; theta_h *= theta_scale; theta_w *= theta_scale; theta_e *= theta_scale; } + + for (uint32_t k = 0; k < n_thetas; k++) { + const uint32_t i0 = block_i0 + 2 * k; + const float ff = freq_factors ? freq_factors[i0 / 2] : 1.0f; + rope_yarn_one(thetas[k] / ff, freq_scale, corr_dims, i0, ext_factor, mscale, cache); + } } #define M_PI 3.1415926535897932384626433 @@ -282,52 +391,54 @@ static void rope_corr_dims(int n_dims, dims[1] = MIN(n_dims - 1, end); } +static inline void hvx_rope_neox_mul(HVX_Vector v0, HVX_Vector v1, HVX_Vector vcos, HVX_Vector vsin, + HVX_Vector * o0, HVX_Vector * o1) { + HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(v0, vcos); + HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(v0, vsin); + HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(v1, vcos); + HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(v1, vsin); + *o0 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s)); + *o1 = Q6_Vsf_equals_Vqf32(Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c)); +} + +// theta_cache full 32-pair blocks are deinterleaved [cos | sin]. static inline void hvx_rope_neox_f32_aa(float * restrict dst, const float * restrict src0, uint32_t ne, const float * restrict theta_cache) { const uint32_t he = ne / 2; const uint32_t nvec = he / 32; const uint32_t nloe = he % 32; - for (uint32_t i = 0; i < nvec; i++) { - HVX_Vector v0 = ((const HVX_Vector *) src0)[i]; - HVX_Vector v1 = hvx_vmemu(src0 + he + i * 32); - - HVX_Vector v2 = ((const HVX_Vector *) theta_cache)[i * 2 + 0]; - HVX_Vector v3 = ((const HVX_Vector *) theta_cache)[i * 2 + 1]; - - HVX_VectorPair vcos_sin = Q6_W_vdeal_VVR(v3, v2, -4); - - HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(v0, Q6_V_lo_W(vcos_sin)); - HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(v0, Q6_V_hi_W(vcos_sin)); - HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(v1, Q6_V_lo_W(vcos_sin)); - HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(v1, Q6_V_hi_W(vcos_sin)); - - HVX_Vector v4 = Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s); - HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c); - - ((HVX_Vector *) dst)[i] = Q6_Vsf_equals_Vqf32(v4); - hvx_vmemu(dst + he + i * 32) = Q6_Vsf_equals_Vqf32(v5); + if (nloe == 0) { + const HVX_Vector * vs = (const HVX_Vector *) src0; + const HVX_Vector * vt = (const HVX_Vector *) theta_cache; + HVX_Vector * vd = (HVX_Vector *) dst; + for (uint32_t i = 0; i < nvec; i++) { + HVX_Vector o0, o1; + hvx_rope_neox_mul(vs[i], vs[nvec + i], vt[i * 2 + 0], vt[i * 2 + 1], &o0, &o1); + vd[i] = o0; + vd[nvec + i] = o1; + } + return; } - if (nloe > 0) { - HVX_Vector v0 = hvx_vmemu(src0 + nvec * 32); - HVX_Vector v1 = hvx_vmemu(src0 + he + nvec * 32); - - HVX_Vector v2 = ((const HVX_Vector *) theta_cache)[nvec * 2 + 0]; - HVX_Vector v3 = ((const HVX_Vector *) theta_cache)[nvec * 2 + 1]; - - HVX_VectorPair vcos_sin = Q6_W_vdeal_VVR(v3, v2, -4); - - HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(v0, Q6_V_lo_W(vcos_sin)); - HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(v0, Q6_V_hi_W(vcos_sin)); - HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(v1, Q6_V_lo_W(vcos_sin)); - HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(v1, Q6_V_hi_W(vcos_sin)); - - HVX_Vector v4 = Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s); - HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c); - - hvx_vec_store_u(dst + nvec * 32, nloe * sizeof(float), Q6_Vsf_equals_Vqf32(v4)); - hvx_vec_store_u(dst + he + nvec * 32, nloe * sizeof(float), Q6_Vsf_equals_Vqf32(v5)); + for (uint32_t i = 0; i < nvec; i++) { + HVX_Vector o0, o1; + hvx_rope_neox_mul(((const HVX_Vector *) src0)[i], + hvx_vmemu(src0 + he + i * 32), + ((const HVX_Vector *) theta_cache)[i * 2 + 0], + ((const HVX_Vector *) theta_cache)[i * 2 + 1], + &o0, &o1); + ((HVX_Vector *) dst)[i] = o0; + hvx_vmemu(dst + he + i * 32) = o1; } + + HVX_Vector v0 = hvx_vmemu(src0 + nvec * 32); + HVX_Vector v1 = hvx_vmemu(src0 + he + nvec * 32); + HVX_Vector vcos = hvx_vmemu(theta_cache + nvec * 64); + HVX_Vector vsin = hvx_vmemu(theta_cache + nvec * 64 + 32); + HVX_Vector o0, o1; + hvx_rope_neox_mul(v0, v1, vcos, vsin, &o0, &o1); + hvx_vec_store_u(dst + nvec * 32, nloe * sizeof(float), o0); + hvx_vec_store_u(dst + he + nvec * 32, nloe * sizeof(float), o1); } static inline void hvx_rope_f32_aa(float * restrict dst, const float * restrict src0, uint32_t ne, const float * restrict theta_cache) { @@ -338,16 +449,15 @@ static inline void hvx_rope_f32_aa(float * restrict dst, const float * restrict HVX_Vector v0 = ((const HVX_Vector *) src0)[i * 2 + 0]; HVX_Vector v1 = ((const HVX_Vector *) src0)[i * 2 + 1]; - HVX_Vector v2 = ((const HVX_Vector *) theta_cache)[i * 2 + 0]; - HVX_Vector v3 = ((const HVX_Vector *) theta_cache)[i * 2 + 1]; + HVX_Vector vcos = ((const HVX_Vector *) theta_cache)[i * 2 + 0]; + HVX_Vector vsin = ((const HVX_Vector *) theta_cache)[i * 2 + 1]; - HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(v1, v0, -4); - HVX_VectorPair vcos_sin = Q6_W_vdeal_VVR(v3, v2, -4); + HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(v1, v0, -4); - HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_hi_W(vcos_sin)); - HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_hi_W(vcos_sin)); + HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vcos); + HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vsin); + HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vcos); + HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vsin); HVX_Vector v4 = Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s); HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c); @@ -361,15 +471,15 @@ static inline void hvx_rope_f32_aa(float * restrict dst, const float * restrict if (nloe > 0) { if (nloe <= 32) { HVX_Vector v0 = hvx_vmemu(src0 + nvec * 64); - HVX_Vector v2 = hvx_vmemu(theta_cache + nvec * 64); + HVX_Vector vcos = hvx_vmemu(theta_cache + nvec * 64); + HVX_Vector vsin = hvx_vmemu(theta_cache + nvec * 64 + 32); - HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(Q6_V_vzero(), v0, -4); - HVX_VectorPair vcos_sin = Q6_W_vdeal_VVR(Q6_V_vzero(), v2, -4); + HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(Q6_V_vzero(), v0, -4); - HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_hi_W(vcos_sin)); - HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_hi_W(vcos_sin)); + HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vcos); + HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vsin); + HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vcos); + HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vsin); HVX_Vector v4 = Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s); HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c); @@ -381,16 +491,15 @@ static inline void hvx_rope_f32_aa(float * restrict dst, const float * restrict HVX_Vector v0 = hvx_vmemu(src0 + nvec * 64); HVX_Vector v1 = hvx_vmemu(src0 + nvec * 64 + 32); - HVX_Vector v2 = hvx_vmemu(theta_cache + nvec * 64); - HVX_Vector v3 = hvx_vmemu(theta_cache + nvec * 64 + 32); + HVX_Vector vcos = hvx_vmemu(theta_cache + nvec * 64); + HVX_Vector vsin = hvx_vmemu(theta_cache + nvec * 64 + 32); - HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(v1, v0, -4); - HVX_VectorPair vcos_sin = Q6_W_vdeal_VVR(v3, v2, -4); + HVX_VectorPair vx0_x1 = Q6_W_vdeal_VVR(v1, v0, -4); - HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), Q6_V_hi_W(vcos_sin)); - HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_lo_W(vcos_sin)); - HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), Q6_V_hi_W(vcos_sin)); + HVX_Vector vx0_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vcos); + HVX_Vector vx0_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_lo_W(vx0_x1), vsin); + HVX_Vector vx1_c = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vcos); + HVX_Vector vx1_s = Q6_Vqf32_vmpy_VsfVsf(Q6_V_hi_W(vx0_x1), vsin); HVX_Vector v4 = Q6_Vqf32_vsub_Vqf32Vqf32(vx0_c, vx1_s); HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(vx0_s, vx1_c); @@ -403,46 +512,23 @@ static inline void hvx_rope_f32_aa(float * restrict dst, const float * restrict } } -static void inline rope_basic_f32(struct htp_rope_context * rctx, uint8_t * restrict dst, uint8_t * restrict src, - uint32_t nr, uint32_t ne0, const float * restrict theta_cache) { +static void inline rope_basic_f32_inplace(struct htp_rope_context * rctx, uint8_t * src, + uint32_t nr, const float * restrict theta_cache) { + const uint32_t n_offs = rctx->n_offs; #pragma unroll(4) for (uint32_t i = 0; i < nr; i++) { - float * d = (float *) (dst + i * rctx->dst_row_size_aligned); float * s = (float *) (src + i * rctx->src0_row_size_aligned); - - hvx_rope_f32_aa(d, s, rctx->n_dims, theta_cache); - - // fill the remain channels with data from src tensor - if (rctx->n_dims < ne0) { - hvx_copy_f32_uu((uint8_t *)(d + rctx->n_dims), (uint8_t *)(s + rctx->n_dims), ne0 - rctx->n_dims); - } + hvx_rope_f32_aa(s + n_offs, s + n_offs, rctx->n_dims, theta_cache); } } -static void inline rope_neox_f32(struct htp_rope_context * rctx, uint8_t * restrict dst, uint8_t * restrict src, - uint32_t nr, uint32_t ne0, const float * restrict theta_cache) { +static void inline rope_neox_f32_inplace(struct htp_rope_context * rctx, uint8_t * src, + uint32_t nr, uint32_t ne, const float * restrict theta_cache) { + const uint32_t n_offs = rctx->n_offs; #pragma unroll(4) for (uint32_t i = 0; i < nr; i++) { - float * d = (float *) (dst + i * rctx->dst_row_size_aligned); float * s = (float *) (src + i * rctx->src0_row_size_aligned); - - hvx_rope_neox_f32_aa(d, s, rctx->n_dims, theta_cache); - - // fill the remain channels with data from src tensor - if (rctx->n_dims < ne0) { - hvx_copy_f32_uu((uint8_t *)(d + rctx->n_dims), (uint8_t *)(s + rctx->n_dims), ne0 - rctx->n_dims); - } - } -} - -static void inline rope_vision_f32(struct htp_rope_context * rctx, uint8_t * restrict dst, uint8_t * restrict src, - uint32_t nr, uint32_t ne0, const float * restrict theta_cache) { - #pragma unroll(4) - for (uint32_t i = 0; i < nr; i++) { - float * d = (float *) (dst + i * rctx->dst_row_size_aligned); - float * s = (float *) (src + i * rctx->src0_row_size_aligned); - - hvx_rope_neox_f32_aa(d, s, ne0, theta_cache); + hvx_rope_neox_f32_aa(s + n_offs, s + n_offs, ne, theta_cache); } } @@ -457,33 +543,31 @@ static void rope_job_f32(unsigned int nth, unsigned int ith, void * data) { htp_rope_preamble; - const uint32_t src0_nrows = rctx->src0_nrows; + const uint32_t src0_nrows = rctx->nrows; const uint32_t src0_nrows_per_thread = rctx->src0_nrows_per_thread; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); + const uint32_t src0_start_row = rctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, rctx->row_start + src0_nrows); // no work for this thread if (src0_start_row >= src0_end_row) { return; } - uint64_t tt = HAP_perf_get_qtimer_count(); - const int32_t mode = rctx->mode; // MROPE, IMROPE and VISION use NEOX-style pairing for the rotation const bool is_neox = (mode & HTP_ROPE_TYPE_NEOX) || (mode & HTP_ROPE_TYPE_MROPE); const bool is_vision = (mode == HTP_ROPE_TYPE_VISION); // VTCM setup - uint8_t * src0_spad_base = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); + uint8_t * src0_spad_base = rctx->vtcm_base + (ith * rctx->spad_per_thread); float * theta_cache = (float *) (src0_spad_base); src0_spad_base = src0_spad_base + rctx->theta_cache_offset; - uint8_t * dst_spad_base = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); - dma_queue * dma_queue = octx->ctx->dma[ith]; - const int32_t * pos = (const int32_t *) src1->data; - const float * freq_factors = src2 ? (const float *) src2->data : NULL; + dma_queue * dma_q = octx->ctx->dma[ith]; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + const int32_t * pos = (const int32_t *) (uintptr_t) src1->data; + const float * freq_factors = rctx->freq_factors; const uint32_t i3_start = fastdiv(src0_start_row, &rctx->div_ne2_ne1); const uint32_t rem = fastmodulo(src0_start_row, ne2 * ne1, &rctx->div_ne2_ne1); @@ -492,6 +576,7 @@ static void rope_job_f32(unsigned int nth, unsigned int ith, void * data) { uint32_t ir = src0_start_row; uint32_t prev_i2 = (uint32_t) -1; + uint32_t cur_slot = 0; for (uint32_t i3 = i3_start; i3 < ne3; i3++) { // batch const uint32_t i2_init = (i3 == i3_start) ? i2_start : 0; @@ -504,35 +589,30 @@ static void rope_job_f32(unsigned int nth, unsigned int ith, void * data) { const uint32_t nrows = MIN(src0_end_row - ir, ne1 - i1); // Depth before prefetch - uint32_t dma_depth = dma_queue_depth(dma_queue); - - // FARF(HIGH, "rope-block %u: ir %u n-rows %u dma-depth %u : usec %u", ith, ir, nrows, dma_depth, - // (unsigned) HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - rctx->t_start)); - - // Prefetch loop - for (uint32_t pnr = 0, pr = 0; pr < nrows && pr < HTP_ROPE_SPAD_NROWS; pr += pnr) { - pnr = MIN(nrows - pr, HTP_ROPE_SPAD_BLOCK); + const uint32_t dma_depth = dma_queue_depth(dma_q); - uint32_t pi1 = i1 + pr; - uint32_t pir = ir + pr; + // Prefetch up to 2 blocks + const uint32_t p_nrows = MIN(nrows, 2 * HTP_ROPE_SPAD_BLOCK); + for (uint32_t pr = 0; pr < p_nrows; pr += HTP_ROPE_SPAD_BLOCK) { + const uint32_t pnr = MIN(nrows - pr, HTP_ROPE_SPAD_BLOCK); + const uint32_t slot = (cur_slot + pr / HTP_ROPE_SPAD_BLOCK) % HTP_ROPE_SPAD_NSLOTS; + uint8_t * spad_slot = rope_spad_slot(src0_spad_base, slot, rctx->src0_row_size_aligned); + const dma_addr_t src0_data = src0->data + i3 * nb03 + i2 * nb02 + (i1 + pr) * nb01; - // Dummy DMA transaction for sequencing (interleaving dst,src,dst,...) - dma_queue_push_vtcm_to_ddr(dma_queue, dma_make_ptr((void *) dst->data, dst_spad_base + pr * rctx->dst_row_size_aligned), 0, 0, 0); + // Dummy DMA transaction for sequencing (interleaving wr, rd, wr, rd, ...) + dma_queue_push(dma_q, dma_make_data(dst->data, spad_slot), 0, 0, 0, 0); - const uint8_t * src_addr = (const uint8_t *) src0->data + i3 * nb03 + i2 * nb02 + pi1 * nb01; - uint8_t * src_spad = src0_spad_base + pr * rctx->src0_row_size_aligned; - - // Copy only the row payload while striding the DDR source - dma_queue_push(dma_queue, dma_make_ptr(src_spad, src_addr), + dma_queue_push(dma_q, dma_make_data(spad_slot, src0_data), rctx->src0_row_size_aligned, rctx->src0_row_stride, rctx->src0_row_size, pnr); - - // FARF(HIGH, "rope-prefetch %u: pr %u i1 %u i2 %u i3 %u src-spad %p src-addr %p pnr %u", ith, pir, pi1, i2, i3, src_spad, src_addr, pnr); } // Update theta cache if (i2 != prev_i2) { prev_i2 = i2; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_A_PREP, i2); + // VISION rotates the full row; other modes only rotate n_dims. + const uint32_t n_cache = is_vision ? ne0 : (uint32_t) rctx->n_dims; const bool is_mrope = (rctx->mode & HTP_ROPE_TYPE_MROPE) != 0; if (is_mrope) { // src1 holds four position arrays stacked along ne0: @@ -545,66 +625,71 @@ static void rope_job_f32(unsigned int nth, unsigned int ith, void * data) { (float) pos[i2 + ne2 * 3], rctx->sections, is_imrope, is_vision, rctx->freq_scale, freq_factors, rctx->corr_dims, - ne0, rctx->ext_factor, rctx->attn_factor, - theta_cache, rctx->theta_scale); + n_cache, rctx->ext_factor, rctx->attn_factor, + theta_cache, rctx->theta_scale, rctx->theta_powers, rctx->theta_scale_32); } else { rope_cache_init(pos[i2], rctx->freq_scale, freq_factors, rctx->corr_dims, - ne0, rctx->ext_factor, rctx->attn_factor, - theta_cache, rctx->theta_scale); + n_cache, rctx->ext_factor, rctx->attn_factor, + theta_cache, rctx->theta_scale, rctx->theta_powers, rctx->theta_scale_32); } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_A_PREP, i2); } // Skip output DMA transactions from prev block (if any) - // No need to wait for those here since we're explicitly waiting for the latest prefecthes below. - for (uint32_t d=0; d < dma_depth; d++) { dma_queue_pop_nowait(dma_queue); } + for (uint32_t d = 0; d < dma_depth; d++) { dma_queue_pop_nowait(dma_q); } // Compute loop - for (uint32_t cnr = 0, cr = 0; cr < nrows; cr += cnr, ir += cnr, i1 += cnr) { - // Number of rows to compute - cnr = MIN(nrows - cr, HTP_ROPE_SPAD_BLOCK); + const uint32_t ne = is_vision ? ne0 : rctx->n_dims; + const uint32_t base_i1 = i1; + const uint32_t base_ir = ir; - uint8_t * dst_spad = (uint8_t *) dma_queue_pop(dma_queue).src; - uint8_t * src_spad = (uint8_t *) dma_queue_pop(dma_queue).dst; + for (uint32_t cnr = 0, cr = 0; cr < nrows; cr += cnr) { + cnr = MIN(nrows - cr, HTP_ROPE_SPAD_BLOCK); + const uint32_t slot = (cur_slot + cr / HTP_ROPE_SPAD_BLOCK) % HTP_ROPE_SPAD_NSLOTS; + const uint32_t cur_ir = base_ir + cr; + const uint32_t cur_i1 = base_i1 + cr; - // FARF(HIGH, "rope-compute %u: ir %u i1 %u i2 %u i3 %u src-spad %p cnr %u : usec %u", ith, ir, i1, i2, i3, src_spad, cnr, - // (unsigned) HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - rctx->t_start)); + dma_queue_pop(dma_q); + uint8_t * cur_spad = (uint8_t *) dma_queue_pop(dma_q).dst; - if (is_vision) { - rope_vision_f32(rctx, dst_spad, src_spad, cnr, ne0, theta_cache); - } else if (is_neox) { - rope_neox_f32(rctx, dst_spad, src_spad, cnr, ne0, theta_cache); + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, cur_ir); + if (is_neox || is_vision) { + rope_neox_f32_inplace(rctx, cur_spad, cnr, ne, theta_cache); } else { - rope_basic_f32(rctx, dst_spad, src_spad, cnr, ne0, theta_cache); + rope_basic_f32_inplace(rctx, cur_spad, cnr, theta_cache); } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, cur_ir); - uint8_t * dst_addr = (uint8_t *) dst->data + i3 * nb3 + i2 * nb2 + i1 * nb1; + const dma_addr_t dst_data = dst->data + i3 * nb3 + i2 * nb2 + cur_i1 * nb1; + dma_queue_push(dma_q, dma_make_data(dst_data, cur_spad), + rctx->dst_row_stride, rctx->src0_row_size_aligned, rctx->dst_row_size, cnr); - // Write only the row payload while striding the DDR dst - dma_queue_push(dma_queue, dma_make_ptr(dst_addr, dst_spad), - rctx->dst_row_stride, rctx->dst_row_size_aligned, rctx->dst_row_size, cnr); + // Prefetch 2 blocks ahead into the slot just freed + if ((cr + 2 * HTP_ROPE_SPAD_BLOCK) < nrows) { + const uint32_t p_cr = cr + 2 * HTP_ROPE_SPAD_BLOCK; + const uint32_t pnr = MIN(nrows - p_cr, HTP_ROPE_SPAD_BLOCK); + const uint32_t p_slot = (cur_slot + p_cr / HTP_ROPE_SPAD_BLOCK) % HTP_ROPE_SPAD_NSLOTS; + uint8_t * p_spad = rope_spad_slot(src0_spad_base, p_slot, rctx->src0_row_size_aligned); + const dma_addr_t p_src0_data = src0->data + i3 * nb03 + i2 * nb02 + (base_i1 + p_cr) * nb01; - // Prefetch more rows (if any) - if ((cr + HTP_ROPE_SPAD_NROWS) < nrows) { - uint32_t pnr = MIN(nrows - (cr + HTP_ROPE_SPAD_NROWS), HTP_ROPE_SPAD_BLOCK); - uint32_t pi1 = i1 + HTP_ROPE_SPAD_NROWS; - uint32_t pir = ir + HTP_ROPE_SPAD_NROWS; - - const uint8_t * src_addr = (const uint8_t *) src0->data + i3 * nb03 + i2 * nb02 + pi1 * nb01; - dma_queue_push(dma_queue, dma_make_ptr(src_spad, src_addr), + dma_queue_push(dma_q, dma_make_data(p_spad, p_src0_data), rctx->src0_row_size_aligned, rctx->src0_row_stride, rctx->src0_row_size, pnr); - - // FARF(HIGH, "rope-prefetch %u: pr %u i1 %u i2 %u i3 %u src-spad %p src-addr %p pnr %u", ith, pir, pi1, i2, i3, src_spad, src_addr, pnr); } } + + const uint32_t n_chunks = (nrows + HTP_ROPE_SPAD_BLOCK - 1) / HTP_ROPE_SPAD_BLOCK; + cur_slot = (cur_slot + n_chunks) % HTP_ROPE_SPAD_NSLOTS; + + ir += nrows; + i1 += nrows; } } } done: - dma_queue_flush(dma_queue); - tt = HAP_perf_get_qtimer_count() - tt; + dma_queue_flush(dma_q); - FARF(HIGH, "rope-f32: %d/%d: (%u:%u) usec %u\n", ith, nth, src0_start_row, src0_end_row, (unsigned) HAP_perf_qtimer_count_to_us(tt)); + FARF(HIGH, "rope-f32: %d/%d: (%u:%u)\n", ith, nth, src0_start_row, src0_end_row); } static int execute_op_rope_f32(struct htp_ops_context * octx) { @@ -615,8 +700,6 @@ static int execute_op_rope_f32(struct htp_ops_context * octx) { const struct htp_tensor * src2 = octx->src[2]; const struct htp_tensor * dst = octx->dst; - const char * op_type = "rope-f32"; - switch (octx->op) { case HTP_OP_ROPE: break; @@ -626,53 +709,66 @@ static int execute_op_rope_f32(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - const uint32_t ne0 = dst->ne[0]; - const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, src0_nrows); + const struct htp_rope_kernel_params * kparams = (const struct htp_rope_kernel_params *) octx->kernel_params; + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } + assert(octx->ctx->vtcm_size >= kparams->vtcm_size); - const size_t src0_row_size = src0->ne[0] * sizeof(float); - const size_t src0_row_stride = src0->nb[1]; - const size_t dst_row_size = dst->ne[0] * sizeof(float); - const size_t dst_row_stride = dst->nb[1]; + if (htp_tensor_is_extended(src1)) { + return HTP_STATUS_NO_SUPPORT; + } - // Aligned row sizes for VTCM - const size_t src0_row_size_aligned = hex_round_up(src0_row_size, VLEN); - const size_t dst_row_size_aligned = hex_round_up(dst_row_stride, VLEN); - const size_t theta_cache_size_aligned = hex_round_up(src0->ne[0] * sizeof(float), 256); - - // Calculate spad sizes per thread - size_t src0_spad_per_thread = theta_cache_size_aligned + HTP_ROPE_SPAD_NROWS * src0_row_size_aligned; - size_t dst_spad_per_thread = HTP_ROPE_SPAD_NROWS * dst_row_size_aligned; - size_t spad_per_thread = src0_spad_per_thread + dst_spad_per_thread; - - // Check if we fit in VTCM - size_t total_vtcm_needed = spad_per_thread * n_threads; - if (octx->ctx->vtcm_size < total_vtcm_needed) { - FARF(ERROR, "%s : current VTCM reservation %zu is too small, needed %zu\n", op_type, octx->ctx->vtcm_size, total_vtcm_needed); - return HTP_STATUS_VTCM_TOO_SMALL; + const uint32_t total_rows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const size_t dst_data_row_size = dst->ne[0] * sizeof(float); + + uint32_t row_start = 0; + uint32_t nrows = total_rows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(float), (uint32_t) dst_data_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition( + total_rows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; } - octx->src0_spad.size_per_thread = src0_spad_per_thread; - octx->dst_spad.size_per_thread = dst_spad_per_thread; - octx->src0_spad.size = n_threads * src0_spad_per_thread; - octx->dst_spad.size = n_threads * dst_spad_per_thread; - octx->src1_spad.size = 0; + if (nrows == 0) { + return HTP_STATUS_OK; + } - octx->src0_spad.data = octx->ctx->vtcm_base; octx->src0_spad.src = NULL; - octx->src1_spad.data = NULL; octx->src1_spad.src = NULL; - octx->dst_spad.data = octx->src0_spad.data + octx->src0_spad.size; octx->dst_spad.src = NULL; + const uint32_t n_threads = octx->n_threads; + + const uint32_t ne0 = dst->ne[0]; + const size_t src0_row_size = src0->ne[0] * sizeof(float); + const size_t src0_row_stride = src0->nb[1]; + const size_t dst_row_size = dst->ne[0] * sizeof(float); + const size_t dst_row_stride = dst->nb[1]; struct htp_rope_context rctx; memset(&rctx, 0, sizeof(struct htp_rope_context)); - rctx.t_start = HAP_perf_get_qtimer_count(); - - rctx.octx = octx; + rctx.octx = octx; + rctx.vtcm_base = (uint8_t *) octx->ctx->vtcm_base; + rctx.spad_per_thread = kparams->spad_per_thread; + rctx.theta_cache_offset = kparams->theta_cache_offset; + + if (src2) { + dma_queue * dma_q = octx->ctx->dma[0]; + const size_t ff_size = src2->ne[0] * sizeof(float); + float * vtcm_freq_factors = (float *) (rctx.vtcm_base + kparams->freq_factors_offset); + dma_queue_push(dma_q, dma_make_data(vtcm_freq_factors, src2->data), + kparams->freq_factors_size, 0, ff_size, 1); + dma_queue_pop(dma_q); + rctx.freq_factors = vtcm_freq_factors; + } const int32_t * op_params = &octx->op_params[0]; rctx.n_dims = ((const int32_t *) op_params)[1]; rctx.mode = ((const int32_t *) op_params)[2]; rctx.n_ctx_orig = ((const int32_t *) op_params)[4]; + rctx.n_offs = ((const int32_t *) op_params)[15]; memcpy(&rctx.freq_base, (int32_t *) op_params + 5, sizeof(float)); memcpy(&rctx.freq_scale, (int32_t *) op_params + 6, sizeof(float)); @@ -683,31 +779,31 @@ static int execute_op_rope_f32(struct htp_ops_context * octx) { memcpy(&rctx.sections, (int32_t *) op_params + 11, sizeof(int) * 4); rctx.theta_scale = powf(rctx.freq_base, -2.0f / rctx.n_dims); + rctx.theta_powers[0] = 1.0f; + for (int j = 1; j < 32; j++) { + rctx.theta_powers[j] = rctx.theta_powers[j - 1] * rctx.theta_scale; + } + rctx.theta_scale_32 = rctx.theta_powers[31] * rctx.theta_scale; rope_corr_dims(rctx.n_dims, rctx.n_ctx_orig, rctx.freq_base, rctx.beta_fast, rctx.beta_slow, rctx.corr_dims); - rctx.src0_row_size = src0_row_size; - rctx.src0_row_stride = src0_row_stride; - rctx.dst_row_size = dst_row_size; - rctx.dst_row_stride = dst_row_stride; - rctx.src0_row_size_aligned = src0_row_size_aligned; - rctx.dst_row_size_aligned = dst_row_size_aligned; - rctx.theta_cache_offset = theta_cache_size_aligned; + rctx.src0_row_size = src0_row_size; + rctx.src0_row_stride = src0_row_stride; + rctx.dst_row_size = dst_row_size; + rctx.dst_row_stride = dst_row_stride; + rctx.src0_row_size_aligned = kparams->src0_row_size_aligned; - rctx.src0_nrows = src0_nrows; - rctx.src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; - - if (src0_nrows > 0) { - rctx.div_ne2_ne1 = init_fastdiv_values(dst->ne[2] * dst->ne[1]); - rctx.div_ne1 = init_fastdiv_values(dst->ne[1]); - } + rctx.src0_nrows = nrows; + rctx.nrows = nrows; + rctx.row_start = row_start; + rctx.src0_nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); + rctx.div_ne2_ne1 = kparams->div_ne2_ne1; + rctx.div_ne1 = kparams->div_ne1; FARF(HIGH, "rope-f32 n-rows %u n-dims %d ne0 %u ext-factor %.6f theta-scale %.6f attn-factor %.6f\n", rctx.src0_nrows, rctx.n_dims, ne0, rctx.ext_factor, rctx.theta_scale, rctx.attn_factor); - if (!(octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - worker_pool_run_func(octx->ctx->worker_pool, rope_job_f32, &rctx, n_threads); - } + work_queue_run(octx->ctx->work_queue, rope_job_f32, &rctx, n_threads); return err; } diff --git a/ggml/src/ggml-hexagon/htp/rope-ops.h b/ggml/src/ggml-hexagon/htp/rope-ops.h new file mode 100644 index 00000000..ee055ccb --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/rope-ops.h @@ -0,0 +1,62 @@ +#ifndef HTP_ROPE_OPS_H +#define HTP_ROPE_OPS_H + +#include "hex-common.h" +#include "hex-fastdiv.h" + +#define HTP_ROPE_SPAD_BLOCK 8 +#define HTP_ROPE_SPAD_NSLOTS 4 +#define HTP_ROPE_SPAD_NROWS (HTP_ROPE_SPAD_BLOCK * HTP_ROPE_SPAD_NSLOTS) + +struct htp_rope_kernel_params { + uint32_t n_threads; + uint32_t src0_nrows; + uint32_t src0_nrows_per_thread; + uint32_t vtcm_size; + uint32_t spad_per_thread; + uint32_t theta_cache_offset; + uint32_t src0_row_size_aligned; + uint32_t freq_factors_offset; + uint32_t freq_factors_size; + + struct fastdiv_values div_ne2_ne1; + struct fastdiv_values div_ne1; +}; + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_rope_kernel_params) <= 128, "htp_rope_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_rope_kernel_params) <= 128, "htp_rope_kernel_params is too large for kernel_params blob"); +#endif + +struct htp_rope_vtcm_layout { + size_t total_bytes; + size_t bytes_per_thread; + size_t theta_cache_size_aligned; + size_t src0_row_size_aligned; + size_t freq_factors_size_aligned; +}; + +static inline void htp_rope_vtcm_layout_build( + struct htp_rope_vtcm_layout * layout, + uint32_t ne00, + uint32_t n_threads, + uint32_t n_freq_factors +) { + const size_t src0_row_size = ne00 * sizeof(float); + const size_t src0_row_size_aligned = hex_round_up((uint32_t) src0_row_size, 128); + const size_t theta_cache_size_aligned = hex_round_up((uint32_t) src0_row_size, 256); + const size_t freq_factors_size_aligned = hex_round_up(n_freq_factors * sizeof(float), 256); + + layout->src0_row_size_aligned = src0_row_size_aligned; + layout->theta_cache_size_aligned = theta_cache_size_aligned; + layout->freq_factors_size_aligned = freq_factors_size_aligned; + layout->bytes_per_thread = theta_cache_size_aligned + HTP_ROPE_SPAD_NROWS * src0_row_size_aligned; + layout->total_bytes = layout->bytes_per_thread * n_threads + freq_factors_size_aligned; +} + +static inline uint8_t * rope_spad_slot(uint8_t * base, uint32_t slot, size_t row_size_aligned) { + return base + (slot * HTP_ROPE_SPAD_BLOCK) * row_size_aligned; +} + +#endif // HTP_ROPE_OPS_H diff --git a/ggml/src/ggml-hexagon/htp/set-rows-ops.c b/ggml/src/ggml-hexagon/htp/set-rows-ops.c index 58c54967..1d72538f 100644 --- a/ggml/src/ggml-hexagon/htp/set-rows-ops.c +++ b/ggml/src/ggml-hexagon/htp/set-rows-ops.c @@ -8,14 +8,21 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" +#include "work-queue.h" #include "hvx-utils.h" +#include "hex-utils.h" +#include "hvx-copy.h" +#include "hvx-quant.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" + +#include "hex-common.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "htp-tensor.h" +#include "htp/set-rows-ops.h" #define set_rows_preamble \ const uint32_t ne00 = octx->src[0]->ne[0]; \ @@ -47,144 +54,207 @@ \ const uint32_t nr = ne01; -struct htp_set_rows_context { +struct set_rows_context { struct htp_ops_context * octx; - struct fastdiv_values div_ne12; - struct fastdiv_values div_ne11; - uint32_t src0_nrows_per_thread; + const struct htp_set_rows_kernel_params * kparams; + struct htp_set_rows_vtcm_layout vtcm_layout; + uint8_t * vtcm_base; + uint32_t task_start; + uint32_t tasks; + uint32_t tasks_per_thread; }; -static void set_rows_thread_f32_f32(unsigned int nth, unsigned int ith, void *data) { - struct htp_set_rows_context * srctx = (struct htp_set_rows_context *)data; - struct htp_ops_context * octx = srctx->octx; - - set_rows_preamble; - - uint64_t qt = HAP_perf_get_qtimer_count(); - - // parallelize by rows of src0 - const uint32_t dr = srctx->src0_nrows_per_thread; - const uint32_t ir0 = dr * ith; - if (ir0 >= nr) { - return; - } - const uint32_t ir1 = (ir0 + dr < nr) ? (ir0 + dr) : nr; - - const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); - - for (uint32_t i03 = 0; i03 < ne03; ++i03) { - for (uint32_t i02 = 0; i02 < ne02; ++i02) { - for (uint32_t i = ir0; i < ir1; ++i) { - const uint32_t i12 = fastmodulo(i03, ne12, &srctx->div_ne12); - const uint32_t i11 = fastmodulo(i02, ne11, &srctx->div_ne11); - const uint32_t i10 = i; - - const uintptr_t src1_addr = octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12; - - uint32_t i1 = is_i32 ? *(int32_t *)src1_addr : *(int64_t *)src1_addr; - if (i1 >= ne1) { - // ignore invalid indices - continue; - } - - const uintptr_t src0_ptr = octx->src[0]->data + i*nb01 + i02*nb02 + i03*nb03; - const uintptr_t dst_ptr = octx->dst->data + i1*nb1 + i02*nb2 + i03*nb3; +#define SET_ROWS_THREAD_DMA_FN(TYPE_NAME, IDX_TYPE, COMPUTE_EXPR) \ +static void set_rows_thread_dma_##TYPE_NAME##_##IDX_TYPE(unsigned int nth, unsigned int ith, void *data) { \ + struct set_rows_context * srctx = (struct set_rows_context *)data; \ + struct htp_ops_context * octx = srctx->octx; \ + const struct htp_set_rows_kernel_params * kparams = srctx->kparams; \ + set_rows_preamble; \ + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ + const uint32_t dr = srctx->tasks_per_thread; \ + const uint32_t ir0 = srctx->task_start + dr * ith; \ + if (ir0 >= srctx->task_start + srctx->tasks) { \ + return; \ + } \ + const uint32_t ir1 = MIN(ir0 + dr, srctx->task_start + srctx->tasks); \ + dma_queue * dma_q = octx->ctx->dma[ith]; \ + const struct htp_set_rows_vtcm_layout * vtcm_layout = &srctx->vtcm_layout; \ + uint8_t * vtcm_src0 = srctx->vtcm_base + vtcm_layout->off_src0 + ith * vtcm_layout->src0_bytes_per_thread; \ + uint8_t * vtcm_dst = srctx->vtcm_base + vtcm_layout->off_dst + ith * vtcm_layout->dst_bytes_per_thread; \ + const uint32_t src0_row_size = ne00 * sizeof(float); \ + const uint32_t dst_row_size = htp_tensor_get_row_size(octx->dst->type, ne00); \ + const uint32_t nrows_per_thread = ir1 - ir0; \ + const uint32_t total_steps = ne03 * ne02 * nrows_per_thread; \ + uint32_t pi_step = 0; \ + uint32_t pi02 = 0; \ + uint32_t pi03 = 0; \ + for (uint32_t step = 0, spad_idx = 0; step < total_steps && spad_idx < 2; ++step, spad_idx++) { \ + uint32_t i = ir0 + pi_step; \ + const dma_addr_t src0_data = octx->src[0]->data + i*nb01 + pi02*nb02 + pi03*nb03; \ + dma_queue_push(dma_q, \ + dma_make_data(octx->dst->data, \ + vtcm_dst + spad_idx * vtcm_layout->dst_spad_half_size), \ + dst_row_size, vtcm_layout->dst_spad_half_size, dst_row_size, 0); \ + dma_queue_push(dma_q, \ + dma_make_data(vtcm_src0 + spad_idx * vtcm_layout->src0_spad_half_size, \ + src0_data), \ + vtcm_layout->src0_spad_half_size, src0_row_size, src0_row_size, 1); \ + pi_step++; \ + if (pi_step == nrows_per_thread) { \ + pi_step = 0; \ + pi02++; \ + if (pi02 == ne02) { \ + pi02 = 0; \ + pi03++; \ + } \ + } \ + } \ + uint32_t ci_step = 0; \ + uint32_t ci02 = 0; \ + uint32_t ci03 = 0; \ + uint32_t ci11_base = 0; \ + uint32_t ci12_base = 0; \ + for (uint32_t step = 0; step < total_steps; ++step) { \ + void * dst_spad = (void *) dma_queue_pop(dma_q).src; \ + void * src_spad = (void *) dma_queue_pop(dma_q).dst; \ + uint32_t i = ir0 + ci_step; \ + const uintptr_t src1_addr = octx->src[1]->data + i*nb10 + ci11_base*nb11 + ci12_base*nb12; \ + const IDX_TYPE i1 = *(const IDX_TYPE *)src1_addr; \ + const bool valid_i1 = ((uint64_t)i1 < (uint64_t)ne1); \ + const uint32_t target_i1 = (uint32_t)i1; \ + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, step); \ + if (valid_i1) { \ + COMPUTE_EXPR; \ + } \ + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, step); \ + if (valid_i1) { \ + const dma_addr_t dst_data = octx->dst->data + target_i1*nb1 + ci02*nb2 + ci03*nb3; \ + dma_queue_push(dma_q, \ + dma_make_data(dst_data, dst_spad), \ + dst_row_size, vtcm_layout->dst_spad_half_size, dst_row_size, 1); \ + } else { \ + dma_queue_push(dma_q, \ + dma_make_data(octx->dst->data, dst_spad), \ + dst_row_size, vtcm_layout->dst_spad_half_size, dst_row_size, 0); \ + } \ + const uint32_t next_step = step + 2; \ + if (next_step < total_steps) { \ + uint32_t ni = ir0 + pi_step; \ + const dma_addr_t psrc0_data = octx->src[0]->data + ni*nb01 + pi02*nb02 + pi03*nb03; \ + dma_queue_push(dma_q, \ + dma_make_data(src_spad, psrc0_data), \ + vtcm_layout->src0_spad_half_size, src0_row_size, src0_row_size, 1); \ + pi_step++; \ + if (pi_step == nrows_per_thread) { \ + pi_step = 0; \ + pi02++; \ + if (pi02 == ne02) { \ + pi02 = 0; \ + pi03++; \ + } \ + } \ + } \ + ci_step++; \ + if (ci_step == nrows_per_thread) { \ + ci_step = 0; \ + ci02++; \ + ci11_base++; \ + if (ci11_base == ne11) { \ + ci11_base = 0; \ + } \ + if (ci02 == ne02) { \ + ci02 = 0; \ + ci03++; \ + ci12_base++; \ + if (ci12_base == ne12) { \ + ci12_base = 0; \ + } \ + } \ + } \ + } \ + dma_queue_flush(dma_q); \ +} - // copy row - hvx_copy_f32_uu((uint8_t *)dst_ptr, (const uint8_t *)src0_ptr, ne00); - } - } - } +SET_ROWS_THREAD_DMA_FN(f32, int32_t, { hvx_copy_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, ne00); }) +SET_ROWS_THREAD_DMA_FN(f32, int64_t, { hvx_copy_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, ne00); }) - qt = HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - qt); - FARF(HIGH, "set-rows-f32-f32 %d/%d: %ux%ux%ux%u (%u:%u) x %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", ith, nth, - ne00, ne01, ne02, ne03, ir0, ir1, ne10, ne11, ne12, ne13, ne0, ne1, ne2, ne3, (unsigned) qt); -} +SET_ROWS_THREAD_DMA_FN(f16, int32_t, { hvx_copy_f16_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, ne00); }) +SET_ROWS_THREAD_DMA_FN(f16, int64_t, { hvx_copy_f16_f32_uu((uint8_t *)dst_spad, (const uint8_t *)src_spad, ne00); }) -static void set_rows_thread_f16_f32(unsigned int nth, unsigned int ith, void *data) { - struct htp_set_rows_context * srctx = (struct htp_set_rows_context *)data; - struct htp_ops_context * octx = srctx->octx; +SET_ROWS_THREAD_DMA_FN(q8_0, int32_t, { hvx_quantize_row_q8_0_f32(dst_spad, (const float *)src_spad, ne00); }) +SET_ROWS_THREAD_DMA_FN(q8_0, int64_t, { hvx_quantize_row_q8_0_f32(dst_spad, (const float *)src_spad, ne00); }) +int op_set_rows(struct htp_ops_context * octx) { + const struct htp_set_rows_kernel_params * kparams = (const struct htp_set_rows_kernel_params *)octx->kernel_params; set_rows_preamble; - uint64_t qt = HAP_perf_get_qtimer_count(); + if (octx->src[0]->type != HTP_TYPE_F32) { + return HTP_STATUS_NO_SUPPORT; + } - // parallelize by rows of src0 - const uint32_t dr = srctx->src0_nrows_per_thread; - const uint32_t ir0 = dr * ith; - if (ir0 >= nr) { - return; + if (octx->dst->type != HTP_TYPE_F32 && octx->dst->type != HTP_TYPE_F16 && octx->dst->type != HTP_TYPE_Q8_0) { + return HTP_STATUS_NO_SUPPORT; } - const uint32_t ir1 = (ir0 + dr < nr) ? (ir0 + dr) : nr; - const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); + if (htp_tensor_is_extended(octx->src[1])) { + return HTP_STATUS_NO_SUPPORT; + } - for (uint32_t i03 = 0; i03 < ne03; ++i03) { - for (uint32_t i02 = 0; i02 < ne02; ++i02) { - for (uint32_t i = ir0; i < ir1; ++i) { - const uint32_t i12 = fastmodulo(i03, ne12, &srctx->div_ne12); - const uint32_t i11 = fastmodulo(i02, ne11, &srctx->div_ne11); - const uint32_t i10 = i; + const struct htp_tensor * dst = octx->dst; + const uint32_t total_tasks = kparams->total_tasks; - const uintptr_t src1_addr = octx->src[1]->data + i10*nb10 + i11*nb11 + i12*nb12; + uint32_t task_start = 0; + uint32_t tasks = total_tasks; - uint32_t i1 = is_i32 ? *(int32_t *)src1_addr : *(int64_t *)src1_addr; - if (i1 >= ne1) { - // ignore invalid indices - continue; - } + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && (dst->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0 && !htp_tensor_is_permuted(dst); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_tasks, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + task_start = range.start; + tasks = range.count; + } - const uint8_t* src0_ptr = (const uint8_t *) octx->src[0]->data + i*nb01 + i02*nb02 + i03*nb03; - uint8_t* dst_ptr = (uint8_t *) octx->dst->data + i1*nb1 + i02*nb2 + i03*nb3; + if (tasks == 0) { + return HTP_STATUS_OK; + } - hvx_copy_f16_f32_uu(dst_ptr, src0_ptr, ne00); - } - } + if (!htp_ops_context_set_n_threads(octx, (uint32_t) kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; } - qt = HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - qt); - FARF(HIGH, "set-rows-f16-f32 %d/%d: %ux%ux%ux%u (%u:%u) x %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", ith, nth, - ne00, ne01, ne02, ne03, ir0, ir1, ne10, ne11, ne12, ne13, ne0, ne1, ne2, ne3, (unsigned) qt); -} + const uint32_t n_threads = octx->n_threads; -int op_set_rows(struct htp_ops_context * octx) { - set_rows_preamble; + // l2fetch the src1 (indices) tensor in the main thread + hex_l2fetch_block((const void *)octx->src[1]->data, octx->src[1]->ne[3] * octx->src[1]->nb[3]); - const uint32_t n_threads = MIN(nr, octx->n_threads); + struct set_rows_context srctx; + srctx.octx = octx; + srctx.kparams = kparams; + srctx.task_start = task_start; + srctx.tasks = tasks; + srctx.tasks_per_thread = fastdiv(tasks + n_threads - 1, &octx->n_threads_div); - if (octx->src[0]->type != HTP_TYPE_F32) { - return HTP_STATUS_NO_SUPPORT; - } + htp_set_rows_vtcm_layout_build(&srctx.vtcm_layout, octx->dst->type, ne00, n_threads); + srctx.vtcm_base = (uint8_t *)octx->ctx->vtcm_base; - if (octx->dst->type != HTP_TYPE_F32 && octx->dst->type != HTP_TYPE_F16) { - return HTP_STATUS_NO_SUPPORT; - } + work_queue_func_t q_func = NULL; + const bool is_i32 = (octx->src[1]->type == HTP_TYPE_I32); - if (octx->src[1]->type != HTP_TYPE_I32 && octx->src[1]->type != HTP_TYPE_I64) { - return HTP_STATUS_NO_SUPPORT; + switch (octx->dst->type) { + case HTP_TYPE_F32: q_func = is_i32 ? set_rows_thread_dma_f32_int32_t : set_rows_thread_dma_f32_int64_t; break; + case HTP_TYPE_F16: q_func = is_i32 ? set_rows_thread_dma_f16_int32_t : set_rows_thread_dma_f16_int64_t; break; + case HTP_TYPE_Q8_0: q_func = is_i32 ? set_rows_thread_dma_q8_0_int32_t : set_rows_thread_dma_q8_0_int64_t; break; + default: return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { - return HTP_STATUS_OK; - } + FARF(HIGH, "set-rows: (%ux%ux%ux%u) x (%ux%ux%ux%u) -> (%ux%ux%ux%u) : src0-vtcm-size %zu dst-vtcm-size %zu n-threads %d\n", + octx->src[0]->ne[0], octx->src[0]->ne[1], octx->src[0]->ne[2], octx->src[0]->ne[3], + octx->src[1]->ne[0], octx->src[1]->ne[1], octx->src[1]->ne[2], octx->src[1]->ne[3], + octx->dst->ne[0], octx->dst->ne[1], octx->dst->ne[2], octx->dst->ne[3], + srctx.vtcm_layout.src0_bytes_per_thread * n_threads, + srctx.vtcm_layout.dst_bytes_per_thread * n_threads, + n_threads); - struct htp_set_rows_context srctx; - srctx.octx = octx; - srctx.div_ne12 = init_fastdiv_values(ne12); - srctx.div_ne11 = init_fastdiv_values(ne11); - - srctx.src0_nrows_per_thread = (nr + n_threads - 1) / n_threads; - - switch(octx->dst->type) { - case HTP_TYPE_F32: - worker_pool_run_func(octx->ctx->worker_pool, set_rows_thread_f32_f32, &srctx, n_threads); - break; - case HTP_TYPE_F16: - worker_pool_run_func(octx->ctx->worker_pool, set_rows_thread_f16_f32, &srctx, n_threads); - break; - default: - return HTP_STATUS_NO_SUPPORT; - } + work_queue_run(octx->ctx->work_queue, q_func, &srctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/set-rows-ops.h b/ggml/src/ggml-hexagon/htp/set-rows-ops.h new file mode 100644 index 00000000..5e98d2cb --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/set-rows-ops.h @@ -0,0 +1,74 @@ +#ifndef HTP_SET_ROWS_OPS_H +#define HTP_SET_ROWS_OPS_H + +#include "hex-fastdiv.h" + +struct htp_set_rows_kernel_params { + int32_t n_threads; + int32_t total_tasks; + int32_t tasks_per_thread; + int32_t vtcm_size; + + // Fastdiv helpers + struct fastdiv_values div_ne11; + struct fastdiv_values div_ne12; + struct fastdiv_values div_tasks_per_thread; + struct fastdiv_values div_ne02; +}; + +struct htp_set_rows_vtcm_layout { + size_t total_bytes; + size_t off_src0; + size_t off_dst; + + size_t src0_bytes_per_thread; + size_t dst_bytes_per_thread; + + size_t src0_spad_half_size; + size_t dst_spad_half_size; +}; + +static inline void htp_set_rows_vtcm_layout_build( + struct htp_set_rows_vtcm_layout * vtcm_layout, + int dst_type, + uint32_t ne00, + uint32_t n_threads) { + + size_t src0_row_size = ne00 * 4; + size_t dst_row_size = 0; + switch (dst_type) { + case 0: // HTP_TYPE_F32 + dst_row_size = ne00 * 4; + break; + case 1: // HTP_TYPE_F16 + dst_row_size = ne00 * 2; + break; + case 8: // HTP_TYPE_Q8_0 + dst_row_size = (ne00 / 32) * 34; + break; + default: + dst_row_size = 0; + break; + } + + size_t src0_row_size_aligned = (src0_row_size + 255) & ~255; + size_t dst_row_size_aligned = (dst_row_size + 255) & ~255; + + vtcm_layout->src0_spad_half_size = src0_row_size_aligned; + vtcm_layout->dst_spad_half_size = dst_row_size_aligned; + + vtcm_layout->src0_bytes_per_thread = src0_row_size_aligned * 2; + vtcm_layout->dst_bytes_per_thread = dst_row_size_aligned * 2; + + vtcm_layout->off_src0 = 0; + vtcm_layout->off_dst = vtcm_layout->off_src0 + vtcm_layout->src0_bytes_per_thread * n_threads; + vtcm_layout->total_bytes = vtcm_layout->off_dst + vtcm_layout->dst_bytes_per_thread * n_threads; +} + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_set_rows_kernel_params) <= 128, "htp_set_rows_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_set_rows_kernel_params) <= 128, "htp_set_rows_kernel_params is too large for kernel_params blob"); +#endif + +#endif // HTP_SET_ROWS_OPS_H diff --git a/ggml/src/ggml-hexagon/htp/softmax-ops.c b/ggml/src/ggml-hexagon/htp/softmax-ops.c index d78bcc0e..48be5d72 100644 --- a/ggml/src/ggml-hexagon/htp/softmax-ops.c +++ b/ggml/src/ggml-hexagon/htp/softmax-ops.c @@ -8,52 +8,49 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" +#include "work-queue.h" #include "hvx-utils.h" #include "hex-fastdiv.h" +#include "hex-common.h" +#include "hex-profile.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" - -#define htp_softmax_preamble3 \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne02 = src0->ne[2]; \ - const uint32_t ne03 = src0->ne[3]; \ - \ - const uint32_t nb00 = src0->nb[0]; \ - const uint32_t nb01 = src0->nb[1]; \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t ne10 = src1 ? src1->ne[0] : 1; \ - const uint32_t ne11 = src1 ? src1->ne[1] : 1; \ - const uint32_t ne12 = src1 ? src1->ne[2] : 1; \ - const uint32_t ne13 = src1 ? src1->ne[3] : 1; \ - \ - const uint32_t nb10 = src1 ? src1->nb[0] : 1; \ - const uint32_t nb11 = src1 ? src1->nb[1] : 1; \ - const uint32_t nb12 = src1 ? src1->nb[2] : 1; \ - const uint32_t nb13 = src1 ? src1->nb[3] : 1; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb0 = dst->nb[0]; \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; +#include "htp-tensor.h" +#include "htp-vtcm.h" +#include "htp/softmax-ops.h" +#include "hvx-flash-attn.h" struct htp_softmax_context { struct htp_ops_context * octx; + const struct htp_softmax_kernel_params * kparams; + + void * compute; + + dma_addr_t data_src0; + dma_addr_t data_src1; + dma_addr_t data_dst; + + uint8_t * vtcm_src0; + uint8_t * vtcm_src1; + uint8_t * vtcm_dst; + + uint32_t vtcm_src0_size_per_thread; + uint32_t vtcm_src1_size_per_thread; + uint32_t vtcm_dst_size_per_thread; + + uint32_t src0_spad_half_size; + uint32_t src1_spad_half_size; + uint32_t dst_spad_half_size; + + uint32_t src0_row_size_aligned; + uint32_t src1_row_size_aligned; + uint32_t dst_row_size_aligned; bool use_f16; - bool use_src1; uint32_t n_head; uint32_t n_head_log2; @@ -63,66 +60,78 @@ struct htp_softmax_context { float m0; float m1; - struct fastdiv_values fastdiv_ne01; - struct fastdiv_values fastdiv_ne02; - struct fastdiv_values fastdiv_ne12; // For mask broadcasting - struct fastdiv_values fastdiv_ne13; // For mask broadcasting + struct fastdiv_values div_ne01; + struct fastdiv_values div_ne02; + struct fastdiv_values div_ne12; + struct fastdiv_values div_ne13; uint32_t src0_nrows_per_thread; + uint32_t row_start; + uint32_t nrows; + + float slopes[512] __attribute__((aligned(128))); }; -static void apply_mask(float * restrict wp0, - const float * restrict mp_f32, - const __fp16 * restrict mp_f16, - uint32_t ne00, - float slope, - bool use_f16) { - if (!mp_f32) { - return; - } - if (use_f16) { - for (uint32_t i = 0; i < ne00; ++i) { - wp0[i] += slope * (float) mp_f16[i]; - } - } else { - for (uint32_t i = 0; i < ne00; ++i) { - wp0[i] += slope * mp_f32[i]; - } - } -} +typedef void (*softmax_compute_fn_t)( + void * restrict dst, + const void * restrict src0, + const void * restrict mask, + uint32_t ne00, + float scale, + float slope +); -static void init_softmax_ctx(struct htp_softmax_context * smctx, struct htp_ops_context * octx) { - const struct htp_tensor * src0 = octx->src[0]; - const struct htp_tensor * src1 = octx->src[1]; +static void hvx_fast_softmax_prep_f16(const uint8_t * restrict src, + uint8_t * restrict dst, + const int num_elems, + float scale, + const uint8_t * restrict mask, + float slope) { + const HVX_Vector * restrict v_src = (const HVX_Vector *) src; + HVX_Vector * restrict v_dst = (HVX_Vector *) dst; + const HVX_Vector * restrict v_mask = (const HVX_Vector *) mask; + + HVX_Vector scale_vec = hvx_vec_splat_f32(scale); + HVX_Vector slope_vec = hvx_vec_splat_f32(slope); - memset(smctx, 0, sizeof(struct htp_softmax_context)); + const int nvec_64 = num_elems / VLEN_FP16; + const int nloe_64 = num_elems % VLEN_FP16; - memcpy(&smctx->scale, (float *) octx->op_params, sizeof(float)); - memcpy(&smctx->max_bias, (float *) octx->op_params + 1, sizeof(float)); + #pragma unroll(2) + for (int i = 0; i < nvec_64; i++) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_mask[i]); + HVX_Vector m0 = Q6_V_lo_W(p); + HVX_Vector m1 = Q6_V_hi_W(p); - smctx->n_head = src0->ne[2]; - smctx->n_head_log2 = 1u << (uint32_t) floor(log2(smctx->n_head)); + HVX_Vector s0 = v_src[2 * i]; + HVX_Vector s1 = v_src[2 * i + 1]; - smctx->m0 = powf(2.0f, -(smctx->max_bias) / smctx->n_head_log2); - smctx->m1 = powf(2.0f, -(smctx->max_bias / 2.0f) / smctx->n_head_log2); + HVX_Vector v0 = Q6_Vqf32_vadd_Vqf32Vqf32(Q6_Vqf32_vmpy_VsfVsf(s0, scale_vec), Q6_Vqf32_vmpy_VsfVsf(m0, slope_vec)); + HVX_Vector v1 = Q6_Vqf32_vadd_Vqf32Vqf32(Q6_Vqf32_vmpy_VsfVsf(s1, scale_vec), Q6_Vqf32_vmpy_VsfVsf(m1, slope_vec)); - smctx->use_src1 = (src1 != 0); - smctx->use_f16 = (src1 != 0) && (src1->type == HTP_TYPE_F16); + v_dst[2 * i] = Q6_Vsf_equals_Vqf32(v0); + v_dst[2 * i + 1] = Q6_Vsf_equals_Vqf32(v1); + } - smctx->octx = octx; + if (nloe_64 > 0) { + HVX_VectorPair p = hvx_vec_f16_to_f32(v_mask[nvec_64]); + HVX_Vector m0 = Q6_V_lo_W(p); - // Initialize fastdiv values - const uint32_t ne01 = src0->ne[1]; - const uint32_t ne02 = src0->ne[2]; + HVX_Vector s0 = v_src[2 * nvec_64]; + HVX_Vector v0 = Q6_Vqf32_vadd_Vqf32Vqf32(Q6_Vqf32_vmpy_VsfVsf(s0, scale_vec), Q6_Vqf32_vmpy_VsfVsf(m0, slope_vec)); - if (ne01 > 0) smctx->fastdiv_ne01 = init_fastdiv_values(ne01); - if (ne02 > 0) smctx->fastdiv_ne02 = init_fastdiv_values(ne02); + if (nloe_64 <= VLEN_FP32) { + hvx_vec_store_a(&v_dst[2 * nvec_64], nloe_64 * sizeof(float), Q6_Vsf_equals_Vqf32(v0)); + } else { + v_dst[2 * nvec_64] = Q6_Vsf_equals_Vqf32(v0); - const uint32_t ne12 = src1 ? src1->ne[2] : 1; - const uint32_t ne13 = src1 ? src1->ne[3] : 1; + HVX_Vector m1 = Q6_V_hi_W(p); + HVX_Vector s1 = v_src[2 * nvec_64 + 1]; + HVX_Vector v1 = Q6_Vqf32_vadd_Vqf32Vqf32(Q6_Vqf32_vmpy_VsfVsf(s1, scale_vec), Q6_Vqf32_vmpy_VsfVsf(m1, slope_vec)); - if (ne12 > 0) smctx->fastdiv_ne12 = init_fastdiv_values(ne12); - if (ne13 > 0) smctx->fastdiv_ne13 = init_fastdiv_values(ne13); + hvx_vec_store_a(&v_dst[2 * nvec_64 + 1], (nloe_64 - VLEN_FP32) * sizeof(float), Q6_Vsf_equals_Vqf32(v1)); + } + } } static void hvx_fast_softmax_prep_f32(const uint8_t * restrict src, @@ -131,57 +140,68 @@ static void hvx_fast_softmax_prep_f32(const uint8_t * restrict src, float scale, const uint8_t * restrict mask, float slope) { - const uint8_t * restrict src_curr = src; - uint8_t * restrict dst_curr = dst; - const uint8_t * restrict mask_curr = mask; + const HVX_Vector * restrict v_src = (const HVX_Vector *) src; + HVX_Vector * restrict v_dst = (HVX_Vector *) dst; + const HVX_Vector * restrict v_mask = (const HVX_Vector *) mask; HVX_Vector scale_vec = hvx_vec_splat_f32(scale); HVX_Vector slope_vec = hvx_vec_splat_f32(slope); - int step_of_1 = num_elems >> 5; + const int nvec = num_elems / VLEN_FP32; + const int nloe = num_elems % VLEN_FP32; #pragma unroll(4) - for (int i = 0; i < step_of_1; i++) { - HVX_Vector v1 = *(HVX_Vector *) src_curr; - - HVX_Vector v3 = *(HVX_Vector *) mask_curr; + for (int i = 0; i < nvec; i++) { + HVX_Vector v1 = v_src[i]; + HVX_Vector v3 = v_mask[i]; HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(v1, scale_vec); - HVX_Vector v4 = Q6_Vqf32_vmpy_VsfVsf(v3, slope_vec); - HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(v2, v4); - *(HVX_Vector *) dst_curr = Q6_Vsf_equals_Vqf32(v5); + v_dst[i] = Q6_Vsf_equals_Vqf32(v5); + } + + if (nloe > 0) { + HVX_Vector v1 = v_src[nvec]; + HVX_Vector v3 = v_mask[nvec]; + + HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(v1, scale_vec); + HVX_Vector v4 = Q6_Vqf32_vmpy_VsfVsf(v3, slope_vec); + HVX_Vector v5 = Q6_Vqf32_vadd_Vqf32Vqf32(v2, v4); - src_curr += VLEN; - dst_curr += VLEN; - mask_curr += VLEN; + hvx_vec_store_a(&v_dst[nvec], nloe * sizeof(float), Q6_Vsf_equals_Vqf32(v5)); } } -static void hvx_fast_softmax_f32(const uint8_t * restrict src, uint8_t * restrict dst, uint8_t * restrict pad, const int num_elems) { - const HVX_Vector * restrict v_src = (HVX_Vector *) src; - HVX_Vector * restrict v_pad = (HVX_Vector *) pad; +static void hvx_fast_softmax_f32(const uint8_t * restrict src, uint8_t * restrict dst, const int num_elems) { + const HVX_Vector * restrict v_src = (const HVX_Vector *) src; HVX_Vector * restrict v_dst = (HVX_Vector *) dst; - HVX_Vector sum_vec = Q6_V_vsplat_R(0x00000000); - HVX_Vector max_vec = hvx_vec_splat_f32(((const float *) src)[0]); - HVX_Vector zero_v = Q6_V_vzero(); - HVX_Vector one_v = hvx_vec_splat_f32(1.0); + const int nvec = num_elems / VLEN_FP32; + const int nloe = num_elems % VLEN_FP32; - int step_of_1 = num_elems >> 5; + HVX_Vector max_vec = hvx_vec_splat_f32(((const float *) src)[0]); - #pragma unroll(4) - for (int i = 0; i < step_of_1; i++) { + #pragma unroll(2) + for (int i = 0; i < nvec; i++) { HVX_Vector v1 = v_src[i]; max_vec = Q6_Vsf_vmax_VsfVsf(max_vec, v1); } - max_vec = hvx_vec_reduce_max_f32(max_vec); // replicated over all lanes + if (nloe > 0) { + HVX_VectorPred q_mask = Q6_Q_vsetq_R(nloe * sizeof(float)); + HVX_Vector neg_inf = hvx_vec_splat_f32(-INFINITY); + HVX_Vector v_tail = Q6_V_vmux_QVV(q_mask, v_src[nvec], neg_inf); + max_vec = Q6_Vsf_vmax_VsfVsf(max_vec, v_tail); + } - #pragma unroll(4) - for (int i = 0; i < step_of_1; i++) { + max_vec = hvx_vec_reduce_max_f32(max_vec); + + HVX_Vector sum_vec = Q6_V_vsplat_R(0x00000000); + + #pragma unroll(2) + for (int i = 0; i < nvec; i++) { HVX_Vector v1 = v_src[i]; HVX_Vector v2 = Q6_Vqf32_vsub_VsfVsf(v1, max_vec); @@ -189,205 +209,392 @@ static void hvx_fast_softmax_f32(const uint8_t * restrict src, uint8_t * restric sum_vec = Q6_Vqf32_vadd_VsfVsf(Q6_Vsf_equals_Vqf32(sum_vec), v3); - v_pad[i] = v3; + v_dst[i] = v3; + } + + if (nloe > 0) { + HVX_VectorPred q_mask = Q6_Q_vsetq_R(nloe * sizeof(float)); + HVX_Vector v1 = v_src[nvec]; + HVX_Vector v2 = Q6_Vqf32_vsub_VsfVsf(v1, max_vec); + HVX_Vector v3 = hvx_vec_exp_f32(Q6_Vsf_equals_Vqf32(v2)); + HVX_Vector v3_pad = Q6_V_vmux_QVV(q_mask, v3, Q6_V_vzero()); + + sum_vec = Q6_Vqf32_vadd_VsfVsf(Q6_Vsf_equals_Vqf32(sum_vec), v3_pad); + v_dst[nvec] = v3_pad; } - sum_vec = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_vec)); // replicated over all lanes + sum_vec = hvx_vec_reduce_sum_f32(Q6_Vsf_equals_Vqf32(sum_vec)); - HVX_VectorPred pos_sum = Q6_Q_vcmp_gt_VwVw(sum_vec, zero_v); + HVX_VectorPred pos_sum = Q6_Q_vcmp_gt_VwVw(sum_vec, Q6_V_vzero()); HVX_Vector v4 = hvx_vec_inverse_f32(sum_vec); - HVX_Vector scale_vec = Q6_V_vmux_QVV(pos_sum, v4, one_v); + HVX_Vector scale_vec = Q6_V_vmux_QVV(pos_sum, v4, hvx_vec_splat_f32(1.0f)); - #pragma unroll(4) - for (int i = 0; i < step_of_1; i++) { - HVX_Vector v1 = v_pad[i]; + #pragma unroll(2) + for (int i = 0; i < nvec; i++) { + HVX_Vector v1 = v_dst[i]; HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(v1, scale_vec); v_dst[i] = Q6_Vsf_equals_Vqf32(v2); } + + if (nloe > 0) { + HVX_Vector v1 = v_dst[nvec]; + HVX_Vector v2 = Q6_Vqf32_vmpy_VsfVsf(v1, scale_vec); + hvx_vec_store_a(&v_dst[nvec], nloe * sizeof(float), Q6_Vsf_equals_Vqf32(v2)); + } } -static float hvx_softmax_f32(const uint8_t * restrict src, uint8_t * restrict dst, uint8_t * restrict spad, const int num_elems, const float max) { - hvx_sub_scalar_f32(spad, src, max, num_elems); +static void compute_fast_softmax_f32_nomask( + void * restrict dst, + const void * restrict src0, + const void * restrict mask, + uint32_t ne00, + float scale, + float slope +) { + (void) mask; + (void) slope; + hvx_scale_f32((uint8_t *) dst, (const uint8_t *) src0, ne00, scale); + hvx_fast_softmax_f32((const uint8_t *) dst, (uint8_t *) dst, ne00); +} - hvx_exp_f32(dst, spad, num_elems, false); - return hvx_reduce_sum_f32(dst, num_elems); +static void compute_fast_softmax_f32_mask_f32( + void * restrict dst, + const void * restrict src0, + const void * restrict mask, + uint32_t ne00, + float scale, + float slope +) { + hvx_fast_softmax_prep_f32((const uint8_t *) src0, (uint8_t *) dst, ne00, scale, (const uint8_t *) mask, slope); + hvx_fast_softmax_f32((const uint8_t *) dst, (uint8_t *) dst, ne00); } -static void softmax_job_f32(unsigned int nth, unsigned int ith, void * data) { - struct htp_softmax_context * smctx = (struct htp_softmax_context *) data; - struct htp_ops_context * octx = smctx->octx; +static void compute_fast_softmax_f32_mask_f16( + void * restrict dst, + const void * restrict src0, + const void * restrict mask, + uint32_t ne00, + float scale, + float slope +) { + hvx_fast_softmax_prep_f16((const uint8_t *) src0, (uint8_t *) dst, ne00, scale, (const uint8_t *) mask, slope); + hvx_fast_softmax_f32((const uint8_t *) dst, (uint8_t *) dst, ne00); +} + +static const softmax_compute_fn_t softmax_kernels[HTP_SOFTMAX_KERNEL_COUNT] = { + [HTP_SOFTMAX_KERNEL_NOMASK] = compute_fast_softmax_f32_nomask, + [HTP_SOFTMAX_KERNEL_MASK_F32] = compute_fast_softmax_f32_mask_f32, + [HTP_SOFTMAX_KERNEL_MASK_F16] = compute_fast_softmax_f32_mask_f16, +}; +static void softmax_thread_dma(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_softmax_context * smctx = (const struct htp_softmax_context *) data; + struct htp_ops_context * octx = smctx->octx; const struct htp_tensor * src0 = octx->src[0]; - const struct htp_tensor * src1 = octx->src[1]; const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; - htp_softmax_preamble3; - - const uint32_t src0_nrows = ne01 * ne02 * ne03; // src0 rows + const uint32_t src0_nrows = smctx->nrows; const uint32_t src0_nrows_per_thread = smctx->src0_nrows_per_thread; - const uint32_t src0_start_row = src0_nrows_per_thread * ith; - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); + const uint32_t src0_start_row = smctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, smctx->row_start + src0_nrows); - // no work for this thread if (src0_start_row >= src0_end_row) { return; } - uint64_t qt = HAP_perf_get_qtimer_count(); + const dma_addr_t data_src0 = smctx->data_src0; + const dma_addr_t data_dst = smctx->data_dst; - int is_aligned = 1; - int opt_path = 0; + const size_t src0_row_size = src0->ne[0] * sizeof(float); + const size_t dst_row_size = src0->ne[0] * sizeof(float); - if (!hex_is_aligned((void *) src0->data, VLEN) || !hex_is_aligned((void *) dst->data, VLEN)) { - is_aligned = 0; - FARF(HIGH, "softmax-f32: unaligned addresses in elementwise op, possibly slower execution\n"); - } + uint8_t * src0_vtcm_base = smctx->vtcm_src0 + (ith * smctx->vtcm_src0_size_per_thread); + uint8_t * dst_vtcm_base = smctx->vtcm_dst + (ith * smctx->vtcm_dst_size_per_thread); - // Only use the fast path when aligned AND row size is multiple of VLEN (128 bytes) - // The fast path (hvx_fast_softmax_f32) doesn't handle tail elements - // The non-opt path uses hvx_softmax_f32 which properly handles all sizes via its helper functions - if ((1 == is_aligned) && !(nb01 & (VLEN - 1))) { - opt_path = 1; - } + const size_t src0_vtcm_half = smctx->src0_spad_half_size; + const size_t dst_vtcm_half = smctx->dst_spad_half_size; - uint8_t * src0_spad_data = octx->src0_spad.data + (ith * octx->src0_spad.size_per_thread); - uint8_t * src1_spad_data = octx->src1_spad.data + (ith * octx->src1_spad.size_per_thread); - uint8_t * dst_spad_data = octx->dst_spad.data + (ith * octx->dst_spad.size_per_thread); + dma_queue * dma_q = octx->ctx->dma[ith]; - float * wp0 = (float *) src0_spad_data; - float * wp1 = (float *) src1_spad_data; - float * wp2 = (float *) dst_spad_data; + for (uint32_t r = src0_start_row, idx = 0; r < src0_end_row && idx < 2; r++, idx++) { + dma_addr_t cur_dst = data_dst + r * dst_row_size; + dma_addr_t cur_src0 = data_src0 + r * src0_row_size; + void * d_spad = dst_vtcm_base + idx * dst_vtcm_half; + void * s_spad = src0_vtcm_base + idx * src0_vtcm_half; - uint32_t prev_i2 = (uint32_t)-1; - float slope = 1.0f; + dma_queue_push(dma_q, dma_make_data(cur_dst, d_spad), + dst_row_size, smctx->dst_row_size_aligned, dst_row_size, 0); + dma_queue_push(dma_q, dma_make_data(s_spad, cur_src0), + smctx->src0_row_size_aligned, src0_row_size, src0_row_size, 1); + } + + softmax_compute_fn_t compute = (softmax_compute_fn_t) smctx->compute; + const uint32_t ne00 = src0->ne[0]; for (uint32_t r = src0_start_row; r < src0_end_row; ++r) { - uint32_t i1 = fastmodulo(r, ne01, &smctx->fastdiv_ne01); - uint32_t r_div_ne01 = fastdiv(r, &smctx->fastdiv_ne01); - uint32_t i2 = fastmodulo(r_div_ne01, ne02, &smctx->fastdiv_ne02); - uint32_t i3 = fastdiv(r_div_ne01, &smctx->fastdiv_ne02); - - // Map to original logic indices - // i01 = i1 - // i02 = i2 - // i03 = i3 - - const uint32_t i11 = i1; - // const uint32_t i12 = i2 % ne12; - // const uint32_t i13 = i3 % ne13; - - uint32_t i12, i13; - if (ne12 == ne02) { - i12 = i2; - } else { - i12 = fastmodulo(i2, ne12, &smctx->fastdiv_ne12); + void * d_spad = (void *) dma_queue_pop(dma_q).src; + void * s_spad = (void *) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); + compute(d_spad, s_spad, NULL, ne00, smctx->scale, 1.0f); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, r); + + dma_addr_t cur_dst = data_dst + r * dst_row_size; + dma_queue_push(dma_q, dma_make_data(cur_dst, d_spad), + dst_row_size, smctx->dst_row_size_aligned, dst_row_size, 1); + + const uint32_t next_r = r + 2; + if (next_r < src0_end_row) { + dma_addr_t next_src0 = data_src0 + next_r * src0_row_size; + dma_queue_push(dma_q, dma_make_data(s_spad, next_src0), + smctx->src0_row_size_aligned, src0_row_size, src0_row_size, 1); } + } - if (ne13 == ne03) { - i13 = i3; - } else { - i13 = fastmodulo(i3, ne13, &smctx->fastdiv_ne13); - } + dma_queue_flush(dma_q); +} - // ALiBi - if (i2 != prev_i2) { - const uint32_t h = i2; // head - slope = (smctx->max_bias > 0.0f) ? h < smctx->n_head_log2 ? powf(smctx->m0, h + 1) : powf(smctx->m1, 2 * (h - smctx->n_head_log2) + 1) : 1.0f; - prev_i2 = i2; - } +static void softmax_thread_mask_dma(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_softmax_context * smctx = (const struct htp_softmax_context *) data; + struct htp_ops_context * octx = smctx->octx; + const struct htp_tensor * src0 = octx->src[0]; + const struct htp_tensor * src1 = octx->src[1]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; - float * sp = (float *) ((char *) src0->data + i1 * nb01 + i2 * nb02 + i3 * nb03); - float * dp = (float *) ((char *) dst->data + i1 * nb1 + i2 * nb2 + i3 * nb3); + const uint32_t src0_nrows = smctx->nrows; + const uint32_t src0_nrows_per_thread = smctx->src0_nrows_per_thread; - // broadcast the mask across rows - __fp16 * mp_f16 = (smctx->use_src1) ? (__fp16 *) ((char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13) : NULL; - float * mp_f32 = (smctx->use_src1) ? (float *) ((char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13) : NULL; + const uint32_t src0_start_row = smctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, smctx->row_start + src0_nrows); - if ((1 == opt_path) && (mp_f32) && !(smctx->use_f16)) { - hvx_fast_softmax_prep_f32((const uint8_t *) sp, (uint8_t *) wp0, ne00, smctx->scale, (const uint8_t *) mp_f32, slope); - hvx_fast_softmax_f32((const uint8_t *) wp0, (uint8_t *) dp, (uint8_t *) wp1, ne00); - } else if (1 == opt_path) { - hvx_scale_f32((uint8_t *) wp0, (const uint8_t *) sp, ne00, smctx->scale); - apply_mask(wp0, mp_f32, mp_f16, ne00, slope, smctx->use_f16); - hvx_fast_softmax_f32((const uint8_t *) wp0, (uint8_t *) dp, (uint8_t *) wp1, ne00); - } else { - // Non-optimized path: uses HVX helper functions that properly handle all tensor sizes - // including non-multiples of 32 (the HVX vector lane count for f32) - hvx_scale_f32((uint8_t *) wp0, (const uint8_t *) sp, ne00, smctx->scale); - apply_mask(wp0, mp_f32, mp_f16, ne00, slope, smctx->use_f16); - float max = hvx_reduce_max_f32((const uint8_t *) wp0, ne00); - float sum = hvx_softmax_f32((const uint8_t *) wp0, (uint8_t *) wp2, (uint8_t *) wp1, ne00, max); - sum = sum > 0.0 ? (1.0 / sum) : 1; - hvx_scale_f32((uint8_t *) dp, (const uint8_t *) wp2, ne00, sum); + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src0 = smctx->data_src0; + const dma_addr_t data_src1 = smctx->data_src1; + const dma_addr_t data_dst = smctx->data_dst; + + const size_t src0_row_size = src0->ne[0] * sizeof(float); + const size_t dst_row_size = src0->ne[0] * sizeof(float); + const size_t mask_row_size = smctx->use_f16 ? (src1->ne[0] * sizeof(__fp16)) : (src1->ne[0] * sizeof(float)); + + uint8_t * src0_vtcm_base = smctx->vtcm_src0 + (ith * smctx->vtcm_src0_size_per_thread); + uint8_t * src1_vtcm_base = smctx->vtcm_src1 + (ith * smctx->vtcm_src1_size_per_thread); + uint8_t * dst_vtcm_base = smctx->vtcm_dst + (ith * smctx->vtcm_dst_size_per_thread); + + const size_t src0_vtcm_half = smctx->src0_spad_half_size; + const size_t src1_vtcm_half = smctx->src1_spad_half_size; + const size_t dst_vtcm_half = smctx->dst_spad_half_size; + + const uint32_t nb11 = src1->nb[1]; + const uint32_t nb12 = src1->nb[2]; + const uint32_t nb13 = src1->nb[3]; + + const uint32_t ne00 = src0->ne[0]; + const uint32_t ne01 = src0->ne[1]; + const uint32_t ne02 = src0->ne[2]; + const uint32_t ne03 = src0->ne[3]; + const uint32_t ne12 = src1->ne[2]; + const uint32_t ne13 = src1->ne[3]; + + const struct fastdiv_values * div_ne01 = &smctx->div_ne01; + const struct fastdiv_values * div_ne02 = &smctx->div_ne02; + const struct fastdiv_values * div_ne12 = &smctx->div_ne12; + const struct fastdiv_values * div_ne13 = &smctx->div_ne13; + + dma_queue * dma_q = octx->ctx->dma[ith]; + + for (uint32_t r = src0_start_row, idx = 0; r < src0_end_row && idx < 2; r++, idx++) { + dma_addr_t cur_dst = data_dst + r * dst_row_size; + dma_addr_t cur_src0 = data_src0 + r * src0_row_size; + + uint32_t i1 = fastmodulo(r, ne01, div_ne01); + uint32_t r_div_ne01 = fastdiv(r, div_ne01); + uint32_t i2 = fastmodulo(r_div_ne01, ne02, div_ne02); + uint32_t i3 = fastdiv(r_div_ne01, div_ne02); + uint32_t i12 = (ne12 == ne02) ? i2 : fastmodulo(i2, ne12, div_ne12); + uint32_t i13 = (ne13 == ne03) ? i3 : fastmodulo(i3, ne13, div_ne13); + dma_addr_t cur_src1 = data_src1 + i1 * nb11 + i12 * nb12 + i13 * nb13; + + void * d_spad = dst_vtcm_base + idx * dst_vtcm_half; + void * s_spad = src0_vtcm_base + idx * src0_vtcm_half; + void * m_spad = src1_vtcm_base + idx * src1_vtcm_half; + + dma_queue_push(dma_q, dma_make_data(cur_dst, d_spad), + dst_row_size, smctx->dst_row_size_aligned, dst_row_size, 0); + dma_queue_push(dma_q, dma_make_data(s_spad, cur_src0), + smctx->src0_row_size_aligned, src0_row_size, src0_row_size, 1); + dma_queue_push(dma_q, dma_make_data(m_spad, cur_src1), + smctx->src1_row_size_aligned, mask_row_size, mask_row_size, 1); + } + + softmax_compute_fn_t compute = (softmax_compute_fn_t) smctx->compute; + const bool has_bias = smctx->max_bias > 0.0f; + uint32_t prev_i2 = (uint32_t)-1; + float slope = 1.0f; + + for (uint32_t r = src0_start_row; r < src0_end_row; ++r) { + void * d_spad = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * s_spad = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + void * m_spad = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + + if (has_bias) { + uint32_t r_div_ne01 = fastdiv(r, div_ne01); + uint32_t i2 = fastmodulo(r_div_ne01, ne02, div_ne02); + if (i2 != prev_i2) { + slope = smctx->slopes[i2]; + prev_i2 = i2; + } + } + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, r); + compute(d_spad, s_spad, m_spad, ne00, smctx->scale, slope); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, r); + + dma_addr_t cur_dst = data_dst + r * dst_row_size; + dma_queue_push(dma_q, dma_make_data(cur_dst, d_spad), + dst_row_size, smctx->dst_row_size_aligned, dst_row_size, 1); + + const uint32_t next_r = r + 2; + if (next_r < src0_end_row) { + dma_addr_t next_src0 = data_src0 + next_r * src0_row_size; + + uint32_t ni1 = fastmodulo(next_r, ne01, div_ne01); + uint32_t nr_div_ne01 = fastdiv(next_r, div_ne01); + uint32_t ni2 = fastmodulo(nr_div_ne01, ne02, div_ne02); + uint32_t ni3 = fastdiv(nr_div_ne01, div_ne02); + uint32_t ni12 = (ne12 == ne02) ? ni2 : fastmodulo(ni2, ne12, div_ne12); + uint32_t ni13 = (ne13 == ne03) ? ni3 : fastmodulo(ni3, ne13, div_ne13); + dma_addr_t next_src1 = data_src1 + ni1 * nb11 + ni12 * nb12 + ni13 * nb13; + + dma_queue_push(dma_q, dma_make_data(s_spad, next_src0), + smctx->src0_row_size_aligned, src0_row_size, src0_row_size, 1); + dma_queue_push(dma_q, dma_make_data(m_spad, next_src1), + smctx->src1_row_size_aligned, mask_row_size, mask_row_size, 1); } } - qt = HAP_perf_qtimer_count_to_us(HAP_perf_get_qtimer_count() - qt); - FARF(HIGH, "softmax-f32 %d/%d: %ux%ux%ux%u (%u:%u) x %ux%ux%ux%u -> %ux%ux%ux%u : opt %u f16 %u usec %u\n", ith, nth, - ne00, ne01, ne02, ne03, src0_start_row, src0_end_row, ne10, ne11, ne12, ne13, - ne0, ne1, ne2, ne3, opt_path, smctx->use_f16, (unsigned) qt); + dma_queue_flush(dma_q); } static int execute_op_softmax_f32(struct htp_ops_context * octx) { - int err = HTP_STATUS_OK; - const struct htp_tensor * src0 = octx->src[0]; - const struct htp_tensor * src1 = octx->src[1]; const struct htp_tensor * dst = octx->dst; - struct htp_softmax_context smctx; const char * op_type = "softmax-f32"; - init_softmax_ctx(&smctx, octx); + const struct htp_softmax_kernel_params * kparams = + (const struct htp_softmax_kernel_params *) octx->kernel_params; - const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = MIN(octx->n_threads, src0_nrows); - - smctx.src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads; - - const size_t src0_row_size = src0->nb[1]; - const size_t src1_row_size = src0_row_size; - const size_t dst_row_size = dst->nb[1]; - - // VTCM scratchpads for all tensors - // 4 rows per thread, padded to HVX vector size - octx->src0_spad.size_per_thread = hex_round_up(4 * src0_row_size, 128); - octx->src1_spad.size_per_thread = hex_round_up(4 * src1_row_size, 128); - octx->dst_spad.size_per_thread = hex_round_up(4 * dst_row_size, 128); - - octx->src0_spad.size = octx->src0_spad.size_per_thread * n_threads; - octx->src1_spad.size = octx->src1_spad.size_per_thread * n_threads; - octx->dst_spad.size = octx->dst_spad.size_per_thread * n_threads; - - size_t spad_size = octx->src0_spad.size + octx->src1_spad.size + octx->dst_spad.size; - - if (src1) { - FARF(HIGH, "%s: %ux%ux%ux%u x %ux%ux%ux%u -> %ux%ux%ux%u : src0-spad-size %u src1-spad-size %u dst-spad-size %u\n", - op_type, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], src1->ne[0], src1->ne[1], src1->ne[2], - src1->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], octx->src0_spad.size, octx->src1_spad.size, - octx->dst_spad.size); - } else { - FARF(HIGH, "%s: %ux%ux%ux%u -> %ux%ux%ux%u : src0-spad-size %u src1-spad-size %u dst-spad-size %u\n", op_type, - src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], - octx->src0_spad.size, octx->src1_spad.size, octx->dst_spad.size); + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; } - // Make sure the reserved vtcm size is sufficient - if (octx->ctx->vtcm_size < spad_size) { - FARF(ERROR, "%s : current VTCM reservation %zu is too small, needed %zu\n", op_type, octx->ctx->vtcm_size, spad_size); + if (kparams->kernel_id >= HTP_SOFTMAX_KERNEL_COUNT) { + return HTP_STATUS_INVAL_PARAMS; + } + + if (octx->ctx->vtcm_size < (size_t) kparams->vtcm_size) { + FARF(ERROR, "%s : current VTCM reservation %zu is too small, needed %u\n", + op_type, octx->ctx->vtcm_size, kparams->vtcm_size); return HTP_STATUS_VTCM_TOO_SMALL; } - octx->src0_spad.data = octx->ctx->vtcm_base; octx->src0_spad.src = NULL; - octx->src1_spad.data = octx->src0_spad.data + octx->src0_spad.size; octx->src1_spad.src = NULL; - octx->dst_spad.data = octx->src1_spad.data + octx->src1_spad.size; octx->dst_spad.src = NULL; + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; + const size_t elem_size = sizeof(float); + const size_t dst_row_size = dst->nb[1]; + + uint32_t row_start = 0; + uint32_t nrows = src0_nrows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, (uint32_t) elem_size, (uint32_t) dst_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition( + src0_nrows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + if (nrows < octx->n_threads) { + htp_ops_context_set_n_threads(octx, nrows ? nrows : 1); + } + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) return err; + const uint32_t n_threads = octx->n_threads; + uint8_t * const vtcm_base = (uint8_t *) octx->ctx->vtcm_base; - worker_pool_run_func(octx->ctx->worker_pool, softmax_job_f32, &smctx, n_threads); + const uint32_t off_src0 = 0; + const uint32_t off_dst = off_src0 + kparams->vtcm_src0_size_per_thread * kparams->n_threads; + const uint32_t off_src1 = off_dst + kparams->vtcm_dst_size_per_thread * kparams->n_threads; - return err; + struct htp_softmax_context smctx = { + .octx = octx, + .kparams = kparams, + .compute = (void *) softmax_kernels[kparams->kernel_id], + + .data_src0 = src0->data, + .data_src1 = kparams->use_src1 ? octx->src[1]->data : 0, + .data_dst = dst->data, + + .vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, vtcm_base, off_src0), + .vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, vtcm_base, off_dst), + .vtcm_src1 = VTCM_LAYOUT_PTR_OPTIONAL(uint8_t, vtcm_base, off_src1, kparams->use_src1), + + .vtcm_src0_size_per_thread = kparams->vtcm_src0_size_per_thread, + .vtcm_src1_size_per_thread = kparams->vtcm_src1_size_per_thread, + .vtcm_dst_size_per_thread = kparams->vtcm_dst_size_per_thread, + + .src0_spad_half_size = kparams->src0_spad_half_size, + .src1_spad_half_size = kparams->src1_spad_half_size, + .dst_spad_half_size = kparams->dst_spad_half_size, + + .src0_row_size_aligned = kparams->src0_row_size_aligned, + .src1_row_size_aligned = kparams->src1_row_size_aligned, + .dst_row_size_aligned = kparams->dst_row_size_aligned, + + .use_f16 = kparams->use_f16 != 0, + + .n_head = kparams->n_head, + .n_head_log2 = kparams->n_head_log2, + + .scale = kparams->scale, + .max_bias = kparams->max_bias, + .m0 = kparams->m0, + .m1 = kparams->m1, + + .div_ne01 = kparams->div_ne01, + .div_ne02 = kparams->div_ne02, + .div_ne12 = kparams->div_ne12, + .div_ne13 = kparams->div_ne13, + + .src0_nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .row_start = row_start, + .nrows = nrows, + }; + + if (kparams->max_bias > 0.0f && kparams->use_src1) { + if (kparams->n_head > 512) { + return HTP_STATUS_INVAL_PARAMS; + } + for (uint32_t h = 0; h < kparams->n_head; h += 32) { + HVX_Vector v_slopes = hvx_alibi_slopes(h, 1, kparams->n_head_log2, kparams->m0, kparams->m1); + hvx_vmem(&smctx.slopes[h]) = v_slopes; + } + } + + work_queue_func_t task_func = kparams->use_src1 ? softmax_thread_mask_dma : softmax_thread_dma; + work_queue_run(octx->ctx->work_queue, task_func, &smctx, n_threads); + + return HTP_STATUS_OK; } int op_softmax(struct htp_ops_context * octx) { diff --git a/ggml/src/ggml-hexagon/htp/softmax-ops.h b/ggml/src/ggml-hexagon/htp/softmax-ops.h new file mode 100644 index 00000000..8d976adb --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/softmax-ops.h @@ -0,0 +1,106 @@ +#ifndef HTP_SOFTMAX_OPS_H +#define HTP_SOFTMAX_OPS_H + +#include +#include +#include +#include +#include "hex-fastdiv.h" +#include "hex-common.h" + +enum htp_softmax_kernel_id { + HTP_SOFTMAX_KERNEL_NOMASK = 0, + HTP_SOFTMAX_KERNEL_MASK_F32, + HTP_SOFTMAX_KERNEL_MASK_F16, + HTP_SOFTMAX_KERNEL_COUNT, +}; + +struct htp_softmax_kernel_params { + uint32_t n_threads; + uint32_t src0_nrows; + uint32_t src0_nrows_per_thread; + uint32_t vtcm_size; + + uint32_t vtcm_src0_size_per_thread; + uint32_t vtcm_src1_size_per_thread; + uint32_t vtcm_dst_size_per_thread; + + uint32_t src0_row_size_aligned; + uint32_t src1_row_size_aligned; + uint32_t dst_row_size_aligned; + + uint32_t src0_spad_half_size; + uint32_t src1_spad_half_size; + uint32_t dst_spad_half_size; + + uint32_t n_head; + uint32_t n_head_log2; + uint32_t use_src1; + uint32_t use_f16; + uint32_t kernel_id; + + float scale; + float max_bias; + float m0; + float m1; + + struct fastdiv_values div_ne01; + struct fastdiv_values div_ne02; + struct fastdiv_values div_ne12; + struct fastdiv_values div_ne13; +}; + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_softmax_kernel_params) <= 128, "htp_softmax_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_softmax_kernel_params) <= 128, "htp_softmax_kernel_params is too large for kernel_params blob"); +#endif + +struct htp_softmax_vtcm_layout { + size_t total_bytes; + size_t off_src0; + size_t off_dst; + size_t off_src1; + + size_t src0_bytes_per_thread; + size_t dst_bytes_per_thread; + size_t src1_bytes_per_thread; + + size_t src0_spad_half_size; + size_t dst_spad_half_size; + size_t src1_spad_half_size; +}; + +static inline void htp_softmax_vtcm_layout_build( + struct htp_softmax_vtcm_layout * layout, + uint32_t ne00, + uint32_t ne10, + bool use_src1, + bool use_f16, + uint32_t n_threads +) { + size_t src0_row_size = ne00 * sizeof(float); + size_t dst_row_size = ne00 * sizeof(float); + size_t src1_row_size = use_src1 ? (ne10 * (use_f16 ? 2 : 4)) : 0; + + size_t src0_row_size_aligned = hex_round_up(src0_row_size, 128); + size_t dst_row_size_aligned = hex_round_up(dst_row_size, 128); + size_t src1_row_size_aligned = use_src1 ? hex_round_up(src1_row_size, 128) : 0; + + layout->src0_spad_half_size = src0_row_size_aligned; + layout->dst_spad_half_size = dst_row_size_aligned; + layout->src1_spad_half_size = src1_row_size_aligned; + + // Double buffering: 2 half-buffers per thread + layout->src0_bytes_per_thread = src0_row_size_aligned * 2; + layout->dst_bytes_per_thread = dst_row_size_aligned * 2; + layout->src1_bytes_per_thread = src1_row_size_aligned * 2; + + layout->off_src0 = 0; + layout->off_dst = layout->off_src0 + layout->src0_bytes_per_thread * n_threads; + layout->off_src1 = layout->off_dst + layout->dst_bytes_per_thread * n_threads; + + layout->total_bytes = layout->off_src1 + layout->src1_bytes_per_thread * n_threads; +} + +#endif // HTP_SOFTMAX_OPS_H diff --git a/ggml/src/ggml-hexagon/htp/solve-tri-ops.c b/ggml/src/ggml-hexagon/htp/solve-tri-ops.c index ae8e1a50..182982fc 100644 --- a/ggml/src/ggml-hexagon/htp/solve-tri-ops.c +++ b/ggml/src/ggml-hexagon/htp/solve-tri-ops.c @@ -1,13 +1,16 @@ #pragma clang diagnostic ignored "-Wunused-but-set-variable" #include -#include #include +#include "hex-common.h" +#include "hex-profile.h" + #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" #include "htp-ops.h" +#include "htp-tensor.h" #include "hvx-types.h" #include "hvx-utils.h" @@ -15,6 +18,7 @@ struct htp_solve_tri_context { struct htp_ops_context * octx; uint32_t jobs_per_thread; uint32_t total_jobs; + uint32_t job_start; uint32_t k_chunks; uint32_t col_block; }; @@ -89,11 +93,11 @@ static void solve_tri_batch_thread_f32(unsigned int nth, unsigned int ith, void const uint32_t col_block = VLEN_FP32; const uint32_t k_full = (k / col_block) * col_block; - const uint32_t start_batch = sctx->jobs_per_thread * ith; - const uint32_t end_batch = MIN(start_batch + sctx->jobs_per_thread, sctx->total_jobs); + const uint32_t start_batch = sctx->job_start + sctx->jobs_per_thread * ith; + const uint32_t end_batch = MIN(start_batch + sctx->jobs_per_thread, sctx->job_start + sctx->total_jobs); - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) start_batch); for (uint32_t batch = start_batch; batch < end_batch; ++batch) { const uint32_t i03 = batch / ne02; @@ -127,11 +131,10 @@ static void solve_tri_batch_thread_f32(unsigned int nth, unsigned int ith, void } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) end_batch); - FARF(HIGH, "solve-tri-batch %d/%d: A=(%ux%u) B=(%ux%u) batch %u:%u usec %u\n", - ith, nth, n, n, k, n, start_batch, end_batch, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + FARF(HIGH, "solve-tri-batch %d/%d: A=(%ux%u) B=(%ux%u) batch %u:%u\n", + ith, nth, n, n, k, n, start_batch, end_batch); } // Chunk-level thread: each job is one (batch, col_chunk) pair. @@ -148,11 +151,11 @@ static void solve_tri_chunk_thread_f32(unsigned int nth, unsigned int ith, void const uint32_t ne02 = src0->ne[2]; - const uint32_t start_job = sctx->jobs_per_thread * ith; - const uint32_t end_job = MIN(start_job + sctx->jobs_per_thread, sctx->total_jobs); + const uint32_t start_job = sctx->job_start + sctx->jobs_per_thread * ith; + const uint32_t end_job = MIN(start_job + sctx->jobs_per_thread, sctx->job_start + sctx->total_jobs); - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) start_job); for (uint32_t job = start_job; job < end_job; ++job) { const uint32_t batch = job / sctx->k_chunks; @@ -161,16 +164,14 @@ static void solve_tri_chunk_thread_f32(unsigned int nth, unsigned int ith, void const uint32_t i03 = batch / ne02; const uint32_t i02 = batch - i03 * ne02; - const uint32_t col0 = chunk * sctx->col_block; - const uint32_t coln = MIN(sctx->col_block, k - col0); - const float * A_batch = (const float *) ((const uint8_t *) (uintptr_t) src0->data + i02 * src0->nb[2] + i03 * src0->nb[3]); const float * B_batch = (const float *) ((const uint8_t *) (uintptr_t) src1->data + i02 * src1->nb[2] + i03 * src1->nb[3]); float * X_batch = (float *) ((uint8_t *) (uintptr_t) dst->data + i02 * dst->nb[2] + i03 * dst->nb[3]); - const bool use_hvx = (coln >= 8); + const uint32_t col0 = chunk * sctx->col_block; + const uint32_t coln = MIN(sctx->col_block, k - col0); for (uint32_t row = 0; row < n; ++row) { const float diag = A_batch[row * n + row]; @@ -179,7 +180,7 @@ static void solve_tri_chunk_thread_f32(unsigned int nth, unsigned int ith, void const float * A_row = A_batch + row * n; const float * B_row = B_batch + row * k; - if (use_hvx) { + if (coln >= 8) { solve_tri_row_hvx(A_row, B_row, X_batch, row, k, col0, coln, inv_diag); } else { solve_tri_row_scalar(A_row, B_row, X_batch, row, k, col0, coln, inv_diag); @@ -187,11 +188,10 @@ static void solve_tri_chunk_thread_f32(unsigned int nth, unsigned int ith, void } } - t2 = HAP_perf_get_qtimer_count(); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) end_job); - FARF(HIGH, "solve-tri-chunk %d/%d: A=(%ux%u) B=(%ux%u) job %u:%u usec %u\n", - ith, nth, n, n, k, n, start_job, end_job, - (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + FARF(HIGH, "solve-tri-chunk %d/%d: A=(%ux%u) B=(%ux%u) jobs %u:%u\n", + ith, nth, n, n, k, n, start_job, end_job); } int op_solve_tri(struct htp_ops_context * octx) { @@ -218,8 +218,8 @@ int op_solve_tri(struct htp_ops_context * octx) { return HTP_STATUS_INVAL_PARAMS; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { - return HTP_STATUS_OK; + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(src1) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; } const uint32_t k = src1->ne[0]; @@ -235,32 +235,64 @@ int op_solve_tri(struct htp_ops_context * octx) { dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], batched); if (batched) { + uint32_t job_start = 0; + uint32_t njobs = total_batches; + + if (octx->ctx->mdev.count > 1) { + const uint32_t batch_size = dst->nb[2]; + const uint32_t batches_per_chunk = (batch_size > 0) ? (HEX_L2_LINE_SIZE / hex_gcd_u32(batch_size, HEX_L2_LINE_SIZE)) : 1; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_batches, htp_tensor_mdev_data_aligned(dst) ? batches_per_chunk : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + job_start = range.start; + njobs = range.count; + } + + if (njobs == 0) { + return HTP_STATUS_OK; + } + // Batch-level parallelism - const uint32_t n_threads = MIN((uint32_t) octx->n_threads, total_batches); + const uint32_t n_threads = octx->n_threads; struct htp_solve_tri_context sctx = { .octx = octx, - .jobs_per_thread = (total_batches + n_threads - 1) / n_threads, - .total_jobs = total_batches, + .jobs_per_thread = fastdiv(njobs + n_threads - 1, &octx->n_threads_div), + .total_jobs = njobs, + .job_start = job_start, .k_chunks = k_chunks, .col_block = col_block, }; - worker_pool_run_func(octx->ctx->worker_pool, solve_tri_batch_thread_f32, &sctx, n_threads); + work_queue_run(octx->ctx->work_queue, solve_tri_batch_thread_f32, &sctx, n_threads); } else { // Chunk-level parallelism const uint32_t total_jobs = total_batches * k_chunks; - const uint32_t n_threads = MIN((uint32_t) octx->n_threads, MAX(total_jobs, 1)); + + uint32_t job_start = 0; + uint32_t njobs = total_jobs; + + if (octx->ctx->mdev.count > 1) { + const bool can_split = htp_tensor_mdev_data_aligned(dst) && ((dst->nb[1] & (HTP_TENSOR_MDEV_LINE_SIZE - 1)) == 0); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(total_jobs, can_split ? 1 : 0, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + job_start = range.start; + njobs = range.count; + } + + if (njobs == 0) { + return HTP_STATUS_OK; + } + + const uint32_t n_threads = octx->n_threads; struct htp_solve_tri_context sctx = { .octx = octx, - .jobs_per_thread = (total_jobs + n_threads - 1) / n_threads, - .total_jobs = total_jobs, + .jobs_per_thread = fastdiv(njobs + n_threads - 1, &octx->n_threads_div), + .total_jobs = njobs, + .job_start = job_start, .k_chunks = k_chunks, .col_block = col_block, }; - worker_pool_run_func(octx->ctx->worker_pool, solve_tri_chunk_thread_f32, &sctx, n_threads); + work_queue_run(octx->ctx->work_queue, solve_tri_chunk_thread_f32, &sctx, n_threads); } return HTP_STATUS_OK; diff --git a/ggml/src/ggml-hexagon/htp/ssm-conv.c b/ggml/src/ggml-hexagon/htp/ssm-conv.c index a48bc9ed..931aa406 100644 --- a/ggml/src/ggml-hexagon/htp/ssm-conv.c +++ b/ggml/src/ggml-hexagon/htp/ssm-conv.c @@ -4,7 +4,6 @@ #include #include -#include #include #include #include @@ -15,121 +14,22 @@ #define GGML_COMMON_DECL_C #include "ggml-common.h" #include "htp-ctx.h" -#include "hex-dma.h" -#include "htp-ops.h" +#include "dma-queue.h" +#include "hex-profile.h" #include "htp-ops.h" +#include "htp-tensor.h" #include "hvx-utils.h" - -#define htp_ssm_conv_tensors_preamble \ - const struct htp_tensor * restrict src0 = octx->src[0]; \ - const struct htp_tensor * restrict src1 = octx->src[1]; \ - const struct htp_tensor * restrict dst = octx->dst; \ - struct htp_spad * restrict src0_spad = &octx->src0_spad; \ - struct htp_spad * restrict src1_spad = &octx->src1_spad; \ - struct htp_spad * restrict dst_spad = &octx->dst_spad; \ - \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne02 = src0->ne[2]; \ - const uint32_t ne03 = src0->ne[3]; \ - \ - const uint32_t ne10 = src1->ne[0]; \ - const uint32_t ne11 = src1->ne[1]; \ - const uint32_t ne12 = src1->ne[2]; \ - const uint32_t ne13 = src1->ne[3]; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb00 = src0->nb[0]; \ - const uint32_t nb01 = src0->nb[1]; \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t nb10 = src1->nb[0]; \ - const uint32_t nb11 = src1->nb[1]; \ - const uint32_t nb12 = src1->nb[2]; \ - const uint32_t nb13 = src1->nb[3]; \ - \ - const uint32_t nb0 = dst->nb[0]; \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; +#include "ssm-conv.h" struct htp_ssm_conv_context { - struct htp_ops_context * octx; - uint32_t nrows_per_thread; - uint32_t d_inner_tile; - uint64_t t_start; + struct htp_ops_context * octx; + const struct htp_ssm_conv_kernel_params * kparams; + uint32_t nrows_per_thread; + uint32_t d_inner_tile; + uint32_t row_start; + uint32_t nrows; }; -#define htp_ssm_conv_preamble \ - struct htp_ssm_conv_context * scctx = (struct htp_ssm_conv_context *) data; \ - struct htp_ops_context * octx = scctx->octx; \ - htp_ssm_conv_tensors_preamble; \ - dma_queue * dma_queue = octx->ctx->dma[ith]; - -// Scalar FP32 SSM_CONV implementation -static void ssm_conv_thread_f32_f32(unsigned int nth, unsigned int ith, void *data) { - htp_ssm_conv_preamble; - - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); - - const uint32_t d_conv = src1->ne[0]; - const uint32_t d_inner = src0->ne[1]; - const uint32_t n_t = dst->ne[1]; - const uint32_t n_s = dst->ne[2]; - - const uint32_t src0_stride_inner = src0->nb[1] / sizeof(float); // stride for inner dimension - const uint32_t src0_stride_seq = src0->nb[2] / sizeof(float); // stride for sequence dimension - const uint32_t src1_stride_inner = src1->nb[1] / sizeof(float); // stride for inner dimension - const uint32_t dst_stride_token = dst->nb[1] / sizeof(float); // stride for token dimension - const uint32_t dst_stride_seq = dst->nb[2] / sizeof(float); // stride for sequence dimension - - const float * src0_data = (const float *) src0->data; - const float * src1_data = (const float *) src1->data; - float * dst_data = (float *) dst->data; - - // Calculate row range for this thread - const uint32_t d_inner_per_thread = scctx->nrows_per_thread; - const uint32_t d_inner_start = d_inner_per_thread * ith; - const uint32_t d_inner_end = MIN(d_inner_start + d_inner_per_thread, d_inner); - - // No work for this thread - if (d_inner_start >= d_inner_end) { - return; - } - - for (uint32_t i3 = 0; i3 < n_s; ++i3) { - for (uint32_t i2 = 0; i2 < n_t; ++i2) { - for (uint32_t i1 = d_inner_start; i1 < d_inner_end; ++i1) { - float sumf = 0.0f; - - for (uint32_t i0 = 0; i0 < d_conv; ++i0) { - const uint32_t src0_idx = (i2 + i0) + i1 * src0_stride_inner + i3 * src0_stride_seq; - const uint32_t src1_idx = i0 + i1 * src1_stride_inner; - - sumf += src0_data[src0_idx] * src1_data[src1_idx]; - } - - const uint32_t dst_idx = i1 + i2 * dst_stride_token + i3 * dst_stride_seq; - dst_data[dst_idx] = sumf; - } - } - } - - t2 = HAP_perf_get_qtimer_count(); - - FARF(HIGH, "ssm-conv-f32 %d/%d: %ux%ux%ux%u (%u:%u) * %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", - ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], d_inner_start, d_inner_end, - src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], dst->ne[0], dst->ne[1], - dst->ne[2], dst->ne[3], (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); -} - - // In-register 32x32 fp32 transpose using std 5-stage HVX vshuff butterfly. static inline void hvx_transpose_32x32_f32(HVX_Vector m[32]) { HVX_Vector tmp[32]; @@ -179,40 +79,69 @@ static inline void hvx_transpose_32x32_f32(HVX_Vector m[32]) { } } -// HVX FP32 SSM_CONV implementation - channel-vectorized HVX kernel with src0/src1 -// transposed into VTCM. -// -// VTCM layouts (per thread): -// src1_T : {d_inner_stride, d_conv} - staged once per launch (small). -// src0_T : {d_inner_tile, ncs} - staged per d_inner-tile. -// -// d_inner_tile is chosen so that per-thread VTCM stays under the budget. -// Each thread iterates ceil(d_inner_per_thread d_inner_tile) tiles serially. -#define HTP_SSM_CONV_VTCM_BUDGET (1u << 20) // 1 MiB per thread - -// Scalar transpose: src1 {d_conv, d_inner} (DDR) -> {d_inner_stride, d_conv} (VTCM) -static inline void transpose_src1(const float * src1_data, - uint32_t src1_stride_inner, - uint32_t i1_off, - uint32_t d_inner_per_thread, - uint32_t d_inner_stride, - uint32_t d_conv, - float * src1_T) { - for (uint32_t i = 0; i < d_inner_per_thread; ++i) { - const float * src_row = src1_data + (i1_off + i) * src1_stride_inner; +// HVX deinterleave for d_conv == 4: channel-major raw VTCM -> tap-major T VTCM +static inline void hvx_ssm_conv_unpack_to_T_4(const float * raw, float * T, uint32_t d_inner_per_thread, uint32_t d_inner_stride) { + for (uint32_t cb = 0; cb < d_inner_per_thread; cb += VLEN_FP32) { + HVX_Vector v0 = *(const HVX_Vector *)(raw + (cb + 0) * 4); + HVX_Vector v1 = *(const HVX_Vector *)(raw + (cb + 8) * 4); + HVX_Vector v2 = *(const HVX_Vector *)(raw + (cb + 16) * 4); + HVX_Vector v3 = *(const HVX_Vector *)(raw + (cb + 24) * 4); + + HVX_VectorPair p01 = Q6_W_vdeal_VVR(v1, v0, -4); + HVX_VectorPair p23 = Q6_W_vdeal_VVR(v3, v2, -4); + + HVX_VectorPair p_w02 = Q6_W_vdeal_VVR(Q6_V_lo_W(p23), Q6_V_lo_W(p01), -4); + HVX_VectorPair p_w13 = Q6_W_vdeal_VVR(Q6_V_hi_W(p23), Q6_V_hi_W(p01), -4); + + *(HVX_Vector *)(T + 0 * d_inner_stride + cb) = Q6_V_lo_W(p_w02); + *(HVX_Vector *)(T + 1 * d_inner_stride + cb) = Q6_V_lo_W(p_w13); + *(HVX_Vector *)(T + 2 * d_inner_stride + cb) = Q6_V_hi_W(p_w02); + *(HVX_Vector *)(T + 3 * d_inner_stride + cb) = Q6_V_hi_W(p_w13); + } +} + +// HVX transpose for general d_conv <= 32: channel-major raw VTCM -> tap-major T VTCM +static inline void hvx_ssm_conv_unpack_to_T_gen(const float * raw, float * T, uint32_t d_inner_per_thread, uint32_t d_inner_stride, uint32_t d_conv) { + uint32_t __attribute__((aligned(VLEN))) mask_buf[VLEN_FP32] = { 0 }; + for (uint32_t j = 0; j < d_conv; ++j) { + mask_buf[j] = 0xFFFFFFFF; + } + const HVX_Vector mask = *(const HVX_Vector *) mask_buf; + + for (uint32_t cb = 0; cb < d_inner_per_thread; cb += VLEN_FP32) { + const uint32_t cb_n = MIN(VLEN_FP32, d_inner_per_thread - cb); + HVX_Vector sub[32]; + for (uint32_t r = 0; r < cb_n; ++r) { + const float * ch_ptr = raw + (cb + r) * d_conv; + sub[r] = Q6_V_vand_VV(*(const HVX_UVector *) ch_ptr, mask); + } + for (uint32_t r = cb_n; r < 32; ++r) { + sub[r] = hvx_vec_splat_f32(0.0f); + } + + hvx_transpose_32x32_f32(sub); + for (uint32_t j = 0; j < d_conv; ++j) { - src1_T[j * d_inner_stride + i] = src_row[j]; + *(HVX_Vector *)(T + j * d_inner_stride + cb) = sub[j]; } } } -// HVX 32x32 src0 transpose: src0 {ncs, d_inner} (DDR) -> src0_T {d_inner_tile, ncs} (VTCM) +static inline void hvx_ssm_conv_unpack_to_T(const float * raw, float * T, uint32_t d_inner_per_thread, uint32_t d_inner_stride, uint32_t d_conv) { + if (d_conv == 4 && (d_inner_per_thread % VLEN_FP32 == 0)) { + hvx_ssm_conv_unpack_to_T_4(raw, T, d_inner_per_thread, d_inner_stride); + } else { + hvx_ssm_conv_unpack_to_T_gen(raw, T, d_inner_per_thread, d_inner_stride, d_conv); + } +} + +// HVX 32x32 src0 transpose for prefill: src0 {tile_n, ncs} (VTCM) -> src0_T {ncs, d_inner_tile} (VTCM) static inline void transpose_src0_block(const float * src0_block, uint32_t ncs, uint32_t cb_n, uint32_t d_inner_tile, float * src0_T_block_dst, - uint32_t cb /* dst column offset */) { + uint32_t cb) { const uint32_t T_TILE = VLEN_FP32; HVX_Vector __attribute__((aligned(VLEN))) sub[32]; @@ -220,20 +149,15 @@ static inline void transpose_src0_block(const float * src0_block, for (uint32_t t0 = 0; t0 < ncs; t0 += T_TILE) { const uint32_t t_n = MIN(T_TILE, ncs - t0); - // Load 32 rows (channels) of T_TILE samples; pad missing channels with zeros. + uint32_t __attribute__((aligned(VLEN))) mask_buf[VLEN_FP32] = { 0 }; + for (uint32_t k = 0; k < t_n; ++k) { + mask_buf[k] = 0xFFFFFFFF; + } + const HVX_Vector mask = *(const HVX_Vector *) mask_buf; + for (uint32_t r = 0; r < cb_n; ++r) { const float * src_row = src0_block + r * ncs + t0; - if (t_n == T_TILE) { - sub[r] = *(const HVX_UVector *) src_row; - } else { - HVX_Vector v = hvx_vec_splat_f32(0.0f); - hvx_vec_store_u(&v, t_n * sizeof(float), hvx_vec_splat_f32(0.0f)); - - float __attribute__((aligned(VLEN))) tmp[VLEN_FP32] = { 0 }; - for (uint32_t k = 0; k < t_n; ++k) tmp[k] = src_row[k]; - v = *(const HVX_Vector *) tmp; - sub[r] = v; - } + sub[r] = (t_n == T_TILE) ? *(const HVX_UVector *) src_row : Q6_V_vand_VV(*(const HVX_UVector *) src_row, mask); } for (uint32_t r = cb_n; r < T_TILE; ++r) { sub[r] = hvx_vec_splat_f32(0.0f); @@ -241,8 +165,6 @@ static inline void transpose_src0_block(const float * src0_block, hvx_transpose_32x32_f32(sub); - // Store transposed sub-tile to src0_T at offsets (t0 + j) * d_inner_tile + cb. - // Only write the valid t_n rows of the transposed result. for (uint32_t r = 0; r < t_n; ++r) { float * dst = src0_T_block_dst + (t0 + r) * d_inner_tile + cb; if (cb_n == T_TILE) { @@ -254,46 +176,165 @@ static inline void transpose_src0_block(const float * src0_block, } } -static void ssm_conv_thread_f32_f32_hvx(unsigned int nth, unsigned int ith, void *data) { - htp_ssm_conv_preamble; +// Single-row decode worker (n_t == 1) +static void ssm_conv_thread_f32_decode(unsigned int nth, unsigned int ith, void * data) { + struct htp_ssm_conv_context * scctx = (struct htp_ssm_conv_context *) data; + struct htp_ops_context * octx = scctx->octx; + const struct htp_ssm_conv_kernel_params * kparams = scctx->kparams; - uint64_t t1, t2; - t1 = HAP_perf_get_qtimer_count(); + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict src1 = octx->src[1]; + const struct htp_tensor * restrict dst = octx->dst; - const uint32_t d_conv = src1->ne[0]; - const uint32_t d_inner = src0->ne[1]; - const uint32_t n_t = dst->ne[1]; - const uint32_t n_s = dst->ne[2]; - const uint32_t ncs = src0->ne[0]; + dma_queue * dma_q = octx->ctx->dma[ith]; - const uint32_t src0_stride_inner = src0->nb[1] / sizeof(float); - const uint32_t src0_stride_seq = src0->nb[2] / sizeof(float); - const uint32_t src1_stride_inner = src1->nb[1] / sizeof(float); - const uint32_t dst_stride_token = dst->nb[1] / sizeof(float); - const uint32_t dst_stride_seq = dst->nb[2] / sizeof(float); + const uint32_t d_conv = kparams->d_conv; + const uint32_t d_inner = kparams->d_inner; + const uint32_t n_s = kparams->n_s; const uint32_t dr = scctx->nrows_per_thread; - const uint32_t ir0 = dr * ith; - const uint32_t ir1 = MIN(ir0 + dr, d_inner); + const uint32_t ir0 = scctx->row_start + dr * ith; + const uint32_t ir1 = MIN(ir0 + dr, scctx->row_start + scctx->nrows); if (ir0 >= ir1) { return; } const uint32_t d_inner_per_thread = ir1 - ir0; - const uint32_t d_inner_stride = scctx->nrows_per_thread; + const uint32_t d_inner_stride = hex_round_up(d_inner_per_thread, VLEN_FP32); + + const size_t src0_stride_seq_bytes = src0->nb[2]; + const size_t dst_stride_seq_bytes = dst->nb[2]; + + uint8_t * src1_spad_base = octx->src1_spad.data + ith * octx->src1_spad.size_per_thread; + uint8_t * src0_spad_base = octx->src0_spad.data + ith * octx->src0_spad.size_per_thread; + uint8_t * dst_spad_base = octx->dst_spad.data + ith * octx->dst_spad.size_per_thread; + + const size_t weight_bytes = (size_t) d_inner_per_thread * d_conv * sizeof(float); + const size_t weight_raw_size = hex_round_up(weight_bytes, 128); + + float * src1_raw = (float *) src1_spad_base; + float * src1_T = (float *) (src1_spad_base + weight_raw_size); + + float * src0_raw = (float *) src0_spad_base; + float * src0_T = (float *) (src0_spad_base + weight_raw_size); + + float * dst_spad = (float *) dst_spad_base; + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + // 1. Fetch weights src1 from DDR into VTCM via DMA (DMA64-safe) + const dma_addr_t src1_ddr = src1->data + ir0 * d_conv * sizeof(float); + dma_queue_push(dma_q, dma_make_data((uint8_t *) src1_raw, src1_ddr), weight_bytes, weight_bytes, weight_bytes, 1); + dma_queue_pop(dma_q); + + // 2. Unpack/transpose src1_raw into src1_T {d_conv, d_inner_stride} + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); + hvx_ssm_conv_unpack_to_T(src1_raw, src1_T, d_inner_per_thread, d_inner_stride, d_conv); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); + + const size_t input_bytes = (size_t) d_inner_per_thread * d_conv * sizeof(float); + const size_t output_bytes = (size_t) d_inner_per_thread * sizeof(float); + + // 3. Process each sequence + for (uint32_t s = 0; s < n_s; ++s) { + const dma_addr_t src0_ddr = src0->data + s * src0_stride_seq_bytes + ir0 * d_conv * sizeof(float); + dma_queue_push(dma_q, dma_make_data((uint8_t *) src0_raw, src0_ddr), input_bytes, input_bytes, input_bytes, 1); + dma_queue_pop(dma_q); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) s); + hvx_ssm_conv_unpack_to_T(src0_raw, src0_T, d_inner_per_thread, d_inner_stride, d_conv); + + for (uint32_t cb = 0; cb < d_inner_per_thread; cb += VLEN_FP32) { + const uint32_t cb_n = MIN(VLEN_FP32, d_inner_per_thread - cb); + HVX_Vector acc = hvx_vec_splat_f32(0.0f); + for (uint32_t j = 0; j < d_conv; ++j) { + HVX_Vector x = *(const HVX_Vector *)(src0_T + j * d_inner_stride + cb); + HVX_Vector w = *(const HVX_Vector *)(src1_T + j * d_inner_stride + cb); + acc = Q6_Vqf32_vadd_Vqf32Vqf32(acc, Q6_Vqf32_vmpy_VsfVsf(x, w)); + } + HVX_Vector y = Q6_Vsf_equals_Vqf32(acc); + if (cb_n == VLEN_FP32) { + *(HVX_Vector *)(dst_spad + cb) = y; + } else { + hvx_vec_store_u(dst_spad + cb, cb_n * sizeof(float), y); + } + } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) s); + + const dma_addr_t dst_ddr = dst->data + s * dst_stride_seq_bytes + ir0 * sizeof(float); + dma_queue_push(dma_q, dma_make_data(dst_ddr, (uint8_t *) dst_spad), output_bytes, output_bytes, output_bytes, 1); + dma_queue_pop(dma_q); + } + + FARF(HIGH, "ssm-conv-f32-decode %d/%d: %ux%ux%ux%u (%u:%u) * %ux%ux%ux%u -> %ux%ux%ux%u\n", + ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ir0, ir1, + src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], dst->ne[0], dst->ne[1], + dst->ne[2], dst->ne[3]); +} + +// Multi-token prefill worker (n_t > 1) +static void ssm_conv_thread_f32_prefill(unsigned int nth, unsigned int ith, void * data) { + struct htp_ssm_conv_context * scctx = (struct htp_ssm_conv_context *) data; + struct htp_ops_context * octx = scctx->octx; + const struct htp_ssm_conv_kernel_params * kparams = scctx->kparams; + + const struct htp_tensor * restrict src0 = octx->src[0]; + const struct htp_tensor * restrict src1 = octx->src[1]; + const struct htp_tensor * restrict dst = octx->dst; + + dma_queue * dma_q = octx->ctx->dma[ith]; + + const uint32_t d_conv = kparams->d_conv; + const uint32_t d_inner = kparams->d_inner; + const uint32_t n_t = kparams->n_t; + const uint32_t n_s = kparams->n_s; + const uint32_t ncs = src0->ne[0]; + + const uint32_t dr = scctx->nrows_per_thread; + const uint32_t ir0 = scctx->row_start + dr * ith; + const uint32_t ir1 = MIN(ir0 + dr, scctx->row_start + scctx->nrows); + + if (ir0 >= ir1) { + return; + } + + const uint32_t d_inner_per_thread = ir1 - ir0; + const uint32_t d_inner_stride = hex_round_up(d_inner_per_thread, VLEN_FP32); const uint32_t d_inner_tile = scctx->d_inner_tile; - const float * src0_data = (const float *) src0->data; - const float * src1_data = (const float *) src1->data; - float * dst_data = (float *) dst->data; + const size_t src0_stride_inner_bytes = src0->nb[1]; + const size_t src0_stride_seq_bytes = src0->nb[2]; + const size_t dst_stride_token_bytes = dst->nb[1]; + const size_t dst_stride_seq_bytes = dst->nb[2]; + + uint8_t * src1_spad_base = octx->src1_spad.data + ith * octx->src1_spad.size_per_thread; + uint8_t * src0_spad_base = octx->src0_spad.data + ith * octx->src0_spad.size_per_thread; + uint8_t * dst_spad_base = octx->dst_spad.data + ith * octx->dst_spad.size_per_thread; - // Per-thread VTCM regions. - float * src0_T = (float *)(octx->src0_spad.data + ith * octx->src0_spad.size_per_thread); - float * src1_T = (float *)(octx->src1_spad.data + ith * octx->src1_spad.size_per_thread); + const size_t weight_bytes = (size_t) d_inner_per_thread * d_conv * sizeof(float); + const size_t weight_raw_size = hex_round_up(weight_bytes, 128); - // Stage src1 weights once into VTCM in {d_inner_stride, d_conv} layout. - transpose_src1(src1_data, src1_stride_inner, ir0, d_inner_per_thread, d_inner_stride, d_conv, src1_T); + float * src1_raw = (float *) src1_spad_base; + float * src1_T = (float *) (src1_spad_base + weight_raw_size); + + const size_t src0_tile_raw_bytes = hex_round_up(d_inner_tile * ncs * sizeof(float), 128); + float * src0_tile_raw = (float *) src0_spad_base; + float * src0_T = (float *) (src0_spad_base + src0_tile_raw_bytes); + + float * dst_tile = (float *) dst_spad_base; + + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + // 1. Fetch weights src1 from DDR into VTCM via DMA (DMA64-safe) + const dma_addr_t src1_ddr = src1->data + ir0 * d_conv * sizeof(float); + dma_queue_push(dma_q, dma_make_data((uint8_t *) src1_raw, src1_ddr), weight_bytes, weight_bytes, weight_bytes, 1); + dma_queue_pop(dma_q); + + // 2. Unpack/transpose src1_raw into src1_T {d_conv, d_inner_stride} + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); + hvx_ssm_conv_unpack_to_T(src1_raw, src1_T, d_inner_per_thread, d_inner_stride, d_conv); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) ir0); const uint32_t C_TILE = VLEN_FP32; @@ -301,14 +342,24 @@ static void ssm_conv_thread_f32_f32_hvx(unsigned int nth, unsigned int ith, void for (uint32_t tile_off = 0; tile_off < d_inner_per_thread; tile_off += d_inner_tile) { const uint32_t tile_n = MIN(d_inner_tile, d_inner_per_thread - tile_off); - // Place src0 chunk into VTCM in {d_inner_tile, ncs} layout. - const float * src0_block = src0_data + i3 * src0_stride_seq + (ir0 + tile_off) * src0_stride_inner; + // Fetch src0 chunk from DDR to VTCM via 2D DMA + const dma_addr_t src0_tile_ddr = src0->data + + i3 * src0_stride_seq_bytes + + (ir0 + tile_off) * src0_stride_inner_bytes; + const size_t row_bytes = ncs * sizeof(float); + + dma_queue_push(dma_q, dma_make_data((uint8_t *) src0_tile_raw, src0_tile_ddr), + row_bytes, src0_stride_inner_bytes, row_bytes, tile_n); + dma_queue_pop(dma_q); + // Transpose src0 chunk in VTCM into {d_inner_tile, ncs} layout + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) tile_off); for (uint32_t cb = 0; cb < tile_n; cb += C_TILE) { const uint32_t cb_n = MIN(C_TILE, tile_n - cb); - transpose_src0_block(src0_block + cb * src0_stride_inner, ncs, cb_n, d_inner_tile, src0_T, cb); + transpose_src0_block(src0_tile_raw + cb * ncs, ncs, cb_n, d_inner_tile, src0_T, cb); } + // Compute convolution for (uint32_t t = 0; t < n_t; ++t) { for (uint32_t cb = 0; cb < tile_n; cb += C_TILE) { const uint32_t cb_n = MIN(C_TILE, tile_n - cb); @@ -319,97 +370,115 @@ static void ssm_conv_thread_f32_f32_hvx(unsigned int nth, unsigned int ith, void HVX_Vector w = *(const HVX_Vector *) (src1_T + j * d_inner_stride + tile_off + cb); acc = Q6_Vqf32_vadd_Vqf32Vqf32(acc, Q6_Vqf32_vmpy_VsfVsf(x, w)); } - HVX_Vector res = Q6_Vsf_equals_Vqf32(acc); - float * dst_ptr = dst_data + i3 * dst_stride_seq + t * dst_stride_token + (ir0 + tile_off + cb); + HVX_Vector y = Q6_Vsf_equals_Vqf32(acc); + float * dst_tile_ptr = dst_tile + t * tile_n + cb; if (cb_n == C_TILE) { - *(HVX_UVector *) dst_ptr = res; + *(HVX_Vector *) dst_tile_ptr = y; } else { - hvx_vec_store_u(dst_ptr, cb_n * sizeof(float), res); + hvx_vec_store_u(dst_tile_ptr, cb_n * sizeof(float), y); } } } + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) tile_off); + + // Writeback dst_tile from VTCM to DDR via 2D DMA + const dma_addr_t dst_tile_ddr = dst->data + + i3 * dst_stride_seq_bytes + + (ir0 + tile_off) * sizeof(float); + const size_t dst_row_bytes = tile_n * sizeof(float); + + dma_queue_push(dma_q, dma_make_data(dst_tile_ddr, (uint8_t *) dst_tile), + dst_stride_token_bytes, dst_row_bytes, dst_row_bytes, n_t); + dma_queue_pop(dma_q); } } - t2 = HAP_perf_get_qtimer_count(); - - FARF(HIGH, "ssm-conv-f32-hvx %d/%d: %ux%ux%ux%u (%u:%u) tile=%u * %ux%ux%ux%u -> %ux%ux%ux%u usec %u\n", - ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ir0, ir1, d_inner_tile, + FARF(HIGH, "ssm-conv-f32-prefill %d/%d: %ux%ux%ux%u (%u:%u) * %ux%ux%ux%u -> %ux%ux%ux%u\n", + ith, nth, src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], ir0, ir1, src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], dst->ne[0], dst->ne[1], - dst->ne[2], dst->ne[3], (unsigned) HAP_perf_qtimer_count_to_us(t2 - t1)); + dst->ne[2], dst->ne[3]); } int op_ssm_conv_f32(struct htp_ops_context * octx) { - htp_ssm_conv_tensors_preamble; + const struct htp_tensor * src0 = octx->src[0]; + const struct htp_tensor * src1 = octx->src[1]; + const struct htp_tensor * dst = octx->dst; if (src0->type != HTP_TYPE_F32 || src1->type != HTP_TYPE_F32 || dst->type != HTP_TYPE_F32) { - FARF(ERROR, "ssm_conv: only (F32 x F32 -> F32) OPs supported"); return HTP_STATUS_NO_SUPPORT; } - struct htp_ssm_conv_context scctx = { 0 }; - scctx.octx = octx; - - const uint32_t d_conv = src1->ne[0]; - const uint32_t d_inner = src0->ne[1]; - const uint32_t n_t = dst->ne[1]; // tokens per sequence - const uint32_t n_s = dst->ne[2]; // number of sequences in the batch - - const uint32_t n_threads = MIN(octx->n_threads, d_inner); - - if (!(octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - uint32_t use_hvx = 0; - if (d_inner >= VLEN_FP32 && n_t >= VLEN_FP32) { - use_hvx = 1; - } - - scctx.nrows_per_thread = hex_round_up((d_inner + n_threads - 1) / n_threads, VLEN_FP32); - - const uint32_t d_inner_per_thread = scctx.nrows_per_thread; - const uint32_t ncs = src0->ne[0]; - - const uint32_t src1_T_size = hex_round_up(d_conv * d_inner_per_thread * sizeof(float), 256); - const uint32_t src0_T_max = HTP_SSM_CONV_VTCM_BUDGET > src1_T_size ? HTP_SSM_CONV_VTCM_BUDGET - src1_T_size : 0; - - uint32_t d_inner_tile = (src0_T_max / sizeof(float)) / ncs; - d_inner_tile -= (d_inner_tile % VLEN_FP32); - if (d_inner_tile == 0) { - FARF(HIGH, "ssm_conv-f32: inner tile rounds to 0 (ncs=%u), falling back to scalar\n", ncs); - use_hvx = 0; - } else { - scctx.d_inner_tile = d_inner_tile; + const struct htp_ssm_conv_kernel_params * kparams = (const struct htp_ssm_conv_kernel_params *) octx->kernel_params; - octx->src0_spad.size_per_thread = hex_round_up(d_inner_tile * ncs * sizeof(float), 256); - octx->src1_spad.size_per_thread = src1_T_size; - octx->dst_spad.size_per_thread = 0; - octx->src0_spad.size = octx->src0_spad.size_per_thread * n_threads; - octx->src1_spad.size = octx->src1_spad.size_per_thread * n_threads; - octx->dst_spad.size = 0; + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } - octx->src0_spad.data = octx->ctx->vtcm_base; - octx->src1_spad.data = octx->src0_spad.data + octx->src0_spad.size; - octx->src0_spad.src = NULL; - octx->src1_spad.src = NULL; + uint32_t row_start = 0; + uint32_t nrows = kparams->d_inner; + + if (octx->ctx->mdev.count > 1) { + const uint32_t elems_per_chunk = VLEN_FP32; + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition( + kparams->d_inner, + htp_tensor_mdev_data_aligned(dst) ? elems_per_chunk : 0, + octx->ctx->mdev.idx, + octx->ctx->mdev.count, + &octx->ctx->mdev.count_div + ); + row_start = range.start; + nrows = range.count; + } - const size_t total_spad = octx->src0_spad.size + octx->src1_spad.size; - if (total_spad > octx->ctx->vtcm_size) { - FARF(HIGH, "ssm_conv-f32: scratchpad %zu exceeds VTCM %zu, falling back to scalar\n", - total_spad, octx->ctx->vtcm_size); - use_hvx = 0; - } - } + if (nrows == 0) { + return HTP_STATUS_OK; + } - FARF(HIGH, "ssm-conv-f32: (%ux%ux%ux%u) x (%ux%ux%ux%u) -> (%ux%ux%ux%u) : use_hvx %d\n", src0->ne[0], - src0->ne[1], src0->ne[2], src0->ne[3], src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], dst->ne[0], - dst->ne[1], dst->ne[2], dst->ne[3], use_hvx); + if (kparams->vtcm_size > octx->ctx->vtcm_size) { + return HTP_STATUS_VTCM_TOO_SMALL; + } - if (use_hvx) { - worker_pool_run_func(octx->ctx->worker_pool, ssm_conv_thread_f32_f32_hvx, &scctx, n_threads); - } else { - worker_pool_run_func(octx->ctx->worker_pool, ssm_conv_thread_f32_f32, &scctx, n_threads); - } + const uint32_t n_threads = octx->n_threads; + + octx->src0_spad.size_per_thread = kparams->vtcm_src0_size_per_thread; + octx->src1_spad.size_per_thread = kparams->vtcm_src1_size_per_thread; + octx->dst_spad.size_per_thread = kparams->vtcm_dst_size_per_thread; + + octx->src0_spad.size = kparams->vtcm_src0_size; + octx->src1_spad.size = kparams->vtcm_src1_size; + octx->dst_spad.size = kparams->vtcm_dst_size; + + octx->src0_spad.data = octx->ctx->vtcm_base; + octx->src1_spad.data = octx->src0_spad.data + octx->src0_spad.size; + octx->dst_spad.data = octx->src1_spad.data + octx->src1_spad.size; + octx->src0_spad.src = NULL; + octx->src1_spad.src = NULL; + octx->dst_spad.src = NULL; + + const uint32_t raw_rpt = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); + const uint32_t d_inner_per_thread = hex_round_up(raw_rpt, VLEN_FP32); + + struct htp_ssm_conv_context scctx = { + .octx = octx, + .kparams = kparams, + .nrows_per_thread = d_inner_per_thread, + .d_inner_tile = kparams->d_inner_tile, + .row_start = row_start, + .nrows = nrows, + }; + + FARF(HIGH, "ssm-conv-f32: (%ux%ux%ux%u) x (%ux%ux%ux%u) -> (%ux%ux%ux%u) : mode %s\n", + src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], + src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], + dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], + kparams->n_t == 1 ? "decode" : "prefill"); + + if (kparams->n_t == 1) { + work_queue_run(octx->ctx->work_queue, ssm_conv_thread_f32_decode, &scctx, n_threads); + } else { + work_queue_run(octx->ctx->work_queue, ssm_conv_thread_f32_prefill, &scctx, n_threads); } return HTP_STATUS_OK; @@ -418,16 +487,10 @@ int op_ssm_conv_f32(struct htp_ops_context * octx) { int op_ssm_conv(struct htp_ops_context * octx) { const struct htp_tensor * dst = octx->dst; - int err = HTP_STATUS_OK; - switch (dst->type) { case HTP_TYPE_F32: - err = op_ssm_conv_f32(octx); - break; + return op_ssm_conv_f32(octx); default: - err = HTP_STATUS_NO_SUPPORT; - break; + return HTP_STATUS_NO_SUPPORT; } - - return err; } diff --git a/ggml/src/ggml-hexagon/htp/ssm-conv.h b/ggml/src/ggml-hexagon/htp/ssm-conv.h new file mode 100644 index 00000000..be62d7bf --- /dev/null +++ b/ggml/src/ggml-hexagon/htp/ssm-conv.h @@ -0,0 +1,40 @@ +#ifndef HTP_SSM_CONV_H +#define HTP_SSM_CONV_H + +#include + +#include "hex-fastdiv.h" +#include "htp-ops.h" + +struct htp_ssm_conv_kernel_params { + uint32_t n_threads; + uint32_t d_conv; + uint32_t d_inner; + uint32_t n_t; + uint32_t n_s; + uint32_t d_inner_per_thread; + uint32_t d_inner_tile; + + uint32_t src0_row_size_aligned; + uint32_t src1_row_size_aligned; + uint32_t dst_row_size_aligned; + + uint32_t vtcm_src0_size_per_thread; + uint32_t vtcm_src1_size_per_thread; + uint32_t vtcm_dst_size_per_thread; + + uint32_t vtcm_src0_size; + uint32_t vtcm_src1_size; + uint32_t vtcm_dst_size; + uint32_t vtcm_size; + + struct fastdiv_values div_n_threads; +}; + +#if defined(__cplusplus) +static_assert(sizeof(struct htp_ssm_conv_kernel_params) <= 128, "htp_ssm_conv_kernel_params is too large for kernel_params blob"); +#else +_Static_assert(sizeof(struct htp_ssm_conv_kernel_params) <= 128, "htp_ssm_conv_kernel_params is too large for kernel_params blob"); +#endif + +#endif // HTP_SSM_CONV_H diff --git a/ggml/src/ggml-hexagon/htp/sum-rows-ops.c b/ggml/src/ggml-hexagon/htp/sum-rows-ops.c index 874c41ab..9b8e04a0 100644 --- a/ggml/src/ggml-hexagon/htp/sum-rows-ops.c +++ b/ggml/src/ggml-hexagon/htp/sum-rows-ops.c @@ -8,40 +8,43 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hvx-utils.h" #define GGML_COMMON_DECL_C #include "ggml-common.h" +#include "hex-common.h" +#include "hex-profile.h" #include "htp-ctx.h" #include "htp-ops.h" -#include "htp-ops.h" +#include "htp-tensor.h" #define sum_rows_preamble \ const struct htp_tensor *src0 = octx->src[0]; \ const struct htp_tensor *dst = octx->dst; \ \ - const uint32_t ne00 = src0->ne[0]; \ - const uint32_t ne01 = src0->ne[1]; \ - const uint32_t ne02 = src0->ne[2]; \ - const uint32_t ne03 = src0->ne[3]; \ - \ - const uint32_t nb00 = src0->nb[0]; \ - const uint32_t nb01 = src0->nb[1]; \ - const uint32_t nb02 = src0->nb[2]; \ - const uint32_t nb03 = src0->nb[3]; \ - \ - const uint32_t ne0 = dst->ne[0]; \ - const uint32_t ne1 = dst->ne[1]; \ - const uint32_t ne2 = dst->ne[2]; \ - const uint32_t ne3 = dst->ne[3]; \ - \ - const uint32_t nb0 = dst->nb[0]; \ - const uint32_t nb1 = dst->nb[1]; \ - const uint32_t nb2 = dst->nb[2]; \ - const uint32_t nb3 = dst->nb[3]; \ + const uint32_t ne00 = src0->ne[0]; \ + const uint32_t ne01 = src0->ne[1]; \ + const uint32_t ne02 = src0->ne[2]; \ + const uint32_t ne03 = src0->ne[3]; \ + \ + const uint32_t nb00 = src0->nb[0]; \ + const uint32_t nb01 = src0->nb[1]; \ + const uint32_t nb02 = src0->nb[2]; \ + const uint32_t nb03 = src0->nb[3]; \ + \ + const uint32_t ne0 = dst->ne[0]; \ + const uint32_t ne1 = dst->ne[1]; \ + const uint32_t ne2 = dst->ne[2]; \ + const uint32_t ne3 = dst->ne[3]; \ + \ + const uint32_t nb0 = dst->nb[0]; \ + const uint32_t nb1 = dst->nb[1]; \ + const uint32_t nb2 = dst->nb[2]; \ + const uint32_t nb3 = dst->nb[3]; \ struct sum_rows_context { + struct htp_ops_context * octx; const uint8_t * src_data; uint8_t * dst_data; uint32_t ne00; @@ -76,6 +79,9 @@ static void sum_rows_thread_f32(unsigned int nth, unsigned int ith, void *data) // Calculate actual number of rows for this thread const uint32_t n_rows = end_row - start_row; + struct htp_thread_trace * tr = &smctx->octx->ctx->trace[ith]; + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) start_row); + for (uint32_t ir = 0; ir < n_rows; ir++) { const float * restrict src_local = src_th + (ir * (src_stride / sizeof(float))); @@ -89,6 +95,8 @@ static void sum_rows_thread_f32(unsigned int nth, unsigned int ith, void *data) dst_th[ir] = hvx_reduce_sum_f32((const uint8_t *) src_local, ne00); } } + + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, (uint16_t) start_row); } int op_sum_rows(struct htp_ops_context * octx) { @@ -98,13 +106,30 @@ int op_sum_rows(struct htp_ops_context * octx) { return HTP_STATUS_NO_SUPPORT; } - if (octx->flags & HTP_OPFLAGS_SKIP_COMPUTE) { + if (htp_tensor_is_extended(src0) || htp_tensor_is_extended(dst)) { + return HTP_STATUS_NO_SUPPORT; + } + + const uint32_t src0_nrows = ne01 * ne02 * ne03; + const size_t dst_data_row_size = dst->ne[0] * sizeof(float); + + uint32_t row_start = 0; + uint32_t nrows = src0_nrows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, sizeof(float), (uint32_t) dst_data_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { return HTP_STATUS_OK; } - const uint32_t src0_nrows = ne01 * ne02 * ne03; - const uint32_t n_threads = MIN(octx->n_threads, src0_nrows); - const uint32_t rows_per_thread = (src0_nrows + n_threads - 1) / n_threads; + const uint32_t n_threads = octx->n_threads; + const uint32_t rows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div); bool opt_path = false; if ((0 == hex_is_aligned((void *) src0->data, VLEN)) && !(nb01 & (VLEN - 1))) { @@ -112,17 +137,18 @@ int op_sum_rows(struct htp_ops_context * octx) { } struct sum_rows_context smctx = { - .src_data = (const uint8_t *) src0->data, - .dst_data = (uint8_t *) dst->data, + .octx = octx, + .src_data = (const uint8_t *) src0->data + row_start * nb01, + .dst_data = (uint8_t *) dst->data + row_start * nb1, .ne00 = ne00, .src_stride = nb01, .dst_stride = nb1, .rows_per_thread = rows_per_thread, - .total_rows = src0_nrows, + .total_rows = nrows, .opt_path = opt_path, }; - worker_pool_run_func(octx->ctx->worker_pool, sum_rows_thread_f32, &smctx, n_threads); + work_queue_run(octx->ctx->work_queue, sum_rows_thread_f32, &smctx, n_threads); return HTP_STATUS_OK; } diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.c b/ggml/src/ggml-hexagon/htp/unary-ops.c index b21415a6..9a1479e2 100644 --- a/ggml/src/ggml-hexagon/htp/unary-ops.c +++ b/ggml/src/ggml-hexagon/htp/unary-ops.c @@ -8,7 +8,7 @@ #include #include -#include "hex-dma.h" +#include "dma-queue.h" #include "hex-fastdiv.h" #include "hvx-exp.h" #include "hvx-sigmoid.h" @@ -23,13 +23,47 @@ #include "htp-vtcm.h" #include "hex-profile.h" +struct htp_unary_context; + +typedef void (*unary_compute_fn_t)(const void * restrict src, + void * restrict dst, + uint32_t num_rows, + const struct htp_unary_context * uctx); + +typedef void (*unary_rms_norm_mul_compute_fn_t)(const void * restrict src, + const void * restrict weight, + void * restrict dst, + uint32_t num_rows, + const struct htp_unary_context * uctx); + +typedef void (*unary_tri_compute_fn_t)(const void * restrict src, + void * restrict dst, + uint32_t num_rows, + uint32_t ir, + const struct htp_unary_context * uctx); + +typedef void (*unary_tile_compute_fn_t)(void * restrict dst, + const void * restrict src, + uint32_t tw, + const struct htp_unary_context * uctx); + +typedef void (*unary_tiled_tri_compute_fn_t)(const void * restrict src, + void * restrict dst, + uint32_t tile_elems, + uint32_t col_start, + uint32_t i01, + uint32_t ne0, + int32_t ttype); + struct htp_unary_context { struct htp_ops_context * octx; const struct htp_unary_kernel_params * kparams; - const uint8_t * data_src0; - const uint8_t * data_src1; // weight/scale tensor for RMS_NORM_MUL - uint8_t * data_dst; + void * compute; + + dma_addr_t data_src0; + dma_addr_t data_src1; // weight/scale tensor for RMS_NORM_MUL + dma_addr_t data_dst; size_t src0_data_row_size; // actual data bytes per row size_t src1_data_row_size; @@ -46,6 +80,7 @@ struct htp_unary_context { uint32_t block; uint32_t src0_nrows; uint32_t src0_nrows_per_thread; + uint32_t row_start; uint32_t nc; uint32_t col_tile; // tiled mode bool broadcast_weight; @@ -120,8 +155,8 @@ static inline uint32_t unary_block_size(uint32_t ir, const size_t src0_row_size_aligned = uctx->src0_row_size_aligned; \ const size_t dst_row_size_aligned = uctx->dst_row_size_aligned; -static void scale_f32(const float * restrict src, - float * restrict dst, +static void scale_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -138,8 +173,8 @@ static void scale_f32(const float * restrict src, } } -static void clamp_f32(const float * restrict src, - float * restrict dst, +static void clamp_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -156,8 +191,24 @@ static void clamp_f32(const float * restrict src, } } -static void rms_norm_f32(const float * restrict src, - float * restrict dst, +static void leaky_relu_f32(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float negative_slope = 0.f; + memcpy(&negative_slope, &op_params[0], sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_leaky_relu_scalar_f32(dst_local, src_local, negative_slope, ne0); + } +} + +static void rms_norm_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -172,9 +223,9 @@ static void rms_norm_f32(const float * restrict src, } } -static void rms_norm_mul_f32(const float * restrict src, - const float * restrict weight, - float * restrict dst, +static void rms_norm_mul_f32(const void * restrict src, + const void * restrict weight, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -190,8 +241,8 @@ static void rms_norm_mul_f32(const float * restrict src, } } -static void norm_f32(const float * restrict src, - float * restrict dst, +static void norm_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -206,8 +257,8 @@ static void norm_f32(const float * restrict src, } } -static void sqr_f32(const float * restrict src, - float * restrict dst, +static void sqr_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -220,8 +271,8 @@ static void sqr_f32(const float * restrict src, } } -static void sqrt_f32(const float * restrict src, - float * restrict dst, +static void sqrt_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -234,8 +285,148 @@ static void sqrt_f32(const float * restrict src, } } -static void neg_f32(const float * restrict src, - float * restrict dst, +static void scale_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float scale = 0.f; + float bias = 0.f; + memcpy(&scale, &op_params[0], sizeof(float)); + memcpy(&bias, &op_params[1], sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_scale_offset_f16_aa((uint8_t *) dst_local, (const uint8_t *) src_local, ne0, scale, bias); + } +} + +static void clamp_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float min = 0.f; + float max = 0.f; + memcpy(&min, &op_params[0], sizeof(float)); + memcpy(&max, &op_params[1], sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_clamp_scalar_f16(dst_local, src_local, (_Float16) min, (_Float16) max, ne0); + } +} + +static void rms_norm_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float epsilon = 0.f; + memcpy(&epsilon, op_params, sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_fast_rms_norm_f16((const uint8_t *) src_local, (uint8_t *) dst_local, ne0, epsilon); + } +} + +static void norm_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float epsilon = 0.f; + memcpy(&epsilon, op_params, sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_fast_norm_f16((const uint8_t *) src_local, (uint8_t *) dst_local, ne0, epsilon); + } +} + +static void sqr_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_sqr_f16_aa((uint8_t *) dst_local, (const uint8_t *) src_local, ne0); + } +} + +static void sqrt_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_sqrt_f16_aa((uint8_t *) dst_local, (const uint8_t *) src_local, ne0); + } +} + +static void abs_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_abs_f16_aa((uint8_t *) dst_local, (const uint8_t *) src_local, ne0); + } +} + +static void log_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_log_f16_aa((uint8_t *) dst_local, (const uint8_t *) src_local, ne0); + } +} + +static void l2_norm_f16(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + float epsilon = 0.f; + memcpy(&epsilon, op_params, sizeof(float)); + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_f = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_f = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_fast_l2_norm_f16((const uint8_t *)src_f, (uint8_t *)dst_f, ne0, epsilon); + } +} + +static void neg_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -248,8 +439,8 @@ static void neg_f32(const float * restrict src, } } -static void exp_f32(const float * restrict src, - float * restrict dst, +static void exp_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -262,8 +453,8 @@ static void exp_f32(const float * restrict src, } } -static void sigmoid_f32(const float * restrict src, - float * restrict dst, +static void sigmoid_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -277,8 +468,8 @@ static void sigmoid_f32(const float * restrict src, } // silu(x) = x * sigmoid(x) -static void silu_f32(const float * restrict src, - float * restrict dst, +static void silu_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -293,8 +484,8 @@ static void silu_f32(const float * restrict src, } // gelu(x) = x * sigmoid(1.702 * x) (quick/sigmoid approximation, matches CPU GELU_QUICK reference) -static void gelu_f32(const float * restrict src, - float * restrict dst, +static void gelu_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -309,8 +500,8 @@ static void gelu_f32(const float * restrict src, } } -static void tri_f32(const float * restrict src, - float * restrict dst, +static void tri_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const uint32_t ir, const struct htp_unary_context * uctx) { @@ -340,7 +531,7 @@ static void tri_f32(const float * restrict src, } if (boundary > ne0) boundary = ne0; - // Full HVX vectors — each starts at a 128-byte aligned offset + // Full HVX vectors - each starts at a 128-byte aligned offset for (uint32_t i = 0; i < nvec; i++) { const uint32_t vec_start = i * VLEN_FP32; const uint32_t vec_end = vec_start + VLEN_FP32; @@ -394,8 +585,8 @@ static void tri_f32(const float * restrict src, } } -static void softplus_f32(const float * restrict src, - float * restrict dst, +static void softplus_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -407,14 +598,14 @@ static void softplus_f32(const float * restrict src, for (uint32_t i = 0; i < ne0; i++) { float x = src_f[i]; - // For x > 20: softplus(x) ≈ x (avoids exp overflow) + // For x > 20: softplus(x) ~ x (avoids exp overflow) dst_f[i] = (x > 20.0f) ? x : logf(1.0f + expf(x)); } } } -static void l2_norm_f32(const float * restrict src, - float * restrict dst, +static void l2_norm_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -422,15 +613,15 @@ static void l2_norm_f32(const float * restrict src, memcpy(&epsilon, op_params, sizeof(float)); for (uint32_t ir = 0; ir < num_rows; ir++) { - const float * restrict src_f = (const float *)((const uint8_t *)src + (ir * src0_row_size_aligned)); - float * restrict dst_f = (float *)((uint8_t *)dst + (ir * dst_row_size_aligned)); + const uint8_t * restrict src_f = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_f = (uint8_t *)dst + (ir * dst_row_size_aligned); hvx_fast_l2_norm_f32((const uint8_t *)src_f, (uint8_t *)dst_f, ne0, epsilon); } } -static void tanh_f32(const float * restrict src, - float * restrict dst, +static void tanh_f32(const void * restrict src, + void * restrict dst, const uint32_t num_rows, const struct htp_unary_context * uctx) { htp_unary_op_preamble; @@ -443,333 +634,142 @@ static void tanh_f32(const float * restrict src, } } -#define DEFINE_UNARY_TASK(NAME, IS_RMS_NORM_MUL, IS_TRI, CORE_EXPR) \ -static void unary_task_f32_##NAME(unsigned int nth, unsigned int ith, void * data) { \ - const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; \ - struct htp_ops_context * octx = uctx->octx; \ - const struct htp_tensor * src = octx->src[0]; \ - const struct htp_tensor * dst = octx->dst; \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - htp_unary_preamble; \ - \ - int32_t * op_params = octx->op_params; \ - uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; \ - \ - const size_t src0_data_row_size = uctx->src0_data_row_size; \ - const size_t dst_data_row_size = uctx->dst_data_row_size; \ - \ - const size_t src0_row_size_aligned = uctx->src0_row_size_aligned; \ - const size_t dst_row_size_aligned = uctx->dst_row_size_aligned; \ - \ - const uint32_t src0_nrows = uctx->src0_nrows; \ - const uint32_t src0_start_row = src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, src0_nrows); \ - \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - const uint8_t * restrict data_src = uctx->data_src0; \ - const uint8_t * restrict data_src1 = uctx->data_src1; \ - uint8_t * restrict data_dst = uctx->data_dst; \ - \ - const struct htp_tensor * src1 = (IS_RMS_NORM_MUL) ? octx->src[1] : NULL; \ - const uint32_t nb11 = src1 ? src1->nb[1] : 0; \ - const uint32_t nb12 = src1 ? src1->nb[2] : 0; \ - const uint32_t nb13 = src1 ? src1->nb[3] : 0; \ - const bool src1_contig = src1 ? ((nb12 == (size_t)ne01 * nb11) && (nb13 == (size_t)ne02 * nb12)) : false; \ - \ - uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); \ - uint8_t * src1_vtcm_data = uctx->vtcm_src1 ? (uctx->vtcm_src1 + (ith * uctx->vtcm_src1_size_per_thread)) : NULL;\ - uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); \ - \ - size_t src0_vtcm_half_size = uctx->src0_vtcm_half_size; \ - size_t src1_vtcm_half_size = uctx->src1_vtcm_half_size; \ - size_t dst_vtcm_half_size = uctx->dst_vtcm_half_size; \ - \ - const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && \ - (nb03 == (size_t)ne02 * nb02); \ - const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && \ - (nb3 == (size_t)ne2 * nb2); \ - \ - const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; \ - const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; \ - const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; \ - \ - const uint32_t src0_max_block = src0_contig ? uctx->block : MIN((uint32_t)uctx->block, ne01); \ - const uint32_t dst_max_block = dst_contig ? uctx->block : MIN((uint32_t)uctx->block, ne1); \ - const uint32_t BLOCK = MIN(src0_max_block, dst_max_block); \ - if (BLOCK == 0) { \ - FARF(ERROR, "unary-f32 : current VTCM reservation %zu is too small, needed at least %zu\n", \ - uctx->vtcm_src0_size_per_thread, src0_row_size_aligned); \ - return; \ - } \ - \ - dma_queue * dma_queue = octx->ctx->dma[ith]; \ - \ - if ((IS_RMS_NORM_MUL) && uctx->broadcast_weight) { \ - dma_queue_push(dma_queue, dma_make_ptr(src1_vtcm_data, data_src1), \ - uctx->src1_row_size_aligned, 0, uctx->src1_data_row_size, 1); \ - dma_queue_flush(dma_queue); \ - } \ - \ - for (uint32_t ir = src0_start_row, vtcm_idx = 0; ir < src0_end_row && vtcm_idx < 2; vtcm_idx++) { \ - const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, ne01, \ - div_ne01); \ - \ - dma_queue_push(dma_queue, \ - dma_make_ptr(data_dst, dst_vtcm_data + (vtcm_idx * dst_vtcm_half_size)), \ - nb1, dst_row_size_aligned, dst_data_row_size, 0); \ - \ - const size_t src0_off = src0_contig ? (ir * nb01) : \ - unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); \ - dma_queue_push(dma_queue, \ - dma_make_ptr(src0_vtcm_data + (vtcm_idx * src0_vtcm_half_size), data_src + src0_off), \ - src0_row_size_aligned, nb01, src0_data_row_size, block_size); \ - \ - if ((IS_RMS_NORM_MUL) && !uctx->broadcast_weight) { \ - const size_t src1_off = src1_contig ? (ir * nb11) : \ - unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb11, nb12, nb13); \ - dma_queue_push(dma_queue, \ - dma_make_ptr(src1_vtcm_data + (vtcm_idx * src1_vtcm_half_size), data_src1 + src1_off), \ - uctx->src1_row_size_aligned, nb11, uctx->src1_data_row_size, block_size); \ - } \ - \ - ir += block_size; \ - } \ - \ - for (uint32_t ir = src0_start_row; ir < src0_end_row; ) { \ - const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, ne01, \ - div_ne01); \ - \ - float * dst_vtcm = (float *) dma_queue_pop(dma_queue).src; \ - float * src0_vtcm = (float *) dma_queue_pop(dma_queue).dst; \ - float * src1_vtcm = NULL; \ - if ((IS_RMS_NORM_MUL) && !uctx->broadcast_weight) { \ - src1_vtcm = (float *) dma_queue_pop(dma_queue).dst; \ - } \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); \ - CORE_EXPR; \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); \ - \ - const size_t dst_off = dst_contig ? (ir * nb1) : \ - unary_row_offset(ir, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3); \ - dma_queue_push(dma_queue, \ - dma_make_ptr(data_dst + dst_off, dst_vtcm), \ - nb1, dst_row_size_aligned, dst_data_row_size, block_size); \ - \ - const uint32_t next_ir = ir + block_size; \ - if (next_ir < src0_end_row) { \ - const uint32_t next_block_size = unary_block_size(next_ir, src0_end_row, BLOCK, src0_contig, dst_contig,\ - ne01, div_ne01); \ - const uint32_t pref_ir = next_ir + next_block_size; \ - if (pref_ir < src0_end_row) { \ - const uint32_t pref_block_size = unary_block_size(pref_ir, src0_end_row, BLOCK, src0_contig, \ - dst_contig, ne01, div_ne01); \ - const size_t src0_pref_off = src0_contig ? (pref_ir * nb01) : \ - unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); \ - dma_queue_push(dma_queue, \ - dma_make_ptr(src0_vtcm, data_src + src0_pref_off), \ - src0_row_size_aligned, nb01, src0_data_row_size, pref_block_size); \ - \ - if ((IS_RMS_NORM_MUL) && !uctx->broadcast_weight) { \ - const size_t src1_pref_off = src1_contig ? (pref_ir * nb11) : \ - unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb11, nb12, nb13); \ - dma_queue_push(dma_queue, \ - dma_make_ptr(src1_vtcm, data_src1 + src1_pref_off), \ - uctx->src1_row_size_aligned, nb11, uctx->src1_data_row_size, pref_block_size); \ - } \ - } \ - } \ - ir += block_size; \ - } \ - \ - dma_queue_flush(dma_queue); \ -} - -DEFINE_UNARY_TASK(norm, false, false, norm_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(rms_norm, false, false, rms_norm_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(rms_norm_mul, true, false, rms_norm_mul_f32(src0_vtcm, uctx->broadcast_weight ? (const float *) src1_vtcm_data : src1_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(scale, false, false, scale_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(clamp, false, false, clamp_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(sqr, false, false, sqr_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(sqrt, false, false, sqrt_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_neg, false, false, neg_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_exp, false, false, exp_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_sigmoid, false, false, sigmoid_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_silu, false, false, silu_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_gelu, false, false, gelu_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_softplus, false, false, softplus_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(unary_tanh, false, false, tanh_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(l2_norm, false, false, l2_norm_f32(src0_vtcm, dst_vtcm, block_size, uctx)) -DEFINE_UNARY_TASK(tri, false, true, tri_f32(src0_vtcm, dst_vtcm, block_size, ir, uctx)) - -// Apply a pointwise unary op to one column tile that is already in VTCM. -#define DEFINE_UNARY_TILED_TASK(NAME, IS_TRI, CORE_TILE_EXPR) \ -static void unary_task_f32_tiled_##NAME(unsigned int nth, unsigned int ith, void * data) { \ - const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; \ - struct htp_ops_context * octx = uctx->octx; \ - const struct htp_tensor * src = octx->src[0]; \ - const struct htp_tensor * dst = octx->dst; \ - struct htp_thread_trace * tr = &octx->ctx->trace[ith]; \ - \ - htp_unary_preamble; \ - \ - int32_t * op_params = octx->op_params; \ - const uint32_t col_tile = uctx->col_tile; \ - \ - const uint32_t src0_nrows = uctx->src0_nrows; \ - const uint32_t src0_start_row = uctx->src0_nrows_per_thread * ith; \ - const uint32_t src0_end_row = MIN(src0_start_row + uctx->src0_nrows_per_thread, src0_nrows); \ - \ - if (src0_start_row >= src0_end_row) { \ - return; \ - } \ - \ - const uint8_t * restrict data_src = uctx->data_src0; \ - uint8_t * restrict data_dst = uctx->data_dst; \ - \ - uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); \ - uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); \ - \ - const size_t src0_half = uctx->src0_vtcm_half_size; \ - const size_t dst_half = uctx->dst_vtcm_half_size; \ - \ - dma_queue * dmaq = octx->ctx->dma[ith]; \ - \ - const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; \ - const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; \ - const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; \ - const struct fastdiv_values * div_tpr = &uctx->kparams->div_tpr; \ - \ - const uint32_t tiles_per_row = (ne0 + col_tile - 1) / col_tile; \ - const int32_t tri_ttype = (IS_TRI) ? op_params[0] : 0; \ - \ - const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && \ - (nb03 == (size_t)ne02 * nb02); \ - const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && \ - (nb3 == (size_t)ne2 * nb2); \ - \ - const uint32_t total_tiles = (src0_end_row - src0_start_row) * tiles_per_row; \ - \ - for (uint32_t t = 0, vtcm_idx = 0; t < total_tiles && vtcm_idx < 2; t++, vtcm_idx++) { \ - const uint32_t row = src0_start_row + t / tiles_per_row; \ - const uint32_t col = (t % tiles_per_row) * col_tile; \ - const uint32_t tw = MIN(col_tile, ne0 - col); \ - const size_t tb = (size_t) tw * sizeof(float); \ - const size_t soff = (src0_contig ? (row * nb01) : \ - unary_row_offset(row, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03)) +\ - (size_t) col * sizeof(float); \ - \ - dma_queue_push(dmaq, dma_make_ptr(data_dst, dst_vtcm_data + (vtcm_idx * dst_half)), 0, 0, 0, 0); \ - dma_queue_push(dmaq, dma_make_ptr(src0_vtcm_data + (vtcm_idx * src0_half), data_src + soff), tb, tb, tb, 1);\ - } \ - \ - uint32_t row = src0_start_row; \ - uint32_t col = 0; \ - uint32_t tile_in_row = 0; \ - uint32_t i01 = fastmodulo(row, ne01, div_ne01); \ - \ - uint32_t prow = src0_start_row + fastdiv(2, div_tpr); \ - uint32_t pcol = fastmodulo(2, tiles_per_row, div_tpr) * col_tile; \ - uint32_t ptile_in_row = fastmodulo(2, tiles_per_row, div_tpr); \ - \ - for (uint32_t t = 0; t < total_tiles; t++) { \ - uint8_t * dst_vtcm = (uint8_t *) dma_queue_pop(dmaq).src; \ - uint8_t * src_vtcm = (uint8_t *) dma_queue_pop(dmaq).dst; \ - \ - const uint32_t tw = MIN(col_tile, ne0 - col); \ - \ - htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, t); \ - CORE_TILE_EXPR; \ - htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, t); \ - \ - const size_t doff = (dst_contig ? (row * nb1) : \ - unary_row_offset(row, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3)) + \ - (size_t) col * sizeof(float); \ - const size_t tb = (size_t) tw * sizeof(float); \ - dma_queue_push(dmaq, dma_make_ptr(data_dst + doff, dst_vtcm), tb, tb, tb, 1); \ - \ - const uint32_t pt = t + 2; \ - if (pt < total_tiles) { \ - const uint32_t ptw = MIN(col_tile, ne0 - pcol); \ - const size_t ptb = (size_t) ptw * sizeof(float); \ - const size_t psoff = (src0_contig ? (prow * nb01) : \ - unary_row_offset(prow, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, \ - nb03)) + \ - (size_t) pcol * sizeof(float); \ - dma_queue_push(dmaq, dma_make_ptr(src_vtcm, data_src + psoff), ptb, ptb, ptb, 1); \ - } \ - \ - tile_in_row++; \ - col += col_tile; \ - if (tile_in_row == tiles_per_row) { \ - tile_in_row = 0; \ - col = 0; \ - row++; \ - i01++; \ - if (i01 == ne01) { \ - i01 = 0; \ - } \ - } \ - \ - ptile_in_row++; \ - pcol += col_tile; \ - if (ptile_in_row == tiles_per_row) { \ - ptile_in_row = 0; \ - pcol = 0; \ - prow++; \ - } \ - } \ - \ - dma_queue_flush(dmaq); \ -} - -static inline void tile_scale_f32(uint8_t * dst_vtcm, const uint8_t * src_vtcm, uint32_t tw, const int32_t * op_params) { +static void abs_f32(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_abs_f32_aa(dst_local, src_local, ne0); + } +} + +static void relu_f32(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_max_scalar_f32(dst_local, src_local, 0.0f, ne0); + } +} + +static void log_f32(const void * restrict src, + void * restrict dst, + const uint32_t num_rows, + const struct htp_unary_context * uctx) { + htp_unary_op_preamble; + + for (uint32_t ir = 0; ir < num_rows; ir++) { + const uint8_t * restrict src_local = (const uint8_t *)src + (ir * src0_row_size_aligned); + uint8_t * restrict dst_local = (uint8_t *)dst + (ir * dst_row_size_aligned); + + hvx_log_f32_aa(dst_local, src_local, ne0); + } +} + +#// Pointwise unary ops on one column tile in VTCM. +static void tile_scale_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { float scale = 0.f; - float bias = 0.f; - memcpy(&scale, &op_params[0], sizeof(float)); - memcpy(&bias, &op_params[1], sizeof(float)); - hvx_scale_offset_f32_aa(dst_vtcm, src_vtcm, tw, scale, bias); + float bias = 0.f; + memcpy(&scale, &uctx->octx->op_params[0], sizeof(float)); + memcpy(&bias, &uctx->octx->op_params[1], sizeof(float)); + hvx_scale_offset_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw, scale, bias); } -static inline void tile_clamp_f32(uint8_t * dst_vtcm, const uint8_t * src_vtcm, uint32_t tw, const int32_t * op_params) { +static void tile_clamp_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { float min = 0.f; float max = 0.f; - memcpy(&min, &op_params[0], sizeof(float)); - memcpy(&max, &op_params[1], sizeof(float)); - hvx_clamp_scalar_f32(dst_vtcm, src_vtcm, min, max, tw); + memcpy(&min, &uctx->octx->op_params[0], sizeof(float)); + memcpy(&max, &uctx->octx->op_params[1], sizeof(float)); + hvx_clamp_scalar_f32((uint8_t *) dst, (const uint8_t *) src, min, max, tw); +} + +static void tile_leaky_relu_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + float negative_slope = 0.f; + memcpy(&negative_slope, &uctx->octx->op_params[0], sizeof(float)); + hvx_leaky_relu_scalar_f32((uint8_t *) dst, (const uint8_t *) src, negative_slope, tw); +} + +static void tile_sqr_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_sqr_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); +} + +static void tile_sqrt_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_sqrt_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); +} + +static void tile_neg_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_scale_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw, -1.0f); } -static inline void tile_unary_softplus_f32(uint8_t * dst_vtcm, const uint8_t * src_vtcm, uint32_t tw) { - const float * restrict sf = (const float *) src_vtcm; - float * restrict df = (float *) dst_vtcm; +static void tile_exp_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_exp_f32((uint8_t *) dst, (const uint8_t *) src, tw, false); +} + +static void tile_sigmoid_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_sigmoid_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); +} + +static void tile_silu_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_sigmoid_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); + hvx_mul_f32_aaa((uint8_t *) dst, (const uint8_t *) src, (uint8_t *) dst, tw); +} + +static void tile_gelu_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_mul_scalar_f32((uint8_t *) dst, (const uint8_t *) src, 1.702f, tw); + hvx_sigmoid_f32_aa((uint8_t *) dst, (uint8_t *) dst, tw); + hvx_mul_f32_aaa((uint8_t *) dst, (const uint8_t *) src, (uint8_t *) dst, tw); +} + +static void tile_softplus_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + const float * restrict sf = (const float *) src; + float * restrict df = (float *) dst; for (uint32_t i = 0; i < tw; i++) { float x = sf[i]; df[i] = (x > 20.0f) ? x : logf(1.0f + expf(x)); } } -// silu(x) = x * sigmoid(x) -static inline void tile_silu_f32(uint8_t * dst_vtcm, const uint8_t * src_vtcm, uint32_t tw) { - hvx_sigmoid_f32_aa(dst_vtcm, src_vtcm, tw); - hvx_mul_f32_aaa(dst_vtcm, src_vtcm, dst_vtcm, tw); +static void tile_tanh_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_tanh_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); } -// gelu(x) = x * sigmoid(1.702 * x) (quick/sigmoid approximation, matches CPU GELU_QUICK reference) -static inline void tile_gelu_f32(uint8_t * dst_vtcm, const uint8_t * src_vtcm, uint32_t tw) { - hvx_mul_scalar_f32(dst_vtcm, src_vtcm, 1.702f, tw); - hvx_sigmoid_f32_aa(dst_vtcm, dst_vtcm, tw); - hvx_mul_f32_aaa(dst_vtcm, src_vtcm, dst_vtcm, tw); +static void tile_abs_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_abs_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); } -// Triangular mask applied to one column tile. Boundary is an absolute column index, so -// each vector compares against its absolute column position (col_start + i*VLEN_FP32). -static inline void tri_apply_tile_f32(const uint8_t * restrict src, uint8_t * restrict dst, - uint32_t tile_elems, uint32_t col_start, uint32_t i01, - uint32_t ne0, int32_t ttype) { +static void tile_log_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_log_f32_aa((uint8_t *) dst, (const uint8_t *) src, tw); +} + +static void tile_relu_f32(void * restrict dst, const void * restrict src, uint32_t tw, const struct htp_unary_context * uctx) { + (void) uctx; + hvx_max_scalar_f32((uint8_t *) dst, (const uint8_t *) src, 0.0f, tw); +} + +static void tri_apply_tile_f32(const void * restrict src, void * restrict dst, + uint32_t tile_elems, uint32_t col_start, uint32_t i01, + uint32_t ne0, int32_t ttype) { const HVX_Vector * restrict v_src = (const HVX_Vector *) src; HVX_Vector * restrict v_dst = (HVX_Vector *) dst; const HVX_Vector zero = hvx_vec_splat_f32(0.0f); @@ -839,61 +839,713 @@ static inline void tri_apply_tile_f32(const uint8_t * restrict src, uint8_t * re } } -DEFINE_UNARY_TILED_TASK(scale, false, tile_scale_f32(dst_vtcm, src_vtcm, tw, op_params)) -DEFINE_UNARY_TILED_TASK(clamp, false, tile_clamp_f32(dst_vtcm, src_vtcm, tw, op_params)) -DEFINE_UNARY_TILED_TASK(sqr, false, hvx_sqr_f32_aa(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(sqrt, false, hvx_sqrt_f32_aa(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(unary_neg, false, hvx_scale_f32_aa(dst_vtcm, src_vtcm, tw, -1.0f)) -DEFINE_UNARY_TILED_TASK(unary_exp, false, hvx_exp_f32(dst_vtcm, src_vtcm, tw, false)) -DEFINE_UNARY_TILED_TASK(unary_sigmoid, false, hvx_sigmoid_f32_aa(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(unary_silu, false, tile_silu_f32(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(unary_gelu, false, tile_gelu_f32(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(unary_softplus, false, tile_unary_softplus_f32(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(unary_tanh, false, hvx_tanh_f32_aa(dst_vtcm, src_vtcm, tw)) -DEFINE_UNARY_TILED_TASK(tri, true, tri_apply_tile_f32(src_vtcm, dst_vtcm, tw, col, i01, ne0, tri_ttype)) +// 1. Standard row-block unary task (F32 and F16). +static void unary_thread_row_block(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; + struct htp_ops_context * octx = uctx->octx; + const struct htp_tensor * src = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + htp_unary_preamble; + + const uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; + const size_t src0_data_row_size = uctx->src0_data_row_size; + const size_t dst_data_row_size = uctx->dst_data_row_size; + const size_t src0_row_size_aligned = uctx->src0_row_size_aligned; + const size_t dst_row_size_aligned = uctx->dst_row_size_aligned; + + const uint32_t src0_nrows = uctx->src0_nrows; + const uint32_t src0_start_row = uctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, uctx->row_start + src0_nrows); + + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src = uctx->data_src0; + const dma_addr_t data_dst = uctx->data_dst; + + uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); + uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); + + const size_t src0_vtcm_half_size = uctx->src0_vtcm_half_size; + const size_t dst_vtcm_half_size = uctx->dst_vtcm_half_size; + + const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && + (nb03 == (size_t)ne02 * nb02); + const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && + (nb3 == (size_t)ne2 * nb2); + + const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; + const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; + const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; + + const uint32_t src0_max_block = src0_contig ? uctx->block : MIN((uint32_t)uctx->block, ne01); + const uint32_t dst_max_block = dst_contig ? uctx->block : MIN((uint32_t)uctx->block, ne1); + const uint32_t BLOCK = MIN(src0_max_block, dst_max_block); + if (BLOCK == 0) { + FARF(ERROR, "unary-row-block : current VTCM reservation %zu is too small, needed at least %zu\n", + uctx->vtcm_src0_size_per_thread, src0_row_size_aligned); + return; + } + + dma_queue * dma_q = octx->ctx->dma[ith]; + + for (uint32_t ir = src0_start_row, vtcm_idx = 0; ir < src0_end_row && vtcm_idx < 2; vtcm_idx++) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, + ne01, div_ne01); + + dma_queue_push(dma_q, + dma_make_data(data_dst, dst_vtcm_data + (vtcm_idx * dst_vtcm_half_size)), + nb1, dst_row_size_aligned, dst_data_row_size, 0); + + const size_t src0_off = src0_contig ? (ir * nb01) : + unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm_data + (vtcm_idx * src0_vtcm_half_size), data_src + src0_off), + src0_row_size_aligned, nb01, src0_data_row_size, block_size); + + ir += block_size; + } + + unary_compute_fn_t compute = (unary_compute_fn_t) uctx->compute; + + for (uint32_t ir = src0_start_row; ir < src0_end_row; ) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, + ne01, div_ne01); + + void * dst_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * src0_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); + compute(src0_vtcm, dst_vtcm, block_size, uctx); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); + + const size_t dst_off = dst_contig ? (ir * nb1) : + unary_row_offset(ir, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3); + dma_queue_push(dma_q, + dma_make_data(data_dst + dst_off, dst_vtcm), + nb1, dst_row_size_aligned, dst_data_row_size, block_size); + + const uint32_t next_ir = ir + block_size; + if (next_ir < src0_end_row) { + const uint32_t next_block_size = unary_block_size(next_ir, src0_end_row, BLOCK, src0_contig, + dst_contig, ne01, div_ne01); + const uint32_t pref_ir = next_ir + next_block_size; + if (pref_ir < src0_end_row) { + const uint32_t pref_block_size = unary_block_size(pref_ir, src0_end_row, BLOCK, src0_contig, + dst_contig, ne01, div_ne01); + const size_t src0_pref_off = src0_contig ? (pref_ir * nb01) : + unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm, data_src + src0_pref_off), + src0_row_size_aligned, nb01, src0_data_row_size, pref_block_size); + } + } + ir += block_size; + } + + dma_queue_flush(dma_q); +} + +// 2. RMS_NORM_MUL row-block task with weight buffer. +static void unary_thread_rms_norm_mul_f32(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; + struct htp_ops_context * octx = uctx->octx; + const struct htp_tensor * src = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + htp_unary_preamble; + + const uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; + const size_t src0_data_row_size = uctx->src0_data_row_size; + const size_t dst_data_row_size = uctx->dst_data_row_size; + const size_t src0_row_size_aligned = uctx->src0_row_size_aligned; + const size_t dst_row_size_aligned = uctx->dst_row_size_aligned; + + const uint32_t src0_nrows = uctx->src0_nrows; + const uint32_t src0_start_row = uctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, uctx->row_start + src0_nrows); + + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src = uctx->data_src0; + const dma_addr_t data_src1 = uctx->data_src1; + const dma_addr_t data_dst = uctx->data_dst; + + const struct htp_tensor * src1 = octx->src[1]; + const uint32_t nb11 = src1->nb[1]; + const uint32_t nb12 = src1->nb[2]; + const uint32_t nb13 = src1->nb[3]; + const uint32_t nb11_bc = (src1->ne[1] > 1) ? nb11 : 0; + const uint32_t nb12_bc = (src1->ne[2] > 1) ? nb12 : 0; + const uint32_t nb13_bc = (src1->ne[3] > 1) ? nb13 : 0; + const bool src1_contig = ((nb12 == (size_t)ne01 * nb11) && (nb13 == (size_t)ne02 * nb12)); + + uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); + uint8_t * src1_vtcm_data = uctx->vtcm_src1 ? (uctx->vtcm_src1 + (ith * uctx->vtcm_src1_size_per_thread)) : NULL; + uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); + + const size_t src0_vtcm_half_size = uctx->src0_vtcm_half_size; + const size_t src1_vtcm_half_size = uctx->src1_vtcm_half_size; + const size_t dst_vtcm_half_size = uctx->dst_vtcm_half_size; + + const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && + (nb03 == (size_t)ne02 * nb02); + const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && + (nb3 == (size_t)ne2 * nb2); + + const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; + const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; + const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; + + const bool src1_needs_row_clip = !uctx->broadcast_weight && !src1_contig; + const bool block_src0_contig = src0_contig && !src1_needs_row_clip; + const bool block_dst_contig = dst_contig && !src1_needs_row_clip; + + const uint32_t src0_max_block = block_src0_contig ? uctx->block : MIN((uint32_t)uctx->block, ne01); + const uint32_t dst_max_block = block_dst_contig ? uctx->block : MIN((uint32_t)uctx->block, ne1); + const uint32_t BLOCK = MIN(src0_max_block, dst_max_block); + if (BLOCK == 0) { + FARF(ERROR, "unary-rms-norm-mul : current VTCM reservation %zu is too small, needed at least %zu\n", + uctx->vtcm_src0_size_per_thread, src0_row_size_aligned); + return; + } + + dma_queue * dma_q = octx->ctx->dma[ith]; + + if (uctx->broadcast_weight) { + dma_queue_push(dma_q, dma_make_data(src1_vtcm_data, data_src1), + uctx->src1_row_size_aligned, 0, uctx->src1_data_row_size, 1); + dma_queue_flush(dma_q); + } + + for (uint32_t ir = src0_start_row, vtcm_idx = 0; ir < src0_end_row && vtcm_idx < 2; vtcm_idx++) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, block_src0_contig, block_dst_contig, + ne01, div_ne01); + + dma_queue_push(dma_q, + dma_make_data(data_dst, dst_vtcm_data + (vtcm_idx * dst_vtcm_half_size)), + nb1, dst_row_size_aligned, dst_data_row_size, 0); + + const size_t src0_off = src0_contig ? (ir * nb01) : + unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm_data + (vtcm_idx * src0_vtcm_half_size), data_src + src0_off), + src0_row_size_aligned, nb01, src0_data_row_size, block_size); + + if (!uctx->broadcast_weight) { + const size_t src1_off = src1_contig ? (ir * nb11) : + unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb11_bc, nb12_bc, nb13_bc); + dma_queue_push(dma_q, + dma_make_data(src1_vtcm_data + (vtcm_idx * src1_vtcm_half_size), data_src1 + src1_off), + uctx->src1_row_size_aligned, nb11, uctx->src1_data_row_size, block_size); + } + + ir += block_size; + } + + unary_rms_norm_mul_compute_fn_t compute = (unary_rms_norm_mul_compute_fn_t) uctx->compute; + + for (uint32_t ir = src0_start_row; ir < src0_end_row; ) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, block_src0_contig, block_dst_contig, + ne01, div_ne01); + + void * dst_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * src0_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + void * src1_vtcm = NULL; + if (!uctx->broadcast_weight) { + src1_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + } + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); + const void * w = uctx->broadcast_weight ? (const void *) src1_vtcm_data : src1_vtcm; + compute(src0_vtcm, w, dst_vtcm, block_size, uctx); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); + + const size_t dst_off = dst_contig ? (ir * nb1) : + unary_row_offset(ir, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3); + dma_queue_push(dma_q, + dma_make_data(data_dst + dst_off, dst_vtcm), + nb1, dst_row_size_aligned, dst_data_row_size, block_size); + + const uint32_t next_ir = ir + block_size; + if (next_ir < src0_end_row) { + const uint32_t next_block_size = unary_block_size(next_ir, src0_end_row, BLOCK, block_src0_contig, + block_dst_contig, ne01, div_ne01); + const uint32_t pref_ir = next_ir + next_block_size; + if (pref_ir < src0_end_row) { + const uint32_t pref_block_size = unary_block_size(pref_ir, src0_end_row, BLOCK, block_src0_contig, + block_dst_contig, ne01, div_ne01); + const size_t src0_pref_off = src0_contig ? (pref_ir * nb01) : + unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm, data_src + src0_pref_off), + src0_row_size_aligned, nb01, src0_data_row_size, pref_block_size); + + if (!uctx->broadcast_weight) { + const size_t src1_pref_off = src1_contig ? (pref_ir * nb11) : + unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb11_bc, nb12_bc, + nb13_bc); + dma_queue_push(dma_q, + dma_make_data(src1_vtcm, data_src1 + src1_pref_off), + uctx->src1_row_size_aligned, nb11, uctx->src1_data_row_size, pref_block_size); + } + } + } + ir += block_size; + } + + dma_queue_flush(dma_q); +} + +// 3. TRI row-block task with row index ir. +static void unary_thread_tri_f32(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; + struct htp_ops_context * octx = uctx->octx; + const struct htp_tensor * src = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + htp_unary_preamble; + + const uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; + const size_t src0_data_row_size = uctx->src0_data_row_size; + const size_t dst_data_row_size = uctx->dst_data_row_size; + const size_t src0_row_size_aligned = uctx->src0_row_size_aligned; + const size_t dst_row_size_aligned = uctx->dst_row_size_aligned; + + const uint32_t src0_nrows = uctx->src0_nrows; + const uint32_t src0_start_row = uctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, uctx->row_start + src0_nrows); + + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src = uctx->data_src0; + const dma_addr_t data_dst = uctx->data_dst; + + uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); + uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); -static int execute_op_unary_f32(struct htp_ops_context * octx) { + const size_t src0_vtcm_half_size = uctx->src0_vtcm_half_size; + const size_t dst_vtcm_half_size = uctx->dst_vtcm_half_size; + + const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && + (nb03 == (size_t)ne02 * nb02); + const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && + (nb3 == (size_t)ne2 * nb2); + + const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; + const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; + const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; + + const uint32_t src0_max_block = src0_contig ? uctx->block : MIN((uint32_t)uctx->block, ne01); + const uint32_t dst_max_block = dst_contig ? uctx->block : MIN((uint32_t)uctx->block, ne1); + const uint32_t BLOCK = MIN(src0_max_block, dst_max_block); + if (BLOCK == 0) { + FARF(ERROR, "unary-tri : current VTCM reservation %zu is too small, needed at least %zu\n", + uctx->vtcm_src0_size_per_thread, src0_row_size_aligned); + return; + } + + dma_queue * dma_q = octx->ctx->dma[ith]; + + for (uint32_t ir = src0_start_row, vtcm_idx = 0; ir < src0_end_row && vtcm_idx < 2; vtcm_idx++) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, + ne01, div_ne01); + + dma_queue_push(dma_q, + dma_make_data(data_dst, dst_vtcm_data + (vtcm_idx * dst_vtcm_half_size)), + nb1, dst_row_size_aligned, dst_data_row_size, 0); + + const size_t src0_off = src0_contig ? (ir * nb01) : + unary_row_offset(ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm_data + (vtcm_idx * src0_vtcm_half_size), data_src + src0_off), + src0_row_size_aligned, nb01, src0_data_row_size, block_size); + + ir += block_size; + } + + unary_tri_compute_fn_t compute = (unary_tri_compute_fn_t) uctx->compute; + + for (uint32_t ir = src0_start_row; ir < src0_end_row; ) { + const uint32_t block_size = unary_block_size(ir, src0_end_row, BLOCK, src0_contig, dst_contig, + ne01, div_ne01); + + void * dst_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * src0_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, ir); + compute(src0_vtcm, dst_vtcm, block_size, ir, uctx); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, ir); + + const size_t dst_off = dst_contig ? (ir * nb1) : + unary_row_offset(ir, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3); + dma_queue_push(dma_q, + dma_make_data(data_dst + dst_off, dst_vtcm), + nb1, dst_row_size_aligned, dst_data_row_size, block_size); + + const uint32_t next_ir = ir + block_size; + if (next_ir < src0_end_row) { + const uint32_t next_block_size = unary_block_size(next_ir, src0_end_row, BLOCK, src0_contig, + dst_contig, ne01, div_ne01); + const uint32_t pref_ir = next_ir + next_block_size; + if (pref_ir < src0_end_row) { + const uint32_t pref_block_size = unary_block_size(pref_ir, src0_end_row, BLOCK, src0_contig, + dst_contig, ne01, div_ne01); + const size_t src0_pref_off = src0_contig ? (pref_ir * nb01) : + unary_row_offset(pref_ir, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03); + dma_queue_push(dma_q, + dma_make_data(src0_vtcm, data_src + src0_pref_off), + src0_row_size_aligned, nb01, src0_data_row_size, pref_block_size); + } + } + ir += block_size; + } + + dma_queue_flush(dma_q); +} + +// 4. Pointwise tiled unary task. +static void unary_thread_tiled(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; + struct htp_ops_context * octx = uctx->octx; + const struct htp_tensor * src = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + htp_unary_preamble; + + const uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; + const uint32_t col_tile = uctx->col_tile; + + const uint32_t src0_nrows = uctx->src0_nrows; + const uint32_t src0_start_row = uctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, uctx->row_start + src0_nrows); + + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src = uctx->data_src0; + const dma_addr_t data_dst = uctx->data_dst; + + uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); + uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); + + const size_t src0_half = uctx->src0_vtcm_half_size; + const size_t dst_half = uctx->dst_vtcm_half_size; + + dma_queue * dma_q = octx->ctx->dma[ith]; + + const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; + const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; + const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; + const struct fastdiv_values * div_tpr = &uctx->kparams->div_tpr; + + const uint32_t tiles_per_row = (ne0 + col_tile - 1) / col_tile; + + const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && + (nb03 == (size_t)ne02 * nb02); + const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && + (nb3 == (size_t)ne2 * nb2); + + const uint32_t total_tiles = (src0_end_row - src0_start_row) * tiles_per_row; + + for (uint32_t t = 0, vtcm_idx = 0; t < total_tiles && vtcm_idx < 2; t++, vtcm_idx++) { + const uint32_t row = src0_start_row + t / tiles_per_row; + const uint32_t col = (t % tiles_per_row) * col_tile; + const uint32_t tw = MIN(col_tile, ne0 - col); + const size_t tb = (size_t) tw * sizeof(float); + const size_t soff = (src0_contig ? (row * nb01) : + unary_row_offset(row, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03)) + + (size_t) col * sizeof(float); + + dma_queue_push(dma_q, dma_make_data(data_dst, dst_vtcm_data + (vtcm_idx * dst_half)), 0, 0, 0, 0); + dma_queue_push(dma_q, dma_make_data(src0_vtcm_data + (vtcm_idx * src0_half), data_src + soff), tb, tb, tb, 1); + } + + unary_tile_compute_fn_t compute = (unary_tile_compute_fn_t) uctx->compute; + + uint32_t row = src0_start_row; + uint32_t col = 0; + uint32_t tile_in_row = 0; + + uint32_t prow = src0_start_row + fastdiv(2, div_tpr); + uint32_t pcol = fastmodulo(2, tiles_per_row, div_tpr) * col_tile; + uint32_t ptile_in_row = fastmodulo(2, tiles_per_row, div_tpr); + + for (uint32_t t = 0; t < total_tiles; t++) { + void * dst_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * src_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + + const uint32_t tw = MIN(col_tile, ne0 - col); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, t); + compute(dst_vtcm, src_vtcm, tw, uctx); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, t); + + const size_t doff = (dst_contig ? (row * nb1) : + unary_row_offset(row, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3)) + + (size_t) col * sizeof(float); + const size_t tb = (size_t) tw * sizeof(float); + dma_queue_push(dma_q, dma_make_data(data_dst + doff, dst_vtcm), tb, tb, tb, 1); + + const uint32_t pt = t + 2; + if (pt < total_tiles) { + const uint32_t ptw = MIN(col_tile, ne0 - pcol); + const size_t ptb = (size_t) ptw * sizeof(float); + const size_t psoff = (src0_contig ? (prow * nb01) : + unary_row_offset(prow, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, + nb03)) + + (size_t) pcol * sizeof(float); + dma_queue_push(dma_q, dma_make_data(src_vtcm, data_src + psoff), ptb, ptb, ptb, 1); + } + + tile_in_row++; + col += col_tile; + if (tile_in_row == tiles_per_row) { + tile_in_row = 0; + col = 0; + row++; + } + + ptile_in_row++; + pcol += col_tile; + if (ptile_in_row == tiles_per_row) { + ptile_in_row = 0; + pcol = 0; + prow++; + } + } + + dma_queue_flush(dma_q); +} + +// 5. TRI tiled task. +static void unary_thread_tiled_tri_f32(unsigned int nth, unsigned int ith, void * data) { + (void) nth; + const struct htp_unary_context * uctx = (const struct htp_unary_context *) data; + struct htp_ops_context * octx = uctx->octx; + const struct htp_tensor * src = octx->src[0]; + const struct htp_tensor * dst = octx->dst; + struct htp_thread_trace * tr = &octx->ctx->trace[ith]; + + htp_unary_preamble; + + const uint32_t src0_nrows_per_thread = uctx->src0_nrows_per_thread; + const int32_t * op_params = octx->op_params; + const uint32_t col_tile = uctx->col_tile; + + const uint32_t src0_nrows = uctx->src0_nrows; + const uint32_t src0_start_row = uctx->row_start + src0_nrows_per_thread * ith; + const uint32_t src0_end_row = MIN(src0_start_row + src0_nrows_per_thread, uctx->row_start + src0_nrows); + + if (src0_start_row >= src0_end_row) { + return; + } + + const dma_addr_t data_src = uctx->data_src0; + const dma_addr_t data_dst = uctx->data_dst; + + uint8_t * src0_vtcm_data = uctx->vtcm_src0 + (ith * uctx->vtcm_src0_size_per_thread); + uint8_t * dst_vtcm_data = uctx->vtcm_dst + (ith * uctx->vtcm_dst_size_per_thread); + + const size_t src0_half = uctx->src0_vtcm_half_size; + const size_t dst_half = uctx->dst_vtcm_half_size; + + dma_queue * dma_q = octx->ctx->dma[ith]; + + const struct fastdiv_values * div_ne01 = &uctx->kparams->div_ne01; + const struct fastdiv_values * div_ne02 = &uctx->kparams->div_ne02; + const struct fastdiv_values * div_ne012 = &uctx->kparams->div_ne012; + const struct fastdiv_values * div_tpr = &uctx->kparams->div_tpr; + + const uint32_t tiles_per_row = (ne0 + col_tile - 1) / col_tile; + const int32_t tri_ttype = op_params[0]; + + const bool src0_contig = (nb02 == (size_t)ne01 * nb01) && + (nb03 == (size_t)ne02 * nb02); + const bool dst_contig = (nb2 == (size_t)ne1 * nb1) && + (nb3 == (size_t)ne2 * nb2); + + const uint32_t total_tiles = (src0_end_row - src0_start_row) * tiles_per_row; + + for (uint32_t t = 0, vtcm_idx = 0; t < total_tiles && vtcm_idx < 2; t++, vtcm_idx++) { + const uint32_t row = src0_start_row + t / tiles_per_row; + const uint32_t col = (t % tiles_per_row) * col_tile; + const uint32_t tw = MIN(col_tile, ne0 - col); + const size_t tb = (size_t) tw * sizeof(float); + const size_t soff = (src0_contig ? (row * nb01) : + unary_row_offset(row, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, nb03)) + + (size_t) col * sizeof(float); + + dma_queue_push(dma_q, dma_make_data(data_dst, dst_vtcm_data + (vtcm_idx * dst_half)), 0, 0, 0, 0); + dma_queue_push(dma_q, dma_make_data(src0_vtcm_data + (vtcm_idx * src0_half), data_src + soff), tb, tb, tb, 1); + } + + unary_tiled_tri_compute_fn_t compute = (unary_tiled_tri_compute_fn_t) uctx->compute; + + uint32_t row = src0_start_row; + uint32_t col = 0; + uint32_t tile_in_row = 0; + uint32_t i01 = fastmodulo(row, ne01, div_ne01); + + uint32_t prow = src0_start_row + fastdiv(2, div_tpr); + uint32_t pcol = fastmodulo(2, tiles_per_row, div_tpr) * col_tile; + uint32_t ptile_in_row = fastmodulo(2, tiles_per_row, div_tpr); + + for (uint32_t t = 0; t < total_tiles; t++) { + void * dst_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).src; + void * src_vtcm = (void *) (uintptr_t) dma_queue_pop(dma_q).dst; + + const uint32_t tw = MIN(col_tile, ne0 - col); + + htp_trace_event_start(tr, HTP_TRACE_EVT_HVX_COMP, t); + compute(src_vtcm, dst_vtcm, tw, col, i01, ne0, tri_ttype); + htp_trace_event_stop(tr, HTP_TRACE_EVT_HVX_COMP, t); + + const size_t doff = (dst_contig ? (row * nb1) : + unary_row_offset(row, ne1, ne2, div_ne01, div_ne02, div_ne012, nb1, nb2, nb3)) + + (size_t) col * sizeof(float); + const size_t tb = (size_t) tw * sizeof(float); + dma_queue_push(dma_q, dma_make_data(data_dst + doff, dst_vtcm), tb, tb, tb, 1); + + const uint32_t pt = t + 2; + if (pt < total_tiles) { + const uint32_t ptw = MIN(col_tile, ne0 - pcol); + const size_t ptb = (size_t) ptw * sizeof(float); + const size_t psoff = (src0_contig ? (prow * nb01) : + unary_row_offset(prow, ne01, ne02, div_ne01, div_ne02, div_ne012, nb01, nb02, + nb03)) + + (size_t) pcol * sizeof(float); + dma_queue_push(dma_q, dma_make_data(src_vtcm, data_src + psoff), ptb, ptb, ptb, 1); + } + + tile_in_row++; + col += col_tile; + if (tile_in_row == tiles_per_row) { + tile_in_row = 0; + col = 0; + row++; + i01++; + if (i01 == ne01) { + i01 = 0; + } + } + + ptile_in_row++; + pcol += col_tile; + if (ptile_in_row == tiles_per_row) { + ptile_in_row = 0; + pcol = 0; + prow++; + } + } + + dma_queue_flush(dma_q); +} + +static int execute_op_unary(struct htp_ops_context * octx) { int err = HTP_STATUS_OK; const struct htp_tensor * src0 = octx->src[0]; const struct htp_tensor * dst = octx->dst; + const bool is_f16 = (src0->type == HTP_TYPE_F16); + const char * op_type = NULL; switch (octx->op) { - case HTP_OP_NORM: op_type = "norm-f32"; break; - case HTP_OP_RMS_NORM: op_type = "rmsnorm-f32"; break; - case HTP_OP_RMS_NORM_MUL: op_type = "rmsnorm-mul-f32"; break; - case HTP_OP_SCALE: op_type = "scale-f32"; break; - case HTP_OP_CLAMP: op_type = "clamp-f32"; break; - case HTP_OP_SQR: op_type = "sqr-f32"; break; - case HTP_OP_SQRT: op_type = "sqrt-f32"; break; - case HTP_OP_UNARY_NEG: op_type = "neg-f32"; break; - case HTP_OP_UNARY_EXP: op_type = "exp-f32"; break; - case HTP_OP_UNARY_SIGMOID: op_type = "sigmoid-f32"; break; - case HTP_OP_UNARY_SILU: op_type = "silu-f32"; break; - case HTP_OP_UNARY_GELU: op_type = "gelu-f32"; break; - case HTP_OP_UNARY_SOFTPLUS: op_type = "softplus-f32"; break; - case HTP_OP_UNARY_TANH: op_type = "tanh-f32"; break; - case HTP_OP_L2_NORM: op_type = "l2norm-f32"; break; - case HTP_OP_TRI: op_type = "tri-f32"; break; - + case HTP_OP_NORM: op_type = is_f16 ? "norm-f16" : "norm-f32"; break; + case HTP_OP_RMS_NORM: op_type = is_f16 ? "rmsnorm-f16" : "rmsnorm-f32"; break; + case HTP_OP_RMS_NORM_MUL: op_type = "rmsnorm-mul-f32"; break; + case HTP_OP_SCALE: op_type = is_f16 ? "scale-f16" : "scale-f32"; break; + case HTP_OP_CLAMP: op_type = is_f16 ? "clamp-f16" : "clamp-f32"; break; + case HTP_OP_LEAKY_RELU: op_type = "leaky-relu-f32"; break; + case HTP_OP_SQR: op_type = is_f16 ? "sqr-f16" : "sqr-f32"; break; + case HTP_OP_SQRT: op_type = is_f16 ? "sqrt-f16" : "sqrt-f32"; break; + case HTP_OP_UNARY_NEG: op_type = "neg-f32"; break; + case HTP_OP_UNARY_EXP: op_type = "exp-f32"; break; + case HTP_OP_UNARY_SIGMOID: op_type = "sigmoid-f32"; break; + case HTP_OP_UNARY_SILU: op_type = "silu-f32"; break; + case HTP_OP_UNARY_GELU: op_type = "gelu-f32"; break; + case HTP_OP_UNARY_SOFTPLUS: op_type = "softplus-f32"; break; + case HTP_OP_UNARY_TANH: op_type = "tanh-f32"; break; + case HTP_OP_UNARY_ABS: op_type = is_f16 ? "abs-f16" : "abs-f32"; break; + case HTP_OP_UNARY_LOG: op_type = is_f16 ? "log-f16" : "log-f32"; break; + case HTP_OP_UNARY_RELU: op_type = "relu-f32"; break; + case HTP_OP_L2_NORM: op_type = is_f16 ? "l2norm-f16" : "l2norm-f32"; break; + case HTP_OP_TRI: op_type = "tri-f32"; break; default: FARF(ERROR, "Unsupported unary Op %u\n", octx->op); return HTP_STATUS_NO_SUPPORT; } + // F16 only has row-block kernels for this subset of ops (see the dispatch switch + // below) - reject everything else up front, before touching kparams/VTCM. + if (is_f16) { + switch (octx->op) { + case HTP_OP_NORM: + case HTP_OP_RMS_NORM: + case HTP_OP_SCALE: + case HTP_OP_CLAMP: + case HTP_OP_SQR: + case HTP_OP_SQRT: + case HTP_OP_L2_NORM: + case HTP_OP_UNARY_ABS: + case HTP_OP_UNARY_LOG: + break; + default: + FARF(ERROR, "unary-%s: not supported for F16\n", op_type); + return HTP_STATUS_NO_SUPPORT; + } + } + const struct htp_unary_kernel_params * kparams = (const struct htp_unary_kernel_params *) octx->kernel_params; + if (!htp_ops_context_set_n_threads(octx, kparams->n_threads)) { + return HTP_STATUS_INVAL_PARAMS; + } + const uint32_t src0_nrows = src0->ne[1] * src0->ne[2] * src0->ne[3]; - const uint32_t n_threads = kparams->n_threads; + const size_t elem_size = is_f16 ? sizeof(_Float16) : sizeof(float); + const size_t src0_data_row_size = src0->ne[0] * elem_size; + const size_t dst_data_row_size = dst->ne[0] * elem_size; + + uint32_t row_start = 0; + uint32_t nrows = src0_nrows; + + if (octx->ctx->mdev.count > 1) { + uint32_t rows_per_chunk = 0; + htp_tensor_mdev_rows_per_chunk(dst, (uint32_t) elem_size, (uint32_t) dst_data_row_size, &rows_per_chunk); + const struct htp_tensor_mdev_range range = htp_tensor_mdev_partition(src0_nrows, rows_per_chunk, octx->ctx->mdev.idx, octx->ctx->mdev.count, &octx->ctx->mdev.count_div); + row_start = range.start; + nrows = range.count; + } + + if (nrows == 0) { + return HTP_STATUS_OK; + } - const size_t src0_data_row_size = src0->ne[0] * sizeof(float); - const size_t dst_data_row_size = dst->ne[0] * sizeof(float); + const uint32_t n_threads = octx->n_threads; const size_t src0_row_size_aligned = kparams->src0_row_size_aligned; const size_t dst_row_size_aligned = kparams->dst_row_size_aligned; + // Always 0 for F16 - htp_unary_vtcm_layout_build() keeps F16 on the row-block path, + // since only F32 has unary_task_f32_tiled_* kernels. const uint32_t col_tile = kparams->col_tile; size_t src1_data_row_size = 0; @@ -901,6 +1553,8 @@ static int execute_op_unary_f32(struct htp_ops_context * octx) { bool broadcast_weight = kparams->broadcast_weight; const struct htp_tensor * src1 = NULL; + // RMS_NORM_MUL fusion is F32-only (its weight tensor is always F32; see + // try_fuse_node()'s type guard), so this never triggers when is_f16 is true. if (octx->op == HTP_OP_RMS_NORM_MUL) { src1 = octx->src[1]; src1_data_row_size = src1->ne[0] * sizeof(float); @@ -919,100 +1573,136 @@ static int execute_op_unary_f32(struct htp_ops_context * octx) { src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], kparams->vtcm_src0_size, kparams->vtcm_src1_size, kparams->vtcm_dst_size); - if (!(octx->flags & HTP_OPFLAGS_SKIP_COMPUTE)) { - uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; - struct htp_unary_context uctx = { - .octx = octx, - .kparams = kparams, - .src0_nrows_per_thread = (src0_nrows + n_threads - 1) / n_threads, - .src0_nrows = src0_nrows, - - .data_src0 = (const uint8_t *)src0->data, - .data_src1 = (octx->op == HTP_OP_RMS_NORM_MUL) ? (const uint8_t *)src1->data : NULL, - .data_dst = (uint8_t *)dst->data, - - .src0_data_row_size = src0_data_row_size, - .src1_data_row_size = src1_data_row_size, - .dst_data_row_size = dst_data_row_size, - - .src0_row_size_aligned = src0_row_size_aligned, - .src1_row_size_aligned = src1_row_size_aligned, - .dst_row_size_aligned = dst_row_size_aligned, - - .src0_vtcm_half_size = kparams->vtcm_src0_size_per_thread / 2, - .src1_vtcm_half_size = (octx->op == HTP_OP_RMS_NORM_MUL) ? (kparams->vtcm_src1_size_per_thread / (broadcast_weight ? 1 : 2)) : 0, - .dst_vtcm_half_size = kparams->vtcm_dst_size_per_thread / 2, - - .block = kparams->block, - .nc = src0->ne[0], - .col_tile = (uint32_t) kparams->col_tile, - .broadcast_weight = broadcast_weight, - - .vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, 0), - .vtcm_src1 = VTCM_LAYOUT_PTR_OPTIONAL(uint8_t, base, kparams->vtcm_src0_size, kparams->vtcm_src1_size > 0), - .vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, kparams->vtcm_src0_size + kparams->vtcm_src1_size), - - .vtcm_src0_size_per_thread = kparams->vtcm_src0_size_per_thread, - .vtcm_src1_size_per_thread = kparams->vtcm_src1_size_per_thread, - .vtcm_dst_size_per_thread = kparams->vtcm_dst_size_per_thread, - }; - - FARF(HIGH, "%s: %s mode (col_tile %u)\n", op_type, col_tile ? "tiled" : "row-block", col_tile); - - worker_callback_t task_func = NULL; - if (col_tile) { - switch (octx->op) { - case HTP_OP_SCALE: task_func = unary_task_f32_tiled_scale; break; - case HTP_OP_CLAMP: task_func = unary_task_f32_tiled_clamp; break; - case HTP_OP_SQR: task_func = unary_task_f32_tiled_sqr; break; - case HTP_OP_SQRT: task_func = unary_task_f32_tiled_sqrt; break; - case HTP_OP_UNARY_NEG: task_func = unary_task_f32_tiled_unary_neg; break; - case HTP_OP_UNARY_EXP: task_func = unary_task_f32_tiled_unary_exp; break; - case HTP_OP_UNARY_SIGMOID: task_func = unary_task_f32_tiled_unary_sigmoid; break; - case HTP_OP_UNARY_SILU: task_func = unary_task_f32_tiled_unary_silu; break; - case HTP_OP_UNARY_GELU: task_func = unary_task_f32_tiled_unary_gelu; break; - case HTP_OP_UNARY_SOFTPLUS: task_func = unary_task_f32_tiled_unary_softplus; break; - case HTP_OP_UNARY_TANH: task_func = unary_task_f32_tiled_unary_tanh; break; - case HTP_OP_TRI: task_func = unary_task_f32_tiled_tri; break; - default: break; - } - } else { - switch (octx->op) { - case HTP_OP_NORM: task_func = unary_task_f32_norm; break; - case HTP_OP_RMS_NORM: task_func = unary_task_f32_rms_norm; break; - case HTP_OP_RMS_NORM_MUL: task_func = unary_task_f32_rms_norm_mul; break; - case HTP_OP_SCALE: task_func = unary_task_f32_scale; break; - case HTP_OP_CLAMP: task_func = unary_task_f32_clamp; break; - case HTP_OP_SQR: task_func = unary_task_f32_sqr; break; - case HTP_OP_SQRT: task_func = unary_task_f32_sqrt; break; - case HTP_OP_UNARY_NEG: task_func = unary_task_f32_unary_neg; break; - case HTP_OP_UNARY_EXP: task_func = unary_task_f32_unary_exp; break; - case HTP_OP_UNARY_SIGMOID: task_func = unary_task_f32_unary_sigmoid; break; - case HTP_OP_UNARY_SILU: task_func = unary_task_f32_unary_silu; break; - case HTP_OP_UNARY_GELU: task_func = unary_task_f32_unary_gelu; break; - case HTP_OP_UNARY_SOFTPLUS: task_func = unary_task_f32_unary_softplus; break; - case HTP_OP_UNARY_TANH: task_func = unary_task_f32_unary_tanh; break; - case HTP_OP_L2_NORM: task_func = unary_task_f32_l2_norm; break; - case HTP_OP_TRI: task_func = unary_task_f32_tri; break; - default: break; - } + uint8_t * const base = (uint8_t *) octx->ctx->vtcm_base; + struct htp_unary_context uctx = { + .octx = octx, + .kparams = kparams, + .src0_nrows_per_thread = fastdiv(nrows + n_threads - 1, &octx->n_threads_div), + .src0_nrows = nrows, + .row_start = row_start, + + .data_src0 = src0->data, + .data_src1 = (octx->op == HTP_OP_RMS_NORM_MUL) ? src1->data : 0, + .data_dst = dst->data, + + .src0_data_row_size = src0_data_row_size, + .src1_data_row_size = src1_data_row_size, + .dst_data_row_size = dst_data_row_size, + + .src0_row_size_aligned = src0_row_size_aligned, + .src1_row_size_aligned = src1_row_size_aligned, + .dst_row_size_aligned = dst_row_size_aligned, + + .src0_vtcm_half_size = kparams->vtcm_src0_size_per_thread / 2, + .src1_vtcm_half_size = (octx->op == HTP_OP_RMS_NORM_MUL) ? (kparams->vtcm_src1_size_per_thread / (broadcast_weight ? 1 : 2)) : 0, + .dst_vtcm_half_size = kparams->vtcm_dst_size_per_thread / 2, + + .block = kparams->block, + .nc = src0->ne[0], + .col_tile = col_tile, + .broadcast_weight = broadcast_weight, + + .vtcm_src0 = VTCM_LAYOUT_PTR(uint8_t, base, 0), + .vtcm_src1 = VTCM_LAYOUT_PTR_OPTIONAL(uint8_t, base, kparams->vtcm_src0_size, kparams->vtcm_src1_size > 0), + .vtcm_dst = VTCM_LAYOUT_PTR(uint8_t, base, kparams->vtcm_src0_size + kparams->vtcm_src1_size), + + .vtcm_src0_size_per_thread = kparams->vtcm_src0_size_per_thread, + .vtcm_src1_size_per_thread = kparams->vtcm_src1_size_per_thread, + .vtcm_dst_size_per_thread = kparams->vtcm_dst_size_per_thread, + }; + + FARF(HIGH, "%s: %s mode (col_tile %u)\n", op_type, col_tile ? "tiled" : "row-block", col_tile); + + worker_callback_t task_func = NULL; + void * compute_func = NULL; + + if (col_tile) { + task_func = unary_thread_tiled; + switch (octx->op) { + case HTP_OP_SCALE: compute_func = (void *) tile_scale_f32; break; + case HTP_OP_CLAMP: compute_func = (void *) tile_clamp_f32; break; + case HTP_OP_LEAKY_RELU: compute_func = (void *) tile_leaky_relu_f32; break; + case HTP_OP_SQR: compute_func = (void *) tile_sqr_f32; break; + case HTP_OP_SQRT: compute_func = (void *) tile_sqrt_f32; break; + case HTP_OP_UNARY_NEG: compute_func = (void *) tile_neg_f32; break; + case HTP_OP_UNARY_EXP: compute_func = (void *) tile_exp_f32; break; + case HTP_OP_UNARY_SIGMOID: compute_func = (void *) tile_sigmoid_f32; break; + case HTP_OP_UNARY_SILU: compute_func = (void *) tile_silu_f32; break; + case HTP_OP_UNARY_GELU: compute_func = (void *) tile_gelu_f32; break; + case HTP_OP_UNARY_SOFTPLUS: compute_func = (void *) tile_softplus_f32; break; + case HTP_OP_UNARY_TANH: compute_func = (void *) tile_tanh_f32; break; + case HTP_OP_UNARY_ABS: compute_func = (void *) tile_abs_f32; break; + case HTP_OP_UNARY_LOG: compute_func = (void *) tile_log_f32; break; + case HTP_OP_UNARY_RELU: compute_func = (void *) tile_relu_f32; break; + case HTP_OP_TRI: + task_func = unary_thread_tiled_tri_f32; + compute_func = (void *) tri_apply_tile_f32; + break; + default: break; } - - if (task_func) { - worker_pool_run_func(octx->ctx->worker_pool, task_func, &uctx, n_threads); - } else { - FARF(ERROR, "execute_op_unary_f32: task function is NULL for op %d\n", octx->op); - err = HTP_STATUS_NO_SUPPORT; + } else if (is_f16) { + task_func = unary_thread_row_block; + switch (octx->op) { + case HTP_OP_NORM: compute_func = (void *) norm_f16; break; + case HTP_OP_RMS_NORM: compute_func = (void *) rms_norm_f16; break; + case HTP_OP_SCALE: compute_func = (void *) scale_f16; break; + case HTP_OP_CLAMP: compute_func = (void *) clamp_f16; break; + case HTP_OP_SQR: compute_func = (void *) sqr_f16; break; + case HTP_OP_SQRT: compute_func = (void *) sqrt_f16; break; + case HTP_OP_L2_NORM: compute_func = (void *) l2_norm_f16; break; + case HTP_OP_UNARY_ABS: compute_func = (void *) abs_f16; break; + case HTP_OP_UNARY_LOG: compute_func = (void *) log_f16; break; + default: break; } + } else { + task_func = unary_thread_row_block; + switch (octx->op) { + case HTP_OP_NORM: compute_func = (void *) norm_f32; break; + case HTP_OP_RMS_NORM: compute_func = (void *) rms_norm_f32; break; + case HTP_OP_RMS_NORM_MUL: + task_func = unary_thread_rms_norm_mul_f32; + compute_func = (void *) rms_norm_mul_f32; + break; + case HTP_OP_SCALE: compute_func = (void *) scale_f32; break; + case HTP_OP_CLAMP: compute_func = (void *) clamp_f32; break; + case HTP_OP_LEAKY_RELU: compute_func = (void *) leaky_relu_f32; break; + case HTP_OP_SQR: compute_func = (void *) sqr_f32; break; + case HTP_OP_SQRT: compute_func = (void *) sqrt_f32; break; + case HTP_OP_UNARY_NEG: compute_func = (void *) neg_f32; break; + case HTP_OP_UNARY_EXP: compute_func = (void *) exp_f32; break; + case HTP_OP_UNARY_SIGMOID: compute_func = (void *) sigmoid_f32; break; + case HTP_OP_UNARY_SILU: compute_func = (void *) silu_f32; break; + case HTP_OP_UNARY_GELU: compute_func = (void *) gelu_f32; break; + case HTP_OP_UNARY_SOFTPLUS: compute_func = (void *) softplus_f32; break; + case HTP_OP_UNARY_TANH: compute_func = (void *) tanh_f32; break; + case HTP_OP_UNARY_ABS: compute_func = (void *) abs_f32; break; + case HTP_OP_UNARY_LOG: compute_func = (void *) log_f32; break; + case HTP_OP_UNARY_RELU: compute_func = (void *) relu_f32; break; + case HTP_OP_L2_NORM: compute_func = (void *) l2_norm_f32; break; + case HTP_OP_TRI: + task_func = unary_thread_tri_f32; + compute_func = (void *) tri_f32; + break; + default: break; + } + } + + if (!task_func || !compute_func) { + FARF(ERROR, "execute_op_unary: task function is NULL for op %d\n", octx->op); + return HTP_STATUS_NO_SUPPORT; } + uctx.compute = compute_func; + work_queue_run(octx->ctx->work_queue, task_func, &uctx, n_threads); + return err; } int op_unary(struct htp_ops_context * octx) { switch (octx->src[0]->type) { case HTP_TYPE_F32: - return execute_op_unary_f32(octx); + case HTP_TYPE_F16: + return execute_op_unary(octx); default: return HTP_STATUS_NO_SUPPORT; diff --git a/ggml/src/ggml-hexagon/htp/unary-ops.h b/ggml/src/ggml-hexagon/htp/unary-ops.h index 1f4c3a5c..e410d7fd 100644 --- a/ggml/src/ggml-hexagon/htp/unary-ops.h +++ b/ggml/src/ggml-hexagon/htp/unary-ops.h @@ -42,6 +42,7 @@ _Static_assert(sizeof(struct htp_unary_kernel_params) <= 128, "htp_unary_kernel_ static inline bool htp_op_is_unary(uint32_t opcode) { switch (opcode) { case HTP_OP_CLAMP: + case HTP_OP_LEAKY_RELU: case HTP_OP_NORM: case HTP_OP_RMS_NORM: case HTP_OP_RMS_NORM_MUL: @@ -55,6 +56,9 @@ static inline bool htp_op_is_unary(uint32_t opcode) { case HTP_OP_UNARY_GELU: case HTP_OP_UNARY_SOFTPLUS: case HTP_OP_UNARY_TANH: + case HTP_OP_UNARY_ABS: + case HTP_OP_UNARY_LOG: + case HTP_OP_UNARY_RELU: case HTP_OP_L2_NORM: case HTP_OP_TRI: return true; @@ -83,17 +87,19 @@ static inline void htp_unary_vtcm_layout_build( bool broadcast_weight, uint32_t n_threads, size_t vtcm_size, + size_t elem_size, uint32_t * out_col_tile, uint32_t * out_vtcm_row_per_thread ) { - const size_t src0_data_row_size = ne00 * sizeof(float); - const size_t dst_data_row_size = ne10 * sizeof(float); + const size_t src0_data_row_size = ne00 * elem_size; + const size_t dst_data_row_size = ne10 * elem_size; const size_t src0_row_size_aligned = hex_round_up(src0_data_row_size, 128); const size_t dst_row_size_aligned = hex_round_up(dst_data_row_size, 128); size_t src1_row_size_aligned = 0; if (op == HTP_OP_RMS_NORM_MUL) { + // RMS_NORM_MUL fusion is F32-only; its weight tensor is always F32. const size_t src1_data_row_size = ne11 * sizeof(float); src1_row_size_aligned = hex_round_up(src1_data_row_size, 128); } @@ -123,12 +129,19 @@ static inline void htp_unary_vtcm_layout_build( const bool is_reduction = (op == HTP_OP_NORM || op == HTP_OP_RMS_NORM || op == HTP_OP_RMS_NORM_MUL || op == HTP_OP_L2_NORM); + // The tiled fallback path below only has F32 task functions (unary_task_f32_tiled_*); + // F16 has no tiled kernels, so it must stay on the row-block path like reduction ops. + // NOTE: if F16 ends up with vtcm_row_per_thread == 0 here (row too large for the VTCM + // budget), execute_op_unary() will see BLOCK == 0 and skip computation for that op + // (logged via FARF(ERROR, ...)) since there is no F16 tiled fallback. This is a known + // limitation; supporting it would require adding F16 tiled kernels. + const bool is_f16 = (elem_size == sizeof(_Float16)); uint32_t col_tile = 0; - if (vtcm_row_per_thread == 0 && !is_reduction) { + if (vtcm_row_per_thread == 0 && !is_reduction && !is_f16) { const size_t per_thread_budget = vtcm_size / n_threads; const size_t col_tile_bytes = hex_align_down(per_thread_budget / 4, 128); - col_tile = (uint32_t) (col_tile_bytes / sizeof(float)); + col_tile = (uint32_t) (col_tile_bytes / elem_size); L->src0_bytes = col_tile_bytes * 2; L->dst_bytes = col_tile_bytes * 2; diff --git a/ggml/src/ggml-hip/CMakeLists.txt b/ggml/src/ggml-hip/CMakeLists.txt index 47f16f56..a6a6b727 100644 --- a/ggml/src/ggml-hip/CMakeLists.txt +++ b/ggml/src/ggml-hip/CMakeLists.txt @@ -70,17 +70,8 @@ list(APPEND GGML_SOURCES_ROCM ${SRCS}) file(GLOB SRCS "../ggml-cuda/template-instances/mmf*.cu") list(APPEND GGML_SOURCES_ROCM ${SRCS}) -if (GGML_CUDA_FA_ALL_QUANTS) - file(GLOB SRCS "../ggml-cuda/template-instances/fattn-vec*.cu") - list(APPEND GGML_SOURCES_ROCM ${SRCS}) - add_compile_definitions(GGML_CUDA_FA_ALL_QUANTS) -else() - list(APPEND GGML_SOURCES_ROCM - ../ggml-cuda/template-instances/fattn-vec-instance-f16-f16.cu - ../ggml-cuda/template-instances/fattn-vec-instance-q4_0-q4_0.cu - ../ggml-cuda/template-instances/fattn-vec-instance-q8_0-q8_0.cu - ../ggml-cuda/template-instances/fattn-vec-instance-bf16-bf16.cu) -endif() +ggml_cuda_fattn_vec_instances(${CMAKE_CURRENT_SOURCE_DIR}/../ggml-cuda SRCS) +list(APPEND GGML_SOURCES_ROCM ${SRCS}) ggml_add_backend_library(ggml-hip ${GGML_HEADERS_ROCM} diff --git a/ggml/src/ggml-impl.h b/ggml/src/ggml-impl.h index 62b76abb..ae26e0c2 100644 --- a/ggml/src/ggml-impl.h +++ b/ggml/src/ggml-impl.h @@ -160,6 +160,18 @@ static float ggml_get_op_params_f32(const struct ggml_tensor * tensor, uint32_t return ((const float *)(tensor->op_params))[i]; } +// [TAG_GGML_PREC] +// - GGML_OP_MUL_MAT +// 0 - acc +// 1 - hint +// 2 - src0 precision +// 3 - src1 precision +// +// - GGML_OP_MUL_MAT_ID +// 0 - acc +// 1 - hint +// 2 - src0 precision +// 3 - src1 precision static void ggml_set_op_params_i32(struct ggml_tensor * tensor, uint32_t i, int32_t value) { assert(i < GGML_MAX_OP_PARAMS / sizeof(int32_t)); ((int32_t *)(tensor->op_params))[i] = value; diff --git a/ggml/src/ggml-metal/CMakeLists.txt b/ggml/src/ggml-metal/CMakeLists.txt index 42054d84..e7afdb69 100644 --- a/ggml/src/ggml-metal/CMakeLists.txt +++ b/ggml/src/ggml-metal/CMakeLists.txt @@ -10,7 +10,9 @@ ggml_add_backend_library(ggml-metal ggml-metal-device.cpp ggml-metal-common.cpp ggml-metal-context.m + ggml-metal-fusion.cpp ggml-metal-ops.cpp + ggml-metal-tuning.cpp ) target_link_libraries(ggml-metal PRIVATE @@ -24,65 +26,144 @@ if (GGML_METAL_NDEBUG) endif() set(METALLIB_COMMON "${CMAKE_CURRENT_SOURCE_DIR}/../ggml-common.h") +set(METALLIB_KERNELS_COMMON "${CMAKE_CURRENT_SOURCE_DIR}/kernels/common.h") +set(METALLIB_KERNELS_DEQUANTIZE "${CMAKE_CURRENT_SOURCE_DIR}/kernels/dequantize.h") +set(METALLIB_KERNELS_QUANTIZE "${CMAKE_CURRENT_SOURCE_DIR}/kernels/quantize.h") + +set(METALLIB_KERNEL_SOURCES + kernels/fa.metal + kernels/mul_mv.metal + kernels/mul_mm.metal + kernels/quantize.metal + kernels/softmax.metal + kernels/norm.metal + kernels/unary.metal + kernels/binbcast.metal + kernels/reduce.metal + kernels/tri.metal + kernels/ssm.metal + kernels/wkv.metal + kernels/gated_delta_net.metal + kernels/solve_tri.metal + kernels/rope.metal + kernels/conv.metal + kernels/upscale.metal + kernels/argsort.metal + kernels/pool.metal + kernels/misc.metal +) + if (GGML_METAL_EMBED_LIBRARY) enable_language(ASM) add_compile_definitions(GGML_METAL_EMBED_LIBRARY) - set(METALLIB_SOURCE "${CMAKE_CURRENT_SOURCE_DIR}/ggml-metal.metal") - set(METALLIB_IMPL "${CMAKE_CURRENT_SOURCE_DIR}/ggml-metal-impl.h") + set(METALLIB_IMPL "${CMAKE_CURRENT_SOURCE_DIR}/ggml-metal-impl.h") file(MAKE_DIRECTORY "${CMAKE_CURRENT_BINARY_DIR}/autogenerated") - # merge ggml-common.h and ggml-metal.metal into a single file - set(METALLIB_EMBED_ASM "${CMAKE_CURRENT_BINARY_DIR}/autogenerated/ggml-metal-embed.s") - set(METALLIB_SOURCE_EMBED "${CMAKE_CURRENT_BINARY_DIR}/autogenerated/ggml-metal-embed.metal") - set(METALLIB_SOURCE_EMBED_TMP "${CMAKE_CURRENT_BINARY_DIR}/autogenerated/ggml-metal-embed.metal.tmp") + set(METALLIB_EMBED_ASM_FILES "") + foreach(src ${METALLIB_KERNEL_SOURCES}) + get_filename_component(kind ${src} NAME_WE) + # symbol names must be valid C identifiers ('-' is not allowed) + string(REPLACE "-" "_" kind_sym ${kind}) + + set(SRC "${CMAKE_CURRENT_SOURCE_DIR}/kernels/${kind}.metal") + set(EMBED "${CMAKE_CURRENT_BINARY_DIR}/autogenerated/ggml-metal-embed-${kind}.metal") + set(ASM "${CMAKE_CURRENT_BINARY_DIR}/autogenerated/ggml-metal-embed-${kind}.s") + + # only prepend headers that this source actually includes + set(HEADERS_FOR_SRC ${METALLIB_KERNELS_COMMON}) + file(STRINGS ${SRC} _has_dequantize REGEX "#include \"dequantize\\.h\"") + file(STRINGS ${SRC} _has_quantize REGEX "#include \"quantize\\.h\"") + if(_has_dequantize) + list(APPEND HEADERS_FOR_SRC ${METALLIB_KERNELS_DEQUANTIZE}) + endif() + if(_has_quantize) + list(APPEND HEADERS_FOR_SRC ${METALLIB_KERNELS_QUANTIZE}) + endif() + + add_custom_command( + OUTPUT "${ASM}" + # Step 1: concatenate shared headers + this kernel source + COMMAND cat ${HEADERS_FOR_SRC} ${SRC} > "${EMBED}.tmp1" + # Step 2: remove internal #include and #pragma once + COMMAND sed -e "/\#include \"common.h\"/d" -e "/\#include \"dequantize.h\"/d" -e "/\#include \"quantize.h\"/d" -e "/\#pragma once/d" < "${EMBED}.tmp1" > "${EMBED}.tmp2" + # Step 3: inline ggml-common.h (replacing __embed_ggml-common.h__ sentinel) + COMMAND sed -e "/__embed_ggml-common.h__/r ${METALLIB_COMMON}" -e "/__embed_ggml-common.h__/d" < "${EMBED}.tmp2" > "${EMBED}.tmp3" + # Step 4: inline ggml-metal-impl.h + COMMAND sed -e "/\#include \"ggml-metal-impl.h\"/r ${METALLIB_IMPL}" -e "/\#include \"ggml-metal-impl.h\"/d" < "${EMBED}.tmp3" > "${EMBED}" + # Step 5: emit an asm chunk with kind-specific start/end symbols + # note: '-' is illegal in C symbols, so we use kind_sym; the macOS + # section name is limited to 16 chars so we keep it shared + # across kinds (__ggml_metallib) and only vary the global symbols. + COMMAND echo ".section __DATA,__ggml_metallib" > "${ASM}" + COMMAND echo ".globl _ggml_metallib_${kind_sym}_start" >> "${ASM}" + COMMAND echo "_ggml_metallib_${kind_sym}_start:" >> "${ASM}" + COMMAND echo .incbin "\"${EMBED}\"" >> "${ASM}" + COMMAND echo ".globl _ggml_metallib_${kind_sym}_end" >> "${ASM}" + COMMAND echo "_ggml_metallib_${kind_sym}_end:" >> "${ASM}" + DEPENDS ../ggml-common.h ggml-metal-impl.h + kernels/common.h kernels/dequantize.h kernels/quantize.h + kernels/${kind}.metal + COMMENT "Generate embedded Metal library for ${kind}" + VERBATIM + ) - add_custom_command( - OUTPUT "${METALLIB_EMBED_ASM}" - COMMAND echo "Embedding Metal library" - COMMAND sed -e "/__embed_ggml-common.h__/r ${METALLIB_COMMON}" -e "/__embed_ggml-common.h__/d" < "${METALLIB_SOURCE}" > "${METALLIB_SOURCE_EMBED_TMP}" - COMMAND sed -e "/\#include \"ggml-metal-impl.h\"/r ${METALLIB_IMPL}" -e "/\#include \"ggml-metal-impl.h\"/d" < "${METALLIB_SOURCE_EMBED_TMP}" > "${METALLIB_SOURCE_EMBED}" - COMMAND echo ".section __DATA,__ggml_metallib" > "${METALLIB_EMBED_ASM}" - COMMAND echo ".globl _ggml_metallib_start" >> "${METALLIB_EMBED_ASM}" - COMMAND echo "_ggml_metallib_start:" >> "${METALLIB_EMBED_ASM}" - COMMAND echo .incbin "\"${METALLIB_SOURCE_EMBED}\"" >> "${METALLIB_EMBED_ASM}" - COMMAND echo ".globl _ggml_metallib_end" >> "${METALLIB_EMBED_ASM}" - COMMAND echo "_ggml_metallib_end:" >> "${METALLIB_EMBED_ASM}" - DEPENDS ../ggml-common.h ggml-metal.metal ggml-metal-impl.h - COMMENT "Generate assembly for embedded Metal library" - VERBATIM - ) + list(APPEND METALLIB_EMBED_ASM_FILES "${ASM}") + endforeach() - target_sources(ggml-metal PRIVATE "${METALLIB_EMBED_ASM}") + target_sources(ggml-metal PRIVATE ${METALLIB_EMBED_ASM_FILES}) else() - # copy metal files to bin directory + # copy header files to bin directory configure_file(../ggml-common.h ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-common.h COPYONLY) - configure_file(ggml-metal.metal ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-metal.metal COPYONLY) configure_file(ggml-metal-impl.h ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-metal-impl.h COPYONLY) + file(MAKE_DIRECTORY "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels") + configure_file(kernels/common.h ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels/common.h COPYONLY) + configure_file(kernels/dequantize.h ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels/dequantize.h COPYONLY) + configure_file(kernels/quantize.h ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels/quantize.h COPYONLY) + + foreach(src ${METALLIB_KERNEL_SOURCES}) + configure_file(${src} ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/${src} COPYONLY) + endforeach() + + # CMAKE_OSX_SYSROOT is an SDK name or path - xcrun accepts both + set(METAL_SDK ${CMAKE_OSX_SYSROOT}) + if (NOT METAL_SDK) + set(METAL_SDK macosx) + endif() + + if (CMAKE_OSX_SYSROOT MATCHES "[Ss]imulator") + set(METAL_TARGET_SIM "-simulator") + else() + set(METAL_TARGET_SIM "") + endif() + if (GGML_METAL_SHADER_DEBUG) - # custom command to do the following: - # xcrun -sdk macosx metal -fno-fast-math -c ggml-metal.metal -o ggml-metal.air - # xcrun -sdk macosx metallib ggml-metal.air -o default.metallib - # - # note: this is the only way I found to disable fast-math in Metal. it's ugly, but at least it works - # disabling fast math is needed in order to pass tests/test-backend-ops + # note: disabling fast math is needed in order to pass tests/test-backend-ops # note: adding -fno-inline fixes the tests when using MTL_SHADER_VALIDATION=1 # note: unfortunately, we have to call it default.metallib instead of ggml.metallib # ref: https://github.com/ggml-org/whisper.cpp/issues/1720 # note: adding -g causes segmentation fault during compile - #set(XC_FLAGS -fno-fast-math -fno-inline -g) set(XC_FLAGS -fno-fast-math -fno-inline) else() set(XC_FLAGS -O3) endif() - # Append macOS metal versioning flags + execute_process(COMMAND xcrun -sdk ${METAL_SDK} --show-sdk-version OUTPUT_VARIABLE METAL_SDK_VERSION OUTPUT_STRIP_TRAILING_WHITESPACE) + if (METAL_SDK_VERSION VERSION_GREATER_EQUAL 26.0) + set(GGML_METAL_HAS_TENSOR_LIB ON) + else() + message(STATUS "Metal SDK ${METAL_SDK_VERSION} does not support the tensor API, skipping ggml-tensor.metallib") + endif() + if (GGML_METAL_MACOSX_VERSION_MIN) message(STATUS "Adding -mmacosx-version-min=${GGML_METAL_MACOSX_VERSION_MIN} flag to metal compilation") list (APPEND XC_FLAGS -mmacosx-version-min=${GGML_METAL_MACOSX_VERSION_MIN}) + elseif (NOT GGML_METAL_TARGET_OS STREQUAL "macos" AND CMAKE_OSX_DEPLOYMENT_TARGET) + message(STATUS "Adding -mtargetos=${GGML_METAL_TARGET_OS}${CMAKE_OSX_DEPLOYMENT_TARGET}${METAL_TARGET_SIM} flag to metal compilation") + list (APPEND XC_FLAGS -mtargetos=${GGML_METAL_TARGET_OS}${CMAKE_OSX_DEPLOYMENT_TARGET}${METAL_TARGET_SIM}) endif() if (GGML_METAL_STD) @@ -90,35 +171,71 @@ else() list (APPEND XC_FLAGS -std=${GGML_METAL_STD}) endif() + # Compile each kernel source to .air, then link into default.metallib + set(AIR_FILES "") + foreach(src ${METALLIB_KERNEL_SOURCES}) + get_filename_component(name ${src} NAME_WE) + set(AIR "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/${name}.air") + list(APPEND AIR_FILES ${AIR}) + add_custom_command( + OUTPUT ${AIR} + COMMAND xcrun -sdk ${METAL_SDK} metal ${XC_FLAGS} -I ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -c ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/${src} -o ${AIR} + DEPENDS ${src} kernels/common.h kernels/dequantize.h kernels/quantize.h ${METALLIB_COMMON} ggml-metal-impl.h + COMMENT "Compiling ${src}" + VERBATIM + ) + endforeach() + + set(METALLIB_FILES ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib) + + # the tensor API kernels go in a separate metallib, loaded only where supported + if (GGML_METAL_HAS_TENSOR_LIB) + set(AIR_MM_TENSOR "${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/mul_mm_tensor.air") + # the tensor API needs OS 26+ + set(XC_FLAGS_TENSOR ${XC_FLAGS} -mtargetos=${GGML_METAL_TARGET_OS}26.0${METAL_TARGET_SIM}) + add_custom_command( + OUTPUT ${AIR_MM_TENSOR} + COMMAND xcrun -sdk ${METAL_SDK} metal ${XC_FLAGS_TENSOR} -DGGML_METAL_HAS_TENSOR -I ${CMAKE_RUNTIME_OUTPUT_DIRECTORY} -c ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels/mul_mm.metal -o ${AIR_MM_TENSOR} + DEPENDS kernels/mul_mm.metal kernels/common.h kernels/dequantize.h ${METALLIB_COMMON} ggml-metal-impl.h + COMMENT "Compiling kernels/mul_mm.metal (tensor API)" + VERBATIM + ) + + add_custom_command( + OUTPUT ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-tensor.metallib + COMMAND xcrun -sdk ${METAL_SDK} metallib ${AIR_MM_TENSOR} -o ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-tensor.metallib + DEPENDS ${AIR_MM_TENSOR} + COMMENT "Linking tensor API Metal kernels into ggml-tensor.metallib" + ) + + list(APPEND METALLIB_FILES ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-tensor.metallib) + endif() + add_custom_command( OUTPUT ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib - COMMAND xcrun -sdk macosx metal ${XC_FLAGS} -c ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-metal.metal -o - | - xcrun -sdk macosx metallib - -o ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib + COMMAND xcrun -sdk ${METAL_SDK} metallib ${AIR_FILES} -o ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib COMMAND rm -f ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-common.h - COMMAND rm -f ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-metal.metal - DEPENDS ggml-metal.metal ${METALLIB_COMMON} - COMMENT "Compiling Metal kernels" - ) + COMMAND rm -f ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/ggml-metal-impl.h + COMMAND rm -rf ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/kernels + DEPENDS ${AIR_FILES} ${AIR_MM_TENSOR} + COMMENT "Linking Metal kernels into default.metallib" + ) - # FIXME: only add to the ggml-metal target? add_custom_target( ggml-metal-lib ALL - DEPENDS ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib - ) + DEPENDS ${METALLIB_FILES} + ) endif() # GGML_METAL_EMBED_LIBRARY if (NOT GGML_METAL_EMBED_LIBRARY) install( - FILES src/ggml-metal/ggml-metal.metal - PERMISSIONS - OWNER_READ - OWNER_WRITE - GROUP_READ - WORLD_READ - DESTINATION ${CMAKE_INSTALL_BINDIR}) - - install( - FILES ${CMAKE_RUNTIME_OUTPUT_DIRECTORY}/default.metallib - DESTINATION ${CMAKE_INSTALL_BINDIR} - ) + DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/kernels/ + DESTINATION ${CMAKE_INSTALL_BINDIR}/kernels + FILES_MATCHING PATTERN "*.metal" PATTERN "*.h" + ) + + install( + FILES ${METALLIB_FILES} + DESTINATION ${CMAKE_INSTALL_BINDIR} + ) endif() diff --git a/ggml/src/ggml-metal/ggml-metal-common.cpp b/ggml/src/ggml-metal/ggml-metal-common.cpp index 2eb9820b..9c0b9474 100644 --- a/ggml/src/ggml-metal/ggml-metal-common.cpp +++ b/ggml/src/ggml-metal/ggml-metal-common.cpp @@ -1,10 +1,46 @@ #include "ggml-metal-common.h" +#include "ggml-metal-fusion.h" +#include "ggml.h" #include "ggml-impl.h" #include "ggml-backend-impl.h" #include +// must stay in sync with the kernel_fwht__ templates in misc.metal +static bool ggml_metal_fwht_supported_size(int64_t n) { + return n == 64 || n == 128 || n == 256 || n == 512; +} + +// the FWHT kernels handle a Hadamard-hinted MUL_MAT only under these conditions. supports_op +// and the dispatch must ask the same question: an F16 src1 that is admitted but then falls +// through reaches the generic path, which has no F32 src0 by F16 src1 kernel. +bool ggml_metal_op_mul_mat_use_fwht(const struct ggml_tensor * op) { + return ggml_get_op_params_i32(op, 1) == GGML_HINT_SRC0_IS_HADAMARD && + op->type == GGML_TYPE_F32 && + (op->src[1]->type == GGML_TYPE_F32 || op->src[1]->type == GGML_TYPE_F16) && + ggml_is_contiguous(op->src[1]) && + ggml_is_contiguous(op) && + ggml_are_same_shape(op->src[1], op) && + ggml_metal_fwht_supported_size(op->src[1]->ne[0]); +} + +bool ggml_metal_op_mul_mat_use_mm(const struct ggml_tensor * op, bool has_simdgroup_mm) { + const int64_t ne00 = op->src[0]->ne[0]; + const int64_t ne11 = op->src[1]->ne[1]; + + return !ggml_is_transposed(op->src[0]) && + !ggml_is_transposed(op->src[1]) && + has_simdgroup_mm && ne00 >= 64 && ne11 > 8; +} + +bool ggml_metal_op_mul_mat_id_use_mm(const struct ggml_tensor * op, bool has_simdgroup_mm) { + const int64_t ne00 = op->src[0]->ne[0]; + const int64_t ne21 = op->src[2]->ne[1]; + + return has_simdgroup_mm && ne00 >= 64 && ne21 >= 32; +} + // represents a memory range (i.e. an interval from a starting address p0 to an ending address p1 in a given buffer pb) // the type indicates whether it is a source range (i.e. ops read data from it) or a destination range (i.e. ops write data to it) struct ggml_mem_range { @@ -204,38 +240,63 @@ struct node_info { void add_fused(ggml_tensor * t) { fused.push_back(t); } + + bool is_output(const ggml_tensor * t) const { + if (t == node) { + return true; + } + for (const auto * f : fused) { + if (t == f) { + return true; + } + } + return false; + } }; static std::vector ggml_metal_graph_optimize_reorder(const std::vector & nodes) { // helper to add node src and dst ranges const auto & h_add = [](ggml_mem_ranges_t mrs, const node_info & node) { + // only external sources matter: sources produced by the fused group are internal for (int i = 0; i < GGML_MAX_SRC; i++) { - if (node.node->src[i]) { - if (!ggml_mem_ranges_add_src(mrs, node.node->src[i])) { + const ggml_tensor * src = node.node->src[i]; + if (src && !node.is_output(src)) { + if (!ggml_mem_ranges_add_src(mrs, src)) { return false; } } } - // keep track of the sources of the fused nodes as well for (const auto * fused : node.fused) { for (int i = 0; i < GGML_MAX_SRC; i++) { - if (fused->src[i]) { - if (!ggml_mem_ranges_add_src(mrs, fused->src[i])) { + const ggml_tensor * src = fused->src[i]; + if (src && !node.is_output(src)) { + if (!ggml_mem_ranges_add_src(mrs, src)) { return false; } } } } - return ggml_mem_ranges_add_dst(mrs, node.dst()); + // all fused tensors are produced by the fused kernel + if (!ggml_mem_ranges_add_dst(mrs, node.node)) { + return false; + } + for (const auto * fused : node.fused) { + if (!ggml_mem_ranges_add_dst(mrs, fused)) { + return false; + } + } + + return true; }; // helper to check if a node can run concurrently with the existing set of nodes const auto & h_check = [](ggml_mem_ranges_t mrs, const node_info & node) { for (int i = 0; i < GGML_MAX_SRC; i++) { - if (node.node->src[i]) { - if (!ggml_mem_ranges_check_src(mrs, node.node->src[i])) { + const ggml_tensor * src = node.node->src[i]; + if (src && !node.is_output(src)) { + if (!ggml_mem_ranges_check_src(mrs, src)) { return false; } } @@ -243,15 +304,25 @@ static std::vector ggml_metal_graph_optimize_reorder(const std::vectorsrc[i]) { - if (!ggml_mem_ranges_check_src(mrs, fused->src[i])) { + const ggml_tensor * src = fused->src[i]; + if (src && !node.is_output(src)) { + if (!ggml_mem_ranges_check_src(mrs, src)) { return false; } } } } - return ggml_mem_ranges_check_dst(mrs, node.dst()); + if (!ggml_mem_ranges_check_dst(mrs, node.node)) { + return false; + } + for (const auto * fused : node.fused) { + if (!ggml_mem_ranges_check_dst(mrs, fused)) { + return false; + } + } + + return true; }; // perform reorders only across these types of ops @@ -373,59 +444,31 @@ static std::vector ggml_metal_graph_optimize_reorder(const std::vectorn_nodes; - enum ggml_op ops[MAX_FUSE]; - std::vector nodes; nodes.reserve(gf->n_nodes); // fuse nodes: // we don't want to make reorders that break fusing, so we first pack all fusable tensors // and perform the reorder over the fused nodes. after the reorder is done, we unfuse + // + // the fusable sequences are declared in the fusion table (ggml-metal-fuse.cpp), so the + // packing here is driven by the same patterns that the op encoders will later use for (int i = 0; i < n; i++) { node_info node = { /*.node =*/ gf->nodes[i], /*.fused =*/ {}, }; - // fuse only ops that start with these operations - // can be expanded when needed - if (node.op() == GGML_OP_ADD || - node.op() == GGML_OP_NORM || - node.op() == GGML_OP_RMS_NORM) { - ops[0] = node.op(); - - int f = i + 1; - while (f < n && f < i + MAX_FUSE) { - // conservatively allow fusing only these ops - // can be expanded when needed - if (gf->nodes[f]->op != GGML_OP_ADD && - gf->nodes[f]->op != GGML_OP_MUL && - gf->nodes[f]->op != GGML_OP_NORM && - gf->nodes[f]->op != GGML_OP_RMS_NORM) { - break; - } - ops[f - i] = gf->nodes[f]->op; - f++; - } - - f -= i; - for (; f > 1; f--) { - if (ggml_can_fuse(gf, i, ops, f)) { - break; - } - } + const int f = ggml_metal_fusion_max(gf, i); - // add the fused tensors into the node info so we can unfuse them later - for (int k = 1; k < f; k++) { - ++i; + // add the fused tensors into the node info so we can unfuse them later + for (int k = 1; k < f; k++) { + ++i; - // the .dst() becomes the last fused tensor - node.add_fused(gf->nodes[i]); - } + // the .dst() becomes the last fused tensor + node.add_fused(gf->nodes[i]); } nodes.push_back(std::move(node)); diff --git a/ggml/src/ggml-metal/ggml-metal-common.h b/ggml/src/ggml-metal/ggml-metal-common.h index 3acbc6ae..e6a28d03 100644 --- a/ggml/src/ggml-metal/ggml-metal-common.h +++ b/ggml/src/ggml-metal/ggml-metal-common.h @@ -47,6 +47,11 @@ bool ggml_mem_ranges_check(ggml_mem_ranges_t mrs, const struct ggml_tensor * ten // if it proves to work well, we can start using it for other backends in the future void ggml_graph_optimize(struct ggml_cgraph * gf); +// mat-mat vs mat-vec dispatch; used by both supports_op and ggml_metal_op_mul_mat* +bool ggml_metal_op_mul_mat_use_fwht (const struct ggml_tensor * op); +bool ggml_metal_op_mul_mat_use_mm (const struct ggml_tensor * op, bool has_simdgroup_mm); +bool ggml_metal_op_mul_mat_id_use_mm(const struct ggml_tensor * op, bool has_simdgroup_mm); + #ifdef __cplusplus } #endif diff --git a/ggml/src/ggml-metal/ggml-metal-context.h b/ggml/src/ggml-metal/ggml-metal-context.h index abf4b06e..b538b1ad 100644 --- a/ggml/src/ggml-metal/ggml-metal-context.h +++ b/ggml/src/ggml-metal/ggml-metal-context.h @@ -33,6 +33,7 @@ ggml_metal_event_t ggml_metal_get_ev_cpy(ggml_metal_t ctx); void ggml_metal_set_n_cb (ggml_metal_t ctx, int n_cb); void ggml_metal_set_abort_callback (ggml_metal_t ctx, ggml_abort_callback abort_callback, void * user_data); + bool ggml_metal_supports_family (ggml_metal_t ctx, int family); void ggml_metal_capture_next_compute(ggml_metal_t ctx); diff --git a/ggml/src/ggml-metal/ggml-metal-context.m b/ggml/src/ggml-metal/ggml-metal-context.m index 32d97cd5..442ed2a0 100644 --- a/ggml/src/ggml-metal/ggml-metal-context.m +++ b/ggml/src/ggml-metal/ggml-metal-context.m @@ -6,6 +6,7 @@ #import "ggml-metal-impl.h" #import "ggml-metal-common.h" #import "ggml-metal-ops.h" +#import "ggml-metal-fusion.h" #import @@ -29,22 +30,20 @@ ggml_metal_device_t dev; ggml_metal_library_t lib; - ggml_metal_event_t ev_cpy; // for async copies + ggml_metal_event_t ev_cpy; // for async copies + ggml_metal_event_t ev_sync; // destination completion signal dispatch_queue_t d_queue; // additional, inference-time compiled pipelines ggml_metal_pipelines_t pipelines_ext; - bool use_fusion; bool use_concurrency; bool use_graph_optimize; int debug_graph; - int debug_fusion; - // how many times a given op was fused - uint64_t fuse_cnt[GGML_OP_COUNT]; + struct ggml_metal_fusion_info * finfo; // capture state int capture_compute; @@ -69,6 +68,10 @@ // extra command buffers for things like getting, setting and copying tensors NSMutableArray * cmd_bufs_ext; + // buffers to release after async Metal operations complete + // if Metal released them, it would do so on a Metal-internal thread without an autorelease pool, which could cause leaks + NSMutableArray * buf_refs; + // the last command buffer queued into the Metal queue with operations relevant to the current Metal backend id cmd_buf_last; @@ -84,106 +87,109 @@ ggml_metal_t ggml_metal_init(ggml_metal_device_t dev) { GGML_LOG_INFO("%s: allocating\n", __func__); + @autoreleasepool { #if TARGET_OS_OSX && !GGML_METAL_NDEBUG - // Show all the Metal device instances in the system - NSArray * devices = MTLCopyAllDevices(); - for (id device in devices) { - GGML_LOG_INFO("%s: found device: %s\n", __func__, [[device name] UTF8String]); - } - [devices release]; // since it was created by a *Copy* C method + // Show all the Metal device instances in the system + NSArray * devices = MTLCopyAllDevices(); + for (id device in devices) { + GGML_LOG_INFO("%s: found device: %s\n", __func__, [[device name] UTF8String]); + } + [devices release]; // since it was created by a *Copy* C method #endif - // init context - ggml_metal_t res = calloc(1, sizeof(struct ggml_metal)); + // init context + ggml_metal_t res = calloc(1, sizeof(struct ggml_metal)); - id device = ggml_metal_device_get_obj(dev); + id device = ggml_metal_device_get_obj(dev); - GGML_LOG_INFO("%s: picking default device: %s\n", __func__, [[device name] UTF8String]); + GGML_LOG_INFO("%s: picking default device: %s\n", __func__, [[device name] UTF8String]); - // TODO: would it be better to have one queue for the backend and one queue for the device? - // the graph encoders and async ops would use the backend queue while the sync ops would use the device queue? - //res->queue = [device newCommandQueue]; [TAG_QUEUE_PER_BACKEND] - id queue = ggml_metal_device_get_queue(dev); - if (queue == nil) { - GGML_LOG_ERROR("%s: error: failed to create command queue\n", __func__); - return NULL; - } - - res->dev = dev; - res->lib = ggml_metal_device_get_library(dev); - if (res->lib == NULL) { - GGML_LOG_WARN("%s: the device does not have a precompiled Metal library - this is unexpected\n", __func__); - GGML_LOG_WARN("%s: will try to compile it on the fly\n", __func__); + // TODO: would it be better to have one queue for the backend and one queue for the device? + // the graph encoders and async ops would use the backend queue while the sync ops would use the device queue? + //res->queue = [device newCommandQueue]; [TAG_QUEUE_PER_BACKEND] + id queue = ggml_metal_device_get_queue(dev); + if (queue == nil) { + GGML_LOG_ERROR("%s: error: failed to create command queue\n", __func__); + free(res); + return NULL; + } - res->lib = ggml_metal_library_init(dev); + res->dev = dev; + res->lib = ggml_metal_device_get_library(dev); if (res->lib == NULL) { - GGML_LOG_ERROR("%s: error: failed to initialize the Metal library\n", __func__); + GGML_LOG_WARN("%s: the device does not have a precompiled Metal library - this is unexpected\n", __func__); + GGML_LOG_WARN("%s: will try to compile it on the fly\n", __func__); - free(res); + res->lib = ggml_metal_library_init(dev); + if (res->lib == NULL) { + GGML_LOG_ERROR("%s: error: failed to initialize the Metal library\n", __func__); - return NULL; - } - } + free(res); - res->ev_cpy = ggml_metal_device_event_init(dev); + return NULL; + } + } - const struct ggml_metal_device_props * props_dev = ggml_metal_device_get_props(dev); + res->ev_cpy = ggml_metal_device_event_init(dev); + res->ev_sync = ggml_metal_device_event_init(dev); - snprintf(res->name, sizeof(res->name), "%s", props_dev->name); + const struct ggml_metal_device_props * props_dev = ggml_metal_device_get_props(dev); - res->d_queue = dispatch_queue_create("ggml-metal", DISPATCH_QUEUE_CONCURRENT); + snprintf(res->name, sizeof(res->name), "%s", props_dev->name); - res->use_fusion = getenv("GGML_METAL_FUSION_DISABLE") == nil; - res->use_concurrency = getenv("GGML_METAL_CONCURRENCY_DISABLE") == nil; + res->d_queue = dispatch_queue_create("ggml-metal", DISPATCH_QUEUE_CONCURRENT); - { - const char * val = getenv("GGML_METAL_GRAPH_DEBUG"); - res->debug_graph = val ? atoi(val) : 0; - } + res->use_concurrency = getenv("GGML_METAL_CONCURRENCY_DISABLE") == nil; - { - const char * val = getenv("GGML_METAL_FUSION_DEBUG"); - res->debug_fusion = val ? atoi(val) : 0; - } + { + const char * val = getenv("GGML_METAL_GRAPH_DEBUG"); + res->debug_graph = val ? atoi(val) : 0; + } - res->use_graph_optimize = true; + res->use_graph_optimize = true; - if (getenv("GGML_METAL_GRAPH_OPTIMIZE_DISABLE") != NULL) { - res->use_graph_optimize = false; - } + if (getenv("GGML_METAL_GRAPH_OPTIMIZE_DISABLE") != NULL) { + res->use_graph_optimize = false; + } - memset(res->fuse_cnt, 0, sizeof(res->fuse_cnt)); + res->finfo = ggml_metal_device_get_fusion_info(dev); + if (ggml_metal_fusion_info_stats(res->finfo)) { + ggml_metal_fusion_info_labels_init(res->finfo); + res->n_cb = 0; + } - GGML_LOG_INFO("%s: use fusion = %s\n", __func__, res->use_fusion ? "true" : "false"); - GGML_LOG_INFO("%s: use concurrency = %s\n", __func__, res->use_concurrency ? "true" : "false"); - GGML_LOG_INFO("%s: use graph optimize = %s\n", __func__, res->use_graph_optimize ? "true" : "false"); + GGML_LOG_INFO("%s: use fusion = %s\n", __func__, ggml_metal_fusion_info_enabled(res->finfo) ? "true" : "false"); + GGML_LOG_INFO("%s: use concurrency = %s\n", __func__, res->use_concurrency ? "true" : "false"); + GGML_LOG_INFO("%s: use graph optimize = %s\n", __func__, res->use_graph_optimize ? "true" : "false"); - res->capture_compute = 0; - res->capture_started = false; - res->capture_scope = nil; + res->capture_compute = 0; + res->capture_started = false; + res->capture_scope = nil; - { - const char * val = getenv("GGML_METAL_CAPTURE_COMPUTE"); - if (val) { - res->capture_compute = atoi(val); + { + const char * val = getenv("GGML_METAL_CAPTURE_COMPUTE"); + if (val) { + res->capture_compute = atoi(val); + } } - } - res->has_error = false; + res->has_error = false; - res->gf = nil; - res->encode_async = nil; - for (int i = 0; i < GGML_METAL_MAX_COMMAND_BUFFERS; ++i) { - res->cmd_bufs[i].obj = nil; - } + res->gf = nil; + res->encode_async = nil; + for (int i = 0; i < GGML_METAL_MAX_COMMAND_BUFFERS; ++i) { + res->cmd_bufs[i].obj = nil; + } - res->cmd_bufs_ext = [[NSMutableArray alloc] init]; + res->cmd_bufs_ext = [[NSMutableArray alloc] init]; + res->buf_refs = [[NSMutableArray alloc] init]; - res->cmd_buf_last = nil; + res->cmd_buf_last = nil; - res->pipelines_ext = ggml_metal_pipelines_init(); + res->pipelines_ext = ggml_metal_pipelines_init(); - return res; + return res; + } } void ggml_metal_free(ggml_metal_t ctx) { @@ -204,20 +210,28 @@ void ggml_metal_free(ggml_metal_t ctx) { [ctx->cmd_bufs_ext removeAllObjects]; [ctx->cmd_bufs_ext release]; + @autoreleasepool { + [ctx->buf_refs removeAllObjects]; + [ctx->buf_refs release]; + } + if (ctx->pipelines_ext) { ggml_metal_pipelines_free(ctx->pipelines_ext); ctx->pipelines_ext = nil; } - if (ctx->debug_fusion > 0) { + if (ggml_metal_fusion_info_debug(ctx->finfo) > 0) { GGML_LOG_DEBUG("%s: fusion stats:\n", __func__); - for (int i = 0; i < GGML_OP_COUNT; i++) { - if (ctx->fuse_cnt[i] == 0) { + + const int n_fusions = ggml_metal_fusion_info_n_fusions(ctx->finfo); + for (int i = 0; i < n_fusions; i++) { + const uint64_t count = ggml_metal_fusion_info_count(ctx->finfo, i); + if (count == 0) { continue; } // note: cannot use ggml_log here - GGML_LOG_DEBUG("%s: - %s: %" PRIu64 "\n", __func__, ggml_op_name((enum ggml_op) i), ctx->fuse_cnt[i]); + GGML_LOG_DEBUG("%s: - %s: %" PRIu64 "\n", __func__, ggml_metal_fusion_info_label(ctx->finfo, i), count); } } @@ -228,6 +242,7 @@ void ggml_metal_free(ggml_metal_t ctx) { dispatch_release(ctx->d_queue); ggml_metal_device_event_free(ctx->dev, ctx->ev_cpy); + ggml_metal_device_event_free(ctx->dev, ctx->ev_sync); free(ctx); } @@ -292,6 +307,10 @@ void ggml_metal_synchronize(ggml_metal_t ctx) { [ctx->cmd_bufs_ext removeAllObjects]; } + + @autoreleasepool { + [ctx->buf_refs removeAllObjects]; + } } static struct ggml_metal_buffer_id ggml_metal_get_buffer_id(const struct ggml_tensor * t) { @@ -335,6 +354,8 @@ void ggml_metal_set_tensor_async(ggml_metal_t ctx, struct ggml_tensor * tensor, [encoder endEncoding]; [cmd_buf commit]; + + [ctx->buf_refs addObject:buf_src]; [buf_src release]; // do not wait here for completion @@ -379,6 +400,8 @@ void ggml_metal_get_tensor_async(ggml_metal_t ctx, const struct ggml_tensor * te [encoder endEncoding]; [cmd_buf commit]; + + [ctx->buf_refs addObject:buf_dst]; [buf_dst release]; // do not wait here for completion @@ -401,10 +424,23 @@ bool ggml_metal_cpy_tensor_async(ggml_metal_t ctx_src, ggml_metal_t ctx_dst, con return false; } + id dst_queue = ggml_metal_device_get_queue(ctx_dst->dev); + id sync_cmd_buf = [dst_queue commandBuffer]; + + ggml_metal_event_encode_signal(ctx_dst->ev_sync, sync_cmd_buf); + + [sync_cmd_buf commit]; + + [ctx_dst->cmd_bufs_ext addObject:sync_cmd_buf]; + ctx_dst->cmd_buf_last = sync_cmd_buf; + + [sync_cmd_buf retain]; + // queue the copy operation into the Metal context // this will be queued at the end, after any currently ongoing GPU operations id queue = ggml_metal_device_get_queue(ctx_src->dev); id cmd_buf = [queue commandBuffer]; + ggml_metal_event_encode_wait(ctx_dst->ev_sync, cmd_buf); id encoder = [cmd_buf blitCommandEncoder]; [encoder copyFromBuffer:bid_src.metal @@ -460,10 +496,17 @@ enum ggml_status ggml_metal_graph_compute(ggml_metal_t ctx, struct ggml_cgraph * @autoreleasepool { ctx->gf = gf; - ctx->n_nodes_0 = MIN(n_main, gf->n_nodes); - ctx->n_nodes_1 = gf->n_nodes - ctx->n_nodes_0; + if (ctx->n_cb == 0) { + // single-threaded encoding: the whole graph is encoded by one command buffer + ctx->n_nodes_0 = gf->n_nodes; + ctx->n_nodes_1 = 0; + ctx->n_nodes_per_cb = 0; + } else { + ctx->n_nodes_0 = MIN(n_main, gf->n_nodes); + ctx->n_nodes_1 = gf->n_nodes - ctx->n_nodes_0; - ctx->n_nodes_per_cb = (ctx->n_nodes_1 + ctx->n_cb - 1) / ctx->n_cb; + ctx->n_nodes_per_cb = (ctx->n_nodes_1 + ctx->n_cb - 1) / ctx->n_cb; + } if (ctx->capture_compute >= 0) { ctx->capture_compute--; @@ -661,6 +704,12 @@ ggml_metal_event_t ggml_metal_get_ev_cpy(ggml_metal_t ctx) { } void ggml_metal_set_n_cb(ggml_metal_t ctx, int n_cb) { + // when fusion stats are collected the graph must be encoded by a single thread so the + // counters are race-free; override whatever the caller requested + if (ggml_metal_fusion_info_stats(ctx->finfo)) { + n_cb = 0; + } + if (ctx->n_cb != n_cb) { ctx->n_cb = MIN(n_cb, GGML_METAL_MAX_COMMAND_BUFFERS); @@ -696,13 +745,12 @@ void ggml_metal_set_n_cb(ggml_metal_t ctx, int n_cb) { ctx->dev, cmd_buf, ctx->gf, + ctx->finfo, idx_start, idx_end, - ctx->use_fusion, ctx->use_concurrency, ctx->capture_compute, - ctx->debug_graph, - ctx->debug_fusion); + ctx->debug_graph); for (int idx = 0; idx < ggml_metal_op_n_nodes(ctx_op); ++idx) { const int res = ggml_metal_op_encode(ctx_op, idx); diff --git a/ggml/src/ggml-metal/ggml-metal-device.cpp b/ggml/src/ggml-metal/ggml-metal-device.cpp index 953c7575..dc6b695e 100644 --- a/ggml/src/ggml-metal/ggml-metal-device.cpp +++ b/ggml/src/ggml-metal/ggml-metal-device.cpp @@ -1,6 +1,7 @@ #include "ggml-metal-device.h" #include "ggml-metal-impl.h" +#include "ggml-metal-tuning.h" #include "ggml-impl.h" @@ -17,10 +18,10 @@ struct ggml_metal_device_deleter { typedef std::unique_ptr ggml_metal_device_ptr; -ggml_metal_device_t ggml_metal_device_get(int device) { +ggml_metal_device_t ggml_metal_device_get(int device, int n_devices) { static std::vector devs; - devs.emplace_back(ggml_metal_device_init(device)); + devs.emplace_back(ggml_metal_device_init(device, n_devices)); return devs.back().get(); } @@ -317,6 +318,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_glu(ggml_metal_l case GGML_GLU_OP_SWIGLU_OAI: op_str = "swiglu_oai"; break; case GGML_GLU_OP_GEGLU_ERF: op_str = "geglu_erf"; break; case GGML_GLU_OP_GEGLU_QUICK: op_str = "geglu_quick"; break; + case GGML_GLU_OP_SWIGLU_CLAMP: op_str = "swiglu_clamp"; break; default: GGML_ABORT("fatal error"); } break; default: GGML_ABORT("fatal error"); @@ -494,25 +496,48 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_lightning_indexe return res; } -ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_dsv4_hc(ggml_metal_library_t lib, ggml_op op) { - const char * name = nullptr; +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_dsv4_hc(ggml_metal_library_t lib, const ggml_tensor * op) { + char name[256]; + const char * base = nullptr; - switch (op) { - case GGML_OP_DSV4_HC_COMB: name = "kernel_dsv4_hc_comb_f32"; break; - case GGML_OP_DSV4_HC_PRE: name = "kernel_dsv4_hc_pre_f32"; break; - case GGML_OP_DSV4_HC_POST: name = "kernel_dsv4_hc_post_f32"; break; - default: GGML_ABORT("fatal error"); + switch (op->op) { + case GGML_OP_DSV4_HC_COMB: + base = "kernel_dsv4_hc_comb_f32"; + snprintf(name, 256, "%s", base); + break; + case GGML_OP_DSV4_HC_PRE: + base = ggml_get_op_params_i32(op, 1) != 0 ? "kernel_dsv4_hc_pre_gated_f32" : "kernel_dsv4_hc_pre_f32"; + snprintf(name, 256, "%s_n_hc=%d", base, (int) op->src[0]->ne[1]); + break; + case GGML_OP_DSV4_HC_POST: + base = op->src[3] ? "kernel_dsv4_hc_post_f32" : "kernel_dsv4_hc_post_nocomb_f32"; + snprintf(name, 256, "%s", base); + break; + default: + GGML_ABORT("fatal error"); } ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { - res = ggml_metal_library_compile_pipeline(lib, name, name, nullptr); + ggml_metal_cv_t cv = nullptr; + + if (op->op == GGML_OP_DSV4_HC_PRE) { + cv = ggml_metal_cv_init(); + ggml_metal_cv_set_int32(cv, (int32_t) op->src[0]->ne[1], FC_DSV4_HC + 0); + } + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + if (cv) { + ggml_metal_cv_free(cv); + } } return res; } -ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv(ggml_metal_library_t lib, const ggml_tensor * op) { +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv( + ggml_metal_library_t lib, const ggml_tensor * op, int32_t nc, bool use_silu) { GGML_ASSERT(op->src[0]->type == GGML_TYPE_F32); GGML_ASSERT(op->src[1]->type == GGML_TYPE_F32); @@ -529,17 +554,24 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv(ggml_me } snprintf(base, 256, "kernel_ssm_conv_%s_%s%s", ggml_type_name(op->src[0]->type), ggml_type_name(op->src[1]->type), suffix); - snprintf(name, 256, "%s", base); + snprintf(name, 256, "%s_nc=%d_silu=%d", base, nc, use_silu ? 1 : 0); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { - res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + ggml_metal_cv_t cv = ggml_metal_cv_init(); + ggml_metal_cv_set_bool(cv, use_silu, FC_SSM_CONV + 1); + ggml_metal_cv_set_int32(cv, nc, FC_SSM_CONV + 2); + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + ggml_metal_cv_free(cv); } return res; } -ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched(ggml_metal_library_t lib, const ggml_tensor * op, int ssm_conv_bs) { +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched( + ggml_metal_library_t lib, const ggml_tensor * op, int ssm_conv_bs, int32_t nc, bool use_silu) { GGML_ASSERT(op->src[0]->type == GGML_TYPE_F32); GGML_ASSERT(op->src[1]->type == GGML_TYPE_F32); @@ -555,13 +587,15 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched } snprintf(base, 256, "kernel_ssm_conv_%s_%s_batched%s", ggml_type_name(op->src[0]->type), ggml_type_name(op->src[1]->type), suffix); - snprintf(name, 256, "%s_ssm_conv_bs=%d", base, ssm_conv_bs); + snprintf(name, 256, "%s_ssm_conv_bs=%d_nc=%d_silu=%d", base, ssm_conv_bs, nc, use_silu ? 1 : 0); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { ggml_metal_cv_t cv = ggml_metal_cv_init(); ggml_metal_cv_set_int16(cv, ssm_conv_bs, FC_SSM_CONV + 0); + ggml_metal_cv_set_bool(cv, use_silu, FC_SSM_CONV + 1); + ggml_metal_cv_set_int32(cv, nc, FC_SSM_CONV + 2); res = ggml_metal_library_compile_pipeline(lib, base, name, cv); @@ -571,7 +605,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched return res; } -ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan(ggml_metal_library_t lib, const ggml_tensor * op) { +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan(ggml_metal_library_t lib, const ggml_tensor * op, bool tail) { GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); char base[256]; @@ -579,7 +613,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan(ggml_me const int nsg = (ne00 + 31)/32; - snprintf(base, 256, "kernel_ssm_scan_%s", ggml_type_name(op->src[0]->type)); + snprintf(base, 256, "kernel_ssm_scan_%s%s", ggml_type_name(op->src[0]->type), tail ? "_tail" : ""); snprintf(name, 256, "%s_nsg=%d", base, nsg); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); @@ -592,7 +626,28 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan(ggml_me // - sgptg floats for shared_x_dt (nsg) // - sgptg floats for shared_dA (nsg) // Total: nsg * (32 + 2) floats - res.smem = (32 + 2)*sizeof(float)*nsg; + res.smem = GGML_PAD((32 + 2)*sizeof(float)*nsg, 16); + + return res; +} + +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan_ssd_mma(ggml_metal_library_t lib, const ggml_tensor * op) { + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_ssm_scan_ssd_mma_%s", ggml_type_name(op->src[0]->type)); + snprintf(name, 256, "%s", base); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + } + + // acs/exp(acs)/state-decay vectors + dtX + SAM rows + two 8x8 tiles per simdgroup + res.smem = (3*OP_SSM_SCAN_SSD_CS + + OP_SSM_SCAN_SSD_CS*OP_SSM_SCAN_SSD_HD + + OP_SSM_SCAN_SSD_NSG*8*OP_SSM_SCAN_SSD_CS + + OP_SSM_SCAN_SSD_NSG*2*8*8)*sizeof(float); return res; } @@ -816,6 +871,8 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv(ggml_meta const char * suffix = ""; + bool split = false; + // use custom matrix x vector kernel switch (tsrc0) { case GGML_TYPE_F32: @@ -907,39 +964,82 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv(ggml_meta nsg = N_SG_IQ2_XXS; nr0 = N_R0_IQ2_XXS; smem = 256*8+128; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_XXS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ2_XS: { nsg = N_SG_IQ2_XS; nr0 = N_R0_IQ2_XS; smem = 512*8+128; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_XS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ3_XXS: { nsg = N_SG_IQ3_XXS; nr0 = N_R0_IQ3_XXS; smem = 256*4+128; + + // split the rows across threads when there are fewer than 32 chunks per row + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ3_XXS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ3_S: { nsg = N_SG_IQ3_S; nr0 = N_R0_IQ3_S; smem = 512*4; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ3_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ2_S: { nsg = N_SG_IQ2_S; nr0 = N_R0_IQ2_S; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ1_S: { nsg = N_SG_IQ1_S; nr0 = N_R0_IQ1_S; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ1_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ1_M: { nsg = N_SG_IQ1_M; nr0 = N_R0_IQ1_M; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ1_M_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ4_NL: { @@ -970,7 +1070,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv(ggml_meta const int16_t r3 = (int16_t) (ne13 / ne03); snprintf(base, 256, "kernel_mul_mv_%s_%s%s", ggml_type_name(tsrc0), ggml_type_name(tsrc1), suffix); - snprintf(name, 256, "%s_nsg=%d_ne12=%d_r2=%d_r3=%d", base, nsg, ne12, r2, r3); + snprintf(name, 256, "%s_nsg=%d_ne12=%d_r2=%d_r3=%d_split=%d", base, nsg, ne12, r2, r3, split); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { @@ -980,6 +1080,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv(ggml_meta ggml_metal_cv_set_int16(cv, (int16_t) ne12, FC_MUL_MV + 2); ggml_metal_cv_set_int16(cv, r2, FC_MUL_MV + 3); ggml_metal_cv_set_int16(cv, r3, FC_MUL_MV + 4); + ggml_metal_cv_set_bool (cv, split, FC_MUL_MV + 5); res = ggml_metal_library_compile_pipeline(lib, base, name, cv); @@ -994,6 +1095,40 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv(ggml_meta return res; } +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_amax_part(ggml_metal_library_t lib) { + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_mul_mm_id_amax_part_f32"); + snprintf(name, 256, "%s", base); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + } + + res.smem = 32*sizeof(float); + + return res; +} + +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_amax(ggml_metal_library_t lib) { + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_mul_mm_id_amax_f32"); + snprintf(name, 256, "%s", base); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + } + + res.smem = 32*sizeof(float); + + return res; +} + ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_map0(ggml_metal_library_t lib, int ne02, int ne20) { char base[256]; char name[256]; @@ -1007,6 +1142,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_map0(g } res.smem = (size_t) ne02*ne20*sizeof(uint16_t); + res.smem = GGML_PAD(res.smem, 16); return res; } @@ -1020,14 +1156,18 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id(ggml_m const bool bc_inp = op->src[0]->ne[0] % 32 != 0; + // src1 prec [TAG_GGML_PREC] + const bool amax = ggml_get_op_params_i32(op, 3) == GGML_PREC_F32; + snprintf(base, 256, "kernel_mul_mm_id_%s_%s", ggml_type_name(tsrc0), ggml_type_name(tsrc1)); - snprintf(name, 256, "%s_bci=%d", base, bc_inp); + snprintf(name, 256, "%s_bci=%d_amax=%d", base, bc_inp, amax); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { ggml_metal_cv_t cv = ggml_metal_cv_init(); ggml_metal_cv_set_bool(cv, bc_inp, FC_MUL_MM + 0); + ggml_metal_cv_set_bool(cv, amax, FC_MUL_MM + 6); res = ggml_metal_library_compile_pipeline(lib, base, name, cv); @@ -1057,6 +1197,8 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_id(ggml_m const char * suffix = ""; + bool split = false; + // use custom matrix x vector kernel switch (tsrc0) { case GGML_TYPE_F32: @@ -1141,39 +1283,82 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_id(ggml_m nsg = N_SG_IQ2_XXS; nr0 = N_R0_IQ2_XXS; smem = 256*8+128; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_XXS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ2_XS: { nsg = N_SG_IQ2_XS; nr0 = N_R0_IQ2_XS; smem = 512*8+128; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_XS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ3_XXS: { nsg = N_SG_IQ3_XXS; nr0 = N_R0_IQ3_XXS; smem = 256*4+128; + + // split the rows across threads when there are fewer than 32 chunks per row + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ3_XXS_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ3_S: { nsg = N_SG_IQ3_S; nr0 = N_R0_IQ3_S; smem = 512*4; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ3_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ2_S: { nsg = N_SG_IQ2_S; nr0 = N_R0_IQ2_S; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ2_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ1_S: { nsg = N_SG_IQ1_S; nr0 = N_R0_IQ1_S; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ1_S_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ1_M: { nsg = N_SG_IQ1_M; nr0 = N_R0_IQ1_M; + + const int nb32 = ne00/32; + if (nb32 < 32 && (32 % nb32) == 0) { + nr0 = N_R0_IQ1_M_SPLIT; + split = true; + } } break; case GGML_TYPE_IQ4_NL: { @@ -1200,7 +1385,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_id(ggml_m }; snprintf(base, 256, "kernel_mul_mv_id_%s_%s%s", ggml_type_name(tsrc0), ggml_type_name(tsrc1), suffix); - snprintf(name, 256, "%s_nsg=%d", base, nsg); + snprintf(name, 256, "%s_nsg=%d_split=%d", base, nsg, split); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { @@ -1210,6 +1395,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_id(ggml_m ggml_metal_cv_set_int16(cv, 1, FC_MUL_MV + 2); ggml_metal_cv_set_int16(cv, 1, FC_MUL_MV + 3); ggml_metal_cv_set_int16(cv, 1, FC_MUL_MV + 4); + ggml_metal_cv_set_bool (cv, split, FC_MUL_MV + 5); res = ggml_metal_library_compile_pipeline(lib, base, name, cv); @@ -1297,11 +1483,11 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_argsort_merge(gg return res; } -ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_fwht(ggml_metal_library_t lib, int n) { +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_fwht(ggml_metal_library_t lib, int n, ggml_type tsrc) { char base[256]; char name[256]; - snprintf(base, 256, "kernel_fwht_f32_%d", n); + snprintf(base, 256, "kernel_fwht_%s_%d", ggml_type_name(tsrc), n); snprintf(name, 256, "%s", base); ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); @@ -1312,7 +1498,7 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_fwht(ggml_metal_ return res; } -// note: reuse the argsort kernel for top_k +// note: reuse the argsort kernel for the bitonic top_k fallback ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k(ggml_metal_library_t lib, const ggml_tensor * op) { assert(op->op == GGML_OP_TOP_K); @@ -1340,6 +1526,23 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k(ggml_metal return res; } +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k_radix(ggml_metal_library_t lib, const ggml_tensor * op) { + assert(op->op == GGML_OP_TOP_K); + + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_top_k_%s_%s", ggml_type_name(op->src[0]->type), ggml_type_name(op->type)); + snprintf(name, 256, "%s", base); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + } + + return res; +} + ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k_merge(ggml_metal_library_t lib, const ggml_tensor * op) { assert(op->op == GGML_OP_TOP_K); @@ -1366,6 +1569,49 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k_merge(ggml return res; } +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_topk_moe( + ggml_metal_library_t lib, int32_t n_expert, int32_t top_k, bool with_norm) { + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_topk_moe_f32"); + snprintf(name, 256, "%s_n_expert=%d_top_k=%d_with_norm=%d", base, n_expert, top_k, with_norm ? 1 : 0); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + ggml_metal_cv_t cv = ggml_metal_cv_init(); + ggml_metal_cv_set_bool (cv, with_norm, FC_TOPK_MOE + 0); + ggml_metal_cv_set_int32(cv, n_expert, FC_TOPK_MOE + 1); + ggml_metal_cv_set_int32(cv, top_k, FC_TOPK_MOE + 2); + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + ggml_metal_cv_free(cv); + } + + return res; +} + +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_moe_reduce(ggml_metal_library_t lib, int32_t n_expert_used) { + char base[256]; + char name[256]; + + snprintf(base, 256, "kernel_moe_reduce_f32"); + snprintf(name, 256, "%s_n_expert_used=%d", base, n_expert_used); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + ggml_metal_cv_t cv = ggml_metal_cv_init(); + ggml_metal_cv_set_int32(cv, n_expert_used, FC_MOE_REDUCE + 0); + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + ggml_metal_cv_free(cv); + } + + return res; +} + ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_pad( ggml_metal_library_t lib, const struct ggml_tensor * op, @@ -1409,6 +1655,23 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_p return res; } +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_kv_f16( + ggml_metal_library_t lib, + const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + char base[256]; + + snprintf(base, 256, "kernel_flash_attn_ext_kv_%s_f16", ggml_type_name(op->src[1]->type)); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, base); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, base, base, nullptr); + } + + return res; +} + ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_blk( ggml_metal_library_t lib, const struct ggml_tensor * op, @@ -1460,7 +1723,10 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext( bool has_bias, bool has_scap, bool has_kvpad, - int32_t nsg) { + int32_t nsg, + bool use_kv_f16, + int32_t ns10, + int32_t ns20) { assert(op->op == GGML_OP_FLASH_ATTN_EXT); char base[256]; @@ -1469,15 +1735,14 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext( const int32_t dk = (int32_t) op->src[1]->ne[0]; const int32_t dv = (int32_t) op->src[2]->ne[0]; - const int32_t ns10 = op->src[1]->nb[1]/op->src[1]->nb[0]; - const int32_t ns20 = op->src[2]->nb[1]/op->src[2]->nb[0]; + const char * type = use_kv_f16 ? "f16" : ggml_type_name(op->src[1]->type); // do bounds checks for the mask? const bool bc_mask = op->src[3] && (op->src[3]->ne[1] % 8 != 0); snprintf(base, 256, "kernel_%s_%s_dk%d_dv%d", "flash_attn_ext", - ggml_type_name(op->src[1]->type), + type, dk, dv); @@ -1517,6 +1782,26 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext( return res; } +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_vec_idx( + ggml_metal_library_t lib, + const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + assert(op->src[3]); + + char name[256]; + + snprintf(name, 256, "kernel_flash_attn_ext_vec_idx"); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + res = ggml_metal_library_compile_pipeline(lib, name, name, nullptr); + } + + GGML_UNUSED(op); + + return res; +} + ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_vec( ggml_metal_library_t lib, const ggml_tensor * op, @@ -1525,8 +1810,14 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_v bool has_bias, bool has_scap, bool has_kvpad, + bool has_sparse, + int32_t nqpsg, + int32_t ne, int32_t nsg, - int32_t nwg) { + int32_t nwg, + bool use_kv_f16, + int32_t ns10, + int32_t ns20) { assert(op->op == GGML_OP_FLASH_ATTN_EXT); char base[256]; @@ -1535,22 +1826,28 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_v const int32_t dk = (int32_t) op->src[1]->ne[0]; const int32_t dv = (int32_t) op->src[2]->ne[0]; - const int32_t ns10 = op->src[1]->nb[1]/op->src[1]->nb[0]; - const int32_t ns20 = op->src[2]->nb[1]/op->src[2]->nb[0]; + const char * type = use_kv_f16 ? "f16" : ggml_type_name(op->src[1]->type); - snprintf(base, 256, "kernel_%s_%s_dk%d_dv%d", + char qne_suffix[16] = {0}; + if (!(nqpsg == 1 && ne == ggml_metal_tuning::fa_vec_baseline_ne(dk, dv))) { + snprintf(qne_suffix, sizeof(qne_suffix), "_q%d_ne%d", nqpsg, ne); + } + + snprintf(base, 256, "kernel_%s_%s_dk%d_dv%d%s", "flash_attn_ext_vec", - ggml_type_name(op->src[1]->type), + type, dk, - dv); + dv, + qne_suffix); - snprintf(name, 256, "%s_mask=%d_sink=%d_bias=%d_scap=%d_kvpad=%d_ns10=%d_ns20=%d_nsg=%d_nwg=%d", + snprintf(name, 256, "%s_mask=%d_sink=%d_bias=%d_scap=%d_kvpad=%d_sparse=%d_ns10=%d_ns20=%d_nsg=%d_nwg=%d", base, has_mask, has_sinks, has_bias, has_scap, has_kvpad, + has_sparse, ns10, ns20, nsg, nwg); @@ -1563,7 +1860,8 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_v ggml_metal_cv_set_bool(cv, has_sinks, FC_FLASH_ATTN_EXT_VEC + 1); ggml_metal_cv_set_bool(cv, has_bias, FC_FLASH_ATTN_EXT_VEC + 2); ggml_metal_cv_set_bool(cv, has_scap, FC_FLASH_ATTN_EXT_VEC + 3); - ggml_metal_cv_set_bool(cv, has_kvpad, FC_FLASH_ATTN_EXT_VEC + 4); + ggml_metal_cv_set_bool(cv, has_kvpad, FC_FLASH_ATTN_EXT_VEC + 4); + ggml_metal_cv_set_bool(cv, has_sparse, FC_FLASH_ATTN_EXT_VEC + 5); ggml_metal_cv_set_int32(cv, ns10, FC_FLASH_ATTN_EXT_VEC + 20); ggml_metal_cv_set_int32(cv, ns20, FC_FLASH_ATTN_EXT_VEC + 21); @@ -1768,7 +2066,48 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_norm(ggml_metal_ ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); if (!res.pipeline) { - res = ggml_metal_library_compile_pipeline(lib, base, name, nullptr); + ggml_metal_cv_t cv = ggml_metal_cv_init(); + ggml_metal_cv_set_bool(cv, false, FC_NORM + 0); + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + ggml_metal_cv_free(cv); + } + + res.smem = 32*sizeof(float); + + return res; +} + +ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_norm_scale(ggml_metal_library_t lib, const ggml_tensor * op) { + assert(op->op == GGML_OP_NORM || op->op == GGML_OP_RMS_NORM); + + GGML_ASSERT(ggml_is_contiguous_rows(op->src[0])); + + char base[256]; + char name[256]; + + const char * suffix = ""; + if (op->ne[0] % 4 == 0) { + suffix = "_4"; + } + + switch (op->op) { + case GGML_OP_NORM: snprintf(base, 256, "kernel_norm_mul_f32%s", suffix); break; + case GGML_OP_RMS_NORM: snprintf(base, 256, "kernel_rms_norm_mul_f32%s", suffix); break; + default: GGML_ABORT("fatal error"); + } + + snprintf(name, 256, "%s_use_scale", base); + + ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name); + if (!res.pipeline) { + ggml_metal_cv_t cv = ggml_metal_cv_init(); + ggml_metal_cv_set_bool(cv, true, FC_NORM + 0); + + res = ggml_metal_library_compile_pipeline(lib, base, name, cv); + + ggml_metal_cv_free(cv); } res.smem = 32*sizeof(float); diff --git a/ggml/src/ggml-metal/ggml-metal-device.h b/ggml/src/ggml-metal/ggml-metal-device.h index 7e1deeaa..1bdaecc7 100644 --- a/ggml/src/ggml-metal/ggml-metal-device.h +++ b/ggml/src/ggml-metal/ggml-metal-device.h @@ -126,10 +126,11 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_cumsum_ad struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_tri (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_soft_max (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_lightning_indexer (ggml_metal_library_t lib, const struct ggml_tensor * op); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_dsv4_hc (ggml_metal_library_t lib, enum ggml_op op); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv (ggml_metal_library_t lib, const struct ggml_tensor * op); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched (ggml_metal_library_t lib, const struct ggml_tensor * op, int ssm_conv_bs); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan (ggml_metal_library_t lib, const struct ggml_tensor * op); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_dsv4_hc (ggml_metal_library_t lib, const struct ggml_tensor * op); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv (ggml_metal_library_t lib, const struct ggml_tensor * op, int32_t nc, bool use_silu); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_conv_batched (ggml_metal_library_t lib, const struct ggml_tensor * op, int ssm_conv_bs, int32_t nc, bool use_silu); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan (ggml_metal_library_t lib, const struct ggml_tensor * op, bool tail); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_ssm_scan_ssd_mma (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_rwkv (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_gated_delta_net (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_solve_tri (ggml_metal_library_t lib, const struct ggml_tensor * op); @@ -137,19 +138,26 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_ex struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_map0 (ggml_metal_library_t lib, int ne02, int ne20); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_amax(ggml_metal_library_t lib); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id_amax_part(ggml_metal_library_t lib); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mm_id (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_mul_mv_id (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_argmax (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_argsort (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_argsort_merge (ggml_metal_library_t lib, const struct ggml_tensor * op); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_fwht (ggml_metal_library_t lib, int n); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_fwht (ggml_metal_library_t lib, int n, enum ggml_type tsrc); + struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k (ggml_metal_library_t lib, const struct ggml_tensor * op); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k_radix (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_top_k_merge (ggml_metal_library_t lib, const struct ggml_tensor * op); -struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_bin (ggml_metal_library_t lib, const struct ggml_tensor * op, int32_t n_fuse ); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_topk_moe (ggml_metal_library_t lib, int32_t n_expert, int32_t top_k, bool with_norm); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_moe_reduce (ggml_metal_library_t lib, int32_t n_expert_used); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_bin (ggml_metal_library_t lib, const struct ggml_tensor * op, int32_t n_fuse); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_bin_one (ggml_metal_library_t lib, enum ggml_op op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_l2_norm (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_group_norm (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_norm (ggml_metal_library_t lib, const struct ggml_tensor * op, int32_t n_fuse); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_norm_scale (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_rope (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_im2col (ggml_metal_library_t lib, const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_conv_transpose_1d (ggml_metal_library_t lib, const struct ggml_tensor * op); @@ -176,6 +184,10 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_att bool has_mask, int32_t ncpsg); +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_kv_f16( + ggml_metal_library_t lib, + const struct ggml_tensor * op); + struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_blk( ggml_metal_library_t lib, const struct ggml_tensor * op, @@ -190,7 +202,14 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_att bool has_bias, bool has_scap, bool has_kvpad, - int32_t nsg); + int32_t nsg, + bool use_kv_f16, + int32_t ns10, + int32_t ns20); + +struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_vec_idx( + ggml_metal_library_t lib, + const struct ggml_tensor * op); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_vec( ggml_metal_library_t lib, @@ -200,8 +219,14 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_att bool has_bias, bool has_scap, bool has_kvpad, + bool has_sparse, + int32_t nqpsg, + int32_t ne, int32_t nsg, - int32_t nwg); + int32_t nwg, + bool use_kv_f16, + int32_t ns10, + int32_t ns20); struct ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_flash_attn_ext_vec_reduce( ggml_metal_library_t lib, @@ -245,10 +270,15 @@ enum ggml_metal_device_id { GGML_METAL_DEVICE_M5_PRO, GGML_METAL_DEVICE_M5_MAX, GGML_METAL_DEVICE_M5_ULTRA, + GGML_METAL_DEVICE_A18_PRO, }; +const char * ggml_metal_device_id_token(enum ggml_metal_device_id id); + struct ggml_metal_device_props { int device; + int device_phys; + int device_virt; char name[128]; char desc[128]; @@ -267,6 +297,7 @@ struct ggml_metal_device_props { bool supports_gpu_family_apple7; enum ggml_metal_device_id device_id; + int gpu_family; int op_offload_min_batch_size; }; @@ -276,10 +307,10 @@ typedef struct ggml_metal_event * ggml_metal_event_t; void ggml_metal_event_encode_signal(ggml_metal_event_t ev, ggml_metal_cmd_buf_t cmd_buf); void ggml_metal_event_encode_wait (ggml_metal_event_t ev, ggml_metal_cmd_buf_t cmd_buf); -ggml_metal_device_t ggml_metal_device_init(int device); +ggml_metal_device_t ggml_metal_device_init(int device, int n_devices); void ggml_metal_device_free(ggml_metal_device_t dev); -ggml_metal_device_t ggml_metal_device_get(int device); +ggml_metal_device_t ggml_metal_device_get(int device, int n_devices); void * ggml_metal_device_get_obj (ggml_metal_device_t dev); // id void * ggml_metal_device_get_queue(ggml_metal_device_t dev); // id @@ -300,6 +331,11 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te const struct ggml_metal_device_props * ggml_metal_device_get_props(ggml_metal_device_t dev); +struct ggml_metal_fusion_info; + +// the device-owned fusion debugging context (NULL unless fusion debugging is enabled) +struct ggml_metal_fusion_info * ggml_metal_device_get_fusion_info(ggml_metal_device_t dev); + // // device buffers // diff --git a/ggml/src/ggml-metal/ggml-metal-device.m b/ggml/src/ggml-metal/ggml-metal-device.m index 312b00dc..81ea7f9d 100644 --- a/ggml/src/ggml-metal/ggml-metal-device.m +++ b/ggml/src/ggml-metal/ggml-metal-device.m @@ -1,8 +1,10 @@ #import "ggml-metal-device.h" +#import "ggml-metal-fusion.h" #import "ggml-impl.h" #import "ggml-backend-impl.h" #import "ggml-metal-impl.h" +#import "ggml-metal-common.h" #include @@ -26,6 +28,9 @@ static const NSInteger MTLGPUFamilyMetal3_GGML = 5001; static const NSInteger MTLGPUFamilyMetal4_GGML = 5002; +// MTLLanguageVersion4_0 is not present in older SDKs +static const NSUInteger MTLLanguageVersion4_0_GGML = 4 << 16; + #if !GGML_METAL_EMBED_LIBRARY // Here to assist with NSBundle Path Hack @interface GGMLMetalClass : NSObject @@ -95,8 +100,66 @@ int ggml_metal_pipeline_max_theads_per_threadgroup(struct ggml_metal_pipeline_wi return pipeline.pipeline->obj.maxTotalThreadsPerThreadgroup; } +// +// MTLLibrary collection (one library per op-source, compiled separately) +// + +// Single source of truth for the per-kind metal libraries. The order here +// defines the enum values and every per-kind table below, so adding a library +// is a one-line change here (plus adding its source to CMakeLists.txt). +// X(suffix, name): name is both the kernels/.metal basename and the +// ggml_metallib__{start,end} embed-symbol stem. +#define GGML_METAL_LIBS \ + X(FA, fa) \ + X(MUL_MV, mul_mv) \ + X(MUL_MM, mul_mm) \ + X(QUANTIZE, quantize) \ + X(SOFTMAX, softmax) \ + X(NORM, norm) \ + X(UNARY, unary) \ + X(BINBCAST, binbcast) \ + X(REDUCE, reduce) \ + X(TRI, tri) \ + X(SSM, ssm) \ + X(WKV, wkv) \ + X(GATED_DELTA_NET, gated_delta_net)\ + X(SOLVE_TRI, solve_tri) \ + X(ROPE, rope) \ + X(CONV, conv) \ + X(UPSCALE, upscale) \ + X(ARGSORT, argsort) \ + X(POOL, pool) \ + X(MISC, misc) + +enum ggml_metal_lib_kind { +#define X(e, s) GGML_METAL_LIB_##e, + GGML_METAL_LIBS +#undef X + GGML_METAL_LIB_COUNT, +}; + +static const char * const k_lib_names[GGML_METAL_LIB_COUNT] = { +#define X(e, s) [GGML_METAL_LIB_##e] = #s, + GGML_METAL_LIBS +#undef X +}; + struct ggml_metal_library { - id obj; + // Per-kind compiled libraries. When single_library is true, the whole library + // (e.g. a pre-compiled default.metallib or a from-source build) lives at + // objs[0] and the remaining slots are nil. + id objs[GGML_METAL_LIB_COUNT]; + bool single_library; // true: combined library at objs[0]; false: per-kind libs in objs[*] + + // Routing table: kernel function name -> objs[] index, populated from each + // compiled library's -[MTLLibrary functionNames]. The actual compiled + // libraries are the single source of truth for which library owns a kernel, + // so adding kernels later requires no manual routing maintenance. + // nil in single_library mode (everything resolves to objs[0]). + NSMutableDictionary * fn_to_lib; + + // kernels from a second metallib, resolved ahead of the combined library + NSSet * override_fns; ggml_metal_device_t dev; ggml_metal_pipelines_t pipelines; // cache of compiled pipelines @@ -104,160 +167,422 @@ int ggml_metal_pipeline_max_theads_per_threadgroup(struct ggml_metal_pipeline_wi NSLock * lock; }; -ggml_metal_library_t ggml_metal_library_init(ggml_metal_device_t dev) { - id library = nil; - id device = ggml_metal_device_get_obj(dev); +// Build the fn_to_lib routing table by querying each compiled library's public +// function names. Call once after all per-kind libraries have been compiled. +static void ggml_metal_library_build_index(ggml_metal_library_t lib) { + @autoreleasepool { + NSMutableDictionary * index = [[NSMutableDictionary alloc] init]; + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + for (NSString * fname in [lib->objs[kind] functionNames]) { + index[fname] = @(kind); + } + } + lib->fn_to_lib = index; + } +} - // load library - // - // - first check if the library is embedded - // - then check if the library is in the bundle - // - if not found, load the source and compile it - // - if that fails, return NULL - // - // TODO: move to a function - { - const int64_t t_start = ggml_time_us(); +// note: defined below, after struct ggml_metal_device +static void ggml_metal_device_disable_tensor(ggml_metal_device_t dev); - NSError * error = nil; - NSString * src = nil; +// the tensor API headers are exposed to the shader compiler only at Metal language version 4.0 +static void ggml_metal_compile_options_set_lang(MTLCompileOptions * options, bool has_tensor) { + if (!has_tensor) { + return; + } -#if GGML_METAL_EMBED_LIBRARY - GGML_LOG_INFO("%s: using embedded metal library\n", __func__); + options.languageVersion = (MTLLanguageVersion) MTLLanguageVersion4_0_GGML; +} - extern const char ggml_metallib_start[]; - extern const char ggml_metallib_end[]; +// Parse a `#include "name"` line. Returns the quoted name in *include_name on +// success. Whitespace-tolerant; ignores `#include <...>` (system headers). +static bool ggml_metal_library_parse_quoted_include(NSString * line, NSString ** include_name) { + NSScanner * scanner = [NSScanner scannerWithString:line]; + scanner.charactersToBeSkipped = [NSCharacterSet whitespaceCharacterSet]; - src = [[NSString alloc] initWithBytes:ggml_metallib_start length:(ggml_metallib_end-ggml_metallib_start) encoding:NSUTF8StringEncoding]; -#else + if (![scanner scanString:@"#" intoString:NULL] || + ![scanner scanString:@"include" intoString:NULL] || + ![scanner scanString:@"\"" intoString:NULL]) { + return false; + } -#ifdef SWIFT_PACKAGE - NSBundle * bundle = SWIFTPM_MODULE_BUNDLE; -#else - NSBundle * bundle = [NSBundle bundleForClass:[GGMLMetalClass class]]; -#endif + NSString * name = nil; + if (![scanner scanUpToString:@"\"" intoString:&name]) { + return false; + } - NSString * path_lib = [bundle pathForResource:@"default" ofType:@"metallib"]; - if (path_lib == nil) { - // Try to find the resource in the directory where the current binary located. - NSString * bin_cur = [[NSProcessInfo processInfo] arguments][0]; - NSString * bin_dir = [bin_cur stringByDeletingLastPathComponent]; - - NSString * path_lib_default = [NSString pathWithComponents:@[bin_dir, @"default.metallib"]]; - if ([[NSFileManager defaultManager] isReadableFileAtPath:path_lib_default]) { - GGML_LOG_INFO("%s: found '%s'\n", __func__, [path_lib_default UTF8String]); - - NSDictionary * atts = [[NSFileManager defaultManager] attributesOfItemAtPath:path_lib_default error:&error]; - if (atts && atts[NSFileType] == NSFileTypeSymbolicLink) { - // Optionally, if this is a symlink, try to resolve it. - path_lib_default = [[NSFileManager defaultManager] destinationOfSymbolicLinkAtPath:path_lib_default error:&error]; - if (path_lib_default && [path_lib_default length] > 0 && ![[path_lib_default substringToIndex:1] isEqualToString:@"/"]) { - // It is a relative path, adding the binary directory as directory prefix. - path_lib_default = [NSString pathWithComponents:@[bin_dir, path_lib_default]]; - } - if (!path_lib_default || ![[NSFileManager defaultManager] isReadableFileAtPath:path_lib_default]) { - // Link to the resource could not be resolved. - path_lib_default = nil; - } else { - GGML_LOG_INFO("%s: symlink resolved '%s'\n", __func__, [path_lib_default UTF8String]); - } + if (include_name) { + *include_name = name; + } + return true; +} + +// Recursively inline `#include "name"` directives. System includes (<...>), +// `#if/#else/#endif`, and other preprocessor lines are passed through to the +// Metal compiler unchanged. `#pragma once` is dropped since `seen` already +// guards against double-inclusion. +static bool ggml_metal_library_flatten_file(NSMutableString * dst, NSString * path, + NSArray * search_paths, + NSMutableSet * seen, NSError ** error) { + NSString * key = [path stringByStandardizingPath]; + if ([seen containsObject:key]) { + return true; + } + [seen addObject:key]; + + NSString * src = [NSString stringWithContentsOfFile:path encoding:NSUTF8StringEncoding error:error]; + if (!src) { + return false; + } + + NSFileManager * fm = [NSFileManager defaultManager]; + for (NSString * line in [src componentsSeparatedByString:@"\n"]) { + NSString * trimmed = [line stringByTrimmingCharactersInSet:[NSCharacterSet whitespaceCharacterSet]]; + if ([trimmed isEqualToString:@"#pragma once"]) { + continue; + } + + NSString * include_name = nil; + if (ggml_metal_library_parse_quoted_include(line, &include_name)) { + NSString * resolved = nil; + for (NSString * dir in search_paths) { + NSString * candidate = [dir stringByAppendingPathComponent:include_name]; + if ([fm isReadableFileAtPath:candidate]) { + resolved = candidate; + break; } - } else { - // The resource couldn't be found in the binary's directory. - path_lib_default = nil; } - - path_lib = path_lib_default; + if (!resolved) { + if (error) { + NSString * msg = [NSString stringWithFormat:@"could not resolve include \"%@\" from '%@'", include_name, path]; + *error = [NSError errorWithDomain:@"ggml-metal-source-flatten" code:1 + userInfo:@{NSLocalizedDescriptionKey: msg}]; + } + return false; + } + if (!ggml_metal_library_flatten_file(dst, resolved, search_paths, seen, error)) { + return false; + } + continue; } - if (path_lib != nil) { - // pre-compiled library found - NSURL * libURL = [NSURL fileURLWithPath:path_lib]; - GGML_LOG_INFO("%s: loading '%s'\n", __func__, [path_lib UTF8String]); + [dst appendString:line]; + [dst appendString:@"\n"]; + } - library = [device newLibraryWithURL:libURL error:&error]; - if (error) { - GGML_LOG_ERROR("%s: error: %s\n", __func__, [[error description] UTF8String]); - return nil; + return true; +} + +static NSString * ggml_metal_library_flatten_source(NSString * path_source, NSError ** error) { + // Search paths cover both runtime layout (build/bin/kernels + build/bin) + // and source-tree layout (ggml/src/ggml-metal/kernels + ggml/src/ggml-metal + ggml/src). + NSString * path_kernels = [path_source stringByDeletingLastPathComponent]; + NSString * path_base = [path_kernels stringByDeletingLastPathComponent]; + NSArray * search_paths = @[ + path_kernels, + path_base, + [path_base stringByDeletingLastPathComponent], + ]; + + NSMutableString * src = [[NSMutableString alloc] init]; + NSMutableSet * seen = [NSMutableSet set]; + + if (!ggml_metal_library_flatten_file(src, path_source, search_paths, seen, error)) { + [src release]; + return nil; + } + return src; +} + +// Compile all per-kind libraries in parallel. `source_for_kind` returns the MSL +// source for a kind (the helper takes ownership and releases it), or nil with +// *err set on failure. On success the objs[] slots are populated and the routing +// index is built; on any failure every error is logged and false is returned +// (the caller is responsible for freeing `res`). +static bool ggml_metal_library_compile_all( + ggml_metal_library_t res, + id device, + NSDictionary * prep, + NSString * (^source_for_kind)(int kind, NSError ** err), + const char * origin) { + const int64_t t_start = ggml_time_us(); + + int64_t * t_per_lib = calloc(GGML_METAL_LIB_COUNT, sizeof(int64_t)); + NSError ** err_per_lib = calloc(GGML_METAL_LIB_COUNT, sizeof(NSError *)); + __block atomic_bool any_failure = false; + + dispatch_group_t group = dispatch_group_create(); + dispatch_queue_t queue = dispatch_get_global_queue(QOS_CLASS_USER_INITIATED, 0); + + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + dispatch_group_async(group, queue, ^{ + + const int64_t t0 = ggml_time_us(); + + NSError * error = nil; + + NSString * src = source_for_kind(kind, &error); + if (!src) { + err_per_lib[kind] = [error retain]; + atomic_store(&any_failure, true); + return; } - } else { - GGML_LOG_INFO("%s: default.metallib not found, loading from source\n", __func__); - NSString * path_source; - NSString * path_resource = [[NSProcessInfo processInfo].environment objectForKey:@"GGML_METAL_PATH_RESOURCES"]; + id lib = nil; - GGML_LOG_INFO("%s: GGML_METAL_PATH_RESOURCES = %s\n", __func__, path_resource ? [path_resource UTF8String] : "nil"); + @autoreleasepool { + MTLCompileOptions * options = [MTLCompileOptions new]; + options.preprocessorMacros = prep; + ggml_metal_compile_options_set_lang(options, ggml_metal_device_get_props(res->dev)->has_tensor); - if (path_resource) { - path_source = [path_resource stringByAppendingPathComponent:@"ggml-metal.metal"]; - } else { - path_source = [bundle pathForResource:@"ggml-metal" ofType:@"metal"]; + lib = [device newLibraryWithSource:src options:options error:&error]; + + [options release]; + + // retain the error before the autorelease pool drains it + if (!lib) { + err_per_lib[kind] = [error retain]; + } } - if (path_source == nil) { - GGML_LOG_WARN("%s: error: could not use bundle path to find ggml-metal.metal, falling back to trying cwd\n", __func__); - path_source = @"ggml-metal.metal"; + [src release]; + + t_per_lib[kind] = ggml_time_us() - t0; + + if (!lib) { + atomic_store(&any_failure, true); + return; } - GGML_LOG_INFO("%s: loading '%s'\n", __func__, [path_source UTF8String]); + res->objs[kind] = lib; + }); + } + dispatch_group_wait(group, DISPATCH_TIME_FOREVER); + dispatch_release(group); + + const bool ok = !atomic_load(&any_failure); + + if (ok) { + const int64_t t_total = ggml_time_us() - t_start; + int64_t t_max = 0; + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + GGML_LOG_DEBUG("%s: compiled '%s' library in %.3f sec\n", + __func__, k_lib_names[kind], t_per_lib[kind] / 1e6); + if (t_per_lib[kind] > t_max) t_max = t_per_lib[kind]; + } + GGML_LOG_INFO("%s: loaded %d libraries from %s in %.3f sec (max single = %.3f sec)\n", + __func__, GGML_METAL_LIB_COUNT, origin, t_total / 1e6, t_max / 1e6); - src = [NSString stringWithContentsOfFile:path_source encoding:NSUTF8StringEncoding error:&error]; - if (error) { - GGML_LOG_ERROR("%s: error: %s\n", __func__, [[error description] UTF8String]); - return nil; + ggml_metal_library_build_index(res); + } else { + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + if (err_per_lib[kind]) { + GGML_LOG_ERROR("%s: failed to build '%s' library: %s\n", __func__, + k_lib_names[kind], [[err_per_lib[kind] description] UTF8String]); + [err_per_lib[kind] release]; } } -#endif + } - if (!library) { - @autoreleasepool { - // dictionary of preprocessor macros - NSMutableDictionary * prep = [NSMutableDictionary dictionary]; + free(err_per_lib); + free(t_per_lib); - if (ggml_metal_device_get_props(dev)->has_bfloat) { - [prep setObject:@"1" forKey:@"GGML_METAL_HAS_BF16"]; - } + return ok; +} - if (ggml_metal_device_get_props(dev)->has_tensor) { - [prep setObject:@"1" forKey:@"GGML_METAL_HAS_TENSOR"]; +// look for .metallib as a bundle resource, then next to the running binary +static NSString * ggml_metal_find_metallib(NSBundle * bundle, NSString * name) { + NSError * error = nil; + + NSString * path_lib = [bundle pathForResource:name ofType:@"metallib"]; + if (path_lib == nil) { + // Try to find the resource in the directory where the current binary located. + NSString * bin_cur = [[NSProcessInfo processInfo] arguments][0]; + NSString * bin_dir = [bin_cur stringByDeletingLastPathComponent]; + + NSString * path_lib_default = [NSString pathWithComponents:@[bin_dir, [name stringByAppendingPathExtension:@"metallib"]]]; + if ([[NSFileManager defaultManager] isReadableFileAtPath:path_lib_default]) { + GGML_LOG_INFO("%s: found '%s'\n", __func__, [path_lib_default UTF8String]); + + NSDictionary * atts = [[NSFileManager defaultManager] attributesOfItemAtPath:path_lib_default error:&error]; + if (atts && atts[NSFileType] == NSFileTypeSymbolicLink) { + // Optionally, if this is a symlink, try to resolve it. + path_lib_default = [[NSFileManager defaultManager] destinationOfSymbolicLinkAtPath:path_lib_default error:&error]; + if (path_lib_default && [path_lib_default length] > 0 && ![[path_lib_default substringToIndex:1] isEqualToString:@"/"]) { + // It is a relative path, adding the binary directory as directory prefix. + path_lib_default = [NSString pathWithComponents:@[bin_dir, path_lib_default]]; + } + if (!path_lib_default || ![[NSFileManager defaultManager] isReadableFileAtPath:path_lib_default]) { + // Link to the resource could not be resolved. + path_lib_default = nil; + } else { + GGML_LOG_INFO("%s: symlink resolved '%s'\n", __func__, [path_lib_default UTF8String]); } + } + } else { + // The resource couldn't be found in the binary's directory. + path_lib_default = nil; + } + + path_lib = path_lib_default; + } + + return path_lib; +} + +ggml_metal_library_t ggml_metal_library_init(ggml_metal_device_t dev) { + id device = ggml_metal_device_get_obj(dev); + ggml_metal_library_t res = calloc(1, sizeof(struct ggml_metal_library)); + res->dev = dev; + res->pipelines = ggml_metal_pipelines_init(); + res->lock = [NSLock new]; + + // shared MTLCompileOptions preprocessor macros (matches the build-time defines) + NSMutableDictionary * prep = [NSMutableDictionary dictionary]; + if (ggml_metal_device_get_props(dev)->has_bfloat) { + [prep setObject:@"1" forKey:@"GGML_METAL_HAS_BF16"]; + } + if (ggml_metal_device_get_props(dev)->has_tensor) { + [prep setObject:@"1" forKey:@"GGML_METAL_HAS_TENSOR"]; + } #if GGML_METAL_EMBED_LIBRARY - [prep setObject:@"1" forKey:@"GGML_METAL_EMBED_LIBRARY"]; + [prep setObject:@"1" forKey:@"GGML_METAL_EMBED_LIBRARY"]; #endif - MTLCompileOptions * options = [MTLCompileOptions new]; - options.preprocessorMacros = prep; +#if GGML_METAL_EMBED_LIBRARY + GGML_LOG_INFO("%s: using embedded metal library\n", __func__); - //[options setFastMathEnabled:false]; + // start/end symbols emitted by CMake (see CMakeLists.txt), one pair per kind +#define X(e, s) extern const char ggml_metallib_##s##_start[]; extern const char ggml_metallib_##s##_end[]; + GGML_METAL_LIBS +#undef X - library = [device newLibraryWithSource:src options:options error:&error]; - if (error) { - GGML_LOG_ERROR("%s: error: %s\n", __func__, [[error description] UTF8String]); - return nil; - } + static const char * const lib_start[GGML_METAL_LIB_COUNT] = { +#define X(e, s) [GGML_METAL_LIB_##e] = ggml_metallib_##s##_start, + GGML_METAL_LIBS +#undef X + }; + static const char * const lib_end[GGML_METAL_LIB_COUNT] = { +#define X(e, s) [GGML_METAL_LIB_##e] = ggml_metallib_##s##_end, + GGML_METAL_LIBS +#undef X + }; -#if !__has_feature(objc_arc) - [options release]; + const bool ok = ggml_metal_library_compile_all(res, device, prep, + ^NSString * (int kind, NSError ** err) { + (void) err; + return [[NSString alloc] initWithBytes:lib_start[kind] + length:(lib_end[kind] - lib_start[kind]) + encoding:NSUTF8StringEncoding]; + }, "embedded data"); + + if (!ok) { + ggml_metal_library_free(res); + return NULL; + } + + return res; +#else +#ifdef SWIFT_PACKAGE + NSBundle * bundle = SWIFTPM_MODULE_BUNDLE; +#else + NSBundle * bundle = [NSBundle bundleForClass:[GGMLMetalClass class]]; #endif - } + + const int64_t t_start = ggml_time_us(); + + NSError * error = nil; + NSString * path_lib = ggml_metal_find_metallib(bundle, @"default"); + + if (path_lib != nil) { + // pre-compiled library found: a single combined default.metallib + NSURL * libURL = [NSURL fileURLWithPath:path_lib]; + GGML_LOG_INFO("%s: loading '%s'\n", __func__, [path_lib UTF8String]); + + res->objs[0] = [device newLibraryWithURL:libURL error:&error]; + res->single_library = true; + if (!res->objs[0]) { + GGML_LOG_ERROR("%s: error: %s\n", __func__, [[error description] UTF8String]); + ggml_metal_library_free(res); + return NULL; } -#if GGML_METAL_EMBED_LIBRARY - [src release]; -#endif // GGML_METAL_EMBED_LIBRARY + // the tensor API kernels are built into a separate metallib + if (ggml_metal_device_get_props(dev)->has_tensor) { + NSString * path_mm = ggml_metal_find_metallib(bundle, @"ggml-tensor"); + + id lib_mm = nil; + if (path_mm != nil) { + lib_mm = [device newLibraryWithURL:[NSURL fileURLWithPath:path_mm] error:&error]; + if (!lib_mm && error) { + GGML_LOG_ERROR("%s: %s\n", __func__, [[error description] UTF8String]); + } + } + + if (lib_mm) { + GGML_LOG_INFO("%s: loaded '%s'\n", __func__, [path_mm UTF8String]); + + res->objs[GGML_METAL_LIB_MUL_MM] = [lib_mm retain]; + res->override_fns = [[NSSet setWithArray:[lib_mm functionNames]] retain]; + } else { + GGML_LOG_INFO("%s: ggml-tensor.metallib not found - disabling the tensor API\n", __func__); + + ggml_metal_device_disable_tensor(dev); + } + } GGML_LOG_INFO("%s: loaded in %.3f sec\n", __func__, (ggml_time_us() - t_start) / 1e6); + return res; } - ggml_metal_library_t res = calloc(1, sizeof(struct ggml_metal_library)); + // no pre-compiled metallib: fall back to compiling each kernel source separately + GGML_LOG_INFO("%s: default.metallib not found, loading kernel sources\n", __func__); - res->obj = library; - res->dev = dev; - res->pipelines = ggml_metal_pipelines_init(); - res->lock = [NSLock new]; + NSString * path_resource = [[NSProcessInfo processInfo].environment objectForKey:@"GGML_METAL_PATH_RESOURCES"]; + if (path_resource) { + GGML_LOG_INFO("%s: GGML_METAL_PATH_RESOURCES = %s\n", __func__, [path_resource UTF8String]); + } + + // resolve each kind's source path up front (file lookup/logging stays on the calling thread) + NSString ** path_per_kind = calloc(GGML_METAL_LIB_COUNT, sizeof(NSString *)); + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + NSString * rel = [NSString stringWithFormat:@"kernels/%s.metal", k_lib_names[kind]]; + + NSString * path_source = nil; + if (path_resource) { + path_source = [path_resource stringByAppendingPathComponent:rel]; + } else { + NSString * stem = [NSString stringWithFormat:@"kernels/%s", k_lib_names[kind]]; + path_source = [bundle pathForResource:stem ofType:@"metal"]; + } + + if (path_source == nil || ![[NSFileManager defaultManager] isReadableFileAtPath:path_source]) { + GGML_LOG_WARN("%s: could not locate %s in bundle, falling back to cwd\n", __func__, [rel UTF8String]); + path_source = rel; + } + + GGML_LOG_DEBUG("%s: loading '%s'\n", __func__, [path_source UTF8String]); + + path_per_kind[kind] = [path_source retain]; + } + + const bool ok = ggml_metal_library_compile_all(res, device, prep, + ^NSString * (int kind, NSError ** err) { + return ggml_metal_library_flatten_source(path_per_kind[kind], err); + }, "source"); + + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + [path_per_kind[kind] release]; + } + free(path_per_kind); + + if (!ok) { + ggml_metal_library_free(res); + return NULL; + } return res; +#endif } ggml_metal_library_t ggml_metal_library_init_from_source(ggml_metal_device_t dev, const char * source, bool verbose) { @@ -285,6 +610,7 @@ ggml_metal_library_t ggml_metal_library_init_from_source(ggml_metal_device_t dev MTLCompileOptions * options = [MTLCompileOptions new]; options.preprocessorMacros = prep; + ggml_metal_compile_options_set_lang(options, ggml_metal_device_get_props(dev)->has_tensor); library = [device newLibraryWithSource:src options:options error:&error]; if (error) { @@ -319,10 +645,11 @@ ggml_metal_library_t ggml_metal_library_init_from_source(ggml_metal_device_t dev return NULL; } - res->obj = library; - res->dev = dev; - res->pipelines = ggml_metal_pipelines_init(); - res->lock = [NSLock new]; + res->objs[0] = library; + res->single_library = true; + res->dev = dev; + res->pipelines = ggml_metal_pipelines_init(); + res->lock = [NSLock new]; return res; } @@ -332,8 +659,18 @@ void ggml_metal_library_free(ggml_metal_library_t lib) { return; } - if (lib->obj) { - [lib->obj release]; + for (int kind = 0; kind < GGML_METAL_LIB_COUNT; ++kind) { + if (lib->objs[kind]) { + [lib->objs[kind] release]; + } + } + + if (lib->fn_to_lib) { + [lib->fn_to_lib release]; + } + + if (lib->override_fns) { + [lib->override_fns release]; } ggml_metal_pipelines_free(lib->pipelines); @@ -394,11 +731,30 @@ struct ggml_metal_pipeline_with_params ggml_metal_library_compile_pipeline(ggml_ GGML_LOG_DEBUG("%s: compiling pipeline: base = '%s', name = '%s'\n", __func__, base, name); + // route to the library that actually defines this kernel; fn_to_lib is + // built from -[MTLLibrary functionNames] so it's always in sync + int lib_idx = 0; + if (lib->override_fns && [lib->override_fns containsObject:base_func]) { + lib_idx = GGML_METAL_LIB_MUL_MM; + } else if (!lib->single_library) { + NSNumber * idx = lib->fn_to_lib[base_func]; + if (!idx) { + [lib->lock unlock]; + + GGML_LOG_ERROR("%s: kernel not found in any metal library: base = '%s', name = '%s'\n", __func__, base, name); + + return res; + } + lib_idx = [idx intValue]; + } + + id mtl_lib = lib->objs[lib_idx]; + id mtl_function; if (!cv) { - mtl_function = [lib->obj newFunctionWithName:base_func]; + mtl_function = [mtl_lib newFunctionWithName:base_func]; } else { - mtl_function = [lib->obj newFunctionWithName:base_func constantValues:cv->obj error:&error]; + mtl_function = [mtl_lib newFunctionWithName:base_func constantValues:cv->obj error:&error]; } if (!mtl_function) { [lib->lock unlock]; @@ -483,7 +839,9 @@ void ggml_metal_encoder_free(ggml_metal_encoder_t encoder) { } void ggml_metal_encoder_debug_group_push(ggml_metal_encoder_t encoder, const char * name) { - [encoder->obj pushDebugGroup:[NSString stringWithCString:name encoding:NSUTF8StringEncoding]]; + @autoreleasepool { + [encoder->obj pushDebugGroup:[NSString stringWithCString:name encoding:NSUTF8StringEncoding]]; + } } void ggml_metal_encoder_debug_group_pop (ggml_metal_encoder_t encoder) { @@ -491,6 +849,10 @@ void ggml_metal_encoder_debug_group_pop (ggml_metal_encoder_t encoder) { } void ggml_metal_encoder_set_pipeline(ggml_metal_encoder_t encoder, struct ggml_metal_pipeline_with_params pipeline) { + if (!pipeline.pipeline) { + GGML_ABORT("%s: nil Metal pipeline (missing kernel; see compile_pipeline log above)\n", __func__); + } + [encoder->obj setComputePipelineState:pipeline.pipeline->obj]; } @@ -503,6 +865,9 @@ void ggml_metal_encoder_set_buffer(ggml_metal_encoder_t encoder, struct ggml_met } void ggml_metal_encoder_set_threadgroup_memory_size(ggml_metal_encoder_t encoder, size_t size, int idx) { + // ref: https://developer.apple.com/documentation/metal/mtlcomputecommandencoder/setthreadgroupmemorylength(_:index:) + GGML_ASSERT(size % 16 == 0); + [encoder->obj setThreadgroupMemoryLength:size atIndex:idx]; } @@ -532,6 +897,9 @@ void ggml_metal_encoder_end_encoding(ggml_metal_encoder_t encoder) { struct ggml_metal_device_props props; + // shared fusion debugging context + struct ggml_metal_fusion_info * finfo; + // virtual address for GPU memory allocations atomic_uintptr_t addr_virt; }; @@ -667,6 +1035,35 @@ void ggml_metal_rsets_free(ggml_metal_rsets_t rsets) { free(rsets); } +static const struct { + const char * name; + const char * token; + enum ggml_metal_device_id id; +} k_metal_devices[] = { +#define DEV(name, id) { name, #id, id } + DEV("M1", GGML_METAL_DEVICE_M1), + DEV("M1 Pro", GGML_METAL_DEVICE_M1_PRO), + DEV("M1 Max", GGML_METAL_DEVICE_M1_MAX), + DEV("M1 Ultra", GGML_METAL_DEVICE_M1_ULTRA), + DEV("M2", GGML_METAL_DEVICE_M2), + DEV("M2 Pro", GGML_METAL_DEVICE_M2_PRO), + DEV("M2 Max", GGML_METAL_DEVICE_M2_MAX), + DEV("M2 Ultra", GGML_METAL_DEVICE_M2_ULTRA), + DEV("M3", GGML_METAL_DEVICE_M3), + DEV("M3 Pro", GGML_METAL_DEVICE_M3_PRO), + DEV("M3 Max", GGML_METAL_DEVICE_M3_MAX), + DEV("M3 Ultra", GGML_METAL_DEVICE_M3_ULTRA), + DEV("M4", GGML_METAL_DEVICE_M4), + DEV("M4 Pro", GGML_METAL_DEVICE_M4_PRO), + DEV("M4 Max", GGML_METAL_DEVICE_M4_MAX), + DEV("M5", GGML_METAL_DEVICE_M5), + DEV("M5 Pro", GGML_METAL_DEVICE_M5_PRO), + DEV("M5 Max", GGML_METAL_DEVICE_M5_MAX), + DEV("M5 Ultra", GGML_METAL_DEVICE_M5_ULTRA), + DEV("A18 Pro", GGML_METAL_DEVICE_A18_PRO), +#undef DEV +}; + static enum ggml_metal_device_id ggml_metal_device_id_parse(const char * name) { if (!name) { return GGML_METAL_DEVICE_GENERIC; @@ -678,274 +1075,282 @@ static enum ggml_metal_device_id ggml_metal_device_id_parse(const char * name) { } const char * suffix = name + sizeof(prefix) - 1; - static const struct { - const char * name; - enum ggml_metal_device_id id; - } table[] = { - {"M1", GGML_METAL_DEVICE_M1}, - {"M1 Pro", GGML_METAL_DEVICE_M1_PRO}, - {"M1 Max", GGML_METAL_DEVICE_M1_MAX}, - {"M1 Ultra", GGML_METAL_DEVICE_M1_ULTRA}, - {"M2", GGML_METAL_DEVICE_M2}, - {"M2 Pro", GGML_METAL_DEVICE_M2_PRO}, - {"M2 Max", GGML_METAL_DEVICE_M2_MAX}, - {"M2 Ultra", GGML_METAL_DEVICE_M2_ULTRA}, - {"M3", GGML_METAL_DEVICE_M3}, - {"M3 Pro", GGML_METAL_DEVICE_M3_PRO}, - {"M3 Max", GGML_METAL_DEVICE_M3_MAX}, - {"M3 Ultra", GGML_METAL_DEVICE_M3_ULTRA}, - {"M4", GGML_METAL_DEVICE_M4}, - {"M4 Pro", GGML_METAL_DEVICE_M4_PRO}, - {"M4 Max", GGML_METAL_DEVICE_M4_MAX}, - {"M5", GGML_METAL_DEVICE_M5}, - {"M5 Pro", GGML_METAL_DEVICE_M5_PRO}, - {"M5 Max", GGML_METAL_DEVICE_M5_MAX}, - {"M5 Ultra", GGML_METAL_DEVICE_M5_ULTRA}, - }; - - for (size_t i = 0; i < sizeof(table)/sizeof(table[0]); ++i) { - if (strcmp(suffix, table[i].name) == 0) { - return table[i].id; + for (size_t i = 0; i < sizeof(k_metal_devices)/sizeof(k_metal_devices[0]); ++i) { + if (strcmp(suffix, k_metal_devices[i].name) == 0) { + return k_metal_devices[i].id; } } return GGML_METAL_DEVICE_GENERIC; } -ggml_metal_device_t ggml_metal_device_init(int device) { +const char * ggml_metal_device_id_token(enum ggml_metal_device_id id) { + for (size_t i = 0; i < sizeof(k_metal_devices)/sizeof(k_metal_devices[0]); ++i) { + if (k_metal_devices[i].id == id) { + return k_metal_devices[i].token; + } + } + return "GGML_METAL_DEVICE_GENERIC"; +} + +ggml_metal_device_t ggml_metal_device_init(int device, int n_devices) { ggml_metal_device_t dev = calloc(1, sizeof(struct ggml_metal_device)); assert(dev != NULL); - if (dev->mtl_device == nil) { - dev->mtl_device = MTLCreateSystemDefaultDevice(); + @autoreleasepool { + if (dev->mtl_device == nil) { + dev->mtl_device = MTLCreateSystemDefaultDevice(); - if (dev->mtl_device) { - dev->mtl_queue = [dev->mtl_device newCommandQueue]; - if (dev->mtl_queue == nil) { - GGML_LOG_ERROR("%s: error: failed to create command queue\n", __func__); - } + if (dev->mtl_device) { + dev->mtl_queue = [dev->mtl_device newCommandQueue]; + if (dev->mtl_queue == nil) { + GGML_LOG_ERROR("%s: error: failed to create command queue\n", __func__); + } - dev->addr_virt = 0x000000400ULL; + dev->addr_virt = 0x000000400ULL; - dev->props.device = device; - dev->props.has_simdgroup_reduction = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; - dev->props.has_simdgroup_reduction |= [dev->mtl_device supportsFamily:MTLGPUFamilyMetal3_GGML]; + dev->props.device = device; - dev->props.has_simdgroup_mm = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; - dev->props.has_unified_memory = dev->mtl_device.hasUnifiedMemory; + // the Metal backend uses the system default device as the single physical device; + // additional (virtual) devices are emulated on top of it via GGML_METAL_DEVICES + dev->props.device_phys = 0; + dev->props.device_virt = device; - dev->props.has_bfloat = [dev->mtl_device supportsFamily:MTLGPUFamilyMetal3_GGML]; - dev->props.has_bfloat |= [dev->mtl_device supportsFamily:MTLGPUFamilyApple6]; - if (getenv("GGML_METAL_BF16_DISABLE") != NULL) { - dev->props.has_bfloat = false; - } + dev->props.has_simdgroup_reduction = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; + dev->props.has_simdgroup_reduction |= [dev->mtl_device supportsFamily:MTLGPUFamilyMetal3_GGML]; - dev->props.has_tensor = [dev->mtl_device supportsFamily:MTLGPUFamilyMetal4_GGML]; - if (getenv("GGML_METAL_TENSOR_DISABLE") != NULL) { - dev->props.has_tensor = false; - } + dev->props.has_simdgroup_mm = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; + dev->props.has_unified_memory = dev->mtl_device.hasUnifiedMemory; - // note: disable the tensor API by default for old chips because with the current implementation it is not useful - // - M2 Ultra: ~5% slower - // - M4, M4 Max: no significant difference - // - // TODO: try to update the tensor API kernels to at least match the simdgroup performance - if (getenv("GGML_METAL_TENSOR_ENABLE") == NULL && - ![[dev->mtl_device name] containsString:@"M5"] && - ![[dev->mtl_device name] containsString:@"M6"] && - ![[dev->mtl_device name] containsString:@"A19"] && - ![[dev->mtl_device name] containsString:@"A20"]) { - GGML_LOG_INFO("%s: tensor API disabled for pre-M5 and pre-A19 devices\n", __func__); - dev->props.has_tensor = false; - } + dev->props.has_bfloat = [dev->mtl_device supportsFamily:MTLGPUFamilyMetal3_GGML]; + dev->props.has_bfloat |= [dev->mtl_device supportsFamily:MTLGPUFamilyApple6]; + if (getenv("GGML_METAL_BF16_DISABLE") != NULL) { + dev->props.has_bfloat = false; + } - // double-check that the tensor API compiles - if (dev->props.has_tensor) { - const char * src_tensor_f16 = "\n" - "#include \n" - "#include \n" - "#include \n" - " \n" - "using namespace metal; \n" - "using namespace mpp::tensor_ops; \n" - " \n" - "kernel void dummy_kernel( \n" - " tensor> A [[buffer(0)]], \n" - " tensor> B [[buffer(1)]], \n" - " device float * C [[buffer(2)]], \n" - " uint2 tgid [[threadgroup_position_in_grid]]) \n" - "{ \n" - " auto tA = A.slice(0, (int)tgid.y); \n" - " auto tB = B.slice((int)tgid.x, 0); \n" - " \n" - " matmul2d< \n" - " matmul2d_descriptor(16, 16, dynamic_extent), \n" - " execution_simdgroups<4>> mm; \n" - " \n" - " auto cT = mm.get_destination_cooperative_tensor(); \n" - " \n" - " auto sA = tA.slice(0, 0); \n" - " auto sB = tB.slice(0, 0); \n" - " mm.run(sB, sA, cT); \n" - " \n" - " auto tC = tensor, tensor_inline>(C, dextents(16, 16)); \n" - " \n" - " cT.store(tC); \n" - "}"; - - GGML_LOG_INFO("%s: testing tensor API for f16 support\n", __func__); - ggml_metal_library_t lib = ggml_metal_library_init_from_source(dev, src_tensor_f16, false); - if (lib == NULL) { - GGML_LOG_WARN("%s: - the tensor API is not supported in this environment - disabling\n", __func__); + dev->props.has_tensor = [dev->mtl_device supportsFamily:MTLGPUFamilyMetal4_GGML]; + if (getenv("GGML_METAL_TENSOR_DISABLE") != NULL) { dev->props.has_tensor = false; - } else { - struct ggml_metal_pipeline_with_params ppl = ggml_metal_library_compile_pipeline(lib, "dummy_kernel", "dummy_kernel", nil); - if (!ppl.pipeline) { + } + + // note: disable the tensor API by default for old chips because with the current implementation it is not useful + // - M2 Ultra: ~5% slower + // - M4, M4 Max: no significant difference + // + // TODO: try to update the tensor API kernels to at least match the simdgroup performance + if (getenv("GGML_METAL_TENSOR_ENABLE") == NULL && + ![[dev->mtl_device name] containsString:@"M5"] && + ![[dev->mtl_device name] containsString:@"M6"] && + ![[dev->mtl_device name] containsString:@"A19"] && + ![[dev->mtl_device name] containsString:@"A20"]) { + GGML_LOG_INFO("%s: tensor API disabled for pre-M5 and pre-A19 devices\n", __func__); + dev->props.has_tensor = false; + } + + // double-check that the tensor API compiles + if (dev->props.has_tensor) { + const char * src_tensor_f16 = "\n" + "#include \n" + "#include \n" + "#include \n" + " \n" + "using namespace metal; \n" + "using namespace mpp::tensor_ops; \n" + " \n" + "kernel void dummy_kernel( \n" + " tensor> A [[buffer(0)]], \n" + " tensor> B [[buffer(1)]], \n" + " device float * C [[buffer(2)]], \n" + " uint2 tgid [[threadgroup_position_in_grid]]) \n" + "{ \n" + " auto tA = A.slice(0, (int)tgid.y); \n" + " auto tB = B.slice((int)tgid.x, 0); \n" + " \n" + " matmul2d< \n" + " matmul2d_descriptor(16, 16, dynamic_extent), \n" + " execution_simdgroups<4>> mm; \n" + " \n" + " auto cT = mm.get_destination_cooperative_tensor(); \n" + " \n" + " auto sA = tA.slice(0, 0); \n" + " auto sB = tB.slice(0, 0); \n" + " mm.run(sB, sA, cT); \n" + " \n" + " auto tC = tensor, tensor_inline>(C, dextents(16, 16)); \n" + " \n" + " cT.store(tC); \n" + "}"; + + GGML_LOG_INFO("%s: testing tensor API for f16 support\n", __func__); + ggml_metal_library_t lib = ggml_metal_library_init_from_source(dev, src_tensor_f16, false); + if (lib == NULL) { GGML_LOG_WARN("%s: - the tensor API is not supported in this environment - disabling\n", __func__); dev->props.has_tensor = false; - } + } else { + struct ggml_metal_pipeline_with_params ppl = ggml_metal_library_compile_pipeline(lib, "dummy_kernel", "dummy_kernel", nil); + if (!ppl.pipeline) { + GGML_LOG_WARN("%s: - the tensor API is not supported in this environment - disabling\n", __func__); + dev->props.has_tensor = false; + } - ggml_metal_library_free(lib); + ggml_metal_library_free(lib); + } } - } - // try to compile a dummy kernel to determine if the tensor API is supported for bfloat - if (dev->props.has_tensor && dev->props.has_bfloat) { - const char * src_tensor_bf16 = "\n" - "#include \n" - "#include \n" - "#include \n" - " \n" - "using namespace metal; \n" - "using namespace mpp::tensor_ops; \n" - " \n" - "kernel void dummy_kernel( \n" - " tensor> A [[buffer(0)]], \n" - " tensor> B [[buffer(1)]], \n" - " device float * C [[buffer(2)]], \n" - " uint2 tgid [[threadgroup_position_in_grid]]) \n" - "{ \n" - " auto tA = A.slice(0, (int)tgid.y); \n" - " auto tB = B.slice((int)tgid.x, 0); \n" - " \n" - " matmul2d< \n" - " matmul2d_descriptor(16, 16, dynamic_extent), \n" - " execution_simdgroups<4>> mm; \n" - " \n" - " auto cT = mm.get_destination_cooperative_tensor(); \n" - " \n" - " auto sA = tA.slice(0, 0); \n" - " auto sB = tB.slice(0, 0); \n" - " mm.run(sB, sA, cT); \n" - " \n" - " auto tC = tensor, tensor_inline>(C, dextents(16, 16)); \n" - " \n" - " cT.store(tC); \n" - "}"; - - GGML_LOG_INFO("%s: testing tensor API for bfloat support\n", __func__); - ggml_metal_library_t lib = ggml_metal_library_init_from_source(dev, src_tensor_bf16, false); - if (lib == NULL) { - GGML_LOG_WARN("%s: - the tensor API does not support bfloat - disabling bfloat support\n", __func__); - dev->props.has_bfloat = false; - } else { - struct ggml_metal_pipeline_with_params ppl = ggml_metal_library_compile_pipeline(lib, "dummy_kernel", "dummy_kernel", nil); - if (!ppl.pipeline) { + // try to compile a dummy kernel to determine if the tensor API is supported for bfloat + if (dev->props.has_tensor && dev->props.has_bfloat) { + const char * src_tensor_bf16 = "\n" + "#include \n" + "#include \n" + "#include \n" + " \n" + "using namespace metal; \n" + "using namespace mpp::tensor_ops; \n" + " \n" + "kernel void dummy_kernel( \n" + " tensor> A [[buffer(0)]], \n" + " tensor> B [[buffer(1)]], \n" + " device float * C [[buffer(2)]], \n" + " uint2 tgid [[threadgroup_position_in_grid]]) \n" + "{ \n" + " auto tA = A.slice(0, (int)tgid.y); \n" + " auto tB = B.slice((int)tgid.x, 0); \n" + " \n" + " matmul2d< \n" + " matmul2d_descriptor(16, 16, dynamic_extent), \n" + " execution_simdgroups<4>> mm; \n" + " \n" + " auto cT = mm.get_destination_cooperative_tensor(); \n" + " \n" + " auto sA = tA.slice(0, 0); \n" + " auto sB = tB.slice(0, 0); \n" + " mm.run(sB, sA, cT); \n" + " \n" + " auto tC = tensor, tensor_inline>(C, dextents(16, 16)); \n" + " \n" + " cT.store(tC); \n" + "}"; + + GGML_LOG_INFO("%s: testing tensor API for bfloat support\n", __func__); + ggml_metal_library_t lib = ggml_metal_library_init_from_source(dev, src_tensor_bf16, false); + if (lib == NULL) { GGML_LOG_WARN("%s: - the tensor API does not support bfloat - disabling bfloat support\n", __func__); dev->props.has_bfloat = false; - } + } else { + struct ggml_metal_pipeline_with_params ppl = ggml_metal_library_compile_pipeline(lib, "dummy_kernel", "dummy_kernel", nil); + if (!ppl.pipeline) { + GGML_LOG_WARN("%s: - the tensor API does not support bfloat - disabling bfloat support\n", __func__); + dev->props.has_bfloat = false; + } - ggml_metal_library_free(lib); + ggml_metal_library_free(lib); + } } - } - dev->props.use_residency_sets = true; + dev->props.use_residency_sets = true; #if defined(GGML_METAL_HAS_RESIDENCY_SETS) - dev->props.use_residency_sets = getenv("GGML_METAL_NO_RESIDENCY") == nil; + dev->props.use_residency_sets = getenv("GGML_METAL_NO_RESIDENCY") == nil; #endif - dev->props.use_shared_buffers = dev->props.has_unified_memory; -#if TARGET_OS_OSX - // In case of eGPU, shared memory may be preferable. - dev->props.use_shared_buffers |= [dev->mtl_device location] == MTLDeviceLocationExternal; + dev->props.use_shared_buffers = dev->props.has_unified_memory; +#if TARGET_OS_OSX && TARGET_CPU_X86_64 + // In case of eGPU, shared memory may be preferable. + dev->props.use_shared_buffers |= [dev->mtl_device location] == MTLDeviceLocationExternal; #endif - if (getenv("GGML_METAL_SHARED_BUFFERS_DISABLE") != NULL) { - dev->props.use_shared_buffers = false; - } - if (getenv("GGML_METAL_SHARED_BUFFERS_ENABLE") != NULL) { - dev->props.use_shared_buffers = true; - } + if (getenv("GGML_METAL_SHARED_BUFFERS_DISABLE") != NULL) { + dev->props.use_shared_buffers = false; + } + if (getenv("GGML_METAL_SHARED_BUFFERS_ENABLE") != NULL) { + dev->props.use_shared_buffers = true; + } - dev->props.supports_gpu_family_apple7 = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; + dev->props.supports_gpu_family_apple7 = [dev->mtl_device supportsFamily:MTLGPUFamilyApple7]; - dev->props.device_id = ggml_metal_device_id_parse([[dev->mtl_device name] UTF8String]); + dev->props.device_id = ggml_metal_device_id_parse([[dev->mtl_device name] UTF8String]); - dev->props.op_offload_min_batch_size = getenv("GGML_OP_OFFLOAD_MIN_BATCH") ? atoi(getenv("GGML_OP_OFFLOAD_MIN_BATCH")) : 32; + dev->props.op_offload_min_batch_size = getenv("GGML_OP_OFFLOAD_MIN_BATCH") ? atoi(getenv("GGML_OP_OFFLOAD_MIN_BATCH")) : 32; - dev->props.max_buffer_size = dev->mtl_device.maxBufferLength; - dev->props.max_theadgroup_memory_size = dev->mtl_device.maxThreadgroupMemoryLength; - if (@available(macOS 10.12, iOS 16.0, *)) { - dev->props.max_working_set_size = dev->mtl_device.recommendedMaxWorkingSetSize; - } else { - dev->props.max_working_set_size = dev->mtl_device.maxBufferLength; - } + dev->props.max_buffer_size = dev->mtl_device.maxBufferLength; + dev->props.max_theadgroup_memory_size = dev->mtl_device.maxThreadgroupMemoryLength; + if (@available(macOS 10.12, iOS 16.0, *)) { + dev->props.max_working_set_size = dev->mtl_device.recommendedMaxWorkingSetSize; + } else { + dev->props.max_working_set_size = dev->mtl_device.maxBufferLength; + } - snprintf(dev->props.name, sizeof(dev->props.name), "%s%d", "MTL", device); - snprintf(dev->props.desc, sizeof(dev->props.desc), "%s", [[dev->mtl_device name] UTF8String]); + { + const char * val = getenv("GGML_METAL_FUSION_DEBUG"); + dev->finfo = ggml_metal_fusion_info_init( + getenv("GGML_METAL_FUSION_DISABLE") == nil, + val ? atoi(val) : 0); + } - dev->library = ggml_metal_library_init(dev); - if (!dev->library) { - GGML_LOG_ERROR("%s: error: failed to create library\n", __func__); - } + snprintf(dev->props.name, sizeof(dev->props.name), "%s%d", "MTL", device); + const char * gpu_name = [[dev->mtl_device name] UTF8String]; + if (n_devices > 1) { + snprintf(dev->props.desc, sizeof(dev->props.desc), "%s (dev p%d/v%d)", + gpu_name, dev->props.device_phys, dev->props.device_virt); + } else { + snprintf(dev->props.desc, sizeof(dev->props.desc), "%s", gpu_name); + } - if (dev->props.use_residency_sets) { - dev->rsets = ggml_metal_rsets_init(dev); - } else { - dev->rsets = nil; - } + dev->library = ggml_metal_library_init(dev); + if (!dev->library) { + GGML_LOG_ERROR("%s: error: failed to create library\n", __func__); + } - // print MTL GPU family: - GGML_LOG_INFO("%s: GPU name: %s (%s)\n", __func__, dev->props.name, dev->props.desc); + if (dev->props.use_residency_sets) { + dev->rsets = ggml_metal_rsets_init(dev); + } else { + dev->rsets = nil; + } - // determine max supported GPU family - // https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf - // https://developer.apple.com/metal/Metal-Feature-Set-Tables.pdf - { - for (int i = MTLGPUFamilyApple1 + 20; i >= MTLGPUFamilyApple1; --i) { - if ([dev->mtl_device supportsFamily:i]) { - GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyApple%d (%d)\n", __func__, i - (int) MTLGPUFamilyApple1 + 1, i); - break; + // print MTL GPU family: + GGML_LOG_INFO("%s: GPU name: %s (%s)\n", __func__, dev->props.name, dev->props.desc); + + // determine max supported GPU family + // https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf + // https://developer.apple.com/metal/Metal-Feature-Set-Tables.pdf + { + for (int i = MTLGPUFamilyApple1 + 20; i >= MTLGPUFamilyApple1; --i) { + if ([dev->mtl_device supportsFamily:i]) { + dev->props.gpu_family = i - (int) MTLGPUFamilyApple1 + 1; + GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyApple%d (%d)\n", __func__, dev->props.gpu_family, i); + break; + } } - } - for (int i = MTLGPUFamilyCommon1 + 5; i >= MTLGPUFamilyCommon1; --i) { - if ([dev->mtl_device supportsFamily:i]) { - GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyCommon%d (%d)\n", __func__, i - (int) MTLGPUFamilyCommon1 + 1, i); - break; +#if TARGET_CPU_X86_64 + for (int i = MTLGPUFamilyCommon1 + 5; i >= MTLGPUFamilyCommon1; --i) { + if ([dev->mtl_device supportsFamily:i]) { + GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyCommon%d (%d)\n", __func__, i - (int) MTLGPUFamilyCommon1 + 1, i); + break; + } } - } +#endif - for (int i = MTLGPUFamilyMetal3_GGML + 5; i >= MTLGPUFamilyMetal3_GGML; --i) { - if ([dev->mtl_device supportsFamily:i]) { - GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyMetal%d (%d)\n", __func__, i - (int) MTLGPUFamilyMetal3_GGML + 3, i); - break; + for (int i = MTLGPUFamilyMetal3_GGML + 5; i >= MTLGPUFamilyMetal3_GGML; --i) { + if ([dev->mtl_device supportsFamily:i]) { + GGML_LOG_INFO("%s: GPU family: MTLGPUFamilyMetal%d (%d)\n", __func__, i - (int) MTLGPUFamilyMetal3_GGML + 3, i); + break; + } } } - } - GGML_LOG_INFO("%s: simdgroup reduction = %s\n", __func__, dev->props.has_simdgroup_reduction ? "true" : "false"); - GGML_LOG_INFO("%s: simdgroup matrix mul. = %s\n", __func__, dev->props.has_simdgroup_mm ? "true" : "false"); - GGML_LOG_INFO("%s: has unified memory = %s\n", __func__, dev->props.has_unified_memory ? "true" : "false"); - GGML_LOG_INFO("%s: has bfloat = %s\n", __func__, dev->props.has_bfloat ? "true" : "false"); - GGML_LOG_INFO("%s: has tensor = %s\n", __func__, dev->props.has_tensor ? "true" : "false"); - GGML_LOG_INFO("%s: use residency sets = %s\n", __func__, dev->props.use_residency_sets ? "true" : "false"); - GGML_LOG_INFO("%s: use shared buffers = %s\n", __func__, dev->props.use_shared_buffers ? "true" : "false"); + GGML_LOG_INFO("%s: simdgroup reduction = %s\n", __func__, dev->props.has_simdgroup_reduction ? "true" : "false"); + GGML_LOG_INFO("%s: simdgroup matrix mul. = %s\n", __func__, dev->props.has_simdgroup_mm ? "true" : "false"); + GGML_LOG_INFO("%s: has unified memory = %s\n", __func__, dev->props.has_unified_memory ? "true" : "false"); + GGML_LOG_INFO("%s: has bfloat = %s\n", __func__, dev->props.has_bfloat ? "true" : "false"); + GGML_LOG_INFO("%s: has tensor = %s\n", __func__, dev->props.has_tensor ? "true" : "false"); + GGML_LOG_INFO("%s: use residency sets = %s\n", __func__, dev->props.use_residency_sets ? "true" : "false"); + GGML_LOG_INFO("%s: use shared buffers = %s\n", __func__, dev->props.use_shared_buffers ? "true" : "false"); #if TARGET_OS_OSX || (TARGET_OS_IOS && __clang_major__ >= 15) - if (@available(macOS 10.12, iOS 16.0, *)) { - GGML_LOG_INFO("%s: recommendedMaxWorkingSetSize = %8.2f MB\n", __func__, dev->props.max_working_set_size / 1e6); - } + if (@available(macOS 10.12, iOS 16.0, *)) { + GGML_LOG_INFO("%s: recommendedMaxWorkingSetSize = %8.2f MB\n", __func__, dev->props.max_working_set_size / 1e6); + } #endif + } } } @@ -955,19 +1360,23 @@ ggml_metal_device_t ggml_metal_device_init(int device) { void ggml_metal_device_free(ggml_metal_device_t dev) { assert(dev != NULL); - ggml_metal_rsets_free(dev->rsets); + @autoreleasepool { + ggml_metal_fusion_info_free(dev->finfo); - ggml_metal_library_free(dev->library); - dev->library = NULL; + ggml_metal_rsets_free(dev->rsets); - if (dev->mtl_queue) { - [dev->mtl_queue release]; - dev->mtl_queue = nil; - } + ggml_metal_library_free(dev->library); + dev->library = NULL; - if (dev->mtl_device) { - [dev->mtl_device release]; - dev->mtl_device = nil; + if (dev->mtl_queue) { + [dev->mtl_queue release]; + dev->mtl_queue = nil; + } + + if (dev->mtl_device) { + [dev->mtl_device release]; + dev->mtl_device = nil; + } } free(dev); @@ -1055,12 +1464,14 @@ ggml_metal_event_t ggml_metal_device_event_init(ggml_metal_device_t dev) { } void ggml_metal_device_event_free(ggml_metal_device_t dev, ggml_metal_event_t ev) { - id event = ev->obj; - [event release]; + @autoreleasepool { + id event = ev->obj; + [event release]; - free(ev); + free(ev); - GGML_UNUSED(dev); + GGML_UNUSED(dev); + } } void ggml_metal_device_event_synchronize(ggml_metal_device_t dev, ggml_metal_event_t ev) { @@ -1075,14 +1486,42 @@ void ggml_metal_device_event_synchronize(ggml_metal_device_t dev, ggml_metal_eve void ggml_metal_device_get_memory(ggml_metal_device_t dev, size_t * free, size_t * total) { if (@available(macOS 10.12, iOS 16.0, *)) { - *total = dev->mtl_device.recommendedMaxWorkingSetSize; - *free = *total - dev->mtl_device.currentAllocatedSize; + *total = dev->mtl_device.recommendedMaxWorkingSetSize; + size_t cur = dev->mtl_device.currentAllocatedSize; + // it's possible to allocate more than `recommendedMaxWorkingSetSize` + *free = *total > cur ? *total - cur : 0; } else { *free = 0; *total = 0; } } +static bool ggml_metal_supports_mul_mat_op( + bool has_simdgroup_reduction, + const struct ggml_tensor * op, + bool src0_f16_has_mv, + bool mm_path) { + if (!has_simdgroup_reduction || + op->src[0]->type == GGML_TYPE_NVFP4 || + op->src[0]->type == GGML_TYPE_TQ1_0) { + return false; + } + + if (op->src[1]->type != GGML_TYPE_F16) { + return true; + } + + if (op->src[0]->type == GGML_TYPE_BF16) { + return false; + } + + if (src0_f16_has_mv && op->src[0]->type == GGML_TYPE_F16) { + return true; + } + + return mm_path; +} + bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_tensor * op) { const bool has_simdgroup_mm = dev->props.has_simdgroup_mm; const bool has_simdgroup_reduction = dev->props.has_simdgroup_reduction; @@ -1154,6 +1593,7 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return ggml_is_contiguous_1(op->src[0]) && (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16); default: return false; @@ -1181,6 +1621,12 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te return true; case GGML_TYPE_BF16: return has_bfloat; + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + return true; default: return false; } @@ -1289,6 +1735,12 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te op->src[0]->ne[0] != 576) { return false; } + if (op->src[1]->ne[0] == 72 && op->src[1]->ne[0] != op->src[2]->ne[0]) { + return false; + } + if (op->src[1]->ne[0] < op->src[2]->ne[0]) { + return false; + } if (op->src[1]->type != op->src[2]->type) { return false; } @@ -1357,8 +1809,6 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32 && - op->src[0]->ne[1] == 4 && - op->src[1]->ne[0] == 4 && ggml_is_contiguous_rows(op->src[0]) && ggml_is_contiguous_rows(op->src[1]); case GGML_OP_DSV4_HC_POST: @@ -1366,16 +1816,15 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && op->src[2]->type == GGML_TYPE_F32 && - op->src[3]->type == GGML_TYPE_F32 && + (op->src[3] == NULL || op->src[3]->type == GGML_TYPE_F32) && op->type == GGML_TYPE_F32 && op->src[1]->ne[1] == 4 && op->src[2]->ne[0] == 4 && - op->src[3]->ne[0] == 4 && - op->src[3]->ne[1] == 4 && + (op->src[3] == NULL || (op->src[3]->ne[0] == 4 && op->src[3]->ne[1] == 4)) && ggml_is_contiguous_rows(op->src[0]) && ggml_is_contiguous_rows(op->src[1]) && ggml_is_contiguous_rows(op->src[2]) && - ggml_is_contiguous_rows(op->src[3]); + (op->src[3] == NULL || ggml_is_contiguous_rows(op->src[3])); case GGML_OP_SSM_SCAN: return has_simdgroup_reduction; case GGML_OP_SSM_CONV: @@ -1386,9 +1835,21 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te case GGML_OP_GATED_DELTA_NET: return has_simdgroup_reduction && op->src[2]->ne[0] % 32 == 0; case GGML_OP_SOLVE_TRI: + return has_simdgroup_reduction && op->src[0]->type == GGML_TYPE_F32; case GGML_OP_MUL_MAT: + // the FWHT kernels read an F16 source directly; every other F16 src1 path + // still goes through ggml_metal_supports_mul_mat_op + if (op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F16 && + ggml_metal_op_mul_mat_use_fwht(op)) { + return has_simdgroup_reduction; + } + return ggml_metal_supports_mul_mat_op( + has_simdgroup_reduction, op, true, + ggml_metal_op_mul_mat_use_mm(op, has_simdgroup_mm)); case GGML_OP_MUL_MAT_ID: - return has_simdgroup_reduction && op->src[0]->type != GGML_TYPE_NVFP4; + return ggml_metal_supports_mul_mat_op( + has_simdgroup_reduction, op, false, + ggml_metal_op_mul_mat_id_use_mm(op, has_simdgroup_mm)); case GGML_OP_SET: case GGML_OP_CPY: case GGML_OP_DUP: @@ -1452,7 +1913,8 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te }; } case GGML_OP_GET_ROWS: - return op->src[0]->type != GGML_TYPE_NVFP4; + return op->src[0]->type != GGML_TYPE_NVFP4 && + op->src[0]->type != GGML_TYPE_TQ1_0; case GGML_OP_SET_ROWS: { if (op->src[0]->type == GGML_TYPE_F16) { @@ -1493,6 +1955,14 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te return &dev->props; } +static void ggml_metal_device_disable_tensor(ggml_metal_device_t dev) { + dev->props.has_tensor = false; +} + +struct ggml_metal_fusion_info * ggml_metal_device_get_fusion_info(ggml_metal_device_t dev) { + return dev->finfo; +} + // // device buffers // @@ -1794,13 +2264,15 @@ ggml_metal_buffer_t ggml_metal_buffer_map(ggml_metal_device_t dev, void * ptr, s } void ggml_metal_buffer_free(ggml_metal_buffer_t buf) { - ggml_metal_device_rsets_rm(buf->dev, buf->rset); + @autoreleasepool { + ggml_metal_device_rsets_rm(buf->dev, buf->rset); - for (int i = 0; i < buf->n_buffers; i++) { - [buf->buffers[i].metal release]; - } + for (int i = 0; i < buf->n_buffers; i++) { + [buf->buffers[i].metal release]; + } - ggml_metal_buffer_rset_free(buf); + ggml_metal_buffer_rset_free(buf); + } if (buf->is_shared && buf->owned) { #if TARGET_OS_OSX diff --git a/ggml/src/ggml-metal/ggml-metal-fusion.cpp b/ggml/src/ggml-metal/ggml-metal-fusion.cpp new file mode 100644 index 00000000..eac3bd6f --- /dev/null +++ b/ggml/src/ggml-metal/ggml-metal-fusion.cpp @@ -0,0 +1,1119 @@ +#include "ggml-metal-fusion.h" + +#include "ggml-backend-impl.h" +#include "ggml-metal-device.h" + +#include +#include +#include +#include +#include +#include + +// derive the non-empty op sequence from the raw `ops_all` sequence +static std::vector ggml_metal_fusion_filter_ops(const std::vector & ops_all) { + std::vector ops; + + for (ggml_op op : ops_all) { + if (!ggml_op_is_empty(op)) { + ops.push_back(op); + } + } + + return ops; +} + +struct ggml_metal_fusion { + ggml_metal_fusion_id id; + + std::vector ops; // non-empty op sequence, derived from ops_all + std::vector ops_all; // full raw op sequence (may include empty RESHAPE/VIEW nodes) + std::vector outs; // additional fused output nodes, relative to ops + + // if unsafe: the generic chain/shape + ggml_can_fuse_subgraph checks are skipped and the + // check callback below is the sole validator (used for patterns that are not elision chains, + // e.g. the gdn + cache-cpy write-through fusion) + bool unsafe; + + // extra backend constraints on top of ggml_can_fuse_subgraph + // nodes[j] is the j-th node of the pattern; node_idxs[idx + j] is its raw graph index + bool (*check)(const struct ggml_metal_fusion * fusion, + const struct ggml_tensor * const * nodes, + const struct ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode); + + ggml_metal_fusion( + ggml_metal_fusion_id id, + const std::vector & ops_all, + const std::vector & outs, + bool unsafe, + bool (*check)(const struct ggml_metal_fusion * fusion, + const struct ggml_tensor * const * nodes, + const struct ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode)) + : id(id), + ops(ggml_metal_fusion_filter_ops(ops_all)), + ops_all(ops_all), + outs(outs), + unsafe(unsafe), + check(check) { + } +}; + +ggml_metal_fusion_id ggml_metal_fusion_get_id(const ggml_metal_fusion * fusion) { + return fusion->id; +} + +// ---- helpers ------------------------------------------------------------- + +// true if two tensors live in the same Metal buffer +static bool ggml_metal_fusion_same_buffer(const ggml_tensor * a, const ggml_tensor * b) { + if (!a || !b) { + return false; + } + + ggml_backend_buffer_t ba = a->view_src ? a->view_src->buffer : a->buffer; + ggml_backend_buffer_t bb = b->view_src ? b->view_src->buffer : b->buffer; + + ggml_metal_buffer_t ca = (ggml_metal_buffer_t) ba->context; + ggml_metal_buffer_t cb = (ggml_metal_buffer_t) bb->context; + + return ggml_metal_buffer_get_id(ca, a).metal == ggml_metal_buffer_get_id(cb, b).metal; +} + +// ---- pattern checks ------------------------------------------------------ + +// NORM/RMS_NORM + MUL + ADD: the weight/bias of each fused step must match the norm input +// width, be contiguous rows, and the fused outputs must stay F32 +static bool ggml_metal_fusion_check_norm( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(mode); + GGML_UNUSED(gf); + GGML_UNUSED(node_idxs); + GGML_UNUSED(idx); + + GGML_ASSERT(fusion->ops.size() >= 2); + + if (fusion->id == GGML_METAL_FUSION_NORM_SCALE) { + GGML_ASSERT(fusion->ops.size() == 2); + + const ggml_tensor * scale = nodes[1]; + if (scale->op != GGML_OP_SCALE || scale->src[0] != nodes[0] || scale->src[1] || + scale->type != GGML_TYPE_F32) { + return false; + } + + return true; + } + + for (int j = 1; j < (int) fusion->ops.size(); j++) { + // the fused MUL/ADD must read the previous node as src0 + if (nodes[j]->src[0] != nodes[j - 1]) { + return false; + } + + // the weight/bias must have the same row width as the norm input + if (nodes[j]->src[1]->ne[0] != nodes[0]->ne[0]) { + return false; + } + + if (!ggml_is_contiguous_rows(nodes[j]->src[1])) { + return false; + } + + if (nodes[j]->type != GGML_TYPE_F32) { + return false; + } + } + + return true; +} + +// SSM_CONV + UNARY (silu) +static bool ggml_metal_fusion_check_ssm_conv_silu( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(fusion); + GGML_UNUSED(gf); + GGML_UNUSED(node_idxs); + GGML_UNUSED(idx); + GGML_UNUSED(mode); + + const ggml_tensor * conv = nodes[0]; + const ggml_tensor * un = nodes[1]; + + if (conv->op != GGML_OP_SSM_CONV || un->op != GGML_OP_UNARY || un->src[0] != conv || un->src[1]) { + return false; + } + + if (ggml_get_unary_op(un) != GGML_UNARY_OP_SILU) { + return false; + } + + if (conv->type != GGML_TYPE_F32 || un->type != GGML_TYPE_F32 || !ggml_is_contiguous_rows(un)) { + return false; + } + + return true; +} + +// ADD x N: each ADD reads the previous ADD as src0, and all addends must share layout +// (and, in FULL mode, live in the same Metal buffer) +static bool ggml_metal_fusion_check_add_chain( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(gf); + GGML_UNUSED(node_idxs); + GGML_UNUSED(idx); + GGML_ASSERT(fusion->ops.size() >= 2); + + for (int j = 1; j < (int) fusion->ops.size(); j++) { + if (nodes[j]->src[0] != nodes[j - 1]) { + return false; + } + + if (!ggml_are_same_layout(nodes[j]->src[1], nodes[j - 1]->src[1])) { + return false; + } + + if (mode == GGML_METAL_FUSION_FULL) { + if (!ggml_metal_fusion_same_buffer(nodes[j]->src[1], nodes[0]->src[1])) { + return false; + } + } + } + + return true; +} + +// GATED_DELTA_NET + CPY: the trailing cpy scatters the gdn state snapshots into the recurrent +// cache, so the gdn kernel writes them straight to the cache and the cpy is elided. +// mirrors ggml_metal_op_can_fuse_gdn_cache (PR #25788). the gdn output has other consumers (the +// attn scores view), so unlike the other patterns this is not an elision chain: the structural +// checks live entirely in this callback (unsafe = true). +static bool ggml_metal_fusion_check_gdn_cache( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(fusion); + GGML_UNUSED(gf); + GGML_UNUSED(node_idxs); + GGML_UNUSED(idx); + + const ggml_tensor * gdn = nodes[0]; + const ggml_tensor * cpy = nodes[1]; + + // the kernel skips the snapshot tail, so the gdn output must not be a graph output + if (gdn->type != GGML_TYPE_F32 || (gdn->flags & GGML_TENSOR_FLAG_OUTPUT)) { + return false; + } + + if (cpy->op != GGML_OP_CPY || (cpy->flags & GGML_TENSOR_FLAG_OUTPUT)) { + return false; + } + + const int64_t S_v = gdn->src[2]->ne[0]; + const int64_t H = gdn->src[2]->ne[1]; + const int64_t n_tokens = gdn->src[2]->ne[2]; + const int64_t n_seqs = gdn->src[2]->ne[3]; + const int64_t K = ggml_get_op_params_i32(gdn, 0); + const size_t tail_off = ggml_row_size(GGML_TYPE_F32, S_v * H * n_tokens * n_seqs); + + const int64_t D = S_v * S_v * H; + const int64_t n_written = std::min(n_tokens, K); + + const ggml_tensor * src = cpy->src[0]; // gdn snapshot tail view + const ggml_tensor * dst = cpy->src[1]; // cache view + + // src must be this gdn's snapshot tail (contiguous, at the tail offset) + if (src->op != GGML_OP_VIEW || src->view_src != gdn || + src->view_offs != tail_off || !ggml_is_contiguous(src)) { + return false; + } + + const int64_t expected_ne[GGML_MAX_DIMS] = { D, n_seqs, n_written, 1 }; + if (dst->type != GGML_TYPE_F32 || + !std::equal(expected_ne, expected_ne + GGML_MAX_DIMS, dst->ne) || + dst->nb[0] != ggml_type_size(GGML_TYPE_F32) || + dst->nb[1] != ggml_row_size(GGML_TYPE_F32, D)) { + return false; + } + + if (mode == GGML_METAL_FUSION_FULL) { + // the cache must be allocated so the kernel can write straight to its buffer + if (dst->data == nullptr) { + return false; + } + } + + return true; +} + +// MUL + SIN + SQR + MUL + ADD (snake activation) +static bool ggml_metal_fusion_check_snake( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(fusion); + GGML_UNUSED(mode); + GGML_UNUSED(gf); + GGML_UNUSED(node_idxs); + GGML_UNUSED(idx); + + const ggml_tensor * mul0 = nodes[0]; + const ggml_tensor * sin_node = nodes[1]; + const ggml_tensor * sqr = nodes[2]; + const ggml_tensor * mul1 = nodes[3]; + const ggml_tensor * add = nodes[4]; + + // x carries the full activation shape, a is the broadcast operand + const ggml_tensor * x = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; + const ggml_tensor * a = (x == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; + + // mul1 reads sqr and inv_b in either operand order + const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; + + // closure check: the trailing add reads the same x as the leading mul + const ggml_tensor * x_in_add = (add->src[0] == mul1) ? add->src[1] : add->src[0]; + + // x is in the supported whitelist and every chain intermediate shares x's type. + // a and inv_b bind as device const float * in the kernel, so they stay F32. + const bool types_ok = + (x->type == GGML_TYPE_F32 || x->type == GGML_TYPE_F16 || x->type == GGML_TYPE_BF16) && + (a->type == GGML_TYPE_F32) && (inv_b->type == GGML_TYPE_F32) && + (mul0->type == x->type) && (sin_node->type == x->type) && + (sqr->type == x->type) && (mul1->type == x->type) && + (add->type == x->type); + + // a / inv_b collapse to [1, C, 1, 1], x and add stay 2D + const bool shape_ok = ggml_are_same_shape(a, inv_b) && a->ne[0] == 1 && a->ne[1] == x->ne[1]; + const bool dim_ok = + (x->ne[2] == 1) && (x->ne[3] == 1) && + (add->ne[2] == 1) && (add->ne[3] == 1) && + (a->ne[2] == 1) && (a->ne[3] == 1) && + (inv_b->ne[2] == 1) && (inv_b->ne[3] == 1); + + // kernel reads x[idx] and a[c] / inv_b[c] linearly, so every operand is contiguous + const bool contig_ok = + ggml_is_contiguous(x) && ggml_is_contiguous(add) && + ggml_is_contiguous(a) && ggml_is_contiguous(inv_b); + + return types_ok && shape_ok && dim_ok && contig_ok && x_in_add == x; +} + +#define GGML_METAL_TOPK_MOE_MAX_EXPERTS 1024 + +// SOFT_MAX + ARGSORT + GET_ROWS (plus optional norm/scale) for MoE routing. +// This is a multi-output elision chain: the fused kernel writes both the selected +// expert ids and the gathered/normalized routing weights. +static const std::vector ops_topk_moe = { + GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS +}; +static const std::vector ops_topk_moe_scale = { + GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS, GGML_OP_SCALE +}; +static const std::vector ops_topk_moe_norm = { + GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS, + GGML_OP_RESHAPE, GGML_OP_SUM_ROWS, GGML_OP_CLAMP, GGML_OP_DIV, GGML_OP_RESHAPE +}; +static const std::vector ops_topk_moe_norm_scale = { + GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS, + GGML_OP_RESHAPE, GGML_OP_SUM_ROWS, GGML_OP_CLAMP, GGML_OP_DIV, GGML_OP_RESHAPE, GGML_OP_SCALE +}; + +static bool ggml_metal_fusion_check_topk_moe( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_ASSERT(fusion->ops.size() >= 3); + GGML_UNUSED(nodes); + + const int n_ops = (int) fusion->ops.size(); + + const bool with_norm = n_ops >= 6; + const bool with_scale = n_ops == 4 || n_ops == 7; + + // the fusion table operates on the non-empty node sequence; the raw graph also + // contains the RESHAPE/VIEW nodes that the fused kernel elides. + const std::vector & ops_all = fusion->ops_all; + + const int raw_start = node_idxs[idx]; + int raw_end = node_idxs[idx + n_ops - 1]; + + // the norm variant ends with a RESHAPE that the non-empty sequence filters out; + // include it so the output use-count check sees the real final routing tensor + if (with_norm && !with_scale) { + if (raw_end + 1 >= gf->n_nodes) { + return false; + } + const ggml_tensor * trailing_reshape = gf->nodes[raw_end + 1]; + if (trailing_reshape->op != GGML_OP_RESHAPE || trailing_reshape->src[0] != gf->nodes[raw_end]) { + return false; + } + raw_end++; + } + + const int raw_count = raw_end - raw_start + 1; + if (raw_count != (int) ops_all.size()) { + return false; + } + + int raw_idxs[GGML_METAL_FUSION_MAX]; + for (int i = 0; i < raw_count; ++i) { + raw_idxs[i] = raw_start + i; + if (gf->nodes[raw_start + i]->op != ops_all[i]) { + return false; + } + } + + const ggml_tensor * softmax = gf->nodes[raw_start]; + const ggml_tensor * probs_reshaped = gf->nodes[raw_start + 1]; + const ggml_tensor * argsort = gf->nodes[raw_start + 2]; + const ggml_tensor * ids = gf->nodes[raw_start + 3]; + const ggml_tensor * get_rows = gf->nodes[raw_start + 4]; + const ggml_tensor * out = gf->nodes[raw_end]; + const ggml_tensor * logits = softmax->src[0]; + + // the fused kernel implements plain softmax only + float scale = 1.0f; + float max_bias = 0.0f; + memcpy(&scale, ((const int32_t *) softmax->op_params) + 0, sizeof(scale)); + memcpy(&max_bias, ((const int32_t *) softmax->op_params) + 1, sizeof(max_bias)); + if (scale != 1.0f || max_bias != 0.0f || softmax->src[1] || softmax->src[2]) { + return false; + } + + if (logits->type != GGML_TYPE_F32 || softmax->type != GGML_TYPE_F32 || + out->type != GGML_TYPE_F32 || ids->type != GGML_TYPE_I32) { + return false; + } + + const int64_t n_expert = logits->ne[0]; + const int64_t n_tokens = logits->ne[1]; + const int64_t n_expert_used = ids->ne[0]; + + if (n_expert <= 0 || n_tokens <= 0 || n_expert_used <= 0 || n_expert_used > n_expert || + n_expert > GGML_METAL_TOPK_MOE_MAX_EXPERTS || n_expert_used > GGML_METAL_TOPK_MOE_MAX_EXPERTS) { + return false; + } + + if (logits->ne[2] != 1 || logits->ne[3] != 1 || + ids->ne[1] != n_tokens || ids->ne[2] != 1 || ids->ne[3] != 1 || + out->ne[0] != 1 || out->ne[1] != n_expert_used || out->ne[2] != n_tokens || out->ne[3] != 1) { + return false; + } + + if (!ggml_is_contiguous(logits) || !ggml_is_contiguous(out) || + ids->nb[0] != ggml_type_size(GGML_TYPE_I32) || + ids->nb[1] != ggml_type_size(GGML_TYPE_I32) * n_expert) { + return false; + } + + if (probs_reshaped->src[0] != softmax || argsort->src[0] != softmax || + ids->src[0] != argsort || get_rows->src[0] != probs_reshaped || get_rows->src[1] != ids) { + return false; + } + + if (with_norm) { + const ggml_tensor * weights_reshaped = gf->nodes[raw_start + 5]; + const ggml_tensor * sum_rows = gf->nodes[raw_start + 6]; + const ggml_tensor * clamp = gf->nodes[raw_start + 7]; + const ggml_tensor * div = gf->nodes[raw_start + 8]; + const ggml_tensor * out_reshaped = gf->nodes[raw_start + 9]; + + if (weights_reshaped->src[0] != get_rows || sum_rows->src[0] != weights_reshaped || + clamp->src[0] != sum_rows || div->src[0] != weights_reshaped || div->src[1] != clamp || + out_reshaped->src[0] != div) { + return false; + } + + if (with_scale) { + const ggml_tensor * scale_node = gf->nodes[raw_start + 10]; + if (scale_node->src[0] != out_reshaped) { + return false; + } + } + } else if (with_scale) { + const ggml_tensor * scale_node = gf->nodes[raw_start + 5]; + if (scale_node->src[0] != get_rows) { + return false; + } + } + + const int outputs[2] = { raw_start + 3, raw_end }; + if (!ggml_can_fuse_subgraph_ext(gf, raw_idxs, raw_count, ops_all.data(), outputs, 2)) { + return false; + } + + if (mode == GGML_METAL_FUSION_FULL) { + if (!logits->data || !out->data || !ids->data) { + return false; + } + } + + return true; +} + +#define GGML_METAL_MOE_REDUCE_MAX_EXPERTS 8 + +struct ggml_metal_moe_reduce_match { + const ggml_tensor * experts; + const ggml_tensor * weights; + const ggml_tensor * dst; + int node_count; +}; + +static bool ggml_metal_fusion_match_moe_reduce( + const ggml_cgraph * gf, int node_idx, const std::vector & ops_all, + ggml_metal_moe_reduce_match * match) { + if (match == nullptr || node_idx < 0 || node_idx + (int) ops_all.size() > gf->n_nodes) { + return false; + } + + const ggml_tensor * mul = gf->nodes[node_idx]; + if (mul->op != GGML_OP_MUL || mul->type != GGML_TYPE_F32) { + return false; + } + + // MUL, then one VIEW per expert, then one ADD per additional expert + const int raw_count = (int) ops_all.size(); + const int n_expert_used = raw_count / 2; + + if (n_expert_used < 2 || n_expert_used > GGML_METAL_MOE_REDUCE_MAX_EXPERTS || + raw_count != 2 * n_expert_used) { + return false; + } + + int n_views = 0; + while (node_idx + 1 + n_views < gf->n_nodes && + gf->nodes[node_idx + 1 + n_views]->op == GGML_OP_VIEW) { + n_views++; + } + + if (n_views != n_expert_used) { + return false; + } + + for (int i = n_expert_used + 1; i < raw_count; ++i) { + if (gf->nodes[node_idx + i]->op != GGML_OP_ADD) { + return false; + } + } + + int raw_idxs[GGML_METAL_FUSION_MAX]; + for (int i = 0; i < raw_count; ++i) { + raw_idxs[i] = node_idx + i; + if (gf->nodes[node_idx + i]->op != ops_all[i]) { + return false; + } + } + + const ggml_tensor * experts = mul->src[0]; + const ggml_tensor * weights = mul->src[1]; + const ggml_tensor * dst = gf->nodes[node_idx + raw_count - 1]; + + if (experts->type != GGML_TYPE_F32 || weights->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { + return false; + } + + const int64_t n_embd = experts->ne[0]; + const int64_t n_tokens = experts->ne[2]; + + if (n_embd <= 0 || n_tokens <= 0 || experts->ne[1] != n_expert_used || experts->ne[3] != 1 || + weights->ne[0] != 1 || weights->ne[1] != n_expert_used || weights->ne[2] != n_tokens || weights->ne[3] != 1 || + dst->ne[0] != n_embd || dst->ne[1] != n_tokens || dst->ne[2] != 1 || dst->ne[3] != 1) { + return false; + } + + if (!ggml_is_contiguous(experts) || !ggml_is_contiguous(weights) || !ggml_is_contiguous(dst)) { + return false; + } + + for (int i = 1; i <= n_expert_used; ++i) { + const ggml_tensor * view = gf->nodes[node_idx + i]; + if (view->view_src != mul || view->src[0] != mul || + view->view_offs != (size_t) (i - 1) * mul->nb[1] || + view->ne[0] != n_embd || view->ne[1] != n_tokens || + view->nb[1] != mul->nb[2]) { + return false; + } + } + + const ggml_tensor * prev_add = nullptr; + for (int j = 1; j < n_expert_used; ++j) { + const ggml_tensor * add = gf->nodes[node_idx + n_expert_used + j]; + const ggml_tensor * rhs = gf->nodes[node_idx + j + 1]; + const ggml_tensor * lhs = j == 1 ? gf->nodes[node_idx + 1] : prev_add; + if (add->src[0] != lhs || add->src[1] != rhs) { + return false; + } + prev_add = add; + } + + const int outputs[1] = { node_idx + raw_count - 1 }; + if (!ggml_can_fuse_subgraph_ext(gf, raw_idxs, raw_count, ops_all.data(), outputs, 1)) { + return false; + } + + match->experts = experts; + match->weights = weights; + match->dst = dst; + match->node_count = raw_count; + return true; +} + +static bool ggml_metal_fusion_check_moe_reduce( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + const ggml_cgraph * gf, + const int * node_idxs, + int idx, + ggml_metal_fusion_mode mode) { + GGML_UNUSED(nodes); + + ggml_metal_moe_reduce_match match; + if (!ggml_metal_fusion_match_moe_reduce(gf, node_idxs[idx], fusion->ops_all, &match)) { + return false; + } + + if ((int) fusion->ops.size() != match.experts->ne[1]) { + return false; + } + + const int raw_end = node_idxs[idx] + match.node_count - 1; + if (node_idxs[idx + (int) fusion->ops.size() - 1] != raw_end) { + return false; + } + + if (mode == GGML_METAL_FUSION_FULL) { + if (!match.experts->data || !match.weights->data || !match.dst->data) { + return false; + } + } + + return true; +} + +// ---- patterns ------------------------------------------------------------ + +static const std::vector ops_norm_mul = { GGML_OP_NORM, GGML_OP_MUL }; +static const std::vector ops_norm_mul_add = { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD }; +static const std::vector ops_norm_scale = { GGML_OP_NORM, GGML_OP_SCALE }; +static const std::vector ops_rms_norm_mul = { GGML_OP_RMS_NORM, GGML_OP_MUL }; +static const std::vector ops_rms_norm_mul_add = { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }; +static const std::vector ops_rms_norm_scale = { GGML_OP_RMS_NORM, GGML_OP_SCALE }; + +static const std::vector ops_add_2 = { GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_add_3 = { GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_add_4 = { GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_add_5 = { GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_add_6 = { GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_add_7 = { GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD }; +static const std::vector ops_snake = { GGML_OP_MUL, GGML_OP_SIN, GGML_OP_SQR, GGML_OP_MUL, GGML_OP_ADD }; + +static const std::vector ops_gdn_cache = { GGML_OP_GATED_DELTA_NET, GGML_OP_CPY }; + +static const std::vector ops_ssm_conv_silu = { GGML_OP_SSM_CONV, GGML_OP_UNARY }; + +static const std::vector ops_moe_reduce_2 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_3 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_ADD, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_4 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, + GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_5 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, + GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_6 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, + GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_7 = { + GGML_OP_MUL, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, + GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD +}; +static const std::vector ops_moe_reduce_8 = { + GGML_OP_MUL, + GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, GGML_OP_VIEW, + GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD, GGML_OP_ADD +}; + +static const std::vector ggml_metal_fusions = { + { GGML_METAL_FUSION_NORM_MUL, ops_norm_mul, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_NORM_MUL_ADD, ops_norm_mul_add, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_NORM_SCALE, ops_norm_scale, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_NORM_MUL, ops_rms_norm_mul, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_NORM_MUL_ADD, ops_rms_norm_mul_add, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_NORM_SCALE, ops_rms_norm_scale, {}, false, ggml_metal_fusion_check_norm }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_2, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_3, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_4, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_5, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_6, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_ADD_CHAIN, ops_add_7, {}, false, ggml_metal_fusion_check_add_chain }, + { GGML_METAL_FUSION_SNAKE, ops_snake, {}, false, ggml_metal_fusion_check_snake }, + { GGML_METAL_FUSION_GDN_CACHE, ops_gdn_cache, {}, true, ggml_metal_fusion_check_gdn_cache }, + { GGML_METAL_FUSION_TOPK_MOE, ops_topk_moe, {1}, true, ggml_metal_fusion_check_topk_moe }, + { GGML_METAL_FUSION_TOPK_MOE, ops_topk_moe_scale, {1}, true, ggml_metal_fusion_check_topk_moe }, + { GGML_METAL_FUSION_TOPK_MOE, ops_topk_moe_norm, {1}, true, ggml_metal_fusion_check_topk_moe }, + { GGML_METAL_FUSION_TOPK_MOE, ops_topk_moe_norm_scale, {1}, true, ggml_metal_fusion_check_topk_moe }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_2, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_3, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_4, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_5, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_6, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_7, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_MOE_REDUCE, ops_moe_reduce_8, {}, true, ggml_metal_fusion_check_moe_reduce }, + { GGML_METAL_FUSION_SSM_CONV_SILU, ops_ssm_conv_silu, {}, false, ggml_metal_fusion_check_ssm_conv_silu }, +}; + +// ---- alloc deps ----------------------------------------------------------- + +static bool ggml_metal_fusion_match_raw_pattern( + const ggml_cgraph * gf, int node_idx, const std::vector & ops) { + if (node_idx < 0 || node_idx + (int) ops.size() > gf->n_nodes) { + return false; + } + + for (int i = 0; i < (int) ops.size(); ++i) { + if (gf->nodes[node_idx + i]->op != ops[i]) { + return false; + } + } + + return true; +} + +static void ggml_metal_fusion_add_pattern_alloc_deps( + void * user_data, + void (*add_alloc_dep)(void *, ggml_tensor *, ggml_tensor *), + ggml_cgraph * gf, + const ggml_metal_fusion * fusion, + int node_idx) { + const int last_node = node_idx + (int) fusion->ops_all.size() - 1; + + // keep all external inputs alive until the fused output + std::set seen; + for (int j = 0; j < (int) fusion->ops_all.size(); ++j) { + ggml_tensor * node = gf->nodes[node_idx + j]; + for (int s = 0; s < GGML_MAX_SRC; ++s) { + ggml_tensor * src = node->src[s]; + if (src && seen.insert(src).second) { + add_alloc_dep(user_data, src, gf->nodes[last_node]); + } + } + seen.insert(node); + } +} + +void ggml_metal_fusion_add_alloc_deps( + void * user_data, + void (*add_alloc_dep)(void *, ggml_tensor *, ggml_tensor *), + ggml_cgraph * gf) { + for (int i = 0; i < gf->n_nodes; ++i) { + const ggml_metal_fusion * best = nullptr; + int best_raw = 0; + + for (const ggml_metal_fusion & fusion : ggml_metal_fusions) { + if ((int) fusion.ops_all.size() <= best_raw) { + continue; + } + if (ggml_metal_fusion_match_raw_pattern(gf, i, fusion.ops_all)) { + best = &fusion; + best_raw = (int) fusion.ops_all.size(); + } + } + + if (best) { + ggml_metal_fusion_add_pattern_alloc_deps(user_data, add_alloc_dep, gf, best, i); + i += best_raw - 1; + } + } +} + +// ---- shared fusion info --------------------------------------------------- + +static std::string ggml_metal_fusion_label(const ggml_metal_fusion * fusion) { + GGML_ASSERT(fusion != nullptr); + + std::string label; + for (int j = 0; j < (int) fusion->ops.size(); j++) { + if (j > 0) { + label += '+'; + } + label += ggml_op_name(fusion->ops[j]); + } + return label; +} + +struct ggml_metal_fusion_info { + std::vector labels; + std::vector counts; + bool enabled; + bool stats; + bool labels_set; + int debug; +}; + +struct ggml_metal_fusion_info * ggml_metal_fusion_info_init(bool enabled, int debug) { + ggml_metal_fusion_info * finfo = new ggml_metal_fusion_info; + finfo->enabled = enabled; + finfo->stats = debug > 0; + finfo->labels_set = false; + finfo->debug = debug; + + if (finfo->stats) { + ggml_metal_fusion_info_labels_init(finfo); + } + + return finfo; +} + +void ggml_metal_fusion_info_free(ggml_metal_fusion_info * finfo) { + delete finfo; +} + +bool ggml_metal_fusion_info_enabled(const ggml_metal_fusion_info * finfo) { + return finfo->enabled; +} + +bool ggml_metal_fusion_info_stats(const ggml_metal_fusion_info * finfo) { + return finfo->stats; +} + +int ggml_metal_fusion_info_debug(const ggml_metal_fusion_info * finfo) { + return finfo->debug; +} + +int ggml_metal_fusion_info_n_fusions(const ggml_metal_fusion_info * finfo) { + return (int) finfo->labels.size(); +} + +const char * ggml_metal_fusion_info_label(const ggml_metal_fusion_info * finfo, int idx) { + GGML_ASSERT(idx >= 0 && idx < (int) finfo->labels.size()); + return finfo->labels[idx].c_str(); +} + +uint64_t ggml_metal_fusion_info_count(const ggml_metal_fusion_info * finfo, int idx) { + GGML_ASSERT(idx >= 0 && idx < (int) finfo->counts.size()); + return finfo->counts[idx]; +} + +void ggml_metal_fusion_info_count_fusion(ggml_metal_fusion_info * finfo, const ggml_metal_fusion * fusion) { + if (!finfo->stats || fusion == nullptr) { + return; + } + + const ptrdiff_t idx = fusion - ggml_metal_fusions.data(); + if (idx >= 0 && idx < (ptrdiff_t) finfo->counts.size()) { + finfo->counts[idx]++; + } +} + +void ggml_metal_fusion_info_set_enabled(ggml_metal_fusion_info * finfo, bool enabled) { + finfo->enabled = enabled; +} + +void ggml_metal_fusion_info_labels_init(ggml_metal_fusion_info * finfo) { + if (finfo->labels_set) { + return; + } + + finfo->labels.clear(); + finfo->counts.assign(ggml_metal_fusions.size(), 0); + finfo->labels.reserve(ggml_metal_fusions.size()); + + for (const ggml_metal_fusion & fusion : ggml_metal_fusions) { + finfo->labels.emplace_back(ggml_metal_fusion_label(&fusion)); + } + + finfo->labels_set = true; +} + +void ggml_metal_fusion_info_stats_init(ggml_metal_fusion_info * finfo) { + finfo->stats = true; + ggml_metal_fusion_info_labels_init(finfo); +} + +void ggml_metal_fusion_info_stats_reset(ggml_metal_fusion_info * finfo) { + std::fill(finfo->counts.begin(), finfo->counts.end(), 0); +} + +int ggml_metal_fusion_info_stats_get(const ggml_metal_fusion_info * finfo, const char ** labels, uint64_t * counts, int n) { + const int n_fusions = (int) finfo->labels.size(); + + if (labels == nullptr) { + return n_fusions; + } + + const int n_fill = std::min(n, n_fusions); + for (int i = 0; i < n_fill; i++) { + labels[i] = finfo->labels[i].c_str(); + if (counts != nullptr) { + counts[i] = finfo->counts[i]; + } + } + + return n_fill; +} + +// ---- memory-range checks ------------------------------------------------- + +// reject fusions where an external source overlaps any fused output. the fused +// kernels elide intermediate nodes, so only sources that are not part of the +// fused subgraph can cause read/write races with the output. +static bool ggml_metal_fusion_check_memory_ranges( + const ggml_metal_fusion * fusion, + const ggml_tensor * const * nodes, + int node_count) { + // some fused kernels write through a tensor that also appears as a source (e.g. the gdn + // cache cpy), so a source that is the same memory as the output is not an external read + // source + auto same_memory = [](const ggml_tensor * a, const ggml_tensor * b) { + if (a->data && b->data && a->data == b->data) { + return true; + } + for (const ggml_tensor * v = a; v; v = v->view_src) { + if (v == b) { + return true; + } + } + for (const ggml_tensor * v = b; v; v = v->view_src) { + if (v == a) { + return true; + } + } + return false; + }; + + auto nodes_overlap = [](const ggml_tensor * a, const ggml_tensor * b) { + if (!a || !b || !a->data || !b->data || !a->buffer || !b->buffer) { + return false; + } + + if (a->buffer != b->buffer) { + return false; + } + + const int64_t a_start = (int64_t) a->data; + const int64_t a_end = a_start + ggml_backend_buft_get_alloc_size(a->buffer->buft, a); + const int64_t b_start = (int64_t) b->data; + const int64_t b_end = b_start + ggml_backend_buft_get_alloc_size(b->buffer->buft, b); + + return (b_start <= a_start && a_start < b_end) || + (a_start <= b_start && b_start < a_end); + }; + + auto is_intermediate = [](const ggml_tensor * src, const ggml_tensor * const * nodes, int j) { + for (int k = 0; k < j; ++k) { + if (src == nodes[k]) { + return true; + } + for (const ggml_tensor * view_src = src->view_src; view_src; view_src = view_src->view_src) { + if (view_src == nodes[k]) { + return true; + } + } + } + return false; + }; + + auto check_dst = [&](const ggml_tensor * dst) { + for (int j = 0; j < node_count; ++j) { + for (int s = 0; s < GGML_MAX_SRC; ++s) { + const ggml_tensor * src = nodes[j]->src[s]; + if (!src || src->op == GGML_OP_NONE || same_memory(src, dst)) { + continue; + } + + if (nodes_overlap(dst, src) && !is_intermediate(src, nodes, j)) { + return false; + } + } + } + return true; + }; + + if (!check_dst(nodes[node_count - 1])) { + return false; + } + + for (int offset : fusion->outs) { + GGML_ASSERT(offset >= 0 && offset < node_count); + if (!check_dst(nodes[offset])) { + return false; + } + } + + return true; +} + +// ---- queries ------------------------------------------------------------- + +// find the longest pattern matching the node sequence starting at idx +// (idx is a position in node_idxs, which maps to graph node indices) +const ggml_metal_fusion * ggml_metal_fusion_next( + const ggml_cgraph * gf, + const int * node_idxs, + int n_idxs, + int idx, + ggml_metal_fusion_mode mode, + int * n_out) { + const ggml_metal_fusion * res = nullptr; + int best = 1; + + for (const ggml_metal_fusion & fusion : ggml_metal_fusions) { + const int n_ops = (int) fusion.ops.size(); + + // only look for a longer match than the current best + if (n_ops <= best) { + continue; + } + if (idx + n_ops > n_idxs) { + continue; + } + + const ggml_tensor * nodes[GGML_METAL_FUSION_MAX]; + + // the op sequence must match exactly + bool ok = true; + for (int j = 0; j < n_ops; j++) { + nodes[j] = gf->nodes[node_idxs[idx + j]]; + if (nodes[j]->op != fusion.ops[j]) { + ok = false; + break; + } + } + if (!ok) { + continue; + } + + if (!fusion.unsafe) { + // common element-wise chain constraints: each node reads the previous one, + // and all nodes have the same shape + for (int j = 1; j < n_ops && ok; j++) { + if (nodes[j]->src[0] != nodes[j - 1] && nodes[j]->src[1] != nodes[j - 1]) { + ok = false; + break; + } + if (!ggml_are_same_shape(nodes[j], nodes[j - 1])) { + ok = false; + break; + } + } + if (!ok) { + continue; + } + + // primary output is the last node; additional outputs come from fusion.outs + int outputs_buf[GGML_METAL_FUSION_MAX]; + outputs_buf[0] = node_idxs[idx + n_ops - 1]; + for (size_t i = 0; i < fusion.outs.size(); ++i) { + const int out_offset = fusion.outs[i]; + GGML_ASSERT(out_offset >= 0 && out_offset < n_ops); + outputs_buf[i + 1] = node_idxs[idx + out_offset]; + } + + const int n_outputs = 1 + (int) fusion.outs.size(); + + // structural subgraph checks (op sequence, elidable uses, view containment) + if (!ggml_can_fuse_subgraph_ext(gf, node_idxs + idx, n_ops, fusion.ops.data(), outputs_buf, n_outputs)) { + continue; + } + } + + // pattern-specific checks (the sole validator for unsafe patterns) + if (fusion.check && !fusion.check(&fusion, nodes, gf, node_idxs, idx, mode)) { + continue; + } + + // the compute phase has allocated tensors and can detect aliasing between + // external sources and fused outputs; the optimizer phase cannot do this yet + if (mode == GGML_METAL_FUSION_FULL && + !ggml_metal_fusion_check_memory_ranges(&fusion, nodes, n_ops)) { + continue; + } + + best = n_ops; + res = &fusion; + } + + *n_out = best; + + return res; +} + +// optimize phase: maximum number of nodes starting at idx (a raw sequential graph index) that +// could be fused, chaining patterns back-to-back. matching runs on the same filtered (view +// transparent) node sequence that the compute phase uses, so the returned count is the raw index +// span from idx to the last matched node (intermediate views are packed along). +int ggml_metal_fusion_max(const ggml_cgraph * gf, int idx) { + // an empty/view node cannot start a pattern - pack it alone + if (ggml_op_is_empty(gf->nodes[idx]->op) || ggml_is_empty(gf->nodes[idx])) { + return 1; + } + + // collect the non-empty node indices starting at idx + int idxs[GGML_METAL_FUSION_MAX]; + int n_idxs = 0; + for (int i = idx; i < gf->n_nodes && n_idxs < GGML_METAL_FUSION_MAX; i++) { + if (!ggml_op_is_empty(gf->nodes[i]->op) && !ggml_is_empty(gf->nodes[i])) { + idxs[n_idxs++] = i; + } + } + if (n_idxs == 0) { + return 1; + } + + int total = 0; + int i_f = 0; + + while (i_f < n_idxs && total < GGML_METAL_FUSION_MAX) { + int len = 1; + const ggml_metal_fusion * fusion = ggml_metal_fusion_next(gf, idxs, n_idxs, i_f, GGML_METAL_FUSION_STRUCTURAL, &len); + if (!fusion || total + len > GGML_METAL_FUSION_MAX) { + break; + } + + total += len; + i_f += len; + } + + if (i_f == 0) { + return 1; + } + + // map the matched non-empty nodes back to the raw index span (views are included) + return std::min(GGML_METAL_FUSION_MAX, idxs[i_f - 1] - idx + 1); +} diff --git a/ggml/src/ggml-metal/ggml-metal-fusion.h b/ggml/src/ggml-metal/ggml-metal-fusion.h new file mode 100644 index 00000000..6b139a69 --- /dev/null +++ b/ggml/src/ggml-metal/ggml-metal-fusion.h @@ -0,0 +1,98 @@ +// single source of truth for the fusions supported by the Metal backend +// +// every fusable subgraph is declared exactly once as a ggml_metal_fusion entry in +// the table in ggml-metal-fusion.cpp. both the graph optimizer (ggml_metal_fusion_max) +// and the op encoders (ggml_metal_fusion_next) consult this same table, so the two +// phases can never disagree about what can be fused. + +#pragma once + +#include "ggml-impl.h" + +#include + +#ifdef __cplusplus +extern "C" { +#endif + +// the maximum number of nodes that can be fused in a single kernel +// (also the maximum length of a packed fusion group during graph optimization) +#define GGML_METAL_FUSION_MAX 16 + +typedef enum ggml_metal_fusion_mode { + // structural checks only; used by the graph optimizer, at which point the graph + // tensors are not allocated yet, so buffer placement cannot be verified + GGML_METAL_FUSION_STRUCTURAL = 0, + // full checks, including buffer placement; used by the op encoders + GGML_METAL_FUSION_FULL, +} ggml_metal_fusion_mode; + +// identifier of each fusion pattern so the op encoders know which kernel to use +typedef enum ggml_metal_fusion_id { + GGML_METAL_FUSION_NONE = 0, + GGML_METAL_FUSION_NORM_MUL, // NORM/RMS_NORM + MUL + GGML_METAL_FUSION_NORM_MUL_ADD, // NORM/RMS_NORM + MUL + ADD + GGML_METAL_FUSION_NORM_SCALE, // NORM/RMS_NORM + SCALE + GGML_METAL_FUSION_ADD_CHAIN, // ADD x N (N in [2, 7]) + GGML_METAL_FUSION_SNAKE, // MUL + SIN + SQR + MUL + ADD + GGML_METAL_FUSION_GDN_CACHE, // GATED_DELTA_NET + CPY (write snapshots into the recurrent cache) + GGML_METAL_FUSION_TOPK_MOE, // SOFT_MAX + ARGSORT + GET_ROWS + norm/scale (MoE routing) + GGML_METAL_FUSION_MOE_REDUCE, // MUL + expert VIEWs + ADD chain (MoE output reduction) + GGML_METAL_FUSION_SSM_CONV_SILU, // SSM_CONV + UNARY (silu) +} ggml_metal_fusion_id; + +struct ggml_metal_fusion; // defined in ggml-metal-fusion.cpp + +typedef struct ggml_metal_fusion ggml_metal_fusion; + +// access the fusion identifier without exposing the full pattern definition +ggml_metal_fusion_id ggml_metal_fusion_get_id(const struct ggml_metal_fusion * fusion); + +// apply any alloc-dependencies required by the fused kernels during graph optimize +void ggml_metal_fusion_add_alloc_deps( + void * user_data, + void (*add_alloc_dep)(void *, struct ggml_tensor *, struct ggml_tensor *), + struct ggml_cgraph * gf); + +// ---- shared fusion info --------------------------------------------------- + +// shared fusion debugging context, owned by the device; newly created backend contexts for that +// device register with it so the fusion counters are race-free and accumulate across contexts. +struct ggml_metal_fusion_info; // defined in ggml-metal-fusion.cpp + +struct ggml_metal_fusion_info * ggml_metal_fusion_info_init(bool enabled, int debug); +void ggml_metal_fusion_info_free(struct ggml_metal_fusion_info * finfo); + +bool ggml_metal_fusion_info_enabled(const struct ggml_metal_fusion_info * finfo); +bool ggml_metal_fusion_info_stats (const struct ggml_metal_fusion_info * finfo); +int ggml_metal_fusion_info_debug (const struct ggml_metal_fusion_info * finfo); + +int ggml_metal_fusion_info_n_fusions(const struct ggml_metal_fusion_info * finfo); +const char * ggml_metal_fusion_info_label (const struct ggml_metal_fusion_info * finfo, int idx); +uint64_t ggml_metal_fusion_info_count (const struct ggml_metal_fusion_info * finfo, int idx); + +void ggml_metal_fusion_info_count_fusion(struct ggml_metal_fusion_info * finfo, const struct ggml_metal_fusion * fusion); +void ggml_metal_fusion_info_set_enabled (struct ggml_metal_fusion_info * finfo, bool enabled); + +void ggml_metal_fusion_info_stats_init ( struct ggml_metal_fusion_info * finfo); +void ggml_metal_fusion_info_stats_reset( struct ggml_metal_fusion_info * finfo); +int ggml_metal_fusion_info_stats_get (const struct ggml_metal_fusion_info * finfo, const char ** labels, uint64_t * counts, int n); +void ggml_metal_fusion_info_labels_init( struct ggml_metal_fusion_info * finfo); + +// compute phase: longest fusion starting at idx (a position in node_idxs) that matches in `mode`. +// returns the matching pattern (nullptr if no fusion) and sets *n_out to the number of nodes consumed. +const ggml_metal_fusion * ggml_metal_fusion_next( + const struct ggml_cgraph * gf, + const int * node_idxs, + int n_idxs, + int idx, + ggml_metal_fusion_mode mode, + int * n_out); + +// optimize phase: maximum number of nodes starting at idx (a raw sequential graph index) that +// could be fused, chaining patterns back-to-back. returns at least 1. +int ggml_metal_fusion_max(const struct ggml_cgraph * gf, int idx); + +#ifdef __cplusplus +} +#endif diff --git a/ggml/src/ggml-metal/ggml-metal-impl.h b/ggml/src/ggml-metal/ggml-metal-impl.h index 1f6e8c48..490dd83a 100644 --- a/ggml/src/ggml-metal/ggml-metal-impl.h +++ b/ggml/src/ggml-metal/ggml-metal-impl.h @@ -14,6 +14,8 @@ #define N_MM_SIMD_GROUP_X 2 #define N_MM_SIMD_GROUP_Y 2 +#define N_MM_NPART_AMAX 256 + // kernel parameters for mat-vec threadgroups // // N_R0: number of src0 rows to process per simdgroup @@ -62,24 +64,31 @@ #define N_R0_IQ1_S 4 #define N_SG_IQ1_S 2 +#define N_R0_IQ1_S_SPLIT 8 #define N_R0_IQ1_M 4 #define N_SG_IQ1_M 2 +#define N_R0_IQ1_M_SPLIT 8 #define N_R0_IQ2_XXS 4 #define N_SG_IQ2_XXS 2 +#define N_R0_IQ2_XXS_SPLIT 8 #define N_R0_IQ2_XS 4 #define N_SG_IQ2_XS 2 +#define N_R0_IQ2_XS_SPLIT 8 #define N_R0_IQ2_S 4 #define N_SG_IQ2_S 2 +#define N_R0_IQ2_S_SPLIT 8 #define N_R0_IQ3_XXS 4 #define N_SG_IQ3_XXS 2 +#define N_R0_IQ3_XXS_SPLIT 8 #define N_R0_IQ3_S 4 #define N_SG_IQ3_S 2 +#define N_R0_IQ3_S_SPLIT 8 #define N_R0_IQ4_NL 2 #define N_SG_IQ4_NL 2 @@ -107,6 +116,10 @@ #define FC_SUM_ROWS 1400 #define FC_UPSCALE 1500 #define FC_GATED_DELTA_NET 1600 +#define FC_NORM 1700 +#define FC_TOPK_MOE 1800 +#define FC_MOE_REDUCE 1900 +#define FC_DSV4_HC 2000 // op-specific constants #define OP_FLASH_ATTN_EXT_NQPSG 8 @@ -158,6 +171,10 @@ #define OP_SUM_ROWS_NUM_SUM_ROWS 10 #define OP_SUM_ROWS_NUM_MEAN 11 +#define OP_SSM_SCAN_SSD_CS 64 // Metal-specific; Chunk Size; 64 is largest multiple of 8 (simdgroup tile) fitting into 32 KiB Metal threadgroup mem limit (~26.75 KiB shared mem; see smem layout comment in kernel_ssm_scan_ssd_mma_f32) +#define OP_SSM_SCAN_SSD_HD 64 // Metal-specific; Head Dim the MMA kernel is specialized for (Mamba-2); use_mma gates on d_inner == this +#define OP_SSM_SCAN_SSD_NSG 4 // Metal-specific; Number of SimdGroups per threadgroup; NSG*32 == threads dispatched per threadgroup + // kernel argument structs // // - element counters (e.g. ne00) typically use int32_t to reduce register usage @@ -329,6 +346,7 @@ typedef struct { uint64_t nb3; int32_t n_past; int32_t n_dims; + int32_t n_offs; int32_t n_ctx_orig; float freq_base; float freq_scale; @@ -341,8 +359,21 @@ typedef struct { int32_t sect_2; int32_t sect_3; bool src2; + bool inplace; } ggml_metal_kargs_rope; +typedef struct { + int32_t ne0; + int32_t ne1; + int32_t ne2; + int32_t ne3; + uint64_t nb0; + uint64_t nb1; + uint64_t nb2; + uint64_t nb3; + int32_t nblocks; +} ggml_metal_kargs_flash_attn_ext_kv_f16; + typedef struct { int32_t ne11; int32_t ne_12_2; // assume K and V are same shape @@ -440,8 +471,21 @@ typedef struct { float m1; int32_t n_head_log2; float logit_softcap; + int32_t n_kv_max_padded; } ggml_metal_kargs_flash_attn_ext_vec; +typedef struct { + int32_t ne30; + int32_t ne31; + int32_t ne32; + int32_t ne33; + uint64_t nb31; + uint64_t nb32; + uint64_t nb33; + int32_t n_kv_max; + int32_t n_kv_max_padded; +} ggml_metal_kargs_flash_attn_ext_vec_idx; + typedef struct { int32_t nrows; } ggml_metal_kargs_flash_attn_ext_vec_reduce; @@ -517,6 +561,14 @@ typedef struct { uint64_t nb21; } ggml_metal_kargs_mul_mm_id_map0; +typedef struct { + int32_t ne00; + int32_t ne01; + int32_t ne02; + uint64_t nb01; + uint64_t nb02; +} ggml_metal_kargs_mul_mm_id_amax; + typedef struct { int32_t ne00; int32_t ne02; @@ -574,6 +626,7 @@ typedef struct { uint64_t nbf1[3]; uint64_t nbf2[3]; uint64_t nbf3[3]; + float scale; } ggml_metal_kargs_norm; typedef struct { @@ -642,6 +695,7 @@ typedef struct { uint64_t nb0; uint64_t nb1; uint64_t nb2; + uint64_t nb3; } ggml_metal_kargs_conv_transpose_2d; typedef struct { @@ -861,7 +915,6 @@ typedef struct { uint64_t nb00; uint64_t nb01; uint64_t nb02; - int64_t ne10; int64_t ne11; uint64_t nb10; uint64_t nb11; @@ -879,6 +932,8 @@ typedef struct { int64_t n_head; int64_t n_group; int64_t n_seq_tokens; + int64_t n_seq_tokens_total; + int64_t token_offset; int64_t n_seqs; int64_t K; uint64_t s_off; @@ -944,6 +999,7 @@ typedef struct { uint64_t nb1; uint64_t nb2; uint64_t nb3; + uint64_t nb_out; // 0 => snapshots are appended after the attn scores (unfused) } ggml_metal_kargs_gated_delta_net; typedef struct { @@ -1168,6 +1224,30 @@ typedef struct { int32_t len; } ggml_metal_kargs_argsort_merge; +typedef struct { + int32_t ne00; // number of columns (elements per row) + int32_t ne01; // rows + int32_t ne02; + int32_t ne03; + uint64_t nb01; // row stride in src0 + uint64_t nb02; + uint64_t nb03; + int32_t top_k; // k +} ggml_metal_kargs_top_k; + +typedef struct { + int32_t ne01; // n_tokens + uint64_t nb01; // logits row stride + uint64_t nb1_ids; // ids row stride + float clamp; + float scale; +} ggml_metal_kargs_topk_moe; + +typedef struct { + int32_t ne00; // n_embd + int32_t ne02; // n_tokens +} ggml_metal_kargs_moe_reduce; + typedef struct { int32_t nrows; } ggml_metal_kargs_fwht; @@ -1220,8 +1300,10 @@ typedef struct { uint64_t nb_x2; uint64_t nb_w0; uint64_t nb_w1; + uint64_t nb_w2; uint64_t nb_d0; uint64_t nb_d1; + float scale; } ggml_metal_kargs_dsv4_hc_pre; typedef struct { diff --git a/ggml/src/ggml-metal/ggml-metal-ops.cpp b/ggml/src/ggml-metal/ggml-metal-ops.cpp index b7f9b2d0..708703e5 100644 --- a/ggml/src/ggml-metal/ggml-metal-ops.cpp +++ b/ggml/src/ggml-metal/ggml-metal-ops.cpp @@ -7,6 +7,8 @@ #include "ggml-metal-impl.h" #include "ggml-metal-common.h" #include "ggml-metal-device.h" +#include "ggml-metal-fusion.h" +#include "ggml-metal-tuning.h" #include #include @@ -30,24 +32,22 @@ struct ggml_metal_op { ggml_metal_device_t dev, ggml_metal_cmd_buf_t cmd_buf, ggml_cgraph * gf, + ggml_metal_fusion_info * finfo, int idx_start, int idx_end, - bool use_fusion, bool use_concurrency, bool use_capture, - int debug_graph, - int debug_fusion) { + int debug_graph) { this->dev = dev; this->lib = ggml_metal_device_get_library(dev); this->enc = ggml_metal_encoder_init(cmd_buf, use_concurrency); this->mem_ranges = ggml_mem_ranges_init(debug_graph); + this->finfo = finfo; this->idx_start = idx_start; this->idx_end = idx_end; - this->use_fusion = use_fusion; this->use_concurrency = use_concurrency; this->use_capture = use_capture; this->debug_graph = debug_graph; - this->debug_fusion = debug_fusion; this->gf = gf; idxs.reserve(gf->n_nodes); @@ -77,15 +77,24 @@ struct ggml_metal_op { return ggml_graph_node(gf, idxs[i]); } - bool can_fuse(int i0, const ggml_op * ops, int n_ops) const { - assert(use_fusion); + // consult the fusion table for the longest pattern starting at i0 + // returns the matching pattern (nullptr if no fusion) and sets *n_out to the number of nodes + const ggml_metal_fusion * can_fuse(int i0, enum ggml_metal_fusion_mode mode, int * n_out) const { + assert(use_fusion()); assert(i0 >= 0 && i0 < n_nodes()); - if (i0 + n_ops > n_nodes()) { - return false; - } + return ggml_metal_fusion_next(gf, idxs.data(), (int) idxs.size(), i0, mode, n_out); + } - return ggml_can_fuse_ext(gf, idxs.data() + i0, ops, n_ops); + // whether to attempt fusion; the toggle lives in the shared fusion debugging context owned + // by the device (initialized from GGML_METAL_FUSION_DISABLE, overridable by the test) + bool use_fusion() const { + return ggml_metal_fusion_info_enabled(finfo); + } + + // record that a fusion fired, indexed by the matching table entry + void count_fusions(const ggml_metal_fusion * fusion) const { + ggml_metal_fusion_info_count_fusion(finfo, fusion); } ggml_metal_device_t dev; @@ -93,12 +102,13 @@ struct ggml_metal_op { ggml_metal_encoder_t enc; ggml_mem_ranges_t mem_ranges; - bool use_fusion; + // shared fusion debugging context + ggml_metal_fusion_info * finfo; + bool use_concurrency; bool use_capture; int debug_graph; - int debug_fusion; private: ggml_cgraph * gf; @@ -114,24 +124,22 @@ ggml_metal_op_t ggml_metal_op_init( ggml_metal_device_t dev, ggml_metal_cmd_buf_t cmd_buf, ggml_cgraph * gf, + ggml_metal_fusion_info * finfo, int idx_start, int idx_end, - bool use_fusion, bool use_concurrency, bool use_capture, - int debug_graph, - int debug_fusion) { + int debug_graph) { ggml_metal_op_t res = new ggml_metal_op( dev, cmd_buf, gf, + finfo, idx_start, idx_end, - use_fusion, use_concurrency, use_capture, - debug_graph, - debug_fusion); + debug_graph); return res; } @@ -218,7 +226,16 @@ static int ggml_metal_op_encode_impl(ggml_metal_op_t ctx, int idx) { // otherwise, we add the new ranges to the encoding context and process the node concurrently // { - const bool is_concurrent = ggml_metal_op_concurrency_check(ctx, node); + bool is_concurrent = ggml_metal_op_concurrency_check(ctx, node); + + if (is_concurrent && ctx->use_fusion()) { + int n_fuse = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n_fuse); + if (fusion) { + // fused kernels write to the last node of the group, not necessarily to the first node's dst + is_concurrent = ggml_mem_ranges_check(ctx->mem_ranges, ctx->node(idx + n_fuse - 1)); + } + } if (!is_concurrent) { ggml_metal_op_concurrency_reset(ctx); @@ -551,8 +568,24 @@ int ggml_metal_op_concat(ggml_metal_op_t ctx, int idx) { const int32_t dim = ((const int32_t *) op->op_params)[0]; + const bool is_q = ggml_is_quantized(op->type); + + // for quantized types, concat is done at the block level (nb0 == type_size == block size) + int32_t ne00_arg = ne00; + int32_t ne10_arg = ne10; + int32_t ne0_arg = ne0; + if (is_q) { + const int32_t blck = ggml_blck_size(op->type); + GGML_ASSERT(ne00 % blck == 0); + GGML_ASSERT(ne10 % blck == 0); + GGML_ASSERT(ne0 % blck == 0); + ne00_arg = ne00/blck; + ne10_arg = ne10/blck; + ne0_arg = ne0/blck; + } + ggml_metal_kargs_concat args = { - /*.ne00 =*/ ne00, + /*.ne00 =*/ ne00_arg, /*.ne01 =*/ ne01, /*.ne02 =*/ ne02, /*.ne03 =*/ ne03, @@ -560,7 +593,7 @@ int ggml_metal_op_concat(ggml_metal_op_t ctx, int idx) { /*.nb01 =*/ nb01, /*.nb02 =*/ nb02, /*.nb03 =*/ nb03, - /*.ne10 =*/ ne10, + /*.ne10 =*/ ne10_arg, /*.ne11 =*/ ne11, /*.ne12 =*/ ne12, /*.ne13 =*/ ne13, @@ -568,7 +601,7 @@ int ggml_metal_op_concat(ggml_metal_op_t ctx, int idx) { /*.nb11 =*/ nb11, /*.nb12 =*/ nb12, /*.nb13 =*/ nb13, - /*.ne0 =*/ ne0, + /*.ne0 =*/ ne0_arg, /*.ne1 =*/ ne1, /*.ne2 =*/ ne2, /*.ne3 =*/ ne3, @@ -587,7 +620,7 @@ int ggml_metal_op_concat(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 2); ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 3); - int nth = std::min(256, ne0); + int nth = std::min(256, ne0_arg); // when rows are small, we can batch them together in a single threadgroup int nrptg = 1; @@ -900,7 +933,7 @@ int ggml_metal_op_glu(ggml_metal_op_t ctx, int idx) { const int64_t nrows = ggml_nrows(op->src[0]); - const int32_t nth = std::min(ggml_metal_pipeline_max_theads_per_threadgroup(pipeline), ne00/2); + const int32_t nth = std::max(1, std::min(ggml_metal_pipeline_max_theads_per_threadgroup(pipeline), ne00/2)); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); @@ -947,7 +980,7 @@ int ggml_metal_op_sum(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1); ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 2); - ggml_metal_encoder_set_threadgroup_memory_size(enc, nsg * sizeof(float), 0); + ggml_metal_encoder_set_threadgroup_memory_size(enc, GGML_PAD(nsg * sizeof(float), 16), 0); ggml_metal_encoder_dispatch_threadgroups(enc, 1, 1, 1, nth, 1, 1); @@ -1381,7 +1414,7 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); ggml_metal_encoder_t enc = ctx->enc; - auto pipeline = ggml_metal_library_get_pipeline_dsv4_hc(ctx->lib, op->op); + auto pipeline = ggml_metal_library_get_pipeline_dsv4_hc(ctx->lib, op); ggml_metal_encoder_set_pipeline(enc, pipeline); @@ -1433,7 +1466,6 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(weights->type == GGML_TYPE_F32); GGML_ASSERT(op->type == GGML_TYPE_F32); - GGML_ASSERT(x->ne[1] == 4); ggml_metal_kargs_dsv4_hc_pre args = { /*.n_embd =*/ (int32_t) x->ne[0], @@ -1443,8 +1475,10 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { /*.nb_x2 =*/ x->nb[2], /*.nb_w0 =*/ weights->nb[0], /*.nb_w1 =*/ weights->nb[1], + /*.nb_w2 =*/ weights->nb[2], /*.nb_d0 =*/ op->nb[0], /*.nb_d1 =*/ op->nb[1], + /*.scale =*/ ggml_get_op_params_f32(op, 0), }; ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); @@ -1467,7 +1501,6 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(residual->type == GGML_TYPE_F32); GGML_ASSERT(post->type == GGML_TYPE_F32); - GGML_ASSERT(comb->type == GGML_TYPE_F32); GGML_ASSERT(op->type == GGML_TYPE_F32); GGML_ASSERT(residual->ne[1] == 4); @@ -1481,9 +1514,9 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { /*.nb_r2 =*/ residual->nb[2], /*.nb_p0 =*/ post->nb[0], /*.nb_p1 =*/ post->nb[1], - /*.nb_c0 =*/ comb->nb[0], - /*.nb_c1 =*/ comb->nb[1], - /*.nb_c2 =*/ comb->nb[2], + /*.nb_c0 =*/ comb ? comb->nb[0] : 0, + /*.nb_c1 =*/ comb ? comb->nb[1] : 0, + /*.nb_c2 =*/ comb ? comb->nb[2] : 0, /*.nb_d0 =*/ op->nb[0], /*.nb_d1 =*/ op->nb[1], /*.nb_d2 =*/ op->nb[2], @@ -1493,8 +1526,12 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(x), 1); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(residual), 2); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(post), 3); - ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(comb), 4); - ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op), 5); + if (comb) { + ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(comb), 4); + ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op), 5); + } else { + ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op), 4); + } const int n_tiles = (args.n_embd + 31)/32; const int nsg = std::min(4, n_tiles); @@ -1511,6 +1548,14 @@ int ggml_metal_op_dsv4_hc(ggml_metal_op_t ctx, int idx) { int ggml_metal_op_soft_max(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); + if (ctx->use_fusion()) { + int n = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n); + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_TOPK_MOE) { + return ggml_metal_op_topk_moe(ctx, idx); + } + } + ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; @@ -1611,6 +1656,20 @@ int ggml_metal_op_ssm_conv(ggml_metal_op_t ctx, int idx) { GGML_TENSOR_LOCALS( int32_t, ne, op, ne); GGML_TENSOR_LOCALS(uint64_t, nb, op, nb); + int n_fuse = 1; + bool use_silu = false; + + if (ctx->use_fusion()) { + int n = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n); + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_SSM_CONV_SILU) { + n_fuse = n; + use_silu = true; + + ctx->count_fusions(fusion); + } + } + ggml_metal_kargs_ssm_conv args = { /*.ne00 =*/ ne00, /*.ne01 =*/ ne01, @@ -1618,7 +1677,6 @@ int ggml_metal_op_ssm_conv(ggml_metal_op_t ctx, int idx) { /*.nb00 =*/ nb00, /*.nb01 =*/ nb01, /*.nb02 =*/ nb02, - /*.ne10 =*/ ne10, /*.ne11 =*/ ne11, /*.nb10 =*/ nb10, /*.nb11 =*/ nb11, @@ -1630,6 +1688,8 @@ int ggml_metal_op_ssm_conv(ggml_metal_op_t ctx, int idx) { /*.nb2 =*/ nb2, }; + const ggml_metal_buffer_id bid_dst = ggml_metal_get_buffer_id(n_fuse > 1 ? ctx->node(idx + n_fuse - 1) : op); + // Use batched kernel for prefill (ne1 > 1) to reduce threadgroup dispatch overhead const bool use_batched = (ne1 > 1); @@ -1644,31 +1704,35 @@ int ggml_metal_op_ssm_conv(ggml_metal_op_t ctx, int idx) { else if (ne1 > 4 ) BATCH_SIZE = 8; else BATCH_SIZE = 2; - auto pipeline = ggml_metal_library_get_pipeline_ssm_conv_batched(lib, op, BATCH_SIZE); + auto pipeline = ggml_metal_library_get_pipeline_ssm_conv_batched(lib, op, BATCH_SIZE, (int32_t) ne10, use_silu); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes(enc, &args, sizeof(args), 0); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 1); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[1]), 2); - ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op), 3); + ggml_metal_encoder_set_buffer(enc, bid_dst, 3); // Dispatch: ne01 rows, ceil(ne1/BATCH_SIZE) token batches, ne02 sequences // Each threadgroup has BATCH_SIZE threads, each handling one token const int n_token_batches = (ne1 + BATCH_SIZE - 1) / BATCH_SIZE; ggml_metal_encoder_dispatch_threadgroups(enc, ne01, n_token_batches, ne02, BATCH_SIZE, 1, 1); } else { - auto pipeline = ggml_metal_library_get_pipeline_ssm_conv(lib, op); + auto pipeline = ggml_metal_library_get_pipeline_ssm_conv(lib, op, (int32_t) ne10, use_silu); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes(enc, &args, sizeof(args), 0); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[0]), 1); ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op->src[1]), 2); - ggml_metal_encoder_set_buffer(enc, ggml_metal_get_buffer_id(op), 3); + ggml_metal_encoder_set_buffer(enc, bid_dst, 3); ggml_metal_encoder_dispatch_threadgroups(enc, ne01, ne1, ne02, 1, 1, 1); } - return 1; + if (n_fuse > 1 && ggml_metal_fusion_info_debug(ctx->finfo) > 1) { + GGML_LOG_DEBUG("%s: fuse: SSM_CONV + UNARY\n", __func__); + } + + return n_fuse; } int ggml_metal_op_ssm_scan(ggml_metal_op_t ctx, int idx) { @@ -1676,6 +1740,7 @@ int ggml_metal_op_ssm_scan(ggml_metal_op_t ctx, int idx) { ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; + const ggml_metal_device_props * props_dev = ggml_metal_device_get_props(ctx->dev); GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); GGML_TENSOR_LOCALS(uint64_t, nb0, op->src[0], nb); @@ -1721,6 +1786,8 @@ int ggml_metal_op_ssm_scan(ggml_metal_op_t ctx, int idx) { /*.n_head =*/ n_head, /*.n_group =*/ n_group, /*.n_seq_tokens =*/ n_seq_tokens, + /*.n_seq_tokens_total =*/ n_seq_tokens, + /*.token_offset =*/ 0, /*.n_seqs =*/ n_seqs, /*.K =*/ K, /*.s_off =*/ ggml_nelements(op->src[1]) * sizeof(float), @@ -1750,26 +1817,53 @@ int ggml_metal_op_ssm_scan(ggml_metal_op_t ctx, int idx) { /*.nb0 =*/ nb0, }; - auto pipeline = ggml_metal_library_get_pipeline_ssm_scan(lib, op); + constexpr int64_t CHUNK = OP_SSM_SCAN_SSD_CS; - GGML_ASSERT(d_state <= ggml_metal_pipeline_max_theads_per_threadgroup(pipeline)); + const int64_t snap_reserve = K > 1 ? K : 0; // tokens reserved for sequential kernel rollback snapshots + const int64_t mma_tokens = ((n_seq_tokens - snap_reserve) / CHUNK) * CHUNK; // largest multiple of CHUNK that leaves snap_reserve for the tail + const bool use_mma = + mma_tokens > 0 && + ne30 == 1 && // checks that A tensor is set to scalar decay per head (A shape {1, n_head}) + props_dev->has_simdgroup_mm && // hardware check for M1 or newer + d_state % 8 == 0 && // d_state must be multiple of 8 to align with simdgroup_float 8x8 tiles + d_inner == OP_SSM_SCAN_SSD_HD; // mma kernel is specialized for the Mamba-2 head dim; this checks it - const size_t smem = pipeline.smem; + const auto dispatch = [&](ggml_metal_pipeline_with_params pipeline, int64_t nth, int64_t n_tg_x) { + GGML_ASSERT(nth <= ggml_metal_pipeline_max_theads_per_threadgroup(pipeline)); + GGML_ASSERT(pipeline.smem <= props_dev->max_theadgroup_memory_size); - ggml_metal_encoder_set_pipeline(enc, pipeline); - ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 2); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 3); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[3]), 4); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[4]), 5); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[5]), 6); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[6]), 7); - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 8); + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), 2); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), 3); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[3]), 4); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[4]), 5); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[5]), 6); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[6]), 7); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 8); + ggml_metal_encoder_set_threadgroup_memory_size(enc, pipeline.smem, 0); + ggml_metal_encoder_dispatch_threadgroups(enc, n_tg_x, n_head, n_seqs, nth, 1, 1); + }; - ggml_metal_encoder_set_threadgroup_memory_size(enc, smem, 0); + if (!use_mma) { + dispatch(ggml_metal_library_get_pipeline_ssm_scan(lib, op, false), d_state, d_inner); + return 1; + } - ggml_metal_encoder_dispatch_threadgroups(enc, d_inner, n_head, n_seqs, d_state, 1, 1); + args.n_seq_tokens = mma_tokens; + dispatch( + ggml_metal_library_get_pipeline_ssm_scan_ssd_mma(lib, op), + OP_SSM_SCAN_SSD_NSG*32, + 1); + + if (mma_tokens < n_seq_tokens) { + ggml_metal_op_concurrency_reset(ctx); + + args.n_seq_tokens = n_seq_tokens - mma_tokens; + args.token_offset = mma_tokens; + dispatch(ggml_metal_library_get_pipeline_ssm_scan(lib, op, true), d_state, d_inner); + } return 1; } @@ -1821,6 +1915,8 @@ int ggml_metal_op_gated_delta_net(ggml_metal_op_t ctx, int idx) { ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; + const bool use_fusion = ctx->use_fusion(); + const int debug_fusion = ggml_metal_fusion_info_debug(ctx->finfo); GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); GGML_TENSOR_LOCALS(uint64_t, nb0, op->src[0], nb); @@ -1833,6 +1929,31 @@ int ggml_metal_op_gated_delta_net(ggml_metal_op_t ctx, int idx) { auto pipeline = ggml_metal_library_get_pipeline_gated_delta_net(lib, op); + // when fused with the trailing cache cpy, the snapshots are written straight into the + // recurrent cache and the cpy is skipped (see GGML_METAL_FUSION_GDN_CACHE) + ggml_metal_buffer_id bid_out = ggml_metal_get_buffer_id(op); + uint64_t nb_out = 0; + int n_fuse = 1; + + if (use_fusion) { + int n = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n); + + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_GDN_CACHE) { + const ggml_tensor * dst_cache = ctx->node(idx + 1)->src[1]; // cache view + + bid_out = ggml_metal_get_buffer_id(dst_cache); + nb_out = dst_cache->nb[2]/sizeof(float); + n_fuse = 2; + + ctx->count_fusions(fusion); + + if (debug_fusion > 1) { + GGML_LOG_DEBUG("%s: fuse: GATED_DELTA_NET + CPY\n", __func__); + } + } + } + int ida = 0; ggml_metal_kargs_gated_delta_net args = { @@ -1871,23 +1992,25 @@ int ggml_metal_op_gated_delta_net(ggml_metal_op_t ctx, int idx) { /*.nb1 =*/ nb1, /*.nb2 =*/ nb2, /*.nb3 =*/ nb3, + /*.nb_out =*/ nb_out, }; ggml_metal_encoder_set_pipeline(enc, pipeline); - ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), ida++); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), ida++); // args ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), ida++); // q ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[1]), ida++); // k ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[2]), ida++); // v ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[3]), ida++); // gate ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[4]), ida++); // beta ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[5]), ida++); // state - ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), ida++); // dst + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), ida++); // dst (attn) + ggml_metal_encoder_set_buffer (enc, bid_out, ida++); // state_out const int nsg = pipeline.nsg; ggml_metal_encoder_dispatch_threadgroups(enc, op->src[2]->ne[0]/nsg, op->src[2]->ne[1], op->src[2]->ne[3], 32, nsg, 1); - return 1; + return n_fuse; } int ggml_metal_op_solve_tri(ggml_metal_op_t ctx, int idx) { @@ -2195,12 +2318,6 @@ int ggml_metal_op_pool_1d(ggml_metal_op_t ctx, int idx) { return 1; } -// supported FWHT sizes, must stay in sync with the -// kernel_fwht_f32_ templates in ggml-metal.metal -static bool ggml_metal_fwht_supported_size(int64_t n) { - return n == 64 || n == 128 || n == 256 || n == 512; -} - int ggml_metal_op_fwht(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); @@ -2216,7 +2333,7 @@ int ggml_metal_op_fwht(ggml_metal_op_t ctx, int idx) { /*.nrows = */ (int32_t) nrows, }; - auto pipeline = ggml_metal_library_get_pipeline_fwht(lib, n); + auto pipeline = ggml_metal_library_get_pipeline_fwht(lib, n, src1->type); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes(enc, &args, sizeof(args), 0); @@ -2302,17 +2419,8 @@ int ggml_metal_op_mul_mat(ggml_metal_op_t ctx, int idx) { ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; - const int32_t hint = ggml_get_op_params_i32(op, 1); - - if (hint == GGML_HINT_SRC0_IS_HADAMARD) { - if (op->src[1]->type == GGML_TYPE_F32 && - op->type == GGML_TYPE_F32 && - ggml_is_contiguous(op->src[1]) && - ggml_is_contiguous(op) && - ggml_are_same_shape(op->src[1], op) && - ggml_metal_fwht_supported_size(op->src[1]->ne[0])) { - return ggml_metal_op_fwht(ctx, idx); - } + if (ggml_metal_op_mul_mat_use_fwht(op)) { + return ggml_metal_op_fwht(ctx, idx); } const ggml_metal_device_props * props_dev = ggml_metal_device_get_props(ctx->dev); @@ -2331,10 +2439,6 @@ int ggml_metal_op_mul_mat(ggml_metal_op_t ctx, int idx) { const int16_t r2 = ne12/ne02; const int16_t r3 = ne13/ne03; - // find the break-even point where the matrix-matrix kernel becomes more efficient compared - // to the matrix-vector kernel - const int ne11_mm_min = 8; - // first try to use small-batch mat-mv kernels // these should be efficient for BS [2, ~8] if (op->src[1]->type == GGML_TYPE_F32 && (ne00%128 == 0) && @@ -2437,12 +2541,7 @@ int ggml_metal_op_mul_mat(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 3); ggml_metal_encoder_dispatch_threadgroups(enc, ((ne01 + r0ptg - 1)/r0ptg), ((ne11 + r1ptg - 1)/r1ptg), ne12*ne13, 32, nsg, 1); - } else if ( - !ggml_is_transposed(op->src[0]) && - !ggml_is_transposed(op->src[1]) && - // for now the matrix-matrix multiplication kernel only works on A14+/M1+ SoCs - // AMD GPU and older A-chips will reuse matrix-vector multiplication kernel - props_dev->has_simdgroup_mm && ne00 >= 64 && ne11 > ne11_mm_min) { + } else if (ggml_metal_op_mul_mat_use_mm(op, props_dev->has_simdgroup_mm)) { //GGML_LOG_INFO("matrix: ne00 = %6d, ne01 = %6d, ne02 = %6d, ne11 = %6d, ne12 = %6d\n", ne00, ne01, ne02, ne11, ne12); // some Metal matrix data types require aligned pointers @@ -2557,6 +2656,15 @@ size_t ggml_metal_op_mul_mat_id_extra_ids(const ggml_tensor * op) { return ggml_type_size(GGML_TYPE_I32)*ne02*ne21; } +size_t ggml_metal_op_mul_mat_id_extra_amax(const ggml_tensor * op) { + assert(op->op == GGML_OP_MUL_MAT_ID); + + GGML_UNUSED(op); + + // 2 scaling factors (8 bytes) + N_MM_NPART_AMAX per-threadgroup scales for stage-1 + return 8 + N_MM_NPART_AMAX*sizeof(float); +} + int ggml_metal_op_mul_mat_id(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); @@ -2591,13 +2699,7 @@ int ggml_metal_op_mul_mat_id(ggml_metal_op_t ctx, int idx) { const uint32_t r2 = 1; const uint32_t r3 = 1; - // find the break-even point where the matrix-matrix kernel becomes more efficient compared - // to the matrix-vector kernel - // ne20 = n_used_experts - // ne21 = n_rows (batch size) - const int ne21_mm_id_min = 32; - - if (props_dev->has_simdgroup_mm && ne00 >= 64 && (ne21 >= ne21_mm_id_min)) { + if (ggml_metal_op_mul_mat_id_use_mm(op, props_dev->has_simdgroup_mm)) { // some Metal matrix data types require aligned pointers // ref: https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf (Table 2.5) //switch (op->src[0]->type) { @@ -2614,6 +2716,39 @@ int ggml_metal_op_mul_mat_id(ggml_metal_op_t ctx, int idx) { ggml_metal_buffer_id bid_ids = bid_tpe; bid_ids.offs += ggml_metal_op_mul_mat_id_extra_tpe(op); + ggml_metal_buffer_id bid_amax = bid_ids; + bid_amax.offs += ggml_metal_op_mul_mat_id_extra_ids(op); + + // src1 prec [TAG_GGML_PREC] + const bool use_amax = ggml_get_op_params_i32(op, 3) == GGML_PREC_F32; + + // src1 rescale factors, computed before the matmul + // ref: https://github.com/ggml-org/llama.cpp/pull/26223 + if (use_amax) { + ggml_metal_kargs_mul_mm_id_amax args = { + /*.ne00 =*/ ne10, + /*.ne01 =*/ ne11, + /*.ne02 =*/ ne12, + /*.nb01 =*/ nb11, + /*.nb02 =*/ nb12, + }; + + auto pipeline = ggml_metal_library_get_pipeline_mul_mm_id_amax_part(lib); + + const size_t smem = pipeline.smem; + + GGML_ASSERT(smem <= props_dev->max_theadgroup_memory_size); + + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); + ggml_metal_encoder_set_buffer (enc, bid_src1, 1); + ggml_metal_encoder_set_buffer (enc, bid_amax, 2); + + ggml_metal_encoder_set_threadgroup_memory_size(enc, smem, 0); + + ggml_metal_encoder_dispatch_threadgroups(enc, N_MM_NPART_AMAX, 1, 1, 256, 1, 1); + } + { ggml_metal_kargs_mul_mm_id_map0 args = { ne02, @@ -2645,9 +2780,20 @@ int ggml_metal_op_mul_mat_id(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_dispatch_threadgroups(enc, 1, 1, 1, ne02, 1, 1); } - // this barrier is always needed because the next kernel has to wait for the id maps to be computed ggml_metal_op_concurrency_reset(ctx); + if (use_amax) { + auto pipeline = ggml_metal_library_get_pipeline_mul_mm_id_amax(lib); + + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_buffer (enc, bid_amax, 0); + + ggml_metal_encoder_dispatch_threadgroups(enc, 1, 1, 1, 32, 1, 1); + + // the next kernel has to wait for the amax data + ggml_metal_op_concurrency_reset(ctx); + } + { auto pipeline = ggml_metal_library_get_pipeline_mul_mm_id(lib, op); @@ -2677,6 +2823,7 @@ int ggml_metal_op_mul_mat_id(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_buffer (enc, bid_tpe, 3); ggml_metal_encoder_set_buffer (enc, bid_ids, 4); ggml_metal_encoder_set_buffer (enc, bid_dst, 5); + ggml_metal_encoder_set_buffer (enc, bid_amax, 6); const size_t smem = pipeline.smem; @@ -2801,6 +2948,111 @@ bool ggml_metal_op_flash_attn_ext_use_vec(const ggml_tensor * op) { return (ne01 < 20) && (ne00 % 32 == 0); } +// ref: https://github.com/ggml-org/llama.cpp/pull/27390 +// dequantize the quantized KV cache to F16 before running the F16 flash attention kernels +static bool ggml_metal_op_flash_attn_ext_use_kv_f16(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + // depending on compute/bandwidth ratio, dequant to f16 kv is not always beneficial + // ref: https://github.com/ggml-org/llama.cpp/pull/27390#issuecomment-5355152767 + // TODO: tune per device + if (op->src[0]->ne[1] < 32) { + return false; + } + + switch (op->src[1]->type) { + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + return true; + default: + return false; + } +} + +// returns the n_kv_max hint if the sparse path is available for this op, or 0 otherwise +// the mask (src[3]) remains the single source of truth: finite entries are the valid KV positions, +// n_kv_max is only an upper bound on their number per mask row, used to size the index lists +static int ggml_metal_op_flash_attn_ext_n_kv_max_sparse(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + int32_t n_kv_max = 0; + memcpy(&n_kv_max, ((const int32_t *) op->op_params) + 4, sizeof(n_kv_max)); + + if (n_kv_max <= 0) { + return 0; + } + + // the sparse indices are gathered from the mask + if (!op->src[3]) { + return 0; + } + + // bound the size of the index lists + if (n_kv_max > 4096) { + return 0; + } + + // vec kernel instantiations exist for these (type, dk, dv) combinations only + const int64_t dk = op->src[1]->ne[0]; + const int64_t dv = op->src[2]->ne[0]; + + const bool dk_dv_ok = (dk == 32 && dv == 32) || + (dk == 64 && dv == 64) || + (dk == 96 && dv == 96) || + (dk == 96 && dv == 64) || + (dk == 128 && dv == 128) || + (dk == 192 && dv == 128) || + (dk == 192 && dv == 192) || + (dk == 256 && dv == 256) || + (dk == 320 && dv == 256) || + (dk == 512 && dv == 512) || + (dk == 576 && dv == 512); + + if (!dk_dv_ok) { + return 0; + } + + switch (op->src[1]->type) { + case GGML_TYPE_F16: + case GGML_TYPE_BF16: + case GGML_TYPE_F32: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + break; + default: + return 0; + } + + return n_kv_max; +} + +// in some models (e.g. MLA-based), V is a view of K (the first ne20 elements of each K row); +// the dequantized V is then a view of the dequantized K and does not need its own dequant or scratch +// - ref: https://github.com/ggml-org/llama.cpp/pull/13435 +static bool ggml_metal_op_flash_attn_ext_v_is_view_of_k(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + const ggml_tensor * K = op->src[1]; + const ggml_tensor * V = op->src[2]; + + return V->view_src && (V->view_src == K || (V->view_src == K->view_src && V->view_offs == K->view_offs)); +} + +// size of the F16 dequantized K tensor; the dequantized V tensor follows it in the same scratch buffer +static size_t ggml_metal_op_flash_attn_ext_kv_f16_k_size(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + GGML_TENSOR_LOCALS( int32_t, ne1, op->src[1], ne); + + return GGML_PAD(sizeof(ggml_fp16_t)*(size_t) ne10*ne11*ne12*ne13, 16); +} + size_t ggml_metal_op_flash_attn_ext_extra_pad(const ggml_tensor * op) { assert(op->op == GGML_OP_FLASH_ATTN_EXT); @@ -2816,6 +3068,18 @@ size_t ggml_metal_op_flash_attn_ext_extra_pad(const ggml_tensor * op) { size_t res = 0; const bool has_mask = op->src[3] != nullptr; + const bool use_kv_f16 = ggml_metal_op_flash_attn_ext_use_kv_f16(op); + + // when the KV is dequantized to F16, the pad kernel copies the tail chunk from the F16 scratch buffer + // note: when V is a view of K, the dequantized V is read from the dequantized K with K's row stride + const bool v_is_view_of_k = use_kv_f16 && ggml_metal_op_flash_attn_ext_v_is_view_of_k(op); + uint64_t nb11_pad = nb11; + uint64_t nb21_pad = nb21; + + if (use_kv_f16) { + nb11_pad = sizeof(ggml_fp16_t)*ne10; + nb21_pad = sizeof(ggml_fp16_t)*(v_is_view_of_k ? ne10 : ne20); + } // note: the non-vec kernel requires more extra memory, so always reserve for it GGML_ASSERT(OP_FLASH_ATTN_EXT_NCPSG >= OP_FLASH_ATTN_EXT_VEC_NCPSG); @@ -2828,8 +3092,8 @@ size_t ggml_metal_op_flash_attn_ext_extra_pad(const ggml_tensor * op) { if (has_kvpad) { res += OP_FLASH_ATTN_EXT_VEC_NCPSG*( - nb11*ne12*ne13 + - nb21*ne22*ne23 + + nb11_pad*ne12*ne13 + + nb21_pad*ne22*ne23 + (has_mask ? ggml_type_size(GGML_TYPE_F16)*ne31*ne32*ne33 : 0)); } } else { @@ -2838,8 +3102,8 @@ size_t ggml_metal_op_flash_attn_ext_extra_pad(const ggml_tensor * op) { if (has_kvpad) { res += OP_FLASH_ATTN_EXT_NCPSG*( - nb11*ne12*ne13 + - nb21*ne22*ne23 + + nb11_pad*ne12*ne13 + + nb21_pad*ne22*ne23 + (has_mask ? ggml_type_size(GGML_TYPE_F16)*ne31*ne32*ne33 : 0)); } } @@ -2915,6 +3179,47 @@ size_t ggml_metal_op_flash_attn_ext_extra_tmp(const ggml_tensor * op) { return res; } +size_t ggml_metal_op_flash_attn_ext_extra_kv_f16(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + // note: always reserve the temp buffer to avoid graph reallocations + //if (!ggml_metal_op_flash_attn_ext_use_kv_f16(op)) { + // return 0; + //} + + GGML_TENSOR_LOCALS( int32_t, ne2, op->src[2], ne); + + const size_t k_size = ggml_metal_op_flash_attn_ext_kv_f16_k_size(op); + + // when V is a view of K, the dequantized V is a view of the dequantized K + const bool v_is_view_of_k = ggml_metal_op_flash_attn_ext_v_is_view_of_k(op); + if (v_is_view_of_k) { + return k_size; + } + + const size_t v_size = GGML_PAD(sizeof(ggml_fp16_t)*(size_t) ne20*ne21*ne22*ne23, 16); + + return k_size + v_size; +} + +// size of the sparse index lists: one list of KV indices per mask row, +// padded with -1 up to a multiple of OP_FLASH_ATTN_EXT_VEC_NCPSG +size_t ggml_metal_op_flash_attn_ext_extra_idx(const ggml_tensor * op) { + assert(op->op == GGML_OP_FLASH_ATTN_EXT); + + GGML_TENSOR_LOCALS( int32_t, ne3, op->src[3], ne); + + const int n_kv_max = ggml_metal_op_flash_attn_ext_n_kv_max_sparse(op); + + if (n_kv_max <= 0) { + return 0; + } + + const int n_kv_max_padded = GGML_PAD(n_kv_max, OP_FLASH_ATTN_EXT_VEC_NCPSG); + + return GGML_PAD(sizeof(int32_t)*(size_t) n_kv_max_padded*ne31*ne32*ne33, 16); +} + int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); @@ -2989,7 +3294,121 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { ggml_metal_buffer_id bid_tmp = bid_blk; bid_tmp.offs += ggml_metal_op_flash_attn_ext_extra_blk(op); - if (!ggml_metal_op_flash_attn_ext_use_vec(op)) { + ggml_metal_buffer_id bid_kv_f16 = bid_tmp; + bid_kv_f16.offs += ggml_metal_op_flash_attn_ext_extra_tmp(op); + + // sparse path: gather the finite mask entries into index lists and run the vec kernels over them + const int n_kv_max_sparse = ggml_metal_op_flash_attn_ext_n_kv_max_sparse(op); + const bool use_sparse = n_kv_max_sparse > 0; + const int n_kv_max_padded = use_sparse ? GGML_PAD(n_kv_max_sparse, OP_FLASH_ATTN_EXT_VEC_NCPSG) : 0; + + // the vec kernels dequantize the KV inline; no need for the F16 dequant pass in the sparse path + const bool use_kv_f16 = !use_sparse && ggml_metal_op_flash_attn_ext_use_kv_f16(op); + + ggml_metal_buffer_id bid_idx = bid_kv_f16; + bid_idx.offs += ggml_metal_op_flash_attn_ext_extra_kv_f16(op); + + ggml_metal_buffer_id bid_k = bid_src1; + ggml_metal_buffer_id bid_v = bid_src2; + + uint64_t nb10_attn = nb10; + uint64_t nb11_attn = nb11; + uint64_t nb12_attn = nb12; + uint64_t nb13_attn = nb13; + uint64_t nb20_attn = nb20; + uint64_t nb21_attn = nb21; + uint64_t nb22_attn = nb22; + uint64_t nb23_attn = nb23; + + if (use_kv_f16) { + assert(ggml_metal_op_flash_attn_ext_extra_kv_f16(op) != 0); + + const bool v_is_view_of_k = ggml_metal_op_flash_attn_ext_v_is_view_of_k(op); + + const int64_t nblocks1_64 = (ne10/ggml_blck_size(op->src[1]->type))*(int64_t) ne11*ne12*ne13; + GGML_ASSERT(nblocks1_64 <= INT32_MAX); + const int32_t nblocks1 = nblocks1_64; + + ggml_metal_buffer_id bid_v_f16 = bid_kv_f16; + bid_v_f16.offs += ggml_metal_op_flash_attn_ext_kv_f16_k_size(op); + + auto pipeline0 = ggml_metal_library_get_pipeline_flash_attn_ext_kv_f16(lib, op); + const int nth = std::min(ggml_metal_pipeline_max_theads_per_threadgroup(pipeline0), 256); + + // K + ggml_metal_kargs_flash_attn_ext_kv_f16 args_k = { + /*.ne0 =*/ ne10, + /*.ne1 =*/ ne11, + /*.ne2 =*/ ne12, + /*.ne3 =*/ ne13, + /*.nb0 =*/ nb10, + /*.nb1 =*/ nb11, + /*.nb2 =*/ nb12, + /*.nb3 =*/ nb13, + /*.nblocks =*/ nblocks1, + }; + + ggml_metal_encoder_set_pipeline(enc, pipeline0); + ggml_metal_encoder_set_bytes (enc, &args_k, sizeof(args_k), 0); + ggml_metal_encoder_set_buffer (enc, bid_src1, 1); + ggml_metal_encoder_set_buffer (enc, bid_kv_f16, 2); + + ggml_metal_encoder_dispatch_threadgroups(enc, (nblocks1 + nth - 1)/nth, 1, 1, nth, 1, 1); + + // V (skip when V is a view of K: the dequantized V is a view of the dequantized K) + if (!v_is_view_of_k) { + const int64_t nblocks2_64 = (ne20/ggml_blck_size(op->src[2]->type))*(int64_t) ne21*ne22*ne23; + GGML_ASSERT(nblocks2_64 <= INT32_MAX); + const int32_t nblocks2 = nblocks2_64; + + ggml_metal_kargs_flash_attn_ext_kv_f16 args_v = { + /*.ne0 =*/ ne20, + /*.ne1 =*/ ne21, + /*.ne2 =*/ ne22, + /*.ne3 =*/ ne23, + /*.nb0 =*/ nb20, + /*.nb1 =*/ nb21, + /*.nb2 =*/ nb22, + /*.nb3 =*/ nb23, + /*.nblocks =*/ nblocks2, + }; + + ggml_metal_encoder_set_pipeline(enc, pipeline0); + ggml_metal_encoder_set_bytes (enc, &args_v, sizeof(args_v), 0); + ggml_metal_encoder_set_buffer (enc, bid_src2, 1); + ggml_metal_encoder_set_buffer (enc, bid_v_f16, 2); + + ggml_metal_encoder_dispatch_threadgroups(enc, (nblocks2 + nth - 1)/nth, 1, 1, nth, 1, 1); + } + + // the pad and attention kernels read the dequantized KV + ggml_metal_op_concurrency_reset(ctx); + + bid_k = bid_kv_f16; + bid_v = v_is_view_of_k ? bid_k : bid_v_f16; + + // contiguous F16 layout of the dequantized K + nb10_attn = sizeof(ggml_fp16_t); + nb11_attn = nb10_attn*ne10; + nb12_attn = nb11_attn*ne11; + nb13_attn = nb12_attn*ne12; + + // if V is a view of K, the dequantized V is read from the dequantized K with K's strides + if (v_is_view_of_k) { + nb20_attn = nb10_attn; + nb21_attn = nb11_attn; + nb22_attn = nb12_attn; + nb23_attn = nb13_attn; + } else { + // contiguous F16 layout of the dequantized V + nb20_attn = sizeof(ggml_fp16_t); + nb21_attn = nb20_attn*ne20; + nb22_attn = nb21_attn*ne21; + nb23_attn = nb22_attn*ne22; + } + } + + if (!use_sparse && !ggml_metal_op_flash_attn_ext_use_vec(op)) { // half8x8 kernel const int nqptg = OP_FLASH_ATTN_EXT_NQPSG; // queries per threadgroup const int ncpsg = OP_FLASH_ATTN_EXT_NCPSG; // cache values per simdgroup @@ -3009,12 +3428,12 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.ne11 =*/ne11, /*.ne_12_2 =*/ne12, /*.ne_12_3 =*/ne13, - /*.nb11 =*/nb11, - /*.nb12 =*/nb12, - /*.nb13 =*/nb13, - /*.nb21 =*/nb21, - /*.nb22 =*/nb22, - /*.nb23 =*/nb23, + /*.nb11 =*/nb11_attn, + /*.nb12 =*/nb12_attn, + /*.nb13 =*/nb13_attn, + /*.nb21 =*/nb21_attn, + /*.nb22 =*/nb22_attn, + /*.nb23 =*/nb23_attn, /*.ne31 =*/ne31, /*.ne32 =*/ne32, /*.ne33 =*/ne33, @@ -3027,8 +3446,8 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_pipeline(enc, pipeline0); ggml_metal_encoder_set_bytes (enc, &args0, sizeof(args0), 0); - ggml_metal_encoder_set_buffer (enc, bid_src1, 1); - ggml_metal_encoder_set_buffer (enc, bid_src2, 2); + ggml_metal_encoder_set_buffer (enc, bid_k, 1); + ggml_metal_encoder_set_buffer (enc, bid_v, 2); ggml_metal_encoder_set_buffer (enc, bid_src3, 3); ggml_metal_encoder_set_buffer (enc, bid_pad, 4); @@ -3073,7 +3492,7 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { ggml_metal_op_concurrency_reset(ctx); } - const int is_q = ggml_is_quantized(op->src[1]->type) ? 1 : 0; + const int is_q = !use_kv_f16 && ggml_is_quantized(op->src[1]->type) ? 1 : 0; // 2*(2*ncpsg) // ncpsg soft_max values + ncpsg mask values @@ -3104,6 +3523,9 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { const size_t smem = FATTN_SMEM(nsg); + const int32_t ns10 = nb11_attn/nb10_attn; + const int32_t ns20 = nb21_attn/nb20_attn; + ggml_metal_kargs_flash_attn_ext args = { /*.ne01 =*/ ne01, /*.ne02 =*/ ne02, @@ -3114,14 +3536,14 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.ne11 =*/ ne11, /*.ne_12_2 =*/ ne12, /*.ne_12_3 =*/ ne13, - /*.ns10 =*/ int32_t(nb11/nb10), - /*.nb11 =*/ nb11, - /*.nb12 =*/ nb12, - /*.nb13 =*/ nb13, - /*.ns20 =*/ int32_t(nb21/nb20), - /*.nb21 =*/ nb21, - /*.nb22 =*/ nb22, - /*.nb23 =*/ nb23, + /*.ns10 =*/ ns10, + /*.nb11 =*/ nb11_attn, + /*.nb12 =*/ nb12_attn, + /*.nb13 =*/ nb13_attn, + /*.ns20 =*/ ns20, + /*.nb21 =*/ nb21_attn, + /*.nb22 =*/ nb22_attn, + /*.nb23 =*/ nb23_attn, /*.ne31 =*/ ne31, /*.ne32 =*/ ne32, /*.ne33 =*/ ne33, @@ -3139,13 +3561,13 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.logit_softcap =*/ logit_softcap, }; - auto pipeline = ggml_metal_library_get_pipeline_flash_attn_ext(lib, op, has_mask, has_sinks, has_bias, has_scap, has_kvpad, nsg); + auto pipeline = ggml_metal_library_get_pipeline_flash_attn_ext(lib, op, has_mask, has_sinks, has_bias, has_scap, has_kvpad, nsg, use_kv_f16, ns10, ns20); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); ggml_metal_encoder_set_buffer (enc, bid_src0, 1); - ggml_metal_encoder_set_buffer (enc, bid_src1, 2); - ggml_metal_encoder_set_buffer (enc, bid_src2, 3); + ggml_metal_encoder_set_buffer (enc, bid_k, 2); + ggml_metal_encoder_set_buffer (enc, bid_v, 3); ggml_metal_encoder_set_buffer (enc, bid_src3, 4); ggml_metal_encoder_set_buffer (enc, bid_src4, 5); ggml_metal_encoder_set_buffer (enc, bid_pad, 6); @@ -3158,17 +3580,59 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { #undef FATTN_SMEM } else { // half4x4 kernel - const int nqptg = OP_FLASH_ATTN_EXT_VEC_NQPSG; // queries per threadgroup + // sparse: the index lists are per query row, so a threadgroup can share KV with Q == 1 only + auto cfg = use_sparse + ? ggml_metal_tuning::fa_vec_baseline_cfg((int) ne00, (int) ne20) + : ggml_metal_tuning::fa_vec_pick( + props_dev->gpu_family, + (int) op->src[1]->type, + (int) ne00, (int) ne20, // dk, dv (ne00 == dk for FA) + ne11, ne01); + + int nqptg = cfg.Q; // queries per threadgroup + const int ncpsg = OP_FLASH_ATTN_EXT_VEC_NCPSG; // cache values per simdgroup !! sync with kernel template arguments !! const int nhptg = 1; // heads per threadgroup GGML_ASSERT(nqptg <= 32); - GGML_ASSERT(nqptg % 1 == 0); + GGML_ASSERT(nqptg == 1 || nqptg == 2 || nqptg == 4); // only instantiated Q values GGML_ASSERT(ncpsg % 32 == 0); bool need_sync = false; - const bool has_kvpad = ne11 % ncpsg != 0; + const bool has_kvpad = !use_sparse && ne11 % ncpsg != 0; + + if (use_sparse) { + assert(ggml_metal_op_flash_attn_ext_extra_idx(op) != 0); + + GGML_ASSERT(ne30 == ne11); + + ggml_metal_kargs_flash_attn_ext_vec_idx args0 = { + /*.ne30 =*/ ne30, + /*.ne31 =*/ ne31, + /*.ne32 =*/ ne32, + /*.ne33 =*/ ne33, + /*.nb31 =*/ nb31, + /*.nb32 =*/ nb32, + /*.nb33 =*/ nb33, + /*.n_kv_max =*/ n_kv_max_sparse, + /*.n_kv_max_padded =*/ n_kv_max_padded, + }; + + auto pipeline0 = ggml_metal_library_get_pipeline_flash_attn_ext_vec_idx(lib, op); + + ggml_metal_encoder_set_pipeline(enc, pipeline0); + ggml_metal_encoder_set_bytes (enc, &args0, sizeof(args0), 0); + ggml_metal_encoder_set_buffer (enc, bid_src3, 1); + ggml_metal_encoder_set_buffer (enc, bid_idx, 2); + + int nth = std::min(ggml_metal_pipeline_max_theads_per_threadgroup(pipeline0), 256); + nth = std::max(32, (nth/32)*32); + + ggml_metal_encoder_dispatch_threadgroups(enc, ne31, ne32, ne33, nth, 1, 1); + + need_sync = true; + } if (has_kvpad) { assert(ggml_metal_op_flash_attn_ext_extra_pad(op) != 0); @@ -3177,12 +3641,12 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.ne11 =*/ne11, /*.ne_12_2 =*/ne12, /*.ne_12_3 =*/ne13, - /*.nb11 =*/nb11, - /*.nb12 =*/nb12, - /*.nb13 =*/nb13, - /*.nb21 =*/nb21, - /*.nb22 =*/nb22, - /*.nb23 =*/nb23, + /*.nb11 =*/nb11_attn, + /*.nb12 =*/nb12_attn, + /*.nb13 =*/nb13_attn, + /*.nb21 =*/nb21_attn, + /*.nb22 =*/nb22_attn, + /*.nb23 =*/nb23_attn, /*.ne31 =*/ne31, /*.ne32 =*/ne32, /*.ne33 =*/ne33, @@ -3195,8 +3659,8 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { ggml_metal_encoder_set_pipeline(enc, pipeline0); ggml_metal_encoder_set_bytes (enc, &args0, sizeof(args0), 0); - ggml_metal_encoder_set_buffer (enc, bid_src1, 1); - ggml_metal_encoder_set_buffer (enc, bid_src2, 2); + ggml_metal_encoder_set_buffer (enc, bid_k, 1); + ggml_metal_encoder_set_buffer (enc, bid_v, 2); ggml_metal_encoder_set_buffer (enc, bid_src3, 3); ggml_metal_encoder_set_buffer (enc, bid_pad, 4); @@ -3222,18 +3686,33 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { // ne20*(nsg) // each simdgroup has a full f32 head vector in shared mem to accumulate results // -#define FATTN_SMEM(nsg) (GGML_PAD(((GGML_PAD(ne00, 128) + 4*ncpsg + 2*GGML_PAD(ne20, 128))*(nsg))*(sizeof(float)/2), 16)) +#define FATTN_SMEM(nsg) (GGML_PAD(((GGML_PAD(ne00, 128) + 4*ncpsg + 2*GGML_PAD(ne20, 128))*(nsg)*nqptg)*(sizeof(float)/2), 16)) int64_t nsg = 1; // workgroups // each workgroup handles nsg*nkpsg cache values int32_t nwg = 1; - if (false) { - // for small KV caches, we could launch a single workgroup and write the results directly to dst/ - // however, this does not lead to significant improvement, so disabled - nwg = 1; - nsg = 4; + if (use_sparse) { + if (ne01 > 32) { + // large sparse batch + nwg = 1; + nsg = 1; + if (n_kv_max_padded == 640) { + nsg = 4; // 640 % (4*32) == 0 + } else { + while (2*nwg*nsg*ncpsg < n_kv_max_padded && nsg < 4) { + nsg *= 2; + } + } + } else { + // small sparse batch + nwg = 32; + nsg = 1; + while (2*nwg*nsg*ncpsg < n_kv_max_padded && nsg < 4) { + nsg *= 2; + } + } } else { nwg = 32; nsg = 1; @@ -3242,6 +3721,15 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { } } + // fall back to baseline (Q=1) if the tuned config exceeds threadgroup memory + if ((size_t) FATTN_SMEM(nsg) > props_dev->max_theadgroup_memory_size) { + cfg = ggml_metal_tuning::fa_vec_baseline_cfg((int) ne00, (int) ne20); + nqptg = cfg.Q; // = 1 + } + + const int32_t ns10 = nb11_attn/nb10_attn; + const int32_t ns20 = nb21_attn/nb20_attn; + ggml_metal_kargs_flash_attn_ext_vec args = { /*.ne01 =*/ ne01, /*.ne02 =*/ ne02, @@ -3249,17 +3737,17 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.nb01 =*/ nb01, /*.nb02 =*/ nb02, /*.nb03 =*/ nb03, - /*.ne11 =*/ ne11, + /*.ne11 =*/ use_sparse ? n_kv_max_padded : ne11, /*.ne_12_2 =*/ ne12, /*.ne_12_3 =*/ ne13, - /*.ns10 =*/ int32_t(nb11/nb10), - /*.nb11 =*/ nb11, - /*.nb12 =*/ nb12, - /*.nb13 =*/ nb13, - /*.ns20 =*/ int32_t(nb21/nb20), - /*.nb21 =*/ nb21, - /*.nb22 =*/ nb22, - /*.nb23 =*/ nb23, + /*.ns10 =*/ ns10, + /*.nb11 =*/ nb11_attn, + /*.nb12 =*/ nb12_attn, + /*.nb13 =*/ nb13_attn, + /*.ns20 =*/ ns20, + /*.nb21 =*/ nb21_attn, + /*.nb22 =*/ nb22_attn, + /*.nb23 =*/ nb23_attn, /*.ne31 =*/ ne31, /*.ne32 =*/ ne32, /*.ne33 =*/ ne33, @@ -3275,19 +3763,21 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { /*.m1 =*/ m1, /*.n_head_log2 =*/ n_head_log2, /*.logit_softcap =*/ logit_softcap, + /*.n_kv_max_padded =*/ n_kv_max_padded, }; - auto pipeline = ggml_metal_library_get_pipeline_flash_attn_ext_vec(lib, op, has_mask, has_sinks, has_bias, has_scap, has_kvpad, nsg, nwg); + auto pipeline = ggml_metal_library_get_pipeline_flash_attn_ext_vec(lib, op, has_mask, has_sinks, has_bias, has_scap, has_kvpad, use_sparse, nqptg, cfg.NE, nsg, nwg, use_kv_f16, ns10, ns20); GGML_ASSERT(nsg*32 <= ggml_metal_pipeline_max_theads_per_threadgroup(pipeline)); ggml_metal_encoder_set_pipeline(enc, pipeline); ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); ggml_metal_encoder_set_buffer (enc, bid_src0, 1); - ggml_metal_encoder_set_buffer (enc, bid_src1, 2); - ggml_metal_encoder_set_buffer (enc, bid_src2, 3); + ggml_metal_encoder_set_buffer (enc, bid_k, 2); + ggml_metal_encoder_set_buffer (enc, bid_v, 3); ggml_metal_encoder_set_buffer (enc, bid_src3, 4); ggml_metal_encoder_set_buffer (enc, bid_src4, 5); + ggml_metal_encoder_set_buffer (enc, use_sparse ? bid_idx : bid_src0, 8); const size_t smem = FATTN_SMEM(nsg); @@ -3295,8 +3785,6 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { GGML_ASSERT(smem <= props_dev->max_theadgroup_memory_size); if (nwg == 1) { - assert(ggml_metal_op_flash_attn_ext_extra_tmp(op) == 0); - // using 1 workgroup -> write the result directly into dst ggml_metal_encoder_set_buffer(enc, bid_pad, 6); ggml_metal_encoder_set_buffer(enc, bid_dst, 7); @@ -3345,56 +3833,26 @@ int ggml_metal_op_flash_attn_ext(ggml_metal_op_t ctx, int idx) { return 1; } -// Snake activation autofuse: mul -> sin -> sqr -> mul -> add -static bool ggml_metal_op_can_fuse_snake(ggml_metal_op_t ctx, int idx) { - static constexpr ggml_op snake_ops[5] = { GGML_OP_MUL, GGML_OP_SIN, GGML_OP_SQR, GGML_OP_MUL, GGML_OP_ADD }; - - if (ctx->node(idx)->op != GGML_OP_MUL || !ctx->can_fuse(idx, snake_ops, 5)) { - return false; - } - - const ggml_tensor * mul0 = ctx->node(idx + 0); - const ggml_tensor * sin_node = ctx->node(idx + 1); - const ggml_tensor * sqr = ctx->node(idx + 2); - const ggml_tensor * mul1 = ctx->node(idx + 3); - const ggml_tensor * add = ctx->node(idx + 4); +int ggml_metal_op_bin(ggml_metal_op_t ctx, int idx) { + int n_fuse = 1; + const ggml_metal_fusion * fusion = nullptr; - // x carries the full activation shape, a is the broadcast operand - const ggml_tensor * x = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; - const ggml_tensor * a = (x == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; + if (ctx->use_fusion()) { + int n = 1; + fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n); + n_fuse = n; - // mul1 reads sqr and inv_b in either operand order - const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; - - // closure check: the trailing add reads the same x as the leading mul - const ggml_tensor * x_in_add = (add->src[0] == mul1) ? add->src[1] : add->src[0]; - - // x is in the supported whitelist and every chain intermediate shares x's type. - // a and inv_b bind as device const float * in the kernel, so they stay F32. - const bool types_ok = - (x->type == GGML_TYPE_F32 || x->type == GGML_TYPE_F16 || x->type == GGML_TYPE_BF16) && - (a->type == GGML_TYPE_F32) && (inv_b->type == GGML_TYPE_F32) && - (mul0->type == x->type) && (sin_node->type == x->type) && - (sqr->type == x->type) && (mul1->type == x->type) && - (add->type == x->type); - // a / inv_b collapse to [1, C, 1, 1], x and add stay 2D - const bool shape_ok = ggml_are_same_shape(a, inv_b) && a->ne[0] == 1 && a->ne[1] == x->ne[1]; - const bool dim_ok = - (x->ne[2] == 1) && (x->ne[3] == 1) && - (add->ne[2] == 1) && (add->ne[3] == 1) && - (a->ne[2] == 1) && (a->ne[3] == 1) && - (inv_b->ne[2] == 1) && (inv_b->ne[3] == 1); - // kernel reads x[idx] and a[c] / inv_b[c] linearly, so every operand is contiguous - const bool contig_ok = - ggml_is_contiguous(x) && ggml_is_contiguous(add) && - ggml_is_contiguous(a) && ggml_is_contiguous(inv_b); - - return types_ok && shape_ok && dim_ok && contig_ok && x_in_add == x; -} + // snake activation autofuse: mul -> sin -> sqr -> mul -> add + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_SNAKE) { + ctx->count_fusions(fusion); + return ggml_metal_op_snake_fused(ctx, idx); + } -int ggml_metal_op_bin(ggml_metal_op_t ctx, int idx) { - if (ctx->use_fusion && ggml_metal_op_can_fuse_snake(ctx, idx)) { - return ggml_metal_op_snake_fused(ctx, idx); + // MoE output reduction: experts * weights -> weighted sum + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_MOE_REDUCE) { + ctx->count_fusions(fusion); + return ggml_metal_op_moe_reduce(ctx, idx); + } } ggml_tensor * op = ctx->node(idx); @@ -3402,9 +3860,9 @@ int ggml_metal_op_bin(ggml_metal_op_t ctx, int idx) { ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; - const bool use_fusion = ctx->use_fusion; + const bool use_fusion = ctx->use_fusion(); - const int debug_fusion = ctx->debug_fusion; + const int debug_fusion = ggml_metal_fusion_info_debug(ctx->finfo); GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); GGML_TENSOR_LOCALS(uint64_t, nb0, op->src[0], nb); @@ -3449,57 +3907,19 @@ int ggml_metal_op_bin(ggml_metal_op_t ctx, int idx) { /*.o1 =*/ { bid_src1.offs }, }; - ggml_op fops[8]; - - int n_fuse = 1; - // c[0] = add(a, b[0]) // c[1] = add(c[0], b[1]) // c[2] = add(c[1], b[2]) // ... - if (use_fusion) { - fops[0] = GGML_OP_ADD; - fops[1] = GGML_OP_ADD; - fops[2] = GGML_OP_ADD; - fops[3] = GGML_OP_ADD; - fops[4] = GGML_OP_ADD; - fops[5] = GGML_OP_ADD; - fops[6] = GGML_OP_ADD; - fops[7] = GGML_OP_ADD; - - // note: in metal, we sometimes encode the graph in parallel so we have to avoid fusing ops - // across splits. idx_end indicates the last node in the current split - for (n_fuse = 0; n_fuse <= 6; ++n_fuse) { - if (!ctx->can_fuse(idx + n_fuse, fops + n_fuse, 2)) { - break; - } - - ggml_tensor * f0 = ctx->node(idx + n_fuse); - ggml_tensor * f1 = ctx->node(idx + n_fuse + 1); - - if (f0 != f1->src[0]) { - break; - } - - // b[0] === b[1] === ... - if (!ggml_are_same_layout(f0->src[1], f1->src[1])) { - break; - } - - // only fuse ops if src1 is in the same Metal buffer - ggml_metal_buffer_id bid_fuse = ggml_metal_get_buffer_id(f1->src[1]); - if (bid_fuse.metal != bid_src1.metal) { - break; - } - - //ctx->fuse_cnt[ops[n_fuse + 1]->op]++; - - args.o1[n_fuse + 1] = bid_fuse.offs; + if (use_fusion && fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_ADD_CHAIN) { + // the offsets of the fused addends are relative to the start of the src1 buffer + for (int i = 1; i < n_fuse; i++) { + args.o1[i] = ggml_metal_get_buffer_id(ctx->node(idx + i)->src[1]).offs; } - ++n_fuse; + ctx->count_fusions(fusion); - if (debug_fusion > 1 && n_fuse > 1) { + if (debug_fusion > 1) { GGML_LOG_DEBUG("%s: fuse: ADD x %d\n", __func__, n_fuse); } } @@ -3707,9 +4127,9 @@ int ggml_metal_op_norm(ggml_metal_op_t ctx, int idx) { ggml_metal_library_t lib = ctx->lib; ggml_metal_encoder_t enc = ctx->enc; - const bool use_fusion = ctx->use_fusion; + const bool use_fusion = ctx->use_fusion(); - const int debug_fusion = ctx->debug_fusion; + const int debug_fusion = ggml_metal_fusion_info_debug(ctx->finfo); GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); GGML_TENSOR_LOCALS(uint64_t, nb0, op->src[0], nb); @@ -3735,67 +4155,61 @@ int ggml_metal_op_norm(ggml_metal_op_t ctx, int idx) { /*.nbf1 =*/ { nb01 }, /*.nbf2 =*/ { nb02 }, /*.nbf3 =*/ { nb03 }, + /*.scale =*/ 1.0f, }; - ggml_op fops[8]; - int n_fuse = 1; + bool fused_norm_scale = false; ggml_metal_buffer_id bid_fuse[2] = { bid_src0, bid_src0 }; // d[0] = norm(a) - // d[1] = mul(d[0], b) + // d[1] = mul(d[0], b) or scale(d[0]) // d[2] = add(d[1], c) if (use_fusion) { - fops[0] = op->op; - fops[1] = GGML_OP_MUL; - fops[2] = GGML_OP_ADD; + int n = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n); - for (n_fuse = 0; n_fuse <= 1; ++n_fuse) { - if (!ctx->can_fuse(idx + n_fuse, fops + n_fuse, 2)) { - break; - } + if (fusion && (ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_NORM_MUL || ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_NORM_MUL_ADD)) { + n_fuse = n; - ggml_tensor * f0 = ctx->node(idx + n_fuse); - ggml_tensor * f1 = ctx->node(idx + n_fuse + 1); + ctx->count_fusions(fusion); - if (f0 != f1->src[0]) { - break; - } + for (int i = 1; i < n_fuse; i++) { + const ggml_tensor * fn = ctx->node(idx + i); - if (f1->src[1]->ne[0] != op->ne[0]) { - break; - } + bid_fuse[i - 1] = ggml_metal_get_buffer_id(fn->src[1]); - if (!ggml_is_contiguous_rows(f1->src[1])) { - break; - } + args.nef1[i] = fn->src[1]->ne[1]; + args.nef2[i] = fn->src[1]->ne[2]; + args.nef3[i] = fn->src[1]->ne[3]; - if (f1->type != GGML_TYPE_F32) { - break; + args.nbf1[i] = fn->src[1]->nb[1]; + args.nbf2[i] = fn->src[1]->nb[2]; + args.nbf3[i] = fn->src[1]->nb[3]; } - //ctx->fuse_cnt[f1->op]++; - - bid_fuse[n_fuse] = ggml_metal_get_buffer_id(f1->src[1]); + if (debug_fusion > 1) { + if (n_fuse == 2) { + GGML_LOG_DEBUG("%s: fuse: %s + MUL\n", __func__, ggml_op_name(op->op)); + } + if (n_fuse == 3) { + GGML_LOG_DEBUG("%s: fuse: %s + MUL + ADD\n", __func__, ggml_op_name(op->op)); + } + } + } - args.nef1[n_fuse + 1] = f1->src[1]->ne[1]; - args.nef2[n_fuse + 1] = f1->src[1]->ne[2]; - args.nef3[n_fuse + 1] = f1->src[1]->ne[3]; + if (fusion && ggml_metal_fusion_get_id(fusion) == GGML_METAL_FUSION_NORM_SCALE) { + n_fuse = n; + fused_norm_scale = true; - args.nbf1[n_fuse + 1] = f1->src[1]->nb[1]; - args.nbf2[n_fuse + 1] = f1->src[1]->nb[2]; - args.nbf3[n_fuse + 1] = f1->src[1]->nb[3]; - } + ctx->count_fusions(fusion); - ++n_fuse; + const ggml_tensor * scale_node = ctx->node(idx + 1); + args.scale = ggml_get_op_params_f32(scale_node, 0); - if (debug_fusion > 1 && n_fuse > 1) { - if (n_fuse == 2) { - GGML_LOG_DEBUG("%s: fuse: %s + MUL\n", __func__, ggml_op_name(op->op)); - } - if (n_fuse == 3) { - GGML_LOG_DEBUG("%s: fuse: %s + MUL + ADD\n", __func__, ggml_op_name(op->op)); + if (debug_fusion > 1) { + GGML_LOG_DEBUG("%s: fuse: %s + SCALE\n", __func__, ggml_op_name(op->op)); } } } @@ -3812,7 +4226,9 @@ int ggml_metal_op_norm(ggml_metal_op_t ctx, int idx) { } } - auto pipeline = ggml_metal_library_get_pipeline_norm(lib, op, n_fuse); + auto pipeline = fused_norm_scale ? + ggml_metal_library_get_pipeline_norm_scale(lib, op) : + ggml_metal_library_get_pipeline_norm(lib, op, n_fuse); int nth = 32; // SIMD width @@ -3884,6 +4300,11 @@ int ggml_metal_op_rope(ggml_metal_op_t ctx, int idx) { const int sect_2 = ((const int32_t *) op->op_params)[13]; const int sect_3 = ((const int32_t *) op->op_params)[14]; + const int n_offs = ((const int32_t *) op->op_params)[15]; + + // when dst aliases src0, the channels outside the rotated window already hold the correct data + const bool inplace = op->data == op->src[0]->data; + ggml_metal_kargs_rope args = { /*.ne00 =*/ ne00, /*.ne01 =*/ ne01, @@ -3903,6 +4324,7 @@ int ggml_metal_op_rope(ggml_metal_op_t ctx, int idx) { /*.nb3 =*/ nb3, /*.n_past =*/ n_past, /*.n_dims =*/ n_dims, + /*.n_offs =*/ n_offs, /*.n_ctx_orig =*/ n_ctx_orig, /*.freq_base =*/ freq_base, /*.freq_scale =*/ freq_scale, @@ -3915,6 +4337,7 @@ int ggml_metal_op_rope(ggml_metal_op_t ctx, int idx) { /* sect_2 =*/ sect_2, /* sect_3 =*/ sect_3, /* src2 =*/ op->src[2] != nullptr, + /* inplace =*/ inplace, }; auto pipeline = ggml_metal_library_get_pipeline_rope(lib, op); @@ -4404,6 +4827,7 @@ int ggml_metal_op_conv_transpose_2d(ggml_metal_op_t ctx, int idx) { const int32_t OW = op->ne[0]; const int32_t OH = op->ne[1]; const int32_t OC = op->ne[2]; + const int32_t N = op->src[1]->ne[3]; ggml_metal_kargs_conv_transpose_2d args = { /*.IC =*/ IC, @@ -4416,6 +4840,7 @@ int ggml_metal_op_conv_transpose_2d(ggml_metal_op_t ctx, int idx) { /*.nb0 =*/ nb0, /*.nb1 =*/ nb1, /*.nb2 =*/ nb2, + /*.nb3 =*/ nb3, }; auto pipeline = ggml_metal_library_get_pipeline_conv_transpose_2d(lib, op); @@ -4430,7 +4855,7 @@ int ggml_metal_op_conv_transpose_2d(ggml_metal_op_t ctx, int idx) { const size_t smem = GGML_PAD(KW * KH * sizeof(float), 16); ggml_metal_encoder_set_threadgroup_memory_size(enc, smem, 0); - ggml_metal_encoder_dispatch_threadgroups(enc, OW, OH, OC, KW, KH, 1); + ggml_metal_encoder_dispatch_threadgroups(enc, OW, OH, OC * N, KW, KH, 1); return 1; } @@ -4863,7 +5288,9 @@ int ggml_metal_op_argsort(ggml_metal_op_t ctx, int idx) { return 1; } -int ggml_metal_op_top_k(ggml_metal_op_t ctx, int idx) { +// bitonic-sort + merge fallback: efficient when k is small and there are few rows, +// where the single-workgroup-per-row radix-select cannot reach enough parallelism +static void ggml_metal_op_top_k_bitonic(ggml_metal_op_t ctx, int idx) { ggml_tensor * op = ctx->node(idx); ggml_metal_library_t lib = ctx->lib; @@ -4971,6 +5398,178 @@ int ggml_metal_op_top_k(ggml_metal_op_t ctx, int idx) { len <<= 1; } +} + +// radix-select: one workgroup per row. Maps each float to an order-preserving unsigned +// key, finds the k-th largest via 4 radix-8 histogram passes, then compacts the top-k +// indices. Fast for large k and/or many rows. +static void ggml_metal_op_top_k_radix(ggml_metal_op_t ctx, int idx) { + ggml_tensor * op = ctx->node(idx); + + ggml_metal_library_t lib = ctx->lib; + ggml_metal_encoder_t enc = ctx->enc; + + GGML_ASSERT(ggml_is_contiguous_rows(op->src[0])); + + GGML_TENSOR_LOCALS( int32_t, ne0, op->src[0], ne); + GGML_TENSOR_LOCALS(uint64_t, nb0, op->src[0], nb); + + auto pipeline = ggml_metal_library_get_pipeline_top_k_radix(lib, op); + + // one workgroup per row; radix-select the k-th largest value + const int nth = std::min(1024, ggml_metal_pipeline_max_theads_per_threadgroup(pipeline)); + + ggml_metal_kargs_top_k args = { + /*.ne00 =*/ ne00, + /*.ne01 =*/ ne01, + /*.ne02 =*/ ne02, + /*.ne03 =*/ ne03, + /*.nb01 =*/ nb01, + /*.nb02 =*/ nb02, + /*.nb03 =*/ nb03, + /*.top_k =*/ (int32_t) op->ne[0], + }; + + // shared memory: 256-entry histogram + bucket/above scalars + output counter + const size_t smem_histo = GGML_PAD(256*sizeof(uint32_t), 16); + const size_t smem_bucket = GGML_PAD( sizeof(uint32_t), 16); + const size_t smem_above = GGML_PAD( sizeof(uint32_t), 16); + const size_t smem_out = GGML_PAD( sizeof(uint32_t), 16); + + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op->src[0]), 1); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(op), 2); + + ggml_metal_encoder_set_threadgroup_memory_size(enc, smem_histo, 0); + ggml_metal_encoder_set_threadgroup_memory_size(enc, smem_bucket, 1); + ggml_metal_encoder_set_threadgroup_memory_size(enc, smem_above, 2); + ggml_metal_encoder_set_threadgroup_memory_size(enc, smem_out, 3); + + ggml_metal_encoder_dispatch_threadgroups(enc, ne01, ne02, ne03, nth, 1, 1); +} + +int ggml_metal_op_topk_moe(ggml_metal_op_t ctx, int idx) { + ggml_metal_library_t lib = ctx->lib; + ggml_metal_encoder_t enc = ctx->enc; + + int n_fuse = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n_fuse); + if (!fusion || ggml_metal_fusion_get_id(fusion) != GGML_METAL_FUSION_TOPK_MOE) { + return 1; + } + + ggml_tensor * softmax = ctx->node(idx); + ggml_tensor * logits = softmax->src[0]; + ggml_tensor * get_rows = ctx->node(idx + 2); + ggml_tensor * ids = get_rows->src[1]; + ggml_tensor * weights = ctx->node(idx + n_fuse - 1); + + const int64_t n_expert = logits->ne[0]; + const int64_t n_tokens = logits->ne[1]; + const int64_t n_expert_used = ids->ne[0]; + + const bool with_norm = n_fuse >= 6; + const bool with_scale = n_fuse == 4 || n_fuse == 7; + + float clamp = -INFINITY; + if (with_norm) { + ggml_tensor * clamp_node = ctx->node(idx + 4); + clamp = ggml_get_op_params_f32(clamp_node, 0); + } + + float scale = 1.0f; + if (with_scale) { + ggml_tensor * scale_node = ctx->node(idx + n_fuse - 1); + scale = ggml_get_op_params_f32(scale_node, 0); + } + + ggml_metal_kargs_topk_moe args = { + /*.ne01 =*/ (int32_t) n_tokens, + /*.nb01 =*/ logits->nb[1], + /*.nb1_ids =*/ ids->nb[1], + /*.clamp =*/ clamp, + /*.scale =*/ scale, + }; + + auto pipeline = ggml_metal_library_get_pipeline_topk_moe(lib, (int32_t) n_expert, (int32_t) n_expert_used, with_norm); + + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(logits), 1); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(weights), 2); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(ids), 3); + + ggml_metal_encoder_dispatch_threadgroups(enc, (uint32_t) n_tokens, 1, 1, 32, 1, 1); + + ctx->count_fusions(fusion); + + if (ggml_metal_fusion_info_debug(ctx->finfo) > 1) { + GGML_LOG_DEBUG("%s: fuse: SOFT_MAX + ARGSORT + GET_ROWS\n", __func__); + } + + return n_fuse; +} + +int ggml_metal_op_moe_reduce(ggml_metal_op_t ctx, int idx) { + ggml_metal_library_t lib = ctx->lib; + ggml_metal_encoder_t enc = ctx->enc; + + int n_fuse = 1; + const ggml_metal_fusion * fusion = ctx->can_fuse(idx, GGML_METAL_FUSION_FULL, &n_fuse); + if (!fusion || ggml_metal_fusion_get_id(fusion) != GGML_METAL_FUSION_MOE_REDUCE) { + return 1; + } + + ggml_tensor * mul = ctx->node(idx); + ggml_tensor * experts = mul->src[0]; + ggml_tensor * weights = mul->src[1]; + ggml_tensor * dst = ctx->node(idx + n_fuse - 1); + + ggml_metal_kargs_moe_reduce args = { + /*.ne00 =*/ (int32_t) experts->ne[0], + /*.ne02 =*/ (int32_t) experts->ne[2], + }; + + auto pipeline = ggml_metal_library_get_pipeline_moe_reduce(lib, (int32_t) experts->ne[1]); + + const int nth = std::min(256, ggml_metal_pipeline_max_theads_per_threadgroup(pipeline)); + const int n_col_tiles = (args.ne00 + nth - 1) / nth; + + ggml_metal_encoder_set_pipeline(enc, pipeline); + ggml_metal_encoder_set_bytes (enc, &args, sizeof(args), 0); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(experts), 1); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(weights), 2); + ggml_metal_encoder_set_buffer (enc, ggml_metal_get_buffer_id(dst), 3); + + ggml_metal_encoder_dispatch_threadgroups(enc, (uint32_t) args.ne02, (uint32_t) n_col_tiles, 1, nth, 1, 1); + + ctx->count_fusions(fusion); + + if (ggml_metal_fusion_info_debug(ctx->finfo) > 1) { + GGML_LOG_DEBUG("%s: fuse: MOE_REDUCE\n", __func__); + } + + return n_fuse; +} + +int ggml_metal_op_top_k(ggml_metal_op_t ctx, int idx) { + ggml_tensor * op = ctx->node(idx); + + // radix-select has a fixed single-workgroup-per-row cost (~50-60us) that is only + // amortized for long rows, many rows, or a large k; otherwise the bitonic path wins + const int ncols = op->src[0]->ne[0]; + const int k = op->ne[0]; + const int nrows = ggml_nrows(op->src[0]); + + const bool use_radix = + ncols > 2048 && (k > 64 || (nrows > 4 && ncols >= 8192)); + + if (use_radix) { + ggml_metal_op_top_k_radix(ctx, idx); + } else { + ggml_metal_op_top_k_bitonic(ctx, idx); + } return 1; } diff --git a/ggml/src/ggml-metal/ggml-metal-ops.h b/ggml/src/ggml-metal/ggml-metal-ops.h index b03b59e0..583d1156 100644 --- a/ggml/src/ggml-metal/ggml-metal-ops.h +++ b/ggml/src/ggml-metal/ggml-metal-ops.h @@ -8,17 +8,18 @@ extern "C" { typedef struct ggml_metal_op * ggml_metal_op_t; +struct ggml_metal_fusion; // forward decl (ggml-metal-device.h) + ggml_metal_op_t ggml_metal_op_init( ggml_metal_device_t dev, ggml_metal_cmd_buf_t cmd_buf, struct ggml_cgraph * gf, + struct ggml_metal_fusion_info * finfo, int idx_start, int idx_end, - bool use_fusion, bool use_concurrency, bool use_capture, - int debug_graph, - int debug_fusion); + int debug_graph); void ggml_metal_op_free(ggml_metal_op_t ctx); @@ -35,6 +36,7 @@ size_t ggml_metal_op_mul_mat_id_extra_tpe(const struct ggml_tensor * op); // id map [n_tokens, n_expert] size_t ggml_metal_op_mul_mat_id_extra_ids(const struct ggml_tensor * op); +size_t ggml_metal_op_mul_mat_id_extra_amax(const struct ggml_tensor * op); // return true if we should use the FA vector kernel for this op bool ggml_metal_op_flash_attn_ext_use_vec(const struct ggml_tensor * op); @@ -42,6 +44,8 @@ bool ggml_metal_op_flash_attn_ext_use_vec(const struct ggml_tensor * op); size_t ggml_metal_op_flash_attn_ext_extra_pad(const struct ggml_tensor * op); size_t ggml_metal_op_flash_attn_ext_extra_blk(const struct ggml_tensor * op); size_t ggml_metal_op_flash_attn_ext_extra_tmp(const struct ggml_tensor * op); +size_t ggml_metal_op_flash_attn_ext_extra_kv_f16(const struct ggml_tensor * op); +size_t ggml_metal_op_flash_attn_ext_extra_idx(const struct ggml_tensor * op); int ggml_metal_op_concat (ggml_metal_op_t ctx, int idx); int ggml_metal_op_repeat (ggml_metal_op_t ctx, int idx); @@ -94,6 +98,8 @@ int ggml_metal_op_timestep_embedding(ggml_metal_op_t ctx, int idx); int ggml_metal_op_argmax (ggml_metal_op_t ctx, int idx); int ggml_metal_op_argsort (ggml_metal_op_t ctx, int idx); int ggml_metal_op_top_k (ggml_metal_op_t ctx, int idx); +int ggml_metal_op_topk_moe (ggml_metal_op_t ctx, int idx); +int ggml_metal_op_moe_reduce (ggml_metal_op_t ctx, int idx); int ggml_metal_op_tri (ggml_metal_op_t ctx, int idx); int ggml_metal_op_opt_step_adamw (ggml_metal_op_t ctx, int idx); int ggml_metal_op_opt_step_sgd (ggml_metal_op_t ctx, int idx); diff --git a/ggml/src/ggml-metal/ggml-metal-tuning.cpp b/ggml/src/ggml-metal/ggml-metal-tuning.cpp new file mode 100644 index 00000000..217ad43b --- /dev/null +++ b/ggml/src/ggml-metal/ggml-metal-tuning.cpp @@ -0,0 +1,1172 @@ +#include "ggml-metal-tuning.h" + +#include +#include +#include + +namespace ggml_metal_tuning { + +int fa_vec_ne11_bucket(int64_t ne11) { + for (int i = 0; i < (int) std::size(FA_VEC_NE11_BUCKETS); ++i) { + if (ne11 < FA_VEC_NE11_BUCKETS[i]) { + return i; + } + } + return (int) std::size(FA_VEC_NE11_BUCKETS); +} + +int fa_vec_ne01_bucket(int64_t ne01) { + for (int i = 0; i < (int) std::size(FA_VEC_NE01_BUCKETS); ++i) { + if (ne01 < FA_VEC_NE01_BUCKETS[i]) { + return i; + } + } + return (int) std::size(FA_VEC_NE01_BUCKETS); +} + +int fa_vec_baseline_ne(int dk, int dv) { + if (dk == 32 && dv == 32) { + return 4; + } + if (dk == 64 && dv == 64) { + return 2; + } + if (dk == 96 && dv == 96) { + return 4; + } + if (dk == 96 && dv == 64) { + return 4; + } + if (dk == 128 && dv == 128) { + return 1; + } + if (dk == 192 && dv == 192) { + return 2; + } + if (dk == 192 && dv == 128) { + return 2; + } + if (dk == 256 && dv == 256) { + return 1; + } + if (dk == 320 && dv == 256) { + return 2; + } + if (dk == 512 && dv == 512) { + return 1; + } + if (dk == 576 && dv == 512) { + return 2; + } + return 4; // template default +} + +fa_vec_cfg_t fa_vec_baseline_cfg(int dk, int dv) { + return { 1, (int8_t) fa_vec_baseline_ne(dk, dv) }; +} + +// Generated by `ggml-metal-tuning fa-vec`; do not hand-edit. +// Keyed by Apple GPU family, retagged from the per-SKU token the tuner emits, and pooled from +// the sweeps listed at the head of each segment. One row per kept bucket, plus +// per-(dtype,dk,dv) ne11-collapsed domain defaults (ne11_b = FA_VEC_NE11_DEFAULT, +// ne01_b = domain). To retune or add a device, see tools/tuning/README.md. +// See ggml-metal-tuning.h for the row/lookup semantics. +// ref: https://github.com/ggml-org/llama.cpp/pull/27824 +// https://github.com/ggml-org/llama.cpp/discussions/27668 +constexpr fa_vec_entry_t fa_vec_tuned_table[] = { + // Apple7 - M1, M1_ULTRA + { { 7, GGML_TYPE_F16, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 64, 64, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 128, 128, 2, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 128, 128, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 128, 128, 2, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 128, 128, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_F16, 128, 128, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_F16, 192, 128, 1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 192, 128, 1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 192, 128, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 192, 128, 1, 3 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 192, 128, 1, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 320, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_F16, 320, 256, 1, 1 }, { 1, 2 } }, + { { 7, GGML_TYPE_F16, 320, 256, 1, 2 }, { 1, 2 } }, + { { 7, GGML_TYPE_F16, 320, 256, 1, 3 }, { 1, 2 } }, + { { 7, GGML_TYPE_F16, 320, 256, 1, 4 }, { 1, 2 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 1, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 3, 2 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q4_0, 32, 32, 3, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 64, 64, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 64, 64, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 64, 64, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 96, 96, 2, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 96, 96, 3, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 128, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 192, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 1, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 2, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 3, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q4_1, 32, 32, 3, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 64, 64, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 64, 64, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 96, 96, 2, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 96, 96, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, 1, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 192, 192, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 192, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 256, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 256, 256, 3, 2 }, { 1, 1 } }, + { { 7, GGML_TYPE_Q4_1, 320, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 512, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 512, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 576, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q4_1, 576, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 32, 32, -1, 1 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_0, 32, 32, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 32, 32, 1, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 32, 32, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 64, 64, -1, 1 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_0, 64, 64, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 64, 64, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 64, 64, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 96, 96, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 96, 96, 1, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, 1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, 1, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 128, 128, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, 1, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 192, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 192, 128, 2, 3 }, { 4, 2 } }, + { { 7, GGML_TYPE_Q5_0, 192, 128, 3, 3 }, { 4, 2 } }, + { { 7, GGML_TYPE_Q5_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 256, 256, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 320, 256, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, -1, 1 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, 1, 2 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, 1, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 64, 64, -1, 1 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_1, 64, 64, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 64, 64, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 64, 64, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, 1, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, 1, 4 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, 2, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_1, 96, 96, 3, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q5_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 128, 128, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 128, 128, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 128, 128, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 192, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 192, 1, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 192, 128, 2, 3 }, { 4, 2 } }, + { { 7, GGML_TYPE_Q5_1, 192, 128, 3, 3 }, { 4, 2 } }, + { { 7, GGML_TYPE_Q5_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 256, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 256, 256, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, 2, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 320, 256, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q5_1, 512, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 512, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 576, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q5_1, 576, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 1, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 7, GGML_TYPE_Q8_0, 32, 32, 3, 4 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, 1, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, 2, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, 2, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 64, 64, 3, 2 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 96, 96, 2, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 128, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 192, 128, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 256, 256, -1, 0 }, { 1, 2 } }, + { { 7, GGML_TYPE_Q8_0, 256, 256, -1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 1, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 1, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 2, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 2, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 3, 1 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 320, 256, 3, 3 }, { 2, 4 } }, + { { 7, GGML_TYPE_Q8_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 7, GGML_TYPE_Q8_0, 576, 512, -1, 1 }, { 1, 4 } }, + // Apple8 - M2, M2_MAX, M2_PRO + { { 8, GGML_TYPE_F16, 64, 64, 1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 64, 64, 2, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 64, 64, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 64, 64, 1, 2 }, { 1, 2 } }, + { { 8, GGML_TYPE_F16, 64, 64, 3, 2 }, { 1, 2 } }, + { { 8, GGML_TYPE_F16, 64, 64, 3, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_F16, 64, 64, 3, 4 }, { 1, 2 } }, + { { 8, GGML_TYPE_F16, 128, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 128, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 128, 128, 3, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_F16, 192, 128, 1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 192, 128, 1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 192, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 320, 256, 2, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 320, 256, 2, 3 }, { 1, 4 } }, + { { 8, GGML_TYPE_F16, 320, 256, 2, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 2, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 3, 2 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 64, 64, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 64, 64, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 64, 64, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 96, 96, 1, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 96, 96, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 96, 96, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, 1, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 128, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 192, 128, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 320, 256, 3, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q4_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 1, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_1, 32, 32, 3, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 64, 64, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 64, 64, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 64, 64, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 64, 64, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 96, 96, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 96, 96, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q4_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 128, 128, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 128, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 128, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 1, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 2, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 2, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 192, 3, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 192, 128, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 3, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 1, 3 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 256, 256, 1, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 320, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 512, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 512, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 576, 512, 1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 576, 512, 1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 576, 512, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 576, 512, 1, 3 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q4_1, 576, 512, 1, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, 1, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, 2, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 64, 64, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_0, 64, 64, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 64, 64, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 64, 64, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 96, 96, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_0, 96, 96, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 128, 128, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, 2, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, 1, 2 }, { 1, 2 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, 1, 4 }, { 1, 2 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 192, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, 2, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 192, 128, 3, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, 1, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 256, 256, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 32, 32, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 32, 32, 1, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 64, 64, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_1, 64, 64, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 64, 64, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 64, 64, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 96, 96, -1, 1 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_1, 96, 96, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 96, 96, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 128, 128, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 192, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 192, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 192, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 192, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, -1, 1 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 1, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 192, 128, 3, 4 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q5_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 256, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 320, 256, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 512, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 512, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 576, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q5_1, 576, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, 2, 4 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, 3, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, 1, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, 2, 2 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, 3, 2 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q8_0, 64, 64, 3, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q8_0, 96, 96, 2, 3 }, { 4, 4 } }, + { { 8, GGML_TYPE_Q8_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 128, 128, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 192, 128, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 256, 256, -1, 0 }, { 1, 2 } }, + { { 8, GGML_TYPE_Q8_0, 256, 256, -1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 1, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 1, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 2, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 2, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 3, 1 }, { 2, 4 } }, + { { 8, GGML_TYPE_Q8_0, 320, 256, 3, 3 }, { 4, 2 } }, + { { 8, GGML_TYPE_Q8_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 8, GGML_TYPE_Q8_0, 576, 512, -1, 1 }, { 1, 4 } }, + // Apple9 - A18_PRO, M3_MAX, M3_PRO, M3_ULTRA, M4, M4_MAX, M4_PRO + { { 9, GGML_TYPE_F16, 32, 32, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 32, 32, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 32, 32, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 32, 32, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 64, 64, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 64, 64, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 64, 64, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 64, 64, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 64, 64, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, 1, 3 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, 1, 4 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 96, 96, 2, 4 }, { 1, 4 } }, + { { 9, GGML_TYPE_F16, 128, 128, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_F16, 128, 128, 1, 1 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 128, 128, 1, 2 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 128, 128, 1, 4 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 128, 128, 2, 2 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 128, 128, 2, 4 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 192, 192, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_F16, 192, 192, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 192, 1, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 192, 1, 4 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 128, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_F16, 192, 128, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 128, 1, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 128, 1, 4 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 192, 128, 2, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 256, 256, 2, 0 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 256, 256, 3, 0 }, { 1, 2 } }, + { { 9, GGML_TYPE_F16, 256, 256, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_F16, 256, 256, 1, 2 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 256, 256, 1, 4 }, { 1, 1 } }, + { { 9, GGML_TYPE_F16, 320, 256, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q4_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 32, 32, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 32, 32, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 64, 64, 2, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_0, 64, 64, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_0, 64, 64, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_0, 96, 96, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 96, 96, 3, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 128, 128, 3, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 192, 1, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 192, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 128, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 128, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 192, 128, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 320, 256, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 320, 256, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 1, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 1, 3 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_0, 576, 512, 1, 4 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 32, 32, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 32, 32, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 32, 32, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_1, 32, 32, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_1, 64, 64, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q4_1, 96, 96, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 96, 96, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 96, 96, 3, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, 1, 2 }, { 1, 1 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, 1, 4 }, { 1, 1 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, 2, 2 }, { 1, 1 } }, + { { 9, GGML_TYPE_Q4_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, 1, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 192, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 192, 128, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 256, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 320, 256, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 320, 256, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 320, 256, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 320, 256, 3, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 1, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 1, 3 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q4_1, 576, 512, 1, 4 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, 1, 2 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, 1, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 64, 64, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 64, 64, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 64, 64, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 64, 64, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 96, 96, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 96, 96, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 96, 96, 1, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 1, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 1, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 2, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 2, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 128, 128, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 192, 192, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, 1, 4 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 256, 256, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 320, 256, 2, 3 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q5_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, 1, 2 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, 1, 3 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q5_0, 576, 512, 1, 4 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, 1, 2 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, 1, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 64, 64, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_1, 64, 64, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 64, 64, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 64, 64, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 96, 96, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_1, 96, 96, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 96, 96, 1, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 96, 96, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, -1, 1 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, 1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, 1, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 128, 128, 3, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 192, 192, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 192, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, -1, 1 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, 1, 4 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 256, 256, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 320, 256, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 320, 256, 1, 3 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q5_1, 320, 256, 3, 3 }, { 2, 2 } }, + { { 9, GGML_TYPE_Q5_1, 512, 512, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 512, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 576, 512, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 576, 512, 2, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 576, 512, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 576, 512, 2, 3 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q5_1, 576, 512, 2, 4 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 32, 32, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q8_0, 32, 32, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q8_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q8_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 64, 64, 3, 2 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q8_0, 64, 64, 3, 3 }, { 4, 4 } }, + { { 9, GGML_TYPE_Q8_0, 96, 96, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 96, 96, 3, 2 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 96, 96, 3, 4 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 128, 128, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 128, 128, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 192, 1, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 192, 2, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 192, 2, 3 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 128, 1, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 128, 2, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 192, 128, 3, 2 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 512, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, 2, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, 3, 0 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, 1, 1 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, 1, 3 }, { 1, 2 } }, + { { 9, GGML_TYPE_Q8_0, 576, 512, 1, 4 }, { 1, 2 } }, + // Apple10 - M5_MAX + { { 10, GGML_TYPE_F16, 32, 32, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 32, 32, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 32, 32, 1, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 32, 32, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 32, 32, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 32, 32, 3, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 64, 64, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 64, 64, 3, 0 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 64, 64, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 64, 64, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 64, 64, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 64, 64, 3, 3 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 64, 64, 3, 4 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 96, 96, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 96, 96, 3, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, 3, 0 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 128, 128, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 128, 128, 1, 2 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, 1, 4 }, { 1, 1 } }, + { { 10, GGML_TYPE_F16, 128, 128, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 128, 128, 3, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 192, 192, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 192, 192, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 192, 192, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 192, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_F16, 192, 128, 1, 4 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 192, 128, 3, 2 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 256, 256, -1, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 256, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 256, 256, 1, 2 }, { 1, 1 } }, + { { 10, GGML_TYPE_F16, 256, 256, 1, 4 }, { 1, 1 } }, + { { 10, GGML_TYPE_F16, 320, 256, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 320, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 320, 256, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 320, 256, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 512, 512, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_F16, 512, 512, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 512, 512, 2, 3 }, { 4, 2 } }, + { { 10, GGML_TYPE_F16, 512, 512, 3, 3 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 512, 512, 3, 4 }, { 4, 1 } }, + { { 10, GGML_TYPE_F16, 576, 512, 2, 0 }, { 4, 4 } }, + { { 10, GGML_TYPE_F16, 576, 512, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_F16, 576, 512, 1, 1 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 576, 512, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 576, 512, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 576, 512, 2, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_F16, 576, 512, 2, 3 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, 1, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 32, 32, 3, 4 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 64, 64, 3, 4 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 1, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 96, 96, 3, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, 1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, 2, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 128, 128, 3, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, 2, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 192, 2, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 128, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 192, 128, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 1, 4 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 2, 3 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 2, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 3, 3 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 256, 256, 3, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_0, 320, 256, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 320, 256, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 320, 256, 2, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 320, 256, 3, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 2, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 2, 3 }, { 4, 1 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 3, 1 }, { 4, 1 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 3, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q4_0, 512, 512, 3, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 2, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 2, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 3, 1 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q4_0, 576, 512, 3, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 1, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 2, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 3, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 32, 32, 3, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, 3, 0 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 64, 64, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, 1, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, 2, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 96, 96, 3, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 1, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 2, 1 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 2, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 3, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 128, 128, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 1, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 1, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 192, 3, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q4_1, 192, 128, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, 3, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, 2, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, 2, 3 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 256, 256, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 2, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 2, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 320, 256, 3, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, -1, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 1, 3 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 2, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 3, 1 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 512, 512, 3, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q4_1, 576, 512, 2, 0 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q4_1, 576, 512, 1, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q4_1, 576, 512, 2, 1 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q4_1, 576, 512, 3, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q5_0, 32, 32, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 32, 32, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 32, 32, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 32, 32, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 32, 32, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 64, 64, 3, 2 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 96, 96, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 96, 96, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 96, 96, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 96, 96, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, 1, 0 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 128, 128, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, 1, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, 1, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, 2, 4 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 192, 3, 4 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 192, 128, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 1, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 1, 2 }, { 1, 1 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 2, 2 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 2, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 256, 256, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 2, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 320, 256, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 3, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 1, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 512, 512, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 2, 1 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 2, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 3, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_0, 576, 512, 3, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 32, 32, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 32, 32, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 32, 32, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 32, 32, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 64, 64, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 96, 96, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 96, 96, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 96, 96, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 96, 96, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 128, 128, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 3, 0 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 192, 3, 3 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, 2, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, -1, 1 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, 1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 192, 128, 3, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, 2, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, 1, 2 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, 2, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 256, 256, 3, 4 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 320, 256, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 320, 256, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 320, 256, -1, 1 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q5_1, 320, 256, 2, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 320, 256, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 2, 0 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 3, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 1, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 1, 4 }, { 1, 1 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 2, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q5_1, 512, 512, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 1, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 2, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 2, 3 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q5_1, 576, 512, 3, 3 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, 1, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 32, 32, 3, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 64, 64, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 64, 64, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 64, 64, 1, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 64, 64, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 96, 96, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 96, 96, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 96, 96, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 96, 96, 3, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 1, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 2, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 2, 4 }, { 2, 2 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 3, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 128, 128, 3, 2 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, 1, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, 2, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, 3, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 192, 3, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q8_0, 192, 128, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 128, -1, 1 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 192, 128, 1, 2 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q8_0, 192, 128, 3, 2 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 256, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 256, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 256, 256, 3, 3 }, { 2, 4 } }, + { { 10, GGML_TYPE_Q8_0, 320, 256, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 320, 256, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 2, 0 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 1, 3 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 2, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 2, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 2, 3 }, { 4, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 2, 4 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 3, 1 }, { 4, 1 } }, + { { 10, GGML_TYPE_Q8_0, 512, 512, 3, 2 }, { 4, 2 } }, + { { 10, GGML_TYPE_Q8_0, 576, 512, -1, 0 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 576, 512, -1, 1 }, { 1, 4 } }, + { { 10, GGML_TYPE_Q8_0, 576, 512, 1, 4 }, { 1, 2 } }, + { { 10, GGML_TYPE_Q8_0, 576, 512, 2, 1 }, { 4, 4 } }, +}; + +static bool g_override_set = false; +static fa_vec_cfg_t g_override_cfg = { 1, 4 }; + +void fa_vec_set_override(fa_vec_cfg_t cfg) { + g_override_cfg = cfg; + g_override_set = true; +} + +void fa_vec_clear_override() { + g_override_set = false; +} + +static const fa_vec_cfg_t * find_cfg(const fa_vec_entry_t * tbl, size_t n, const fa_vec_key_t & k) { + for (size_t i = 0; i < n; ++i) { + if (memcmp(&tbl[i].key, &k, sizeof(k)) == 0) { + return &tbl[i].cfg; + } + } + return nullptr; +} + +fa_vec_cfg_t fa_vec_pick(int gpu_family, int dtype, int dk, int dv, int64_t ne11, int64_t ne01) { + if (g_override_set) { + return g_override_cfg; + } + + const fa_vec_cfg_t baseline = fa_vec_baseline_cfg(dk, dv); + + const int ne11_b = fa_vec_ne11_bucket(ne11); + if (ne11_b == 0) { + return baseline; // short KV: attention is a small slice of the step, left to baseline + } + + const int ne01_b = fa_vec_ne01_bucket(ne01); + + fa_vec_key_t k{}; + k.family = (int8_t) gpu_family; + k.dtype = (int8_t) dtype; + k.dk = (int16_t) dk; + k.dv = (int16_t) dv; + k.ne11_b = (int8_t) ne11_b; + k.ne01_b = (int8_t) ne01_b; + + // exact bucket, then the ne01 domain default (ne11 collapsed) + if (auto * c = find_cfg(fa_vec_tuned_table, std::size(fa_vec_tuned_table), k)) { + return *c; + } + k.ne11_b = FA_VEC_NE11_DEFAULT; + k.ne01_b = (ne01_b == 0) ? FA_VEC_DOMAIN_DECODE : FA_VEC_DOMAIN_BATCH; + if (auto * c = find_cfg(fa_vec_tuned_table, std::size(fa_vec_tuned_table), k)) { + return *c; + } + + return baseline; +} + +} // namespace ggml_metal_tuning diff --git a/ggml/src/ggml-metal/ggml-metal-tuning.h b/ggml/src/ggml-metal/ggml-metal-tuning.h new file mode 100644 index 00000000..003b4d6b --- /dev/null +++ b/ggml/src/ggml-metal/ggml-metal-tuning.h @@ -0,0 +1,75 @@ +#pragma once + +#include "ggml.h" + +#include +#include + +namespace ggml_metal_tuning { + +// FA vec selection buckets. ne01 (query rows) splits decode (==1) from batch (>=2), the +// batch side refined into {2,3,4,5}: Q>1 reuses one K/V load across rows, so it only pays +// off once ne01 aligns with Q. ne11 (KV length) is bucketed too, as the Q>1 crossover is +// head-size dependent (small dk crosses late, large dk wins even at short KV). +constexpr int FA_VEC_NE11_BUCKETS[] = { 1024, 4096, 16384 }; +constexpr int FA_VEC_NE01_BUCKETS[] = { 2, 3, 4, 5 }; + +int fa_vec_ne11_bucket(int64_t ne11); +int fa_vec_ne01_bucket(int64_t ne01); + +// NE baked into each (dk,dv) baseline instantiation in kernels/fa.metal. +// Hand-maintained mirror; keep in sync with those instantiations. +// The Metal test slice covers every legal config for dk=128 and dk=576. +int fa_vec_baseline_ne(int dk, int dv); + +// Tuned table has two row kinds. Exact rows key a (ne11_b, ne01_b) bucket. Default rows +// collapse ne11 over one ne01 domain: ne11_b == FA_VEC_NE11_DEFAULT and ne01_b holds the +// domain. fa_vec_pick tries exact bucket -> domain default -> baseline; short KV +// (ne11 < FA_VEC_NE11_BUCKETS[0]) always uses baseline. +constexpr int8_t FA_VEC_NE11_DEFAULT = -1; +constexpr int8_t FA_VEC_DOMAIN_DECODE = 0; // ne01 == 1 +constexpr int8_t FA_VEC_DOMAIN_BATCH = 1; // ne01 >= 2 + +struct fa_vec_key_t { + int8_t family; + int8_t dtype; + int16_t dk; + int16_t dv; + int8_t ne11_b; + int8_t ne01_b; +}; + +static_assert(sizeof(fa_vec_key_t) == 8, "fa_vec_key_t must be tightly packed for memcmp"); + +struct fa_vec_cfg_t { + int8_t Q; + int8_t NE; +}; + +struct fa_vec_entry_t { + fa_vec_key_t key; + fa_vec_cfg_t cfg; +}; + +// legal NE values for a (dk,dv): NL = 32/NE, require (dk/4)%NL==0 && (dv/4)%NL==0. +// single source shared by the offline tuner and test-backend-ops. +inline std::vector fa_vec_legal_ne(int dk, int dv) { + std::vector r; + for (int ne : { 1, 2, 4 }) { + const int nl = 32 / ne; + if ((dk / 4) % nl == 0 && (dv / 4) % nl == 0) { + r.push_back(ne); + } + } + return r; +} + +// test/tune-only override; when set, fa_vec_pick returns it directly. +void fa_vec_set_override(fa_vec_cfg_t cfg); +void fa_vec_clear_override(); +fa_vec_cfg_t fa_vec_baseline_cfg(int dk, int dv); + +// Keyed by Apple GPU family; an untuned family matches no row and gets the baseline. +fa_vec_cfg_t fa_vec_pick(int gpu_family, int dtype, int dk, int dv, int64_t ne11, int64_t ne01); + +} // namespace ggml_metal_tuning diff --git a/ggml/src/ggml-metal/ggml-metal.cpp b/ggml/src/ggml-metal/ggml-metal.cpp index ef3c92f2..c6c8ce83 100644 --- a/ggml/src/ggml-metal/ggml-metal.cpp +++ b/ggml/src/ggml-metal/ggml-metal.cpp @@ -4,8 +4,10 @@ #include "ggml-backend-impl.h" #include "ggml-metal-device.h" +#include "ggml-metal-fusion.h" #include "ggml-metal-context.h" #include "ggml-metal-ops.h" +#include "ggml-metal-tuning.h" #include #include @@ -203,6 +205,11 @@ static ggml_backend_buffer_t ggml_backend_metal_buffer_type_alloc_buffer(ggml_ba ggml_metal_device_t ctx_dev = (ggml_metal_device_t)buft->device->context; ggml_metal_buffer_t res = ggml_metal_buffer_init(ctx_dev, size, shared); + if (res == NULL) { + GGML_LOG_ERROR("%s: failed to allocate Metal buffer of %zu bytes (out of memory)\n", __func__, size); + return NULL; + } + ggml_backend_buffer_i buf_i = ggml_metal_buffer_is_shared(res) ? ggml_backend_metal_buffer_shared_i : ggml_backend_metal_buffer_private_i; @@ -219,12 +226,15 @@ static size_t ggml_backend_metal_buffer_type_get_alloc_size(ggml_backend_buffer_ { res += ggml_metal_op_mul_mat_id_extra_tpe(tensor); res += ggml_metal_op_mul_mat_id_extra_ids(tensor); + res += ggml_metal_op_mul_mat_id_extra_amax(tensor); } break; case GGML_OP_FLASH_ATTN_EXT: { res += ggml_metal_op_flash_attn_ext_extra_pad(tensor); res += ggml_metal_op_flash_attn_ext_extra_blk(tensor); res += ggml_metal_op_flash_attn_ext_extra_tmp(tensor); + res += ggml_metal_op_flash_attn_ext_extra_kv_f16(tensor); + res += ggml_metal_op_flash_attn_ext_extra_idx(tensor); } break; case GGML_OP_CUMSUM: case GGML_OP_ARGSORT: @@ -551,7 +561,13 @@ static void ggml_backend_metal_event_wait(ggml_backend_t backend, ggml_backend_e ggml_metal_event_wait(ctx, ev); } -static void ggml_backend_metal_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph) { +static void ggml_backend_metal_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph, ggml_backend_graph_optimize_params * params) { + GGML_ASSERT(params && params->add_alloc_dep); + + // keep the MoE weighted-reduction inputs alive until the fused output so the + // allocator cannot reuse them while the fused kernel is still reading them + ggml_metal_fusion_add_alloc_deps(params->user_data, params->add_alloc_dep, cgraph); + ggml_metal_t ctx = (ggml_metal_t)backend->context; ggml_metal_graph_optimize(ctx, cgraph); @@ -869,10 +885,96 @@ static ggml_backend_feature * ggml_backend_metal_get_features(ggml_backend_reg_t GGML_UNUSED(reg); } +// test/tune-only override for the FA vec (Q, NE) selection, reached via proc_address. +static void ggml_backend_metal_tuning_set_fa_vec_override(int Q, int NE) { + ggml_metal_tuning::fa_vec_set_override({ (int8_t) Q, (int8_t) NE }); +} + +static void ggml_backend_metal_tuning_clear_fa_vec_override(void) { + ggml_metal_tuning::fa_vec_clear_override(); +} + +static int ggml_backend_metal_tuning_fa_vec_ne11_bucket(int64_t ne11) { + return ggml_metal_tuning::fa_vec_ne11_bucket(ne11); +} + +static int ggml_backend_metal_tuning_fa_vec_ne01_bucket(int64_t ne01) { + return ggml_metal_tuning::fa_vec_ne01_bucket(ne01); +} + +static int ggml_backend_metal_tuning_fa_vec_baseline_ne(int dk, int dv) { + return ggml_metal_tuning::fa_vec_baseline_ne(dk, dv); +} + +static const char * ggml_backend_metal_tuning_device_token(ggml_backend_dev_t dev) { + ggml_metal_device_t ctx_dev = (ggml_metal_device_t)dev->context; + + return ggml_metal_device_id_token(ggml_metal_device_get_props(ctx_dev)->device_id); +} + +// generic fusion debugging API (ad-hoc proc-address mechanism): the test resolves the device +// fusion context once and passes that opaque handle to the rest of the functions +typedef void * ggml_backend_fusion_t; + +static ggml_backend_fusion_t ggml_backend_metal_fusion_get(ggml_backend_dev_t dev) { + return ggml_metal_device_get_fusion_info((ggml_metal_device_t)dev->context); +} + +static void ggml_backend_metal_fusion_stats_init(ggml_backend_fusion_t finfo) { + ggml_metal_fusion_info_stats_init((struct ggml_metal_fusion_info *) finfo); +} + +static void ggml_backend_metal_fusion_stats_reset(ggml_backend_fusion_t finfo) { + ggml_metal_fusion_info_stats_reset((struct ggml_metal_fusion_info *) finfo); +} + +static int ggml_backend_metal_fusion_stats_get(ggml_backend_fusion_t finfo, const char ** labels, uint64_t * counts, int n) { + return ggml_metal_fusion_info_stats_get((struct ggml_metal_fusion_info *) finfo, labels, counts, n); +} + +static void ggml_backend_metal_fusion_set_enabled(ggml_backend_fusion_t finfo, bool enabled) { + ggml_metal_fusion_info_set_enabled((struct ggml_metal_fusion_info *) finfo, enabled); +} + static void * ggml_backend_metal_get_proc_address(ggml_backend_reg_t reg, const char * name) { if (strcmp(name, "ggml_backend_get_features") == 0) { return (void *)ggml_backend_metal_get_features; } + if (strcmp(name, "ggml_backend_metal_tuning_set_fa_vec_override") == 0) { + return (void *)ggml_backend_metal_tuning_set_fa_vec_override; + } + if (strcmp(name, "ggml_backend_metal_tuning_clear_fa_vec_override") == 0) { + return (void *)ggml_backend_metal_tuning_clear_fa_vec_override; + } + if (strcmp(name, "ggml_backend_metal_tuning_fa_vec_ne11_bucket") == 0) { + return (void *)ggml_backend_metal_tuning_fa_vec_ne11_bucket; + } + if (strcmp(name, "ggml_backend_metal_tuning_fa_vec_ne01_bucket") == 0) { + return (void *)ggml_backend_metal_tuning_fa_vec_ne01_bucket; + } + if (strcmp(name, "ggml_backend_metal_tuning_fa_vec_baseline_ne") == 0) { + return (void *)ggml_backend_metal_tuning_fa_vec_baseline_ne; + } + if (strcmp(name, "ggml_backend_metal_tuning_device_token") == 0) { + return (void *)ggml_backend_metal_tuning_device_token; + } + // generic fusion debugging API (ad-hoc proc-address mechanism, not part of the official + // ggml backend interface yet; a backend that adopts it exports these exact names) + if (strcmp(name, "ggml_backend_fusion_get") == 0) { + return (void *)ggml_backend_metal_fusion_get; + } + if (strcmp(name, "ggml_backend_fusion_stats_init") == 0) { + return (void *)ggml_backend_metal_fusion_stats_init; + } + if (strcmp(name, "ggml_backend_fusion_stats_reset") == 0) { + return (void *)ggml_backend_metal_fusion_stats_reset; + } + if (strcmp(name, "ggml_backend_fusion_stats_get") == 0) { + return (void *)ggml_backend_metal_fusion_stats_get; + } + if (strcmp(name, "ggml_backend_fusion_set_enabled") == 0) { + return (void *)ggml_backend_metal_fusion_set_enabled; + } return NULL; @@ -890,7 +992,7 @@ static ggml_backend_dev_t ggml_backend_metal_device_init(ggml_backend_reg_t reg, return new ggml_backend_device { /* .iface = */ ggml_backend_metal_device_i, /* .reg = */ reg, - /* .context = */ ggml_metal_device_get(device), + /* .context = */ ggml_metal_device_get(device, g_devices), }; } diff --git a/ggml/src/ggml-metal/ggml-metal.metal b/ggml/src/ggml-metal/ggml-metal.metal deleted file mode 100644 index 243c997f..00000000 --- a/ggml/src/ggml-metal/ggml-metal.metal +++ /dev/null @@ -1,11820 +0,0 @@ -#define GGML_COMMON_DECL_METAL -#define GGML_COMMON_IMPL_METAL -#if defined(GGML_METAL_EMBED_LIBRARY) -__embed_ggml-common.h__ -#else -#include "ggml-common.h" -#endif -#include "ggml-metal-impl.h" - -#include - -#ifdef GGML_METAL_HAS_TENSOR -#include - -#include -#endif - -using namespace metal; - -#define MAX(x, y) ((x) > (y) ? (x) : (y)) -#define MIN(x, y) ((x) < (y) ? (x) : (y)) -#define SWAP(x, y) { auto tmp = (x); (x) = (y); (y) = tmp; } - -#define PAD2(x, n) (((x) + (n) - 1) & ~((n) - 1)) - -#define FOR_UNROLL(x) _Pragma("clang loop unroll(full)") for (x) - -#define N_SIMDWIDTH 32 // assuming SIMD group size is 32 - -// ref: https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf -// -// cmd: -// .../usr/bin/metal -dM -E -c ggml/src/ggml-metal/ggml-metal.metal -// .../usr/bin/metal -dM -E -c -target air64-apple-ios14.0 ggml/src/ggml-metal/ggml-metal.metal -// -#if __METAL_VERSION__ < 310 && defined(GGML_METAL_HAS_BF16) -#undef GGML_METAL_HAS_BF16 -#endif - -#if defined(GGML_METAL_HAS_BF16) -typedef matrix bfloat4x4; -typedef matrix bfloat2x4; -#endif - -#define QK_NL 16 - -constexpr constant static float kvalues_iq4nl_f[16] = { - -127.f, -104.f, -83.f, -65.f, -49.f, -35.f, -22.f, -10.f, 1.f, 13.f, 25.f, 38.f, 53.f, 69.f, 89.f, 113.f -}; - -constexpr constant static float kvalues_mxfp4_f[16] = { - 0, .5f, 1.f, 1.5f, 2.f, 3.f, 4.f, 6.f, -0, -.5f, -1.f, -1.5f, -2.f, -3.f, -4.f, -6.f -}; - -static inline int best_index_int8(int n, constant float * val, float x) { - if (x <= val[0]) return 0; - if (x >= val[n-1]) return n-1; - int ml = 0, mu = n-1; - while (mu-ml > 1) { - int mav = (ml+mu)/2; - if (x < val[mav]) mu = mav; else ml = mav; - } - return x - val[mu-1] < val[mu] - x ? mu-1 : mu; -} - -static inline float e8m0_to_fp32(uint8_t x) { - uint32_t bits; - - if (x == 0) { - bits = 0x00400000; - } else { - bits = (uint32_t) x << 23; - } - - return as_type(bits); -} - -static inline float dot(float x, float y) { - return x*y; -} - -static inline float sum(float x) { - return x; -} - -static inline float sum(float4 x) { - return x[0] + x[1] + x[2] + x[3]; -} - -// NOTE: this is not dequantizing - we are simply fitting the template -template -void dequantize_f32(device const float4x4 * src, short il, thread type4x4 & reg) { - reg = (type4x4)(*src); -} - -template -void dequantize_f32_t4(device const float4 * src, short il, thread type4 & reg) { - reg = (type4)(*src); -} - -template -void dequantize_f16(device const half4x4 * src, short il, thread type4x4 & reg) { - reg = (type4x4)(*src); -} - -template -void dequantize_f16_t4(device const half4 * src, short il, thread type4 & reg) { - reg = (type4)(*(src)); -} - -#if defined(GGML_METAL_HAS_BF16) -template -void dequantize_bf16(device const bfloat4x4 * src, short il, thread type4x4 & reg) { - reg = (type4x4)(*src); -} - -template -void dequantize_bf16_t4(device const bfloat4 * src, short il, thread type4 & reg) { - reg = (type4)(*(src)); -} -#endif - -template -void dequantize_q1_0(device const block_q1_0 * xb, short il, thread type4x4 & reg) { - device const uint8_t * qs = xb->qs; - const float d = xb->d; - const float neg_d = -d; - - const int byte_offset = il * 2; // il*16 bits = il*2 bytes - const uint8_t b0 = qs[byte_offset]; - const uint8_t b1 = qs[byte_offset + 1]; - - float4x4 reg_f; - - reg_f[0][0] = select(neg_d, d, bool(b0 & 0x01)); - reg_f[0][1] = select(neg_d, d, bool(b0 & 0x02)); - reg_f[0][2] = select(neg_d, d, bool(b0 & 0x04)); - reg_f[0][3] = select(neg_d, d, bool(b0 & 0x08)); - reg_f[1][0] = select(neg_d, d, bool(b0 & 0x10)); - reg_f[1][1] = select(neg_d, d, bool(b0 & 0x20)); - reg_f[1][2] = select(neg_d, d, bool(b0 & 0x40)); - reg_f[1][3] = select(neg_d, d, bool(b0 & 0x80)); - - reg_f[2][0] = select(neg_d, d, bool(b1 & 0x01)); - reg_f[2][1] = select(neg_d, d, bool(b1 & 0x02)); - reg_f[2][2] = select(neg_d, d, bool(b1 & 0x04)); - reg_f[2][3] = select(neg_d, d, bool(b1 & 0x08)); - reg_f[3][0] = select(neg_d, d, bool(b1 & 0x10)); - reg_f[3][1] = select(neg_d, d, bool(b1 & 0x20)); - reg_f[3][2] = select(neg_d, d, bool(b1 & 0x40)); - reg_f[3][3] = select(neg_d, d, bool(b1 & 0x80)); - - reg = (type4x4) reg_f; -} - -template -void dequantize_q1_0_t4(device const block_q1_0 * xb, short il, thread type4 & reg) { - const float d = xb->d; - const float neg_d = -d; - const int base = il * 4; - const uint8_t byte = xb->qs[base / 8]; - const int s = base % 8; - - float4 reg_f; - reg_f[0] = select(neg_d, d, bool((byte >> (s )) & 1)); - reg_f[1] = select(neg_d, d, bool((byte >> (s + 1)) & 1)); - reg_f[2] = select(neg_d, d, bool((byte >> (s + 2)) & 1)); - reg_f[3] = select(neg_d, d, bool((byte >> (s + 3)) & 1)); - - reg = (type4) reg_f; -} - -template -void dequantize_q2_0(device const block_q2_0 * xb, short il, thread type4x4 & reg) { - device const uint8_t * qs = xb->qs; - const float d = xb->d; - - const int byte_offset = il * 4; // il*16 elements = il*4 bytes (4 elements per byte) - float4x4 reg_f; - - for (int i = 0; i < 4; i++) { - const uint8_t b = qs[byte_offset + i]; - reg_f[i][0] = ((float)((b >> 0) & 3) - 1.0f) * d; - reg_f[i][1] = ((float)((b >> 2) & 3) - 1.0f) * d; - reg_f[i][2] = ((float)((b >> 4) & 3) - 1.0f) * d; - reg_f[i][3] = ((float)((b >> 6) & 3) - 1.0f) * d; - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q2_0_t4(device const block_q2_0 * xb, short il, thread type4 & reg) { - const float d = xb->d; - const uint8_t b = xb->qs[il]; - - float4 reg_f; - reg_f[0] = ((float)((b >> 0) & 3) - 1.0f) * d; - reg_f[1] = ((float)((b >> 2) & 3) - 1.0f) * d; - reg_f[2] = ((float)((b >> 4) & 3) - 1.0f) * d; - reg_f[3] = ((float)((b >> 6) & 3) - 1.0f) * d; - - reg = (type4) reg_f; -} - -template -void dequantize_q4_0(device const block_q4_0 * xb, short il, thread type4x4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 1); - const float d1 = il ? (xb->d / 16.h) : xb->d; - const float d2 = d1 / 256.f; - const float md = -8.h * xb->d; - const ushort mask0 = il ? 0x00F0 : 0x000F; - const ushort mask1 = mask0 << 8; - - float4x4 reg_f; - - for (int i = 0; i < 8; i++) { - reg_f[i/2][2*(i%2) + 0] = d1 * (qs[i] & mask0) + md; - reg_f[i/2][2*(i%2) + 1] = d2 * (qs[i] & mask1) + md; - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q4_0_t4(device const block_q4_0 * xb, short il, thread type4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 1); - const float d1 = (il/4) ? (xb->d / 16.h) : xb->d; - const float d2 = d1 / 256.f; - const float md = -8.h * xb->d; - const ushort mask0 = (il/4) ? 0x00F0 : 0x000F; - const ushort mask1 = mask0 << 8; - - for (int i = 0; i < 2; i++) { - reg[2*i + 0] = d1 * (qs[2*(il%4) + i] & mask0) + md; - reg[2*i + 1] = d2 * (qs[2*(il%4) + i] & mask1) + md; - } -} - -void quantize_q1_0(device const float * src, device block_q1_0 & dst) { - float sum_abs = 0.0f; - for (int j = 0; j < QK1_0; j++) { - sum_abs += fabs(src[j]); - } - dst.d = sum_abs / QK1_0; - - for (int j = 0; j < QK1_0 / 8; j++) { - dst.qs[j] = 0; - } - for (int j = 0; j < QK1_0; j++) { - if (src[j] >= 0.0f) { - dst.qs[j / 8] |= (1 << (j % 8)); - } - } -} - -void quantize_q2_0(device const float * src, device block_q2_0 & dst) { - float amax = 0.0f; - for (int j = 0; j < QK2_0; j++) { - float a = fabs(src[j]); - if (a > amax) amax = a; - } - const float d = amax; - dst.d = d; - - const float id = d > 0.0f ? 1.0f / d : 0.0f; - - for (int j = 0; j < QK2_0 / 4; j++) { - dst.qs[j] = 0; - } - for (int j = 0; j < QK2_0; j++) { - int q = (int)round(src[j] * id) + 1; - q = max(0, min(3, q)); - dst.qs[j / 4] |= (q << (2 * (j % 4))); - } -} - -void quantize_q4_0(device const float * src, device block_q4_0 & dst) { -#pragma METAL fp math_mode(safe) - float amax = 0.0f; // absolute max - float max = 0.0f; - - for (int j = 0; j < QK4_0; j++) { - const float v = src[j]; - if (amax < fabs(v)) { - amax = fabs(v); - max = v; - } - } - - const float d = max / -8; - const float id = d ? 1.0f/d : 0.0f; - - dst.d = d; - - for (int j = 0; j < QK4_0/2; ++j) { - const float x0 = src[0 + j]*id; - const float x1 = src[QK4_0/2 + j]*id; - - const uint8_t xi0 = MIN(15, (int8_t)(x0 + 8.5f)); - const uint8_t xi1 = MIN(15, (int8_t)(x1 + 8.5f)); - - dst.qs[j] = xi0; - dst.qs[j] |= xi1 << 4; - } -} - -void quantize_q4_1(device const float * src, device block_q4_1 & dst) { -#pragma METAL fp math_mode(safe) - float min = FLT_MAX; - float max = -FLT_MAX; - - for (int j = 0; j < QK4_1; j++) { - const float v = src[j]; - if (min > v) min = v; - if (max < v) max = v; - } - - const float d = (max - min) / ((1 << 4) - 1); - const float id = d ? 1.0f/d : 0.0f; - - dst.d = d; - dst.m = min; - - for (int j = 0; j < QK4_1/2; ++j) { - const float x0 = (src[0 + j] - min)*id; - const float x1 = (src[QK4_1/2 + j] - min)*id; - - const uint8_t xi0 = MIN(15, (int8_t)(x0 + 0.5f)); - const uint8_t xi1 = MIN(15, (int8_t)(x1 + 0.5f)); - - dst.qs[j] = xi0; - dst.qs[j] |= xi1 << 4; - } -} - -void quantize_q5_0(device const float * src, device block_q5_0 & dst) { -#pragma METAL fp math_mode(safe) - float amax = 0.0f; // absolute max - float max = 0.0f; - - for (int j = 0; j < QK5_0; j++) { - const float v = src[j]; - if (amax < fabs(v)) { - amax = fabs(v); - max = v; - } - } - - const float d = max / -16; - const float id = d ? 1.0f/d : 0.0f; - - dst.d = d; - - uint32_t qh = 0; - for (int j = 0; j < QK5_0/2; ++j) { - const float x0 = src[0 + j]*id; - const float x1 = src[QK5_0/2 + j]*id; - - const uint8_t xi0 = MIN(31, (int8_t)(x0 + 16.5f)); - const uint8_t xi1 = MIN(31, (int8_t)(x1 + 16.5f)); - - dst.qs[j] = (xi0 & 0xf) | ((xi1 & 0xf) << 4); - qh |= ((xi0 & 0x10u) >> 4) << (j + 0); - qh |= ((xi1 & 0x10u) >> 4) << (j + QK5_0/2); - } - - thread const uint8_t * qh8 = (thread const uint8_t *)&qh; - - for (int j = 0; j < 4; ++j) { - dst.qh[j] = qh8[j]; - } -} - -void quantize_q5_1(device const float * src, device block_q5_1 & dst) { -#pragma METAL fp math_mode(safe) - float max = src[0]; - float min = src[0]; - - for (int j = 1; j < QK5_1; j++) { - const float v = src[j]; - min = v < min ? v : min; - max = v > max ? v : max; - } - - const float d = (max - min) / 31; - const float id = d ? 1.0f/d : 0.0f; - - dst.d = d; - dst.m = min; - - uint32_t qh = 0; - for (int j = 0; j < QK5_1/2; ++j) { - const float x0 = (src[0 + j] - min)*id; - const float x1 = (src[QK5_1/2 + j] - min)*id; - - const uint8_t xi0 = (uint8_t)(x0 + 0.5f); - const uint8_t xi1 = (uint8_t)(x1 + 0.5f); - - dst.qs[j] = (xi0 & 0xf) | ((xi1 & 0xf) << 4); - qh |= ((xi0 & 0x10u) >> 4) << (j + 0); - qh |= ((xi1 & 0x10u) >> 4) << (j + QK5_1/2); - } - - thread const uint8_t * qh8 = (thread const uint8_t *)&qh; - - for (int j = 0; j < 4; ++j) { - dst.qh[j] = qh8[j]; - } -} - -void quantize_q8_0(device const float * src, device block_q8_0 & dst) { -#pragma METAL fp math_mode(safe) - float amax = 0.0f; // absolute max - - for (int j = 0; j < QK8_0; j++) { - const float v = src[j]; - amax = MAX(amax, fabs(v)); - } - - const float d = amax / ((1 << 7) - 1); - const float id = d ? 1.0f/d : 0.0f; - - dst.d = d; - - for (int j = 0; j < QK8_0; ++j) { - const float x0 = src[j]*id; - - dst.qs[j] = round(x0); - } -} - -void quantize_iq4_nl(device const float * src, device block_iq4_nl & dst) { -#pragma METAL fp math_mode(safe) - float amax = 0.0f; // absolute max - float max = 0.0f; - - for (int j = 0; j < QK4_NL; j++) { - const float v = src[j]; - if (amax < fabs(v)) { - amax = fabs(v); - max = v; - } - } - - const float d = max / kvalues_iq4nl_f[0]; - const float id = d ? 1.0f/d : 0.0f; - - float sumqx = 0, sumq2 = 0; - for (int j = 0; j < QK4_NL/2; ++j) { - const float x0 = src[0 + j]*id; - const float x1 = src[QK4_NL/2 + j]*id; - - const uint8_t xi0 = best_index_int8(16, kvalues_iq4nl_f, x0); - const uint8_t xi1 = best_index_int8(16, kvalues_iq4nl_f, x1); - - dst.qs[j] = xi0 | (xi1 << 4); - - const float v0 = kvalues_iq4nl_f[xi0]; - const float v1 = kvalues_iq4nl_f[xi1]; - const float w0 = src[0 + j]*src[0 + j]; - const float w1 = src[QK4_NL/2 + j]*src[QK4_NL/2 + j]; - sumqx += w0*v0*src[j] + w1*v1*src[QK4_NL/2 + j]; - sumq2 += w0*v0*v0 + w1*v1*v1; - - } - - dst.d = sumq2 > 0 ? sumqx/sumq2 : d; -} - -void quantize_tq2_0(device const float * src, device block_tq2_0 & dst) { -#pragma METAL fp math_mode(safe) - float amax = 0.0f; // absolute max - - for (int j = 0; j < QK_K; j++) { - const float v = src[j]; - amax = MAX(amax, fabs(v)); - } - - const float d = amax; - const float id = d ? 1.0f/d : 0.0f; - - dst.d = (half) d; - - for (int j = 0; j < QK_K/4; j += 32) { - for (int m = 0; m < 32; ++m) { - uint8_t q = 0; - for (int n = 0; n < 4; ++n) { - // -1, 0, 1 -> 0, 1, 2 - int xi = (int)round(src[m + n*32] * id) + 1; - q += (uint8_t)((xi & 3) << (2*n)); - } - dst.qs[j + m] = q; - } - src += 4*32; - } -} - -template -void dequantize_q4_1(device const block_q4_1 * xb, short il, thread type4x4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 2); - const float d1 = il ? (xb->d / 16.h) : xb->d; - const float d2 = d1 / 256.f; - const float m = xb->m; - const ushort mask0 = il ? 0x00F0 : 0x000F; - const ushort mask1 = mask0 << 8; - - float4x4 reg_f; - - for (int i = 0; i < 8; i++) { - reg_f[i/2][2*(i%2) + 0] = ((qs[i] & mask0) * d1) + m; - reg_f[i/2][2*(i%2) + 1] = ((qs[i] & mask1) * d2) + m; - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q4_1_t4(device const block_q4_1 * xb, short il, thread type4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 2); - const float d1 = (il/4) ? (xb->d / 16.h) : xb->d; - const float d2 = d1 / 256.f; - const float m = xb->m; - const ushort mask0 = (il/4) ? 0x00F0 : 0x000F; - const ushort mask1 = mask0 << 8; - - for (int i = 0; i < 2; i++) { - reg[2*i + 0] = d1 * (qs[2*(il%4) + i] & mask0) + m; - reg[2*i + 1] = d2 * (qs[2*(il%4) + i] & mask1) + m; - } -} - -template -void dequantize_q5_0(device const block_q5_0 * xb, short il, thread type4x4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 3); - const float d = xb->d; - const float md = -16.h * xb->d; - const ushort mask = il ? 0x00F0 : 0x000F; - - const uint32_t qh = *((device const uint32_t *)xb->qh); - - const int x_mv = il ? 4 : 0; - - const int gh_mv = il ? 12 : 0; - const int gh_bk = il ? 0 : 4; - - float4x4 reg_f; - - for (int i = 0; i < 8; i++) { - // extract the 5-th bits for x0 and x1 - const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; - const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; - - // combine the 4-bits from qs with the 5th bit - const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); - const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); - - reg_f[i/2][2*(i%2) + 0] = d * x0 + md; - reg_f[i/2][2*(i%2) + 1] = d * x1 + md; - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q5_0_t4(device const block_q5_0 * xb, short il, thread type4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 3); - const float d = xb->d; - const float md = -16.h * xb->d; - const ushort mask = (il/4) ? 0x00F0 : 0x000F; - - const uint32_t qh = *((device const uint32_t *)xb->qh); - - const int x_mv = (il/4) ? 4 : 0; - - const int gh_mv = (il/4) ? 12 : 0; - const int gh_bk = (il/4) ? 0 : 4; - - for (int ii = 0; ii < 2; ii++) { - int i = 2*(il%4) + ii; - - // extract the 5-th bits for x0 and x1 - const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; - const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; - - // combine the 4-bits from qs with the 5th bit - const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); - const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); - - reg[2*ii + 0] = d * x0 + md; - reg[2*ii + 1] = d * x1 + md; - } -} - -template -void dequantize_q5_1(device const block_q5_1 * xb, short il, thread type4x4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 4); - const float d = xb->d; - const float m = xb->m; - const ushort mask = il ? 0x00F0 : 0x000F; - - const uint32_t qh = *((device const uint32_t *)xb->qh); - - const int x_mv = il ? 4 : 0; - - const int gh_mv = il ? 12 : 0; - const int gh_bk = il ? 0 : 4; - - float4x4 reg_f; - - for (int i = 0; i < 8; i++) { - // extract the 5-th bits for x0 and x1 - const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; - const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; - - // combine the 4-bits from qs with the 5th bit - const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); - const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); - - reg_f[i/2][2*(i%2) + 0] = d * x0 + m; - reg_f[i/2][2*(i%2) + 1] = d * x1 + m; - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q5_1_t4(device const block_q5_1 * xb, short il, thread type4 & reg) { - device const uint16_t * qs = ((device const uint16_t *)xb + 4); - const float d = xb->d; - const float m = xb->m; - const ushort mask = (il/4) ? 0x00F0 : 0x000F; - - const uint32_t qh = *((device const uint32_t *)xb->qh); - - const int x_mv = (il/4) ? 4 : 0; - - const int gh_mv = (il/4) ? 12 : 0; - const int gh_bk = (il/4) ? 0 : 4; - - for (int ii = 0; ii < 2; ii++) { - int i = 2*(il%4) + ii; - - // extract the 5-th bits for x0 and x1 - const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; - const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; - - // combine the 4-bits from qs with the 5th bit - const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); - const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); - - reg[2*ii + 0] = d * x0 + m; - reg[2*ii + 1] = d * x1 + m; - } -} - -template -void dequantize_q8_0(device const block_q8_0 *xb, short il, thread type4x4 & reg) { - device const int8_t * qs = ((device const int8_t *)xb->qs); - const float d = xb->d; - - float4x4 reg_f; - - for (int i = 0; i < 16; i++) { - reg_f[i/4][i%4] = (qs[i + 16*il] * d); - } - - reg = (type4x4) reg_f; -} - -template -void dequantize_q8_0_t4(device const block_q8_0 *xb, short il, thread type4 & reg) { - device const int8_t * qs = ((device const int8_t *)xb->qs); - const float d = xb->d; - - for (int i = 0; i < 4; i++) { - reg[i] = (qs[4*(il%4) + i + 16*(il/4)] * d); - } -} - -template -void dequantize_mxfp4(device const block_mxfp4 * xb, short il, thread type4x4 & reg) { - device const uint8_t * q2 = (device const uint8_t *)xb->qs; - - const float d = e8m0_to_fp32(xb->e); - const uint8_t shr = il >= 1 ? 4 : 0; - - for (int i = 0; i < 4; ++i) { - reg[i][0] = d * kvalues_mxfp4_f[(q2[4*i + 0] >> shr) & 0x0F]; - reg[i][1] = d * kvalues_mxfp4_f[(q2[4*i + 1] >> shr) & 0x0F]; - reg[i][2] = d * kvalues_mxfp4_f[(q2[4*i + 2] >> shr) & 0x0F]; - reg[i][3] = d * kvalues_mxfp4_f[(q2[4*i + 3] >> shr) & 0x0F]; - } -} - -template -void dequantize_mxfp4_t4(device const block_mxfp4 * xb, short il, thread type4 & reg) { - device const uint8_t * q2 = (device const uint8_t *)xb->qs; - - const float d = e8m0_to_fp32(xb->e); - const short il4 = il%4; - - const uint8_t shr = il >= 4 ? 4 : 0; - - reg[0] = d * kvalues_mxfp4_f[(q2[4*il4 + 0] >> shr) & 0x0F]; - reg[1] = d * kvalues_mxfp4_f[(q2[4*il4 + 1] >> shr) & 0x0F]; - reg[2] = d * kvalues_mxfp4_f[(q2[4*il4 + 2] >> shr) & 0x0F]; - reg[3] = d * kvalues_mxfp4_f[(q2[4*il4 + 3] >> shr) & 0x0F]; -} - -template -void dequantize_q2_K(device const block_q2_K *xb, short il, thread type4x4 & reg) { - const float d = xb->d; - const float min = xb->dmin; - device const uint8_t * q = (device const uint8_t *)xb->qs; - float dl, ml; - uint8_t sc = xb->scales[il]; - - q = q + 32*(il/8) + 16*(il&1); - il = (il/2)%4; - - half coef = il>1 ? (il>2 ? 1/64.h : 1/16.h) : (il>0 ? 1/4.h : 1.h); - uchar mask = il>1 ? (il>2 ? 192 : 48) : (il>0 ? 12 : 3); - dl = d * (sc & 0xF) * coef, ml = min * (sc >> 4); - for (int i = 0; i < 16; ++i) { - reg[i/4][i%4] = dl * (q[i] & mask) - ml; - } -} - -template -void dequantize_q3_K(device const block_q3_K *xb, short il, thread type4x4 & reg) { - const half d_all = xb->d; - device const uint8_t * q = (device const uint8_t *)xb->qs; - device const uint8_t * h = (device const uint8_t *)xb->hmask; - device const int8_t * scales = (device const int8_t *)xb->scales; - - q = q + 32 * (il/8) + 16 * (il&1); - h = h + 16 * (il&1); - uint8_t m = 1 << (il/2); - uint16_t kmask1 = (il/4)>1 ? ((il/4)>2 ? 192 : 48) : \ - ((il/4)>0 ? 12 : 3); - uint16_t kmask2 = il/8 ? 0xF0 : 0x0F; - uint16_t scale_2 = scales[il%8], scale_1 = scales[8 + il%4]; - int16_t dl_int = (il/4)&1 ? (scale_2&kmask2) | ((scale_1&kmask1) << 2) - : (scale_2&kmask2) | ((scale_1&kmask1) << 4); - float dl = il<8 ? d_all * (dl_int - 32.f) : d_all * (dl_int / 16.f - 32.f); - const float ml = 4.f * dl; - - il = (il/2) & 3; - const half coef = il>1 ? (il>2 ? 1/64.h : 1/16.h) : (il>0 ? 1/4.h : 1.h); - const uint8_t mask = il>1 ? (il>2 ? 192 : 48) : (il>0 ? 12 : 3); - dl *= coef; - - for (int i = 0; i < 16; ++i) { - reg[i/4][i%4] = dl * (q[i] & mask) - (h[i] & m ? 0 : ml); - } -} - -static inline uchar2 get_scale_min_k4_just2(int j, int k, device const uchar * q) { - return j < 4 ? uchar2{uchar(q[j+0+k] & 63), uchar(q[j+4+k] & 63)} - : uchar2{uchar((q[j+4+k] & 0xF) | ((q[j-4+k] & 0xc0) >> 2)), uchar((q[j+4+k] >> 4) | ((q[j-0+k] & 0xc0) >> 2))}; -} - -template -void dequantize_q4_K(device const block_q4_K * xb, short il, thread type4x4 & reg) { - device const uchar * q = xb->qs; - - short is = (il/4) * 2; - q = q + (il/4) * 32 + 16 * (il&1); - il = il & 3; - const uchar2 sc = get_scale_min_k4_just2(is, il/2, xb->scales); - const float d = il < 2 ? xb->d : xb->d / 16.h; - const float min = xb->dmin; - const float dl = d * sc[0]; - const float ml = min * sc[1]; - - const ushort mask = il < 2 ? 0x0F : 0xF0; - for (int i = 0; i < 16; ++i) { - reg[i/4][i%4] = dl * (q[i] & mask) - ml; - } -} - -template -void dequantize_q5_K(device const block_q5_K *xb, short il, thread type4x4 & reg) { - device const uint8_t * q = xb->qs; - device const uint8_t * qh = xb->qh; - - short is = (il/4) * 2; - q = q + 32 * (il/4) + 16 * (il&1); - qh = qh + 16 * (il&1); - uint8_t ul = 1 << (il/2); - il = il & 3; - const uchar2 sc = get_scale_min_k4_just2(is, il/2, xb->scales); - const float d = il < 2 ? xb->d : xb->d / 16.f; - const float min = xb->dmin; - const float dl = d * sc[0]; - const float ml = min * sc[1]; - - const ushort mask = il<2 ? 0x0F : 0xF0; - const float qh_val = il<2 ? 16.f : 256.f; - for (int i = 0; i < 16; ++i) { - reg[i/4][i%4] = dl * ((q[i] & mask) + (qh[i] & ul ? qh_val : 0)) - ml; - } -} - -template -void dequantize_q6_K(device const block_q6_K *xb, short il, thread type4x4 & reg) { - const half d_all = xb->d; - device const uint16_t * ql = (device const uint16_t *)xb->ql; - device const uint16_t * qh = (device const uint16_t *)xb->qh; - device const int8_t * scales = (device const int8_t *)xb->scales; - - ql = ql + 32*(il/8) + 16*((il/2)&1) + 8*(il&1); - qh = qh + 16*(il/8) + 8*(il&1); - float sc = scales[(il%2) + 2 * ((il/2))]; - il = (il/2) & 3; - - const uint32_t kmask1 = il>1 ? (il>2 ? 0xC0C0C0C0 : 0x30303030) : (il>0 ? 0x0C0C0C0C : 0x03030303); - const uint32_t kmask2 = il>1 ? 0xF0F0F0F0 : 0x0F0F0F0F; - const float ml = d_all * sc * 32.f; - const float dl0 = d_all * sc; - const float dl1 = dl0 / 256.f; - const float dl2 = dl0 / (256.f * 256.f); - const float dl3 = dl0 / (256.f * 256.f * 256.f); - const uint8_t shr_h = il>2 ? 2 : 0; - const uint8_t shl_h = il>1 ? 0 : (il>0 ? 2 : 4); - const uint8_t shr_l = il>1 ? 4 : 0; - for (int i = 0; i < 4; ++i) { - const uint32_t low = (ql[2*i] | (uint32_t)(ql[2*i+1] << 16)) & kmask2; - const uint32_t high = (qh[2*i] | (uint32_t)(qh[2*i+1] << 16)) & kmask1; - const uint32_t q = ((high << shl_h) >> shr_h) | (low >> shr_l); - reg[i][0] = dl0 * ((half)(q & 0xFF)) - ml; - reg[i][1] = dl1 * ((float)(q & 0xFF00)) - ml; - reg[i][2] = dl2 * ((float)(q & 0xFF0000)) - ml; - reg[i][3] = dl3 * ((float)(q & 0xFF000000)) - ml; - } -} - -template -void dequantize_iq2_xxs(device const block_iq2_xxs * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const float d = xb->d; - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - // each block of 32 needs 2 uint32_t's for the quants & scale, so 4 uint16_t's. - device const uint16_t * q2 = xb->qs + 4*ib32; - const uint32_t aux32_g = q2[0] | (q2[1] << 16); - const uint32_t aux32_s = q2[2] | (q2[3] << 16); - thread const uint8_t * aux8 = (thread const uint8_t *)&aux32_g; - const float dl = d * (0.5f + (aux32_s >> 28)) * 0.25f; - constant uint8_t * grid = (constant uint8_t *)(iq2xxs_grid + aux8[2*il+0]); - uint8_t signs = ksigns_iq2xs[(aux32_s >> 14*il) & 127]; - for (int i = 0; i < 8; ++i) { - reg[i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); - } - grid = (constant uint8_t *)(iq2xxs_grid + aux8[2*il+1]); - signs = ksigns_iq2xs[(aux32_s >> (14*il+7)) & 127]; - for (int i = 0; i < 8; ++i) { - reg[2+i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); - } -} - -template -void dequantize_iq2_xs(device const block_iq2_xs * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const float d = xb->d; - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - device const uint16_t * q2 = xb->qs + 4*ib32; - const float dl = d * (0.5f + ((xb->scales[ib32] >> 4*il) & 0xf)) * 0.25f; - constant uint8_t * grid = (constant uint8_t *)(iq2xs_grid + (q2[2*il+0] & 511)); - uint8_t signs = ksigns_iq2xs[q2[2*il+0] >> 9]; - for (int i = 0; i < 8; ++i) { - reg[i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); - } - grid = (constant uint8_t *)(iq2xs_grid + (q2[2*il+1] & 511)); - signs = ksigns_iq2xs[q2[2*il+1] >> 9]; - for (int i = 0; i < 8; ++i) { - reg[2+i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); - } -} - -template -void dequantize_iq3_xxs(device const block_iq3_xxs * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const float d = xb->d; - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - device const uint8_t * q3 = xb->qs + 8*ib32; - device const uint16_t * gas = (device const uint16_t *)(xb->qs + QK_K/4) + 2*ib32; - const uint32_t aux32 = gas[0] | (gas[1] << 16); - const float dl = d * (0.5f + (aux32 >> 28)) * 0.5f; - constant uint8_t * grid1 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+0]); - constant uint8_t * grid2 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+1]); - uint8_t signs = ksigns_iq2xs[(aux32 >> 14*il) & 127]; - for (int i = 0; i < 4; ++i) { - reg[0][i] = dl * grid1[i] * (signs & kmask_iq2xs[i+0] ? -1.f : 1.f); - reg[1][i] = dl * grid2[i] * (signs & kmask_iq2xs[i+4] ? -1.f : 1.f); - } - grid1 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+2]); - grid2 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+3]); - signs = ksigns_iq2xs[(aux32 >> (14*il+7)) & 127]; - for (int i = 0; i < 4; ++i) { - reg[2][i] = dl * grid1[i] * (signs & kmask_iq2xs[i+0] ? -1.f : 1.f); - reg[3][i] = dl * grid2[i] * (signs & kmask_iq2xs[i+4] ? -1.f : 1.f); - } -} - -template -void dequantize_iq3_s(device const block_iq3_s * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const float d = xb->d; - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - device const uint8_t * qs = xb->qs + 8*ib32; - device const uint8_t * signs = xb->signs + 4*ib32 + 2*il; - const uint8_t qh = xb->qh[ib32] >> 4*il; - const float dl = d * (1 + 2*((xb->scales[ib32/2] >> 4*(ib32%2)) & 0xf)); - constant uint8_t * grid1 = (constant uint8_t *)(iq3s_grid + (qs[4*il+0] | ((qh << 8) & 256))); - constant uint8_t * grid2 = (constant uint8_t *)(iq3s_grid + (qs[4*il+1] | ((qh << 7) & 256))); - for (int i = 0; i < 4; ++i) { - reg[0][i] = dl * grid1[i] * select(1, -1, signs[0] & kmask_iq2xs[i+0]); - reg[1][i] = dl * grid2[i] * select(1, -1, signs[0] & kmask_iq2xs[i+4]); - } - grid1 = (constant uint8_t *)(iq3s_grid + (qs[4*il+2] | ((qh << 6) & 256))); - grid2 = (constant uint8_t *)(iq3s_grid + (qs[4*il+3] | ((qh << 5) & 256))); - for (int i = 0; i < 4; ++i) { - reg[2][i] = dl * grid1[i] * select(1, -1, signs[1] & kmask_iq2xs[i+0]); - reg[3][i] = dl * grid2[i] * select(1, -1, signs[1] & kmask_iq2xs[i+4]); - } -} - -template -void dequantize_iq2_s(device const block_iq2_s * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const float d = xb->d; - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; - device const uint8_t * signs = qs + QK_K/8; - const uint8_t qh = xb->qh[ib32] >> 4*il; - const float dl = d * (0.5f + ((xb->scales[ib32] >> 4*il) & 0xf)) * 0.25f; - constant uint8_t * grid1 = (constant uint8_t *)(iq2s_grid + (qs[0] | ((qh << 8) & 0x300))); - constant uint8_t * grid2 = (constant uint8_t *)(iq2s_grid + (qs[1] | ((qh << 6) & 0x300))); - for (int i = 0; i < 8; ++i) { - reg[i/4+0][i%4] = dl * grid1[i] * select(1, -1, signs[0] & kmask_iq2xs[i]); - reg[i/4+2][i%4] = dl * grid2[i] * select(1, -1, signs[1] & kmask_iq2xs[i]); - } -} - -template -void dequantize_iq1_s(device const block_iq1_s * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const int ib32 = il/2; - il = il%2; - const float d = xb->d; - device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; - device const uint16_t * qh = xb->qh; - const float dl = d * (2*((qh[ib32] >> 12) & 7) + 1); - const float ml = dl * (qh[ib32] & 0x8000 ? -1 - IQ1S_DELTA : -1 + IQ1S_DELTA); - const uint16_t h = qh[ib32] >> 6*il; - constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((h << 8) & 0x700))); - constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((h << 5) & 0x700))); - for (int i = 0; i < 4; ++i) { - reg[0][i] = dl * (grid1[i] & 0xf) + ml; - reg[1][i] = dl * (grid1[i] >> 4) + ml; - reg[2][i] = dl * (grid2[i] & 0xf) + ml; - reg[3][i] = dl * (grid2[i] >> 4) + ml; - } -} - -template -void dequantize_iq1_m(device const block_iq1_m * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const int ib32 = il/2; - il = il%2; - device const uint16_t * sc = (device const uint16_t *)xb->scales; - - iq1m_scale_t scale; - scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000); - const float d = scale.f16; - - device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; - device const uint8_t * qh = xb->qh + 2*ib32 + il; - - const float dl = d * (2*((sc[ib32/2] >> (6*(ib32%2)+3*il)) & 7) + 1); - const float ml1 = dl * (qh[0] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); - const float ml2 = dl * (qh[0] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); - constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); - constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 4) & 0x700))); - for (int i = 0; i < 4; ++i) { - reg[0][i] = dl * (grid1[i] & 0xf) + ml1; - reg[1][i] = dl * (grid1[i] >> 4) + ml1; - reg[2][i] = dl * (grid2[i] & 0xf) + ml2; - reg[3][i] = dl * (grid2[i] >> 4) + ml2; - } -} - -template -void dequantize_iq4_nl(device const block_iq4_nl * xb, short il, thread type4x4 & reg) { - device const uint16_t * q4 = (device const uint16_t *)xb->qs; - const float d = xb->d; - uint32_t aux32; - thread const uint8_t * q8 = (thread const uint8_t *)&aux32; - for (int i = 0; i < 4; ++i) { - aux32 = ((q4[2*i] | (q4[2*i+1] << 16)) >> 4*il) & 0x0f0f0f0f; - reg[i][0] = d * kvalues_iq4nl_f[q8[0]]; - reg[i][1] = d * kvalues_iq4nl_f[q8[1]]; - reg[i][2] = d * kvalues_iq4nl_f[q8[2]]; - reg[i][3] = d * kvalues_iq4nl_f[q8[3]]; - } -} - -template -void dequantize_iq4_nl_t4(device const block_iq4_nl * xb, short il, thread type4 & reg) { - device const uint16_t * q4 = (device const uint16_t *)xb->qs; - const float d = xb->d; - uint32_t aux32; - thread const uint8_t * q8 = (thread const uint8_t *)&aux32; - aux32 = ((q4[2*(il%4)] | (q4[2*(il%4)+1] << 16)) >> 4*(il/4)) & 0x0f0f0f0f; - reg[0] = d * kvalues_iq4nl_f[q8[0]]; - reg[1] = d * kvalues_iq4nl_f[q8[1]]; - reg[2] = d * kvalues_iq4nl_f[q8[2]]; - reg[3] = d * kvalues_iq4nl_f[q8[3]]; -} - -template -void dequantize_iq4_xs(device const block_iq4_xs * xb, short il, thread type4x4 & reg) { - // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 - const int ib32 = il/2; - il = il%2; - // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 - device const uint32_t * q4 = (device const uint32_t *)xb->qs + 4*ib32; - const int ls = ((xb->scales_l[ib32/2] >> 4*(ib32%2)) & 0xf) | (((xb->scales_h >> 2*ib32) & 3) << 4); - const float d = (float)xb->d * (ls - 32); - uint32_t aux32; - thread const uint8_t * q8 = (thread const uint8_t *)&aux32; - for (int i = 0; i < 4; ++i) { - aux32 = (q4[i] >> 4*il) & 0x0f0f0f0f; - reg[i][0] = d * kvalues_iq4nl_f[q8[0]]; - reg[i][1] = d * kvalues_iq4nl_f[q8[1]]; - reg[i][2] = d * kvalues_iq4nl_f[q8[2]]; - reg[i][3] = d * kvalues_iq4nl_f[q8[3]]; - } -} - -template -void dequantize_tq2_0(device const block_tq2_0 * xb, short il, thread type4x4 & reg) { - device const uint8_t * qs = xb->qs; - const float d = xb->d; - - float4x4 reg_f; - - // 2 bits per element, 4 elements per byte, 128 elements per 32-byte group - const short base = il * 16; - for (int k = 0; k < 16; k++) { - const int i = base + k; - const int byte = ((i >> 7) & 1) * 32 + (i & 31); - const int l = (i >> 5) & 3; - reg_f[k/4][k%4] = d * (float)(((qs[byte] >> (2*l)) & 3) - 1); - } - - reg = (type4x4) reg_f; -} - -enum ggml_sort_order { - GGML_SORT_ORDER_ASC, - GGML_SORT_ORDER_DESC, -}; - -constant float GELU_COEF_A = 0.044715f; -constant float GELU_QUICK_COEF = -1.702f; -constant float SQRT_2_OVER_PI = 0.79788456080286535587989211986876f; -constant float SQRT_2_INV = 0.70710678118654752440084436210484f; - -// based on Abramowitz and Stegun formula 7.1.26 or similar Hastings' approximation -// ref: https://www.johndcook.com/blog/python_erf/ -constant float p_erf = 0.3275911f; -constant float a1_erf = 0.254829592f; -constant float a2_erf = -0.284496736f; -constant float a3_erf = 1.421413741f; -constant float a4_erf = -1.453152027f; -constant float a5_erf = 1.061405429f; - -template -inline T erf_approx(T x) { - T sign_x = sign(x); - x = fabs(x); - T t = 1.0f / (1.0f + p_erf * x); - T y = 1.0f - (((((a5_erf * t + a4_erf) * t) + a3_erf) * t + a2_erf) * t + a1_erf) * t * exp(-x * x); - return sign_x * y; -} - -template T elu_approx(T x); - -template<> inline float elu_approx(float x) { - return (x > 0.f) ? x : (exp(x) - 1); -} - -template<> inline float4 elu_approx(float4 x) { - float4 res; - - res[0] = (x[0] > 0.0f) ? x[0] : (exp(x[0]) - 1.0f); - res[1] = (x[1] > 0.0f) ? x[1] : (exp(x[1]) - 1.0f); - res[2] = (x[2] > 0.0f) ? x[2] : (exp(x[2]) - 1.0f); - res[3] = (x[3] > 0.0f) ? x[3] : (exp(x[3]) - 1.0f); - - return res; -} - -constant short FC_unary_op [[function_constant(FC_UNARY + 0)]]; -constant bool FC_unary_cnt[[function_constant(FC_UNARY + 1)]]; - -template -kernel void kernel_unary_impl( - constant ggml_metal_kargs_unary & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { -#define FC_OP FC_unary_op -#define FC_CNT FC_unary_cnt - - device const T0 * src0_ptr; - device T * dst_ptr; - - int i0; - - if (FC_CNT) { - i0 = tgpig.x; - - src0_ptr = (device const T0 *) (src0); - dst_ptr = (device T *) (dst); - } else { - const int i03 = tgpig.z; - const int i02 = tgpig.y; - const int k0 = tgpig.x/args.ne01; - const int i01 = tgpig.x - k0*args.ne01; - - i0 = k0*ntg.x + tpitg.x; - - src0_ptr = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); - dst_ptr = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1 ); - } - - { - //threadgroup_barrier(mem_flags::mem_none); - - if (!FC_CNT) { - if (i0 >= args.ne0) { - return; - } - } - - const TC x = (TC) src0_ptr[i0]; - - if (FC_OP == OP_UNARY_NUM_SCALE) { - dst_ptr[i0] = (T) (args.scale * x + args.bias); - } - - if (FC_OP == OP_UNARY_NUM_FILL) { - dst_ptr[i0] = (T) args.val; - } - - if (FC_OP == OP_UNARY_NUM_CLAMP) { - dst_ptr[i0] = (T) clamp(x, args.min, args.max); - } - - if (FC_OP == OP_UNARY_NUM_SQR) { - dst_ptr[i0] = (T) (x * x); - } - - if (FC_OP == OP_UNARY_NUM_SQRT) { - dst_ptr[i0] = (T) sqrt(x); - } - - if (FC_OP == OP_UNARY_NUM_SIN) { - dst_ptr[i0] = (T) sin(x); - } - - if (FC_OP == OP_UNARY_NUM_COS) { - dst_ptr[i0] = (T) cos(x); - } - - if (FC_OP == OP_UNARY_NUM_LOG) { - dst_ptr[i0] = (T) log(x); - } - - if (FC_OP == OP_UNARY_NUM_LEAKY_RELU) { - dst_ptr[i0] = (T) (TC(x > 0)*x + TC(x <= 0)*(x * args.slope)); - } - - if (FC_OP == OP_UNARY_NUM_TANH) { - dst_ptr[i0] = (T) precise::tanh(x); - } - - if (FC_OP == OP_UNARY_NUM_RELU) { - dst_ptr[i0] = (T) fmax(0, x); - } - - if (FC_OP == OP_UNARY_NUM_SIGMOID) { - dst_ptr[i0] = (T) (1 / (1 + exp(-x))); - } - - if (FC_OP == OP_UNARY_NUM_GELU) { - dst_ptr[i0] = (T) (0.5*x*(1 + precise::tanh(SQRT_2_OVER_PI*x*(1 + GELU_COEF_A*x*x)))); - } - - if (FC_OP == OP_UNARY_NUM_GELU_ERF) { - dst_ptr[i0] = (T) (0.5*x*(1 + erf_approx(SQRT_2_INV*x))); - } - - if (FC_OP == OP_UNARY_NUM_GELU_QUICK) { - dst_ptr[i0] = (T) (x * (1/(1 + exp(GELU_QUICK_COEF*x)))); - } - - if (FC_OP == OP_UNARY_NUM_SILU) { - dst_ptr[i0] = (T) (x / (1 + exp(-x))); - } - - if (FC_OP == OP_UNARY_NUM_ELU) { - dst_ptr[i0] = (T) elu_approx(x); - } - - if (FC_OP == OP_UNARY_NUM_NEG) { - dst_ptr[i0] = (T) -x; - } - - if (FC_OP == OP_UNARY_NUM_ABS) { - dst_ptr[i0] = (T) fabs(x); - } - - if (FC_OP == OP_UNARY_NUM_SGN) { - dst_ptr[i0] = T(x > 0) - T(x < 0); - } - - if (FC_OP == OP_UNARY_NUM_STEP) { - dst_ptr[i0] = T(x > 0); - } - - if (FC_OP == OP_UNARY_NUM_HARDSWISH) { - dst_ptr[i0] = (T) (x * fmax(0, fmin(1, x/6 + 0.5))); - } - - if (FC_OP == OP_UNARY_NUM_HARDSIGMOID) { - dst_ptr[i0] = (T) fmax(0, fmin(1, x/6 + 0.5)); - } - - if (FC_OP == OP_UNARY_NUM_EXP) { - dst_ptr[i0] = (T) exp(x); - } - - if (FC_OP == OP_UNARY_NUM_SOFTPLUS) { - dst_ptr[i0] = (T) select(log(1 + exp(x)), x, x > 20); - } - - if (FC_OP == OP_UNARY_NUM_EXPM1) { - // TODO: precise implementation - dst_ptr[i0] = (T) (exp(x) - 1); - } - - if (FC_OP == OP_UNARY_NUM_FLOOR) { - dst_ptr[i0] = (T) floor(x); - } - - if (FC_OP == OP_UNARY_NUM_CEIL) { - dst_ptr[i0] = (T) ceil(x); - } - - if (FC_OP == OP_UNARY_NUM_ROUND) { - dst_ptr[i0] = (T) round(x); - } - - if (FC_OP == OP_UNARY_NUM_TRUNC) { - dst_ptr[i0] = (T) trunc(x); - } - - if (FC_OP == OP_UNARY_NUM_XIELU) { - const TC xi = x; - const TC gate = TC(xi > TC(0.0f)); - const TC clamped = fmin(xi, TC(args.val)); - const TC y_pos = TC(args.scale) * xi * xi + TC(args.bias) * xi; - const TC y_neg = (exp(clamped) - TC(1.0f) - xi) * TC(args.slope) + TC(args.bias) * xi; - dst_ptr[i0] = (T) (gate * y_pos + (TC(1.0f) - gate) * y_neg); - } - } - -#undef FC_OP -#undef FC_CNT -} - -typedef decltype(kernel_unary_impl) kernel_unary_t; - -template [[host_name("kernel_unary_f32_f32")]] kernel kernel_unary_t kernel_unary_impl; -template [[host_name("kernel_unary_f32_f32_4")]] kernel kernel_unary_t kernel_unary_impl; -template [[host_name("kernel_unary_f16_f16")]] kernel kernel_unary_t kernel_unary_impl; -template [[host_name("kernel_unary_f16_f16_4")]] kernel kernel_unary_t kernel_unary_impl; - -kernel void kernel_silu_back_f32( - constant ggml_metal_kargs_silu_back & args, - device const float * dy, - device const float * x, - device float * dx, - uint gid [[thread_position_in_grid]]) { - if (gid >= args.ne) { - return; - } - - const float s = 1.0f / (1.0f + exp(-x[gid])); - dx[gid] = dy[gid] * s * (1.0f + x[gid] * (1.0f - s)); -} - -// OP: 0 - add, 1 - sub, 2 - mul, 3 - div -constant short FC_bin_op [[function_constant(FC_BIN + 0)]]; -constant short FC_bin_f [[function_constant(FC_BIN + 1)]]; -constant bool FC_bin_rb [[function_constant(FC_BIN + 2)]]; -constant bool FC_bin_cb [[function_constant(FC_BIN + 3)]]; - -template -kernel void kernel_bin_fuse_impl( - constant ggml_metal_kargs_bin & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { -#define FC_OP FC_bin_op -#define FC_F FC_bin_f -#define FC_RB FC_bin_rb -#define FC_CB FC_bin_cb - - if (FC_RB) { - // row broadcast - const uint i0 = tgpig.y*args.ne00 + tgpig.x; - const uint i1 = FC_CB ? tgpig.x%args.ne10 : tgpig.x; - - device const T0 * src0_row = (device const T0 *) (src0); - device T * dst_row = (device T *) (dst); - - if (FC_F == 1) { - device const T1 * src1_row = (device const T1 *) (src1 + args.o1[0]); - - if (FC_OP == 0) { - dst_row[i0] = src0_row[i0] + src1_row[i1]; - } - - if (FC_OP == 1) { - dst_row[i0] = src0_row[i0] - src1_row[i1]; - } - - if (FC_OP == 2) { - dst_row[i0] = src0_row[i0] * src1_row[i1]; - } - - if (FC_OP == 3) { - dst_row[i0] = src0_row[i0] / src1_row[i1]; - } - } else { - T0 res = src0_row[i0]; - - if (FC_OP == 0) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res += ((device const T1 *) (src1 + args.o1[j]))[i1]; - } - } - - if (FC_OP == 1) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res -= ((device const T1 *) (src1 + args.o1[j]))[i1]; - } - } - - if (FC_OP == 2) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res *= ((device const T1 *) (src1 + args.o1[j]))[i1]; - } - } - - if (FC_OP == 3) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res /= ((device const T1 *) (src1 + args.o1[j]))[i1]; - } - } - - dst_row[i0] = res; - } - } else { - const int i03 = tgpig.z; - const int i02 = tgpig.y; - const int i01 = tgpig.x; - - if (i01 >= args.ne01) { - return; - } - - const int i13 = i03%args.ne13; - const int i12 = i02%args.ne12; - const int i11 = i01%args.ne11; - - device const T0 * src0_ptr = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + args.offs); - device T * dst_ptr = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1 + args.offs); - - if (FC_F == 1) { - device const T1 * src1_ptr = (device const T1 *) (src1 + args.o1[0] + i13*args.nb13 + i12*args.nb12 + i11*args.nb11); - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const int i10 = FC_CB ? i0%args.ne10 : i0; - - if (FC_OP == 0) { - dst_ptr[i0] = src0_ptr[i0] + src1_ptr[i10]; - } - - if (FC_OP == 1) { - dst_ptr[i0] = src0_ptr[i0] - src1_ptr[i10]; - } - - if (FC_OP == 2) { - dst_ptr[i0] = src0_ptr[i0] * src1_ptr[i10]; - } - - if (FC_OP == 3) { - dst_ptr[i0] = src0_ptr[i0] / src1_ptr[i10]; - } - } - } else { - device const T1 * src1_ptr[8]; - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - src1_ptr[j] = (device const T1 *) (src1 + args.o1[j] + i13*args.nb13 + i12*args.nb12 + i11*args.nb11); - } - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const int i10 = FC_CB ? i0%args.ne10 : i0; - - T res = src0_ptr[i0]; - - if (FC_OP == 0) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res += src1_ptr[j][i10]; - } - } - - if (FC_OP == 1) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res -= src1_ptr[j][i10]; - } - } - - if (FC_OP == 2) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res *= src1_ptr[j][i10]; - } - } - - if (FC_OP == 3) { - FOR_UNROLL (short j = 0; j < FC_F; ++j) { - res /= src1_ptr[j][i10]; - } - } - - dst_ptr[i0] = res; - } - } - } - -#undef FC_OP -#undef FC_F -#undef FC_RB -#undef FC_CB -} - -typedef decltype(kernel_bin_fuse_impl) kernel_bin_fuse_t; - -template [[host_name("kernel_bin_fuse_f32_f32_f32")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; -template [[host_name("kernel_bin_fuse_f32_f32_f32_4")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; -template [[host_name("kernel_bin_fuse_f16_f16_f16")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; -template [[host_name("kernel_bin_fuse_f16_f16_f16_4")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; - -kernel void kernel_add_id( - constant ggml_metal_kargs_add_id & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int i1 = tgpig.x; - const int i2 = tgpig.y; - - const int i11 = *((device const int32_t *) (src2 + i1*sizeof(int32_t) + i2*args.nb21)); - - const size_t nb1 = args.ne0 * sizeof(float); - const size_t nb2 = args.ne1 * nb1; - - device float * dst_row = (device float *)((device char *)dst + i1*nb1 + i2*nb2); - device const float * src0_row = (device const float *)((device char *)src0 + i1*args.nb01 + i2*args.nb02); - device const float * src1_row = (device const float *)((device char *)src1 + i11*args.nb11); - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - dst_row[i0] = src0_row[i0] + src1_row[i0]; - } -} - -template -kernel void kernel_repeat( - constant ggml_metal_kargs_repeat & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int i3 = tgpig.z; - const int i2 = tgpig.y; - const int i1 = tgpig.x; - - const int i03 = i3%args.ne03; - const int i02 = i2%args.ne02; - const int i01 = i1%args.ne01; - - device const char * src0_ptr = src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01; - device char * dst_ptr = dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1; - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const int i00 = i0%args.ne00; - *((device T *)(dst_ptr + i0*args.nb0)) = *((device T *)(src0_ptr + i00*args.nb00)); - } -} - -typedef decltype(kernel_repeat) kernel_repeat_t; - -template [[host_name("kernel_repeat_f32")]] kernel kernel_repeat_t kernel_repeat; -template [[host_name("kernel_repeat_f16")]] kernel kernel_repeat_t kernel_repeat; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_repeat_bf16")]] kernel kernel_repeat_t kernel_repeat; -#endif -template [[host_name("kernel_repeat_i32")]] kernel kernel_repeat_t kernel_repeat; -template [[host_name("kernel_repeat_i16")]] kernel kernel_repeat_t kernel_repeat; - -template -kernel void kernel_reglu( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - const float x0 = src0_row[i0]; - const float x1 = src1_row[i0]; - - dst_row[i0] = (T)(x0*x1*(x0 > 0.0f)); - } -} - -typedef decltype(kernel_reglu) kernel_reglu_t; - -template [[host_name("kernel_reglu_f32")]] kernel kernel_reglu_t kernel_reglu; -template [[host_name("kernel_reglu_f16")]] kernel kernel_reglu_t kernel_reglu; - -template -kernel void kernel_geglu( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - const float x0 = src0_row[i0]; - const float x1 = src1_row[i0]; - - const float gelu = 0.5f*x0*(1.0f + precise::tanh(SQRT_2_OVER_PI*x0*(1.0f + GELU_COEF_A*x0*x0))); - - dst_row[i0] = (T)(gelu*x1); - } -} - -typedef decltype(kernel_geglu) kernel_geglu_t; - -template [[host_name("kernel_geglu_f32")]] kernel kernel_geglu_t kernel_geglu; -template [[host_name("kernel_geglu_f16")]] kernel kernel_geglu_t kernel_geglu; - -template -kernel void kernel_swiglu( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - const float x0 = src0_row[i0]; - const float x1 = src1_row[i0]; - - const float silu = x0 / (1.0f + exp(-x0)); - - dst_row[i0] = (T)(silu*x1); - } -} - -typedef decltype(kernel_swiglu) kernel_swiglu_t; - -template [[host_name("kernel_swiglu_f32")]] kernel kernel_swiglu_t kernel_swiglu; -template [[host_name("kernel_swiglu_f16")]] kernel kernel_swiglu_t kernel_swiglu; - -template -kernel void kernel_swiglu_oai( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - float x0 = src0_row[i0]; - float x1 = src1_row[i0]; - - x0 = min(x0, args.limit); - x1 = max(min(x1, args.limit), -args.limit); - - float out_glu = x0 / (1.0f + exp(-x0 * args.alpha)); - out_glu = out_glu * (1.0f + x1); - - dst_row[i0] = (T)out_glu; - } -} - -typedef decltype(kernel_swiglu_oai) kernel_swiglu_oai_t; - -template [[host_name("kernel_swiglu_oai_f32")]] kernel kernel_swiglu_oai_t kernel_swiglu_oai; -template [[host_name("kernel_swiglu_oai_f16")]] kernel kernel_swiglu_oai_t kernel_swiglu_oai; - -template -kernel void kernel_geglu_erf( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - const float x0 = src0_row[i0]; - const float x1 = src1_row[i0]; - - const float gelu_erf = 0.5f*x0*(1.0f+erf_approx(x0*SQRT_2_INV)); - - dst_row[i0] = (T)(gelu_erf*x1); - } -} - -typedef decltype(kernel_geglu_erf) kernel_geglu_erf_t; - -template [[host_name("kernel_geglu_erf_f32")]] kernel kernel_geglu_erf_t kernel_geglu_erf; -template [[host_name("kernel_geglu_erf_f16")]] kernel kernel_geglu_erf_t kernel_geglu_erf; - -template -kernel void kernel_geglu_quick( - constant ggml_metal_kargs_glu & args, - device const char * src0, - device const char * src1, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; - device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; - device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); - - for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { - const float x0 = src0_row[i0]; - const float x1 = src1_row[i0]; - - const float gelu_quick = x0*(1.0f/(1.0f+exp(GELU_QUICK_COEF*x0))); - - dst_row[i0] = (T)(gelu_quick*x1); - } -} - -typedef decltype(kernel_geglu_quick) kernel_geglu_quick_t; - -template [[host_name("kernel_geglu_quick_f32")]] kernel kernel_geglu_quick_t kernel_geglu_quick; -template [[host_name("kernel_geglu_quick_f16")]] kernel kernel_geglu_quick_t kernel_geglu_quick; - -kernel void kernel_op_sum_f32( - constant ggml_metal_kargs_sum & args, - device const float * src0, - device float * dst, - threadgroup float * shmem_f32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - - if (args.np == 0) { - return; - } - - // TODO: become function constant - const uint nsg = (ntg.x + 31) / 32; - - float sumf = 0; - - for (uint64_t i0 = tpitg.x; i0 < args.np; i0 += ntg.x) { - sumf += src0[i0]; - } - - sumf = simd_sum(sumf); - - if (tiisg == 0) { - shmem_f32[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - float total = 0; - - if (sgitg == 0) { - float v = 0; - - if (tpitg.x < nsg) { - v = shmem_f32[tpitg.x]; - } - - total = simd_sum(v); - - if (tpitg.x == 0) { - dst[0] = total; - } - } -} - -constant short FC_sum_rows_op [[function_constant(FC_SUM_ROWS + 0)]]; - -template -kernel void kernel_sum_rows_impl( - constant ggml_metal_kargs_sum_rows & args, - device const char * src0, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { -#define FC_OP FC_sum_rows_op - - const int i3 = tgpig.z; - const int i2 = tgpig.y; - const int i1 = tgpig.x; - - threadgroup T0 * shmem_t = (threadgroup T0 *) shmem; - - if (sgitg == 0) { - shmem_t[tiisg] = 0.0f; - } - - device const T0 * src_row = (device const T0 *) (src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03); - device T * dst_row = (device T *) (dst + i1*args.nb1 + i2*args.nb2 + i3*args.nb3); - - T0 sumf = T0(0.0f); - - for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { - sumf += src_row[i0]; - } - - sumf = simd_sum(sumf); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shmem_t[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sumf = shmem_t[tiisg]; - sumf = simd_sum(sumf); - - if (tpitg.x == 0) { - if (FC_OP == OP_SUM_ROWS_NUM_MEAN) { - if (is_same::value) { - dst_row[0] = sum(sumf) / (4*args.ne00); - } else { - dst_row[0] = sum(sumf) / args.ne00; - } - } else { - dst_row[0] = sum(sumf); - } - } - -#undef FC_OP -} - -typedef decltype(kernel_sum_rows_impl) kernel_sum_rows_t; - -template [[host_name("kernel_sum_rows_f32_f32")]] kernel kernel_sum_rows_t kernel_sum_rows_impl; -template [[host_name("kernel_sum_rows_f32_f32_4")]] kernel kernel_sum_rows_t kernel_sum_rows_impl; - -template -kernel void kernel_cumsum_blk( - constant ggml_metal_kargs_cumsum_blk & args, - device const char * src0, - device char * tmp, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int ib = tgpig[0]/args.ne01; - - const int i00 = ib*ntg.x; - const int i01 = tgpig[0]%args.ne01; - const int i02 = tgpig[1]; - const int i03 = tgpig[2]; - - device const float * src0_row = (device const float *) (src0 + - args.nb01*i01 + - args.nb02*i02 + - args.nb03*i03); - - threadgroup float * shmem_f32 = (threadgroup float *) shmem; - - float v = 0.0f; - - if (i00 + tpitg.x < args.ne00) { - v = src0_row[i00 + tpitg.x]; - } - - float s = simd_prefix_inclusive_sum(v); - - if (tiisg == N_SIMDWIDTH - 1) { - shmem_f32[sgitg] = s; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (sgitg == 0) { - shmem_f32[tiisg] = simd_prefix_exclusive_sum(shmem_f32[tiisg]); - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - s += shmem_f32[sgitg]; - - device float * dst_row = (device float *) dst + - args.ne00*i01 + - args.ne00*args.ne01*i02 + - args.ne00*args.ne01*args.ne02*i03; - - if (i00 + tpitg.x < args.ne00) { - dst_row[i00 + tpitg.x] = s; - } - - if (args.outb && tpitg.x == ntg.x - 1) { - device float * tmp_row = (device float *) tmp + - args.net0*i01 + - args.net0*args.net1*i02 + - args.net0*args.net1*args.net2*i03; - - tmp_row[ib] = s; - } -} - -typedef decltype(kernel_cumsum_blk) kernel_cumsum_blk_t; - -template [[host_name("kernel_cumsum_blk_f32")]] kernel kernel_cumsum_blk_t kernel_cumsum_blk; - -template -kernel void kernel_cumsum_add( - constant ggml_metal_kargs_cumsum_add & args, - device const char * tmp, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int ib = tgpig[0]/args.ne01; - - if (ib == 0) { - return; - } - - const int i00 = ib*ntg.x; - const int i01 = tgpig[0]%args.ne01; - const int i02 = tgpig[1]; - const int i03 = tgpig[2]; - - device const float * tmp_row = (device const float *) (tmp + - args.nbt1*i01 + - args.nbt2*i02 + - args.nbt3*i03); - - device float * dst_row = (device float *) dst + - args.ne00*i01 + - args.ne00*args.ne01*i02 + - args.ne00*args.ne01*args.ne02*i03; - - if (i00 + tpitg.x < args.ne00) { - dst_row[i00 + tpitg.x] += tmp_row[ib - 1]; - } -} - -typedef decltype(kernel_cumsum_add) kernel_cumsum_add_t; - -template [[host_name("kernel_cumsum_add_f32")]] kernel kernel_cumsum_add_t kernel_cumsum_add; - - -template -bool _ggml_vec_tri_cmp(const int i, const int r); - -template<> -bool _ggml_vec_tri_cmp(const int i, const int r) { - return i < r; -} - -template<> -bool _ggml_vec_tri_cmp(const int i, const int r) { - return i <= r; -} - -template<> -bool _ggml_vec_tri_cmp(const int i, const int r) { - return i > r; -} - -template<> -bool _ggml_vec_tri_cmp(const int i, const int r) { - return i >= r; -} - -template -kernel void kernel_tri( - constant ggml_metal_kargs_tri & args, - device const char * src0, - device const char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int i3 = tgpig.z; - const int i2 = tgpig.y; - const int i1 = tgpig.x; - - if (i3 >= args.ne03 || i2 >= args.ne02 || i1 >= args.ne01) { - return; - } - - device const T * src_row = (device const T *) ((device const char *) src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03); - device T * dst_row = (device T *) ((device char *) dst + i1*args.nb1 + i2*args.nb2 + i3*args.nb3); - - // Each thread is a single element of the row if ne00 < max threads per - // threadgroup, so this will loop once for each index that this thread is - // responsible for - for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { - // Use the comparison as a mask for branchless - dst_row[i0] = static_cast(_ggml_vec_tri_cmp(i0, i1)) * src_row[i0]; - } -} - -typedef decltype(kernel_tri) kernel_tri_t; - -template [[host_name("kernel_tri_f32_0")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f32_1")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f32_2")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f32_3")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f16_0")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f16_1")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f16_2")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_f16_3")]] kernel kernel_tri_t kernel_tri; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_tri_bf16_0")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_bf16_1")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_bf16_2")]] kernel kernel_tri_t kernel_tri; -template [[host_name("kernel_tri_bf16_3")]] kernel kernel_tri_t kernel_tri; -#endif - -template -kernel void kernel_soft_max( - constant ggml_metal_kargs_soft_max & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - threadgroup float * buf [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint sgitg[[simdgroup_index_in_threadgroup]], - uint tiisg[[thread_index_in_simdgroup]], - uint3 tptg[[threads_per_threadgroup]]) { - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - const int32_t i01 = tgpig.x; - - const int32_t i13 = i03%args.ne13; - const int32_t i12 = i02%args.ne12; - const int32_t i11 = i01; - - device const float * psrc0 = (device const float *) (src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); - device const T * pmask = src1 != src0 ? (device const T * ) (src1 + i11*args.nb11 + i12*args.nb12 + i13*args.nb13) : nullptr; - device const float * psrc2 = src2 != src0 ? (device const float *) (src2) : nullptr; - device float * pdst = (device float *) (dst + i01*args.nb1 + i02*args.nb2 + i03*args.nb3); - - float slope = 1.0f; - - // ALiBi - if (args.max_bias > 0.0f) { - const int32_t h = i02; - - const float base = h < args.n_head_log2 ? args.m0 : args.m1; - const int exp = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; - - slope = pow(base, exp); - } - - // parallel max - float lmax = psrc2 ? psrc2[i02] : -INFINITY; - - for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { - lmax = MAX(lmax, psrc0[i00]*args.scale + (pmask ? slope*pmask[i00] : 0.0f)); - } - - // find the max value in the block - float max_val = simd_max(lmax); - if (tptg.x > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = -INFINITY; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = max_val; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - max_val = buf[tiisg]; - max_val = simd_max(max_val); - } - - // parallel sum - float lsum = 0.0f; - for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { - const float exp_psrc0 = exp((psrc0[i00]*args.scale + (pmask ? slope*pmask[i00] : 0.0f)) - max_val); - lsum += exp_psrc0; - pdst[i00] = exp_psrc0; - } - - // This barrier fixes a failing test - // ref: https://github.com/ggml-org/ggml/pull/621#discussion_r1425156335 - threadgroup_barrier(mem_flags::mem_none); - - float sum = simd_sum(lsum); - - if (tptg.x > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = 0.0f; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = sum; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sum = buf[tiisg]; - sum = simd_sum(sum); - } - - if (psrc2) { - sum += exp(psrc2[i02] - max_val); - } - - const float inv_sum = 1.0f/sum; - - for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { - pdst[i00] *= inv_sum; - } -} - -template -kernel void kernel_soft_max_4( - constant ggml_metal_kargs_soft_max & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - threadgroup float * buf [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint sgitg[[simdgroup_index_in_threadgroup]], - uint tiisg[[thread_index_in_simdgroup]], - uint3 tptg[[threads_per_threadgroup]]) { - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - const int32_t i01 = tgpig.x; - - const int32_t i13 = i03%args.ne13; - const int32_t i12 = i02%args.ne12; - const int32_t i11 = i01; - - device const float4 * psrc4 = (device const float4 *) (src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); - device const T * pmask = src1 != src0 ? (device const T * ) (src1 + i11*args.nb11 + i12*args.nb12 + i13*args.nb13) : nullptr; - device const float * psrc2 = src2 != src0 ? (device const float * ) (src2) : nullptr; - device float4 * pdst4 = (device float4 *) (dst + i01*args.nb1 + i02*args.nb2 + i03*args.nb3); - - float slope = 1.0f; - - if (args.max_bias > 0.0f) { - const int32_t h = i02; - - const float base = h < args.n_head_log2 ? args.m0 : args.m1; - const int exp = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; - - slope = pow(base, exp); - } - - // parallel max - float4 lmax4 = psrc2 ? psrc2[i02] : -INFINITY; - - for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { - lmax4 = fmax(lmax4, psrc4[i00]*args.scale + (float4)((pmask ? slope*pmask[i00] : 0.0f))); - } - - const float lmax = MAX(MAX(lmax4[0], lmax4[1]), MAX(lmax4[2], lmax4[3])); - - float max_val = simd_max(lmax); - if (tptg.x > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = -INFINITY; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = max_val; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - max_val = buf[tiisg]; - max_val = simd_max(max_val); - } - - // parallel sum - float4 lsum4 = 0.0f; - for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { - const float4 exp_psrc4 = exp((psrc4[i00]*args.scale + (float4)((pmask ? slope*pmask[i00] : 0.0f))) - max_val); - lsum4 += exp_psrc4; - pdst4[i00] = exp_psrc4; - } - - const float lsum = lsum4[0] + lsum4[1] + lsum4[2] + lsum4[3]; - - // This barrier fixes a failing test - // ref: https://github.com/ggml-org/ggml/pull/621#discussion_r1425156335 - threadgroup_barrier(mem_flags::mem_none); - - float sum = simd_sum(lsum); - - if (tptg.x > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = 0.0f; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = sum; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sum = buf[tiisg]; - sum = simd_sum(sum); - } - - if (psrc2) { - sum += exp(psrc2[i02] - max_val); - } - - const float inv_sum = 1.0f/sum; - - for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { - pdst4[i00] *= inv_sum; - } -} - -typedef decltype(kernel_soft_max) kernel_soft_max_t; -typedef decltype(kernel_soft_max_4) kernel_soft_max_4_t; - -template [[host_name("kernel_soft_max_f16")]] kernel kernel_soft_max_t kernel_soft_max; -template [[host_name("kernel_soft_max_f32")]] kernel kernel_soft_max_t kernel_soft_max; -template [[host_name("kernel_soft_max_f16_4")]] kernel kernel_soft_max_4_t kernel_soft_max_4; -template [[host_name("kernel_soft_max_f32_4")]] kernel kernel_soft_max_4_t kernel_soft_max_4; - -// ref: ggml.c:ggml_compute_forward_ssm_conv_f32 -kernel void kernel_ssm_conv_f32_f32( - constant ggml_metal_kargs_ssm_conv & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - const int64_t ir = tgpig.x; - const int64_t i2 = tgpig.y; - const int64_t i3 = tgpig.z; - - const int64_t nc = args.ne10; - //const int64_t ncs = args.ne00; - //const int64_t nr = args.ne01; - //const int64_t n_t = args.ne1; - //const int64_t n_s = args.ne2; - - device const float * s = (device const float *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); - device const float * c = (device const float *) ((device const char *) src1 + ir*args.nb11); - device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); - - float sumf = 0.0f; - - for (int64_t i0 = 0; i0 < nc; ++i0) { - sumf += s[i0] * c[i0]; - } - - x[0] = sumf; -} - -kernel void kernel_ssm_conv_f32_f32_4( - constant ggml_metal_kargs_ssm_conv & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - const int64_t ir = tgpig.x; - const int64_t i2 = tgpig.y; - const int64_t i3 = tgpig.z; - - const int64_t nc = args.ne10; - //const int64_t ncs = args.ne00; - //const int64_t nr = args.ne01; - //const int64_t n_t = args.ne1; - //const int64_t n_s = args.ne2; - - device const float4 * s = (device const float4 *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); - device const float4 * c = (device const float4 *) ((device const char *) src1 + ir*args.nb11); - device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); - - float sumf = 0.0f; - - for (int64_t i0 = 0; i0 < nc/4; ++i0) { - sumf += dot(s[i0], c[i0]); - } - - x[0] = sumf; -} - -constant short FC_ssm_conv_bs [[function_constant(FC_SSM_CONV + 0)]]; - -// Batched version: each threadgroup processes multiple tokens for better efficiency -// Thread layout: each thread handles one token, threadgroup covers BATCH_SIZE tokens -kernel void kernel_ssm_conv_f32_f32_batched( - constant ggml_metal_kargs_ssm_conv & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - // tgpig.x = row index (ir) - // tgpig.y = batch of tokens (i2_base / BATCH_SIZE) - // tgpig.z = sequence index (i3) - // tpitg.x = thread within batch (0..BATCH_SIZE-1) - const short BATCH_SIZE = FC_ssm_conv_bs; - - const int64_t ir = tgpig.x; - const int64_t i2_base = tgpig.y * BATCH_SIZE; - const int64_t i3 = tgpig.z; - const int64_t i2_off = tpitg.x; - const int64_t i2 = i2_base + i2_off; - - const int64_t nc = args.ne10; // conv kernel size (typically 4) - const int64_t n_t = args.ne1; // number of tokens - - // Bounds check for partial batches at the end - if (i2 >= n_t) { - return; - } - - // Load conv weights (shared across all tokens for this row) - device const float * c = (device const float *) ((device const char *) src1 + ir*args.nb11); - - // Load source for this specific token - device const float * s = (device const float *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); - - // Output location for this token - device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); - - float sumf = 0.0f; - for (int64_t i0 = 0; i0 < nc; ++i0) { - sumf += s[i0] * c[i0]; - } - - x[0] = sumf; -} - -kernel void kernel_ssm_conv_f32_f32_batched_4( - constant ggml_metal_kargs_ssm_conv & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - // tgpig.x = row index (ir) - // tgpig.y = batch of tokens (i2_base / BATCH_SIZE) - // tgpig.z = sequence index (i3) - // tpitg.x = thread within batch (0..BATCH_SIZE-1) - const short BATCH_SIZE = FC_ssm_conv_bs; - - const int64_t ir = tgpig.x; - const int64_t i2_base = tgpig.y * BATCH_SIZE; - const int64_t i3 = tgpig.z; - const int64_t i2_off = tpitg.x; - const int64_t i2 = i2_base + i2_off; - - const int64_t nc = args.ne10; // conv kernel size (typically 4) - const int64_t n_t = args.ne1; // number of tokens - - // Bounds check for partial batches at the end - if (i2 >= n_t) { - return; - } - - // Load conv weights (shared across all tokens for this row) - device const float4 * c = (device const float4 *) ((device const char *) src1 + ir*args.nb11); - - // Load source for this specific token - device const float4 * s = (device const float4 *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); - - // Output location for this token - device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); - - float sumf = 0.0f; - for (int64_t i0 = 0; i0 < nc/4; ++i0) { - sumf += dot(s[i0], c[i0]); - } - - x[0] = sumf; -} - -// ref: ggml.c:ggml_compute_forward_ssm_scan_f32, Mamba-2 part -// Optimized version: reduces redundant memory loads by having one thread load shared values -kernel void kernel_ssm_scan_f32( - constant ggml_metal_kargs_ssm_scan & args, - device const void * src0, - device const void * src1, - device const void * src2, - device const void * src3, - device const void * src4, - device const void * src5, - device const void * src6, - device float * dst, - threadgroup float * shared [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgptg[[simdgroups_per_threadgroup]], - uint3 tgpg[[threadgroups_per_grid]]) { - constexpr short NW = N_SIMDWIDTH; - - // Shared memory layout: - // [0..sgptg*NW-1]: partial sums for reduction (existing) - // [sgptg*NW..sgptg*NW+sgptg-1]: pre-computed x_dt values for each token in batch - // [sgptg*NW+sgptg..sgptg*NW+2*sgptg-1]: pre-computed dA values for each token in batch - threadgroup float * shared_sums = shared; - threadgroup float * shared_x_dt = shared + sgptg * NW; - threadgroup float * shared_dA = shared + sgptg * NW + sgptg; - - shared_sums[tpitg.x] = 0.0f; - - const int32_t i0 = tpitg.x; - const int32_t i1 = tgpig.x; - const int32_t ir = tgpig.y; // current head - const int32_t i3 = tgpig.z; // current seq - - const int32_t nc = args.d_state; - const int32_t nr = args.d_inner; - const int32_t nh = args.n_head; - const int32_t ng = args.n_group; - const int32_t n_t = args.n_seq_tokens; - const int32_t n_s = args.n_seqs; - const int32_t K = args.K; - - const int32_t s_off = args.s_off; - - device const int32_t * ids = (device const int32_t *) src6; - - device const float * s0_buff = (device const float *) ((device const char *) src0 + ir*args.nb02 + ids[i3]*args.nb03); - device float * s_buff = (device float *) ((device char *) dst + ir*args.nb02 + i3*args.nb03 + s_off); - - const int32_t i = i0 + i1*nc; - const int32_t g = ir / (nh / ng); // repeat_interleave - - float s0 = s0_buff[i]; - float s = 0.0f; - - device const float * A = (device const float *) ((device const char *) src3 + ir*args.nb31); // {ne30, nh} - - const float A0 = A[i0%args.ne30]; - - device const float * x = (device const float *)((device const char *) src1 + i1*args.nb10 + ir*args.nb11 + i3*args.nb13); // {dim, nh, nt, ns} - device const float * dt = (device const float *)((device const char *) src2 + ir*args.nb20 + i3*args.nb22); // {nh, nt, ns} - device const float * B = (device const float *)((device const char *) src4 + g*args.nb41 + i3*args.nb43); // {d_state, ng, nt, ns} - device const float * C = (device const float *)((device const char *) src5 + g*args.nb51 + i3*args.nb53); // {d_state, ng, nt, ns} - - device float * y = dst + (i1 + ir*(nr) + i3*(n_t*nh*nr)); // {dim, nh, nt, ns} - - for (int i2 = 0; i2 < n_t; i2 += sgptg) { - threadgroup_barrier(mem_flags::mem_threadgroup); - - // Pre-compute x_dt and dA for this batch of tokens - // Only first sgptg threads do the loads and expensive math - if (i0 < sgptg && i2 + i0 < n_t) { - // ns12 and ns21 are element strides (nb12/nb10, nb21/nb20) - device const float * x_t = x + i0 * args.ns12; - device const float * dt_t = dt + i0 * args.ns21; - - const float dt0 = dt_t[0]; - const float dtsp = dt0 <= 20.0f ? log(1.0f + exp(dt0)) : dt0; - shared_x_dt[i0] = x_t[0] * dtsp; - shared_dA[i0] = dtsp; // Store dtsp, compute exp(dtsp * A0) per-thread since A0 varies - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (int t = 0; t < sgptg && i2 + t < n_t; t++) { - const float x_dt = shared_x_dt[t]; - const float dA = exp(shared_dA[t] * A0); - - s = (s0 * dA) + (B[i0] * x_dt); - - const float sumf = simd_sum(s * C[i0]); - - if (tiisg == 0) { - shared_sums[t*NW + sgitg] = sumf; - } - - // recurse - s0 = s; - - const int32_t slot = n_t - 1 - (i2 + t); - if (slot > 0 && slot < K) { - device float * s_snapshot = (device float *) ((device char *) s_buff + (int64_t) slot*n_s*args.nb03); - s_snapshot[i] = s; - } - - B += args.ns42; - C += args.ns52; - } - - // Advance pointers for next batch - x += sgptg * args.ns12; - dt += sgptg * args.ns21; - - threadgroup_barrier(mem_flags::mem_threadgroup); - - const float sumf = simd_sum(shared_sums[sgitg*NW + tiisg]); - - if (tiisg == 0 && i2 + sgitg < n_t) { - y[sgitg*nh*nr] = sumf; - } - - y += sgptg*nh*nr; - } - - s_buff[i] = s; -} - -kernel void kernel_rwkv_wkv6_f32( - device const float * k, - device const float * v, - device const float * r, - device const float * tf, - device const float * td, - device const float * state_in, - device float * dst, - constant uint & B, - constant uint & T, - constant uint & C, - constant uint & H, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const uint head_size = 64; // TODO: support head_size = 128 - const uint batch_id = tgpig.x / H; - const uint head_id = tgpig.x % H; - const uint tid = tpitg.x; - - if (batch_id >= B || head_id >= H) { - return; - } - - const uint state_size = C * head_size; - const uint n_seq_tokens = T / B; - - threadgroup float _k[head_size]; - threadgroup float _r[head_size]; - threadgroup float _tf[head_size]; - threadgroup float _td[head_size]; - - float state[head_size]; - - for (uint i = 0; i < head_size; i++) { - state[i] = state_in[batch_id * state_size + head_id * head_size * head_size - + i * head_size + tid]; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - _tf[tid] = tf[head_id * head_size + tid]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - const uint start_t = batch_id * n_seq_tokens * C + head_id * head_size + tid; - const uint end_t = (batch_id + 1) * n_seq_tokens * C + head_id * head_size + tid; - - for (uint t = start_t; t < end_t; t += C) { - threadgroup_barrier(mem_flags::mem_threadgroup); - _k[tid] = k[t]; - _r[tid] = r[t]; - _td[tid] = td[t]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - const float v_val = v[t]; - float y = 0.0; - - for (uint j = 0; j < head_size; j += 4) { - float4 k_vec = float4(_k[j], _k[j+1], _k[j+2], _k[j+3]); - float4 r_vec = float4(_r[j], _r[j+1], _r[j+2], _r[j+3]); - float4 tf_vec = float4(_tf[j], _tf[j+1], _tf[j+2], _tf[j+3]); - float4 td_vec = float4(_td[j], _td[j+1], _td[j+2], _td[j+3]); - float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); - - float4 kv = k_vec * v_val; - - float4 temp = tf_vec * kv + s_vec; - y += dot(r_vec, temp); - - s_vec = s_vec * td_vec + kv; - state[j] = s_vec[0]; - state[j+1] = s_vec[1]; - state[j+2] = s_vec[2]; - state[j+3] = s_vec[3]; - } - - dst[t] = y; - } - - for (uint i = 0; i < head_size; i++) { - dst[T * C + batch_id * state_size + head_id * head_size * head_size - + i * head_size + tid] = state[i]; - } -} - -kernel void kernel_rwkv_wkv7_f32( - device const float * r, - device const float * w, - device const float * k, - device const float * v, - device const float * a, - device const float * b, - device const float * state_in, - device float * dst, - constant uint & B, - constant uint & T, - constant uint & C, - constant uint & H, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const uint head_size = 64; // TODO: support head_size = 128 - const uint batch_id = tgpig.x / H; - const uint head_id = tgpig.x % H; - const uint tid = tpitg.x; - - if (batch_id >= B || head_id >= H) { - return; - } - - const uint state_size = C * head_size; - const uint n_seq_tokens = T / B; - - threadgroup float _r[head_size]; - threadgroup float _w[head_size]; - threadgroup float _k[head_size]; - threadgroup float _a[head_size]; - threadgroup float _b[head_size]; - - float state[head_size]; - - for (uint i = 0; i < head_size; i++) { - state[i] = state_in[batch_id * state_size + head_id * head_size * head_size - + tid * head_size + i]; - } - - const uint start_t = batch_id * n_seq_tokens * C + head_id * head_size + tid; - const uint end_t = (batch_id + 1) * n_seq_tokens * C + head_id * head_size + tid; - - for (uint t = start_t; t < end_t; t += C) { - threadgroup_barrier(mem_flags::mem_threadgroup); - _r[tid] = r[t]; - _w[tid] = w[t]; - _k[tid] = k[t]; - _a[tid] = a[t]; - _b[tid] = b[t]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - const float v_val = v[t]; - float y = 0.0, sa = 0.0; - - float4 sa_vec(0.0); - - for (uint j = 0; j < head_size; j += 4) { - float4 a_vec = float4(_a[j], _a[j+1], _a[j+2], _a[j+3]); - float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); - sa_vec += a_vec * s_vec; - } - sa = sa_vec[0] + sa_vec[1] + sa_vec[2] + sa_vec[3]; - - for (uint j = 0; j < head_size; j += 4) { - float4 r_vec = float4(_r[j], _r[j+1], _r[j+2], _r[j+3]); - float4 w_vec = float4(_w[j], _w[j+1], _w[j+2], _w[j+3]); - float4 k_vec = float4(_k[j], _k[j+1], _k[j+2], _k[j+3]); - float4 b_vec = float4(_b[j], _b[j+1], _b[j+2], _b[j+3]); - float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); - - float4 kv = k_vec * v_val; - - s_vec = s_vec * w_vec + kv + sa * b_vec; - y += dot(s_vec, r_vec); - - state[j] = s_vec[0]; - state[j+1] = s_vec[1]; - state[j+2] = s_vec[2]; - state[j+3] = s_vec[3]; - } - - dst[t] = y; - } - - for (uint i = 0; i < head_size; i++) { - dst[T * C + batch_id * state_size + head_id * head_size * head_size - + tid * head_size + i] = state[i]; - } -} - -constant short FC_gated_delta_net_ne20 [[function_constant(FC_GATED_DELTA_NET + 0)]]; -constant short FC_gated_delta_net_ne30 [[function_constant(FC_GATED_DELTA_NET + 1)]]; -constant short FC_gated_delta_net_K [[function_constant(FC_GATED_DELTA_NET + 2)]]; - -#if 1 -template -kernel void kernel_gated_delta_net_impl( - constant ggml_metal_kargs_gated_delta_net & args, - device const char * q, - device const char * k, - device const char * v, - device const char * g, - device const char * b, - device const char * s, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { -#define S_v FC_gated_delta_net_ne20 -#define G FC_gated_delta_net_ne30 -#define K FC_gated_delta_net_K - - const uint tx = tpitg.x; - const uint ty = tpitg.y; - - const uint i23 = tgpig.z; // B (n_seqs) - const uint i21 = tgpig.y; // H (head) - const uint i20 = tgpig.x*NSG + ty; // row within S_v - - const uint i01 = i21 % args.ne01; - const uint i11 = i21 % args.ne11; - - const float scale = 1.0f / sqrt((float)S_v); - - // input state layout [S_v, S_v, H, n_seqs] (s0 only): per-seq stride is H*D. - // state is stored transposed: M[i20][is] = S[is][i20], so row i20 is contiguous - const uint state_in_base = (i23*args.ne21 + i21)*S_v*S_v + i20*S_v; - device const float * s_ptr = (device const float *) (s) + state_in_base; - - float ls[NSG]; - - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - ls[j] = s_ptr[is]; - } - - device float * dst_attn = (device float *) (dst) + (i23*args.ne22*args.ne21 + i21)*S_v + i20; - - device const float * q_ptr = (device const float *) (q + i23*args.nb03 + i01*args.nb01); - device const float * k_ptr = (device const float *) (k + i23*args.nb13 + i11*args.nb11); - device const float * v_ptr = (device const float *) (v + i23*args.nb23 + i21*args.nb21); - - device const float * b_ptr = (device const float *) (b) + (i23*args.ne22*args.ne21 + i21); - device const float * g_ptr = (device const float *) (g) + (i23*args.ne22*args.ne21 + i21)*G; - - // snapshot slot mapping: slot 0 = most recent state, slot s = s tokens back. - // When n_tokens < K, only slots 0..n_tokens-1 are written; older slots are caller-owned. - - // output state base offset: after attention scores - const uint attn_size = args.ne22 * args.ne21 * S_v * args.ne23; - // output state per-slot size: S_v * S_v * H * n_seqs - const uint state_size_per_snap = S_v * S_v * args.ne21 * args.ne23; - // per-(seq,head) offset within a slot - const uint state_out_base = (i23*args.ne21 + i21)*S_v*S_v + i20*S_v; - - for (short t = 0; t < args.ne22; t++) { - float s_k = 0.0f; - - if (G == 1) { - const float g_exp = exp(g_ptr[0]); - - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - ls[j] *= g_exp; - - s_k += ls[j]*k_ptr[is]; - } - } else { - // KDA - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - ls[j] *= exp(g_ptr[is]); - - s_k += ls[j]*k_ptr[is]; - } - } - - s_k = simd_sum(s_k); - - const float d = (v_ptr[i20] - s_k)*b_ptr[0]; - - float y = 0.0f; - - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - ls[j] += k_ptr[is]*d; - - y += ls[j]*q_ptr[is]; - } - - y = simd_sum(y); - - if (tx == 0) { - dst_attn[t*args.ne21*S_v] = y*scale; - } - - q_ptr += args.ns02; - k_ptr += args.ns12; - v_ptr += args.ns22; - - b_ptr += args.ne21; - g_ptr += args.ne21*G; - - if (K > 1) { - const int target_slot = (int)args.ne22 - 1 - (int)t; - if (target_slot >= 0 && target_slot < (int)K) { - device float * dst_state = (device float *) (dst) + attn_size + (uint)target_slot * state_size_per_snap + state_out_base; - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - dst_state[is] = ls[j]; - } - } - } - } - - if (K == 1) { - device float * dst_state = (device float *) (dst) + attn_size + state_out_base; - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - dst_state[is] = ls[j]; - } - } - -#undef S_v -#undef G -#undef K -} - -typedef decltype(kernel_gated_delta_net_impl<4>) kernel_gated_delta_net_t; - -template [[host_name("kernel_gated_delta_net_f32_1")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<1>; -template [[host_name("kernel_gated_delta_net_f32_2")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<2>; -template [[host_name("kernel_gated_delta_net_f32_4")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<4>; - -#else -// a simplified version of the above -// no performance improvement, so keep the above version for now - -template -kernel void kernel_gated_delta_net_impl( - constant ggml_metal_kargs_gated_delta_net & args, - device const char * q, - device const char * k, - device const char * v, - device const char * g, - device const char * b, - device const char * s, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { -#define S_v FC_gated_delta_net_ne20 -#define G FC_gated_delta_net_ne30 - - const uint tx = tpitg.x; - const uint ty = tpitg.y; - - const uint i23 = tgpig.z; // B - const uint i21 = tgpig.y; // H - const uint i20 = tgpig.x*NSG + ty; - - const uint i01 = i21 % args.ne01; - const uint i11 = i21 % args.ne11; - - const float scale = 1.0f / sqrt((float)S_v); - - device const float * s_ptr = (device const float *) (s) + (i23*args.ne21 + i21)*S_v*S_v + i20; - - float lsf[NSG]; - - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - lsf[j] = s_ptr[is*S_v]; - } - - thread T * ls = (thread T *) (lsf); - - device float * dst_attn = (device float *) (dst) + (i23*args.ne22*args.ne21 + i21)*S_v + i20; - - device const float * q_ptr = (device const float *) (q + i23*args.nb03 + i01*args.nb01); - device const float * k_ptr = (device const float *) (k + i23*args.nb13 + i11*args.nb11); - device const float * v_ptr = (device const float *) (v + i23*args.nb23 + i21*args.nb21); - - device const float * b_ptr = (device const float *) (b) + (i23*args.ne22*args.ne21 + i21); - device const float * g_ptr = (device const float *) (g) + (i23*args.ne22*args.ne21 + i21)*G; - - for (short t = 0; t < args.ne22; t++) { - device const T * qt_ptr = (device const T *) (q_ptr); - device const T * kt_ptr = (device const T *) (k_ptr); - device const T * gt_ptr = (device const T *) (g_ptr); - - if (G == 1) { - *ls *= exp(g_ptr[0]); - } else { - // KDA - *ls *= exp(gt_ptr[tx]); - } - - const float s_k = simd_sum(dot(*ls, kt_ptr[tx])); - - const float d = (v_ptr[i20] - s_k)*b_ptr[0]; - - *ls += kt_ptr[tx]*d; - - const float y = simd_sum(dot(*ls, qt_ptr[tx])); - - if (tx == 0) { - *dst_attn = y*scale; - } - - q_ptr += args.ns02; - k_ptr += args.ns12; - v_ptr += args.ns22; - - b_ptr += args.ne21; - g_ptr += args.ne21*G; - - dst_attn += args.ne21*S_v; - } - - device float * dst_state = (device float *) (dst) + args.ne23*args.ne22*args.ne21*S_v + (i23*args.ne21 + i21)*S_v*S_v + i20; - device T * dstt_state = (device T *) (dst_state); - - FOR_UNROLL (short j = 0; j < NSG; j++) { - const short is = tx*NSG + j; - dst_state[is*S_v] = lsf[j]; - } - -#undef S_v -#undef G -} - -typedef decltype(kernel_gated_delta_net_impl) kernel_gated_delta_net_t; - -template [[host_name("kernel_gated_delta_net_f32_1")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; -template [[host_name("kernel_gated_delta_net_f32_2")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; -template [[host_name("kernel_gated_delta_net_f32_4")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; -#endif - -constant short FC_solve_tri_nsg [[function_constant(FC_SOLVE_TRI + 0)]]; -constant short FC_solve_tri_n [[function_constant(FC_SOLVE_TRI + 1)]]; -constant short FC_solve_tri_k [[function_constant(FC_SOLVE_TRI + 2)]]; - -kernel void kernel_solve_tri_f32( - constant ggml_metal_kargs_solve_tri & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - ushort3 tgpig[[threadgroup_position_in_grid]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - constexpr short NW = N_SIMDWIDTH; - - const short NSG = FC_solve_tri_nsg; - const short N = FC_solve_tri_n; - const short K = FC_solve_tri_k; - const short NP = PAD2(N, NW); - - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - const int32_t i01 = tgpig.x*NSG + sgitg; - - threadgroup float * sh0 = (threadgroup float *) shmem; - - device const float * src0_ptr = (device const float *)(src0 + i02 * args.nb02 + i03 * args.nb03) + sgitg*N; - device const float * src1_ptr = (device const float *)(src1 + i02 * args.nb12 + i03 * args.nb13) + i01; - device float * dst_ptr = (device float *)(dst + i02 * args.nb2 + i03 * args.nb3) + i01; - - for (short rr = 0; rr < N; rr += NSG) { - threadgroup_barrier(mem_flags::mem_threadgroup); - - { - threadgroup float * sh0_cur = sh0 + sgitg*NP; - - for (short t = 0; t*NW < N; ++t) { - const short idx = t*NW + tiisg; - sh0_cur[idx] = src0_ptr[idx]; - } - - src0_ptr += NSG*N; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (i01 >= args.ne10) { - continue; - } - - for (short ir = 0; ir < NSG && rr + ir < N; ++ir) { - const short r = rr + ir; - - threadgroup float * sh0_cur = sh0 + ir*NP; - - float sum = 0.0f; - - for (short t = 0; t*NW < r; ++t) { - const short idx = t*NW + tiisg; - sum += sh0_cur[idx] * dst_ptr[idx*K] * (idx < r); - } - - sum = simd_sum(sum); - - if (tiisg == 0) { - const float diag = sh0_cur[r]; - - dst_ptr[r*K] = (src1_ptr[r*K] - sum) / diag; - } - } - } -} - -kernel void kernel_argmax_f32( - constant ggml_metal_kargs_argmax & args, - device const char * src0, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint sgitg[[simdgroup_index_in_threadgroup]], - uint tiisg[[thread_index_in_simdgroup]], - uint ntg[[threads_per_threadgroup]]) { - device const float * x_row = (device const float *) ((device const char *) src0 + tgpig * args.nb01); - - float lmax = -INFINITY; - int32_t larg = -1; - - for (int i00 = tpitg; i00 < args.ne00; i00 += ntg) { - if (x_row[i00] > lmax) { - lmax = x_row[i00]; - larg = i00; - } - } - - // find the argmax value in the block - float max_val = simd_max(lmax); - int32_t arg_val = simd_max(select(-1, larg, lmax == max_val)); - - device int32_t * dst_i32 = (device int32_t *) dst; - - threadgroup float * shared_maxval = (threadgroup float *) shmem; - threadgroup int32_t * shared_argmax = (threadgroup int32_t *) shmem + N_SIMDWIDTH; - - if (ntg > N_SIMDWIDTH) { - if (sgitg == 0) { - shared_maxval[tiisg] = -INFINITY; - shared_argmax[tiisg] = -1; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shared_maxval[sgitg] = max_val; - shared_argmax[sgitg] = arg_val; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - max_val = shared_maxval[tiisg]; - arg_val = shared_argmax[tiisg]; - - float max_val_reduced = simd_max(max_val); - int32_t arg_val_reduced = simd_max(select(-1, arg_val, max_val == max_val_reduced)); - - dst_i32[tgpig] = arg_val_reduced; - - return; - } - - dst_i32[tgpig] = arg_val; -} - -// F == 1 : norm (no fuse) -// F == 2 : norm + mul -// F == 3 : norm + mul + add -template -kernel void kernel_norm_fuse_impl( - constant ggml_metal_kargs_norm & args, - device const char * src0, - device const char * src1_0, - device const char * src1_1, - device char * dst, - threadgroup float * shmem_f32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - if (sgitg == 0) { - shmem_f32[tiisg] = 0.0f; - } - - const int i01 = tgpig.x; - const int i02 = tgpig.y; - const int i03 = tgpig.z; - - device const T * x = (device const T *) (src0 + i03*args.nbf3[0] + i02*args.nbf2[0] + i01*args.nbf1[0]); - - device const T * f0 = (device const T *) (src1_0 + (i03%args.nef3[1])*args.nbf3[1] + (i02%args.nef2[1])*args.nbf2[1] + (i01%args.nef1[1])*args.nbf1[1]); - device const T * f1 = (device const T *) (src1_1 + (i03%args.nef3[2])*args.nbf3[2] + (i02%args.nef2[2])*args.nbf2[2] + (i01%args.nef1[2])*args.nbf1[2]); - - T sumft(0.0f); - - float sumf = 0.0f; - - for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { - sumft += x[i00]; - } - sumf = dot(sumft, T(1.0f)); - sumf = simd_sum(sumf); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shmem_f32[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sumf = shmem_f32[tiisg]; - sumf = simd_sum(sumf); - - const float mean = sumf/args.ne00; - - device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); - - sumf = 0.0f; - for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { - y[i00] = x[i00] - mean; - sumf += dot(y[i00], y[i00]); - } - sumf = simd_sum(sumf); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shmem_f32[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sumf = shmem_f32[tiisg]; - sumf = simd_sum(sumf); - - const float variance = sumf/args.ne00; - - const float scale = 1.0f/sqrt(variance + args.eps); - for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { - if (F == 1) { - y[i00] = (y[i00]*scale); - } - if (F == 2) { - y[i00] = (y[i00]*scale)*f0[i00]; - } - if (F == 3) { - y[i00] = (y[i00]*scale)*f0[i00] + f1[i00]; - } - } -} - -typedef decltype(kernel_norm_fuse_impl) kernel_norm_fuse_t; - -template [[host_name("kernel_norm_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; -template [[host_name("kernel_norm_mul_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; -template [[host_name("kernel_norm_mul_add_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; - -template [[host_name("kernel_norm_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; -template [[host_name("kernel_norm_mul_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; -template [[host_name("kernel_norm_mul_add_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; - -// F == 1 : rms_norm (no fuse) -// F == 2 : rms_norm + mul -// F == 3 : rms_norm + mul + add -template -kernel void kernel_rms_norm_fuse_impl( - constant ggml_metal_kargs_norm & args, - device const char * src0, - device const char * src1_0, - device const char * src1_1, - device char * dst, - threadgroup float * shmem_f32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - if (sgitg == 0) { - shmem_f32[tiisg] = 0.0f; - } - - const int i01 = tgpig.x; - const int i02 = tgpig.y; - const int i03 = tgpig.z; - - device const T * x = (device const T *) (src0 + i03*args.nbf3[0] + i02*args.nbf2[0] + i01*args.nbf1[0]); - - device const T * f0 = (device const T *) (src1_0 + (i03%args.nef3[1])*args.nbf3[1] + (i02%args.nef2[1])*args.nbf2[1] + (i01%args.nef1[1])*args.nbf1[1]); - device const T * f1 = (device const T *) (src1_1 + (i03%args.nef3[2])*args.nbf3[2] + (i02%args.nef2[2])*args.nbf2[2] + (i01%args.nef1[2])*args.nbf1[2]); - - float sumf = 0.0f; - - // parallel sum - for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { - sumf += dot(x[i00], x[i00]); - } - sumf = simd_sum(sumf); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shmem_f32[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sumf = shmem_f32[tiisg]; - sumf = simd_sum(sumf); - - const float mean = sumf/args.ne00; - const float scale = 1.0f/sqrt(mean + args.eps); - - device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); - for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { - if (F == 1) { - y[i00] = (x[i00]*scale); - } - if (F == 2) { - y[i00] = (x[i00]*scale)*f0[i00]; - } - if (F == 3) { - y[i00] = (x[i00]*scale)*f0[i00] + f1[i00]; - } - } -} - -typedef decltype(kernel_rms_norm_fuse_impl) kernel_rms_norm_fuse_t; - -template [[host_name("kernel_rms_norm_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; -template [[host_name("kernel_rms_norm_mul_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; -template [[host_name("kernel_rms_norm_mul_add_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; - -template [[host_name("kernel_rms_norm_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; -template [[host_name("kernel_rms_norm_mul_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; -template [[host_name("kernel_rms_norm_mul_add_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; - -template -kernel void kernel_l2_norm_impl( - constant ggml_metal_kargs_l2_norm & args, - device const char * src0, - device char * dst, - threadgroup float * shmem_f32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int i03 = tgpig.z; - const int i02 = tgpig.y; - const int i01 = tgpig.x; - - if (sgitg == 0) { - shmem_f32[tiisg] = 0.0f; - } - - device const T0 * x = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); - device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); - - float sumf = 0.0f; - - // parallel sum - for (int i00 = tpitg.x; i00 < args.ne00; i00 += ntg.x) { - sumf += dot(x[i00], x[i00]); - } - sumf = simd_sum(sumf); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - shmem_f32[sgitg] = sumf; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - sumf = shmem_f32[tiisg]; - sumf = simd_sum(sumf); - - const float scale = 1.0f/max(sqrt(sumf), args.eps); - - for (int i00 = tpitg.x; i00 < args.ne00; i00 += ntg.x) { - y[i00] = x[i00] * scale; - } -} - -typedef decltype(kernel_l2_norm_impl) kernel_l2_norm_t; - -template [[host_name("kernel_l2_norm_f32_f32")]] kernel kernel_l2_norm_t kernel_l2_norm_impl; -template [[host_name("kernel_l2_norm_f32_f32_4")]] kernel kernel_l2_norm_t kernel_l2_norm_impl; - -kernel void kernel_group_norm_f32( - constant ggml_metal_kargs_group_norm & args, - device const float * src0, - device float * dst, - threadgroup float * buf [[threadgroup(0)]], - uint tgpig[[threadgroup_position_in_grid]], - uint tpitg[[thread_position_in_threadgroup]], - uint sgitg[[simdgroup_index_in_threadgroup]], - uint tiisg[[thread_index_in_simdgroup]], - uint ntg[[threads_per_threadgroup]]) { - const int64_t ne = args.ne00*args.ne01*args.ne02; - const int64_t gs = args.ne00*args.ne01*((args.ne02 + args.ngrp - 1) / args.ngrp); - - int start = tgpig * gs; - int end = start + gs; - - start += tpitg; - - if (end >= ne) { - end = ne; - } - - float tmp = 0.0f; // partial sum for thread in warp - - for (int j = start; j < end; j += ntg) { - tmp += src0[j]; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - tmp = simd_sum(tmp); - if (ntg > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = 0.0f; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = tmp; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - tmp = buf[tiisg]; - tmp = simd_sum(tmp); - } - - const float mean = tmp / gs; - tmp = 0.0f; - - for (int j = start; j < end; j += ntg) { - float xi = src0[j] - mean; - dst[j] = xi; - tmp += xi * xi; - } - - tmp = simd_sum(tmp); - if (ntg > N_SIMDWIDTH) { - if (sgitg == 0) { - buf[tiisg] = 0.0f; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tiisg == 0) { - buf[sgitg] = tmp; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - tmp = buf[tiisg]; - tmp = simd_sum(tmp); - } - - const float variance = tmp / gs; - const float scale = 1.0f/sqrt(variance + args.eps); - for (int j = start; j < end; j += ntg) { - dst[j] *= scale; - } -} - -// Q1_0 dot product: dot = d * (2 * Σ(yl[i] where bit=1) - sumy) -inline float block_q_n_dot_y(device const block_q1_0 * qb_curr, float sumy, thread float * yl, int il) { - device const uint8_t * qs = qb_curr->qs + il / 8; - const uint8_t b0 = qs[0]; - const uint8_t b1 = qs[1]; - - float acc = 0.0f; - - acc += select(0.0f, yl[ 0], bool(b0 & 0x01)); - acc += select(0.0f, yl[ 1], bool(b0 & 0x02)); - acc += select(0.0f, yl[ 2], bool(b0 & 0x04)); - acc += select(0.0f, yl[ 3], bool(b0 & 0x08)); - acc += select(0.0f, yl[ 4], bool(b0 & 0x10)); - acc += select(0.0f, yl[ 5], bool(b0 & 0x20)); - acc += select(0.0f, yl[ 6], bool(b0 & 0x40)); - acc += select(0.0f, yl[ 7], bool(b0 & 0x80)); - - acc += select(0.0f, yl[ 8], bool(b1 & 0x01)); - acc += select(0.0f, yl[ 9], bool(b1 & 0x02)); - acc += select(0.0f, yl[10], bool(b1 & 0x04)); - acc += select(0.0f, yl[11], bool(b1 & 0x08)); - acc += select(0.0f, yl[12], bool(b1 & 0x10)); - acc += select(0.0f, yl[13], bool(b1 & 0x20)); - acc += select(0.0f, yl[14], bool(b1 & 0x40)); - acc += select(0.0f, yl[15], bool(b1 & 0x80)); - - return qb_curr->d * (2.0f * acc - sumy); -} - -// Q2_0 dot: d * (sum_lo(y) + 2*sum_hi(y) - sumy) via per-bit conditional adds -inline float block_q_n_dot_y(device const block_q2_0 * qb_curr, float sumy, thread float * yl, int il) { - device const uint8_t * qs = qb_curr->qs + (il / 4); - const uint8_t b0 = qs[0]; - const uint8_t b1 = qs[1]; - const uint8_t b2 = qs[2]; - const uint8_t b3 = qs[3]; - - // Accumulate where low bit is set (bits 0,2,4,6 of each byte) - float acc_lo = 0.0f; - acc_lo += select(0.0f, yl[ 0], bool(b0 & 0x01)); - acc_lo += select(0.0f, yl[ 1], bool(b0 & 0x04)); - acc_lo += select(0.0f, yl[ 2], bool(b0 & 0x10)); - acc_lo += select(0.0f, yl[ 3], bool(b0 & 0x40)); - acc_lo += select(0.0f, yl[ 4], bool(b1 & 0x01)); - acc_lo += select(0.0f, yl[ 5], bool(b1 & 0x04)); - acc_lo += select(0.0f, yl[ 6], bool(b1 & 0x10)); - acc_lo += select(0.0f, yl[ 7], bool(b1 & 0x40)); - acc_lo += select(0.0f, yl[ 8], bool(b2 & 0x01)); - acc_lo += select(0.0f, yl[ 9], bool(b2 & 0x04)); - acc_lo += select(0.0f, yl[10], bool(b2 & 0x10)); - acc_lo += select(0.0f, yl[11], bool(b2 & 0x40)); - acc_lo += select(0.0f, yl[12], bool(b3 & 0x01)); - acc_lo += select(0.0f, yl[13], bool(b3 & 0x04)); - acc_lo += select(0.0f, yl[14], bool(b3 & 0x10)); - acc_lo += select(0.0f, yl[15], bool(b3 & 0x40)); - - // Accumulate where high bit is set (bits 1,3,5,7 of each byte) - float acc_hi = 0.0f; - acc_hi += select(0.0f, yl[ 0], bool(b0 & 0x02)); - acc_hi += select(0.0f, yl[ 1], bool(b0 & 0x08)); - acc_hi += select(0.0f, yl[ 2], bool(b0 & 0x20)); - acc_hi += select(0.0f, yl[ 3], bool(b0 & 0x80)); - acc_hi += select(0.0f, yl[ 4], bool(b1 & 0x02)); - acc_hi += select(0.0f, yl[ 5], bool(b1 & 0x08)); - acc_hi += select(0.0f, yl[ 6], bool(b1 & 0x20)); - acc_hi += select(0.0f, yl[ 7], bool(b1 & 0x80)); - acc_hi += select(0.0f, yl[ 8], bool(b2 & 0x02)); - acc_hi += select(0.0f, yl[ 9], bool(b2 & 0x08)); - acc_hi += select(0.0f, yl[10], bool(b2 & 0x20)); - acc_hi += select(0.0f, yl[11], bool(b2 & 0x80)); - acc_hi += select(0.0f, yl[12], bool(b3 & 0x02)); - acc_hi += select(0.0f, yl[13], bool(b3 & 0x08)); - acc_hi += select(0.0f, yl[14], bool(b3 & 0x20)); - acc_hi += select(0.0f, yl[15], bool(b3 & 0x80)); - - return qb_curr->d * (acc_lo + 2.0f * acc_hi - sumy); -} - -// function for calculate inner product between half a q4_0 block and 16 floats (yl), sumy is SUM(yl[i]) -// il indicates where the q4 quants begin (0 or QK4_0/4) -// we assume that the yl's have been multiplied with the appropriate scale factor -// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) -inline float block_q_n_dot_y(device const block_q4_0 * qb_curr, float sumy, thread float * yl, int il) { - float d = qb_curr->d; - - float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; - - device const uint16_t * qs = ((device const uint16_t *) qb_curr + 1 + il/2); - - for (int i = 0; i < 8; i += 2) { - acc[0] += yl[i + 0] * (qs[i / 2] & 0x000F); - acc[1] += yl[i + 1] * (qs[i / 2] & 0x0F00); - acc[2] += yl[i + 8] * (qs[i / 2] & 0x00F0); - acc[3] += yl[i + 9] * (qs[i / 2] & 0xF000); - } - - return d * (sumy * -8.f + acc[0] + acc[1] + acc[2] + acc[3]); -} - -// function for calculate inner product between half a q4_1 block and 16 floats (yl), sumy is SUM(yl[i]) -// il indicates where the q4 quants begin (0 or QK4_0/4) -// we assume that the yl's have been multiplied with the appropriate scale factor -// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) -inline float block_q_n_dot_y(device const block_q4_1 * qb_curr, float sumy, thread float * yl, int il) { - float d = qb_curr->d; - float m = qb_curr->m; - - float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; - - device const uint16_t * qs = ((device const uint16_t *) qb_curr + 2 + il/2); - - for (int i = 0; i < 8; i+=2) { - acc[0] += yl[i + 0] * (qs[i / 2] & 0x000F); - acc[1] += yl[i + 1] * (qs[i / 2] & 0x0F00); - acc[2] += yl[i + 8] * (qs[i / 2] & 0x00F0); - acc[3] += yl[i + 9] * (qs[i / 2] & 0xF000); - } - - return d * (acc[0] + acc[1] + acc[2] + acc[3]) + sumy * m; -} - -// function for calculate inner product between half a q5_0 block and 16 floats (yl), sumy is SUM(yl[i]) -// il indicates where the q5 quants begin (0 or QK5_0/4) -// we assume that the yl's have been multiplied with the appropriate scale factor -// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) -inline float block_q_n_dot_y(device const block_q5_0 * qb_curr, float sumy, thread float * yl, int il) { - float d = qb_curr->d; - - float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; - - device const uint16_t * qs = ((device const uint16_t *)qb_curr + 3 + il/2); - const uint32_t qh = *((device const uint32_t *)qb_curr->qh); - - for (int i = 0; i < 8; i+=2) { - acc[0] += yl[i + 0] * ((qs[i / 2] & 0x000F) | ((qh >> (i+0+il ) << 4 ) & 0x00010)); - acc[1] += yl[i + 1] * ((qs[i / 2] & 0x0F00) | ((qh >> (i+1+il ) << 12) & 0x01000)); - acc[2] += yl[i + 8] * ((qs[i / 2] & 0x00F0) | ((qh >> (i+0+il+QK5_0/2) << 8 ) & 0x00100)); - acc[3] += yl[i + 9] * ((qs[i / 2] & 0xF000) | ((qh >> (i+1+il+QK5_0/2) << 16) & 0x10000)); - } - - return d * (sumy * -16.f + acc[0] + acc[1] + acc[2] + acc[3]); -} - -// function for calculate inner product between half a q5_1 block and 16 floats (yl), sumy is SUM(yl[i]) -// il indicates where the q5 quants begin (0 or QK5_1/4) -// we assume that the yl's have been multiplied with the appropriate scale factor -// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) -inline float block_q_n_dot_y(device const block_q5_1 * qb_curr, float sumy, thread float * yl, int il) { - float d = qb_curr->d; - float m = qb_curr->m; - - float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; - - device const uint16_t * qs = ((device const uint16_t *)qb_curr + 4 + il/2); - const uint32_t qh = *((device const uint32_t *)qb_curr->qh); - - for (int i = 0; i < 8; i+=2) { - acc[0] += yl[i + 0] * ((qs[i / 2] & 0x000F) | ((qh >> (i+0+il ) << 4 ) & 0x00010)); - acc[1] += yl[i + 1] * ((qs[i / 2] & 0x0F00) | ((qh >> (i+1+il ) << 12) & 0x01000)); - acc[2] += yl[i + 8] * ((qs[i / 2] & 0x00F0) | ((qh >> (i+0+il+QK5_0/2) << 8 ) & 0x00100)); - acc[3] += yl[i + 9] * ((qs[i / 2] & 0xF000) | ((qh >> (i+1+il+QK5_0/2) << 16) & 0x10000)); - } - - return d * (acc[0] + acc[1] + acc[2] + acc[3]) + sumy * m; -} - -template -static inline void helper_mv_reduce_and_write( - device float * dst_f32, - float sumf[NR0], - const int r0, - const int ne01, - ushort tiisg, - ushort sgitg, - threadgroup char * shmem) { - constexpr short NW = N_SIMDWIDTH; - - threadgroup float * shmem_f32[NR0]; - - for (short row = 0; row < NR0; ++row) { - shmem_f32[row] = (threadgroup float *) shmem + NW*row; - - if (sgitg == 0) { - shmem_f32[row][tiisg] = 0.0f; - } - - sumf[row] = simd_sum(sumf[row]); - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (short row = 0; row < NR0; ++row) { - if (tiisg == 0) { - shmem_f32[row][sgitg] = sumf[row]; - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (short row = 0; row < NR0 && r0 + row < ne01; ++row) { - float tot = simd_sum(shmem_f32[row][tiisg]); - - if (tiisg == 0 && sgitg == 0) { - dst_f32[r0 + row] = tot; - } - } -} - -constant short FC_mul_mv_nsg [[function_constant(FC_MUL_MV + 0)]]; -constant short FC_mul_mv_nxpsg [[function_constant(FC_MUL_MV + 1)]]; -constant short FC_mul_mv_ne12 [[function_constant(FC_MUL_MV + 2)]]; -constant short FC_mul_mv_r2 [[function_constant(FC_MUL_MV + 3)]]; -constant short FC_mul_mv_r3 [[function_constant(FC_MUL_MV + 4)]]; - -template -void mul_vec_q_n_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NQ = 16; - - const int nb = args.ne00/QK4_0; - - const int r0 = (tgpig.x*NSG + sgitg)*NR0; - //const int r0 = tgpig.x*NR0; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - //device const block_q_type * x = (device const block_q_type *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - // pointers to src0 rows - device const block_q_type * ax[NR0]; - FOR_UNROLL (int row = 0; row < NR0; ++row) { - const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - - ax[row] = (device const block_q_type *) ((device char *) src0 + offset0); - } - - float sumf[NR0] = {0.f}; - - const short ix = (tiisg/(NW/NQ)); - const short il = (tiisg%(NW/NQ))*8; - - //const int ib0 = sgitg*NQ + ix; - const int ib0 = ix; - - float yl[16]; // src1 vector cache - - //device const float * yb = y + ix*QK4_0 + il; - device const float * yb = y + ib0*QK4_0 + il; - - // each thread in a SIMD group deals with half a block. - //for (int ib = ib0; ib < nb; ib += NSG*NQ) { - for (int ib = ib0; ib < nb; ib += NQ) { - float sumy[2] = { 0.f, 0.f }; - - FOR_UNROLL (short i = 0; i < 8; i += 2) { - sumy[0] += yb[i + 0] + yb[i + 1]; - yl[i + 0] = yb[i + 0]; - yl[i + 1] = yb[i + 1]/256.f; - - sumy[1] += yb[i + 16] + yb[i + 17]; - yl[i + 8] = yb[i + 16]/16.f; - yl[i + 9] = yb[i + 17]/4096.f; - } - - FOR_UNROLL (short row = 0; row < NR0; row++) { - sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy[0] + sumy[1], yl, il); - } - - yb += QK4_0 * 16; - //yb += NSG*NQ*QK4_0; - } - - device float * dst_f32 = (device float *) dst + im*args.ne0*args.ne1 + r1*args.ne0; - - //helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); - - for (int row = 0; row < NR0; ++row) { - const float tot = simd_sum(sumf[row]); - - if (tiisg == 0 && r0 + row < args.ne01) { - dst_f32[r0 + row] = tot; - } - } -} - -template -void kernel_mul_mv_q1_0_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK1_0; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset1 = r1*args.nb11 + (i12)*args.nb12 + (i13)*args.nb13; - - device const float * y = (device const float *) (src1 + offset1); - - device const block_q1_0 * ax[nr0]; - for (int row = 0; row < nr0; ++row) { - const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - ax[row] = (device const block_q1_0 *) ((device char *) src0 + offset0); - } - - float yl[16]; - float sumf[nr0] = {0.f}; - - const short ix = (tiisg/8); - const short il = (tiisg%8)*16; - - device const float * yb = y + ix*QK1_0 + il; - - for (int ib = ix; ib < nb; ib += N_SIMDWIDTH/8) { - float sumy = 0.f; - - FOR_UNROLL (short i = 0; i < 16; i++) { - yl[i] = yb[i]; - sumy += yb[i]; - } - - FOR_UNROLL (short row = 0; row < nr0; row++) { - sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy, yl, il); - } - - yb += QK1_0 * (N_SIMDWIDTH/8); - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0; ++row) { - const float tot = simd_sum(sumf[row]); - - if (tiisg == 0 && first_row + row < args.ne01) { - dst_f32[first_row + row] = tot; - } - } -} - -[[host_name("kernel_mul_mv_q1_0_f32")]] -kernel void kernel_mul_mv_q1_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_q1_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q2_0_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK2_0; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset1 = r1*args.nb11 + (i12)*args.nb12 + (i13)*args.nb13; - - device const float * y = (device const float *) (src1 + offset1); - - device const block_q2_0 * ax[nr0]; - for (int row = 0; row < nr0; ++row) { - const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - ax[row] = (device const block_q2_0 *) ((device char *) src0 + offset0); - } - - float yl[16]; - float sumf[nr0] = {0.f}; - - // group 64: 4 sub-blocks of 16 weights per Q2_0 block - const short ix = (tiisg/4); - const short il = (tiisg%4)*16; - - device const float * yb = y + ix*QK2_0 + il; - - for (int ib = ix; ib < nb; ib += N_SIMDWIDTH/4) { - float sumy = 0.f; - - FOR_UNROLL (short i = 0; i < 16; i++) { - yl[i] = yb[i]; - sumy += yb[i]; - } - - FOR_UNROLL (short row = 0; row < nr0; row++) { - sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy, yl, il); - } - - yb += QK2_0 * (N_SIMDWIDTH/4); - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0; ++row) { - const float tot = simd_sum(sumf[row]); - - if (tiisg == 0 && first_row + row < args.ne01) { - dst_f32[first_row + row] = tot; - } - } -} - -[[host_name("kernel_mul_mv_q2_0_f32")]] -kernel void kernel_mul_mv_q2_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_q2_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -kernel void kernel_mul_mv_q4_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -kernel void kernel_mul_mv_q4_1_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -kernel void kernel_mul_mv_q5_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -kernel void kernel_mul_mv_q5_1_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q8_0_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NQ = 8; - - const int nb = args.ne00/QK8_0; - - const int r0 = tgpig.x*NR0; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - //device const block_q8_0 * x = (device const block_q8_0 *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - // pointers to src0 rows - device const block_q8_0 * ax[NR0]; - FOR_UNROLL (short row = 0; row < NR0; ++row) { - const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - - ax[row] = (device const block_q8_0 *) ((device char *) src0 + offset0); - } - - float sumf[NR0] = { 0.f }; - - const short ix = tiisg/(NW/NQ); - const short il = tiisg%(NW/NQ); - - const int ib0 = sgitg*NQ + ix; - - float yl[NQ]; - - device const float * yb = y + ib0*QK8_0 + il*NQ; - - // each thread in a SIMD group deals with NQ quants at a time - for (int ib = ib0; ib < nb; ib += NSG*NQ) { - for (short i = 0; i < NQ; ++i) { - yl[i] = yb[i]; - } - - for (short row = 0; row < NR0; row++) { - device const int8_t * qs = ax[row][ib].qs + il*NQ; - - float sumq = 0.f; - FOR_UNROLL (short i = 0; i < NQ; ++i) { - sumq += qs[i] * yl[i]; - } - - sumf[row] += sumq*ax[row][ib].d; - } - - yb += NSG*NQ*QK8_0; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); -} - -[[host_name("kernel_mul_mv_q8_0_f32")]] -kernel void kernel_mul_mv_q8_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_q8_0_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -// mat-vec kernel processing in chunks of float4 -// chpb - chunks per quantization block -template -void kernel_mul_mv_ext_q4_f32_impl( - constant ggml_metal_kargs_mul_mv_ext & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - const short NSG = FC_mul_mv_nsg; - const short nxpsg = FC_mul_mv_nxpsg; - - const short chpt = 4; // chunks per thread - - //const short nxpsg = (32); - const short nypsg = (32/nxpsg); - - const short tx = tiisg%nxpsg; - const short ty = tiisg/nxpsg; - - const int i01 = tgpig.x*(nypsg*NSG) + nypsg*sgitg + ty; - const int i11 = tgpig.y*r1ptg; - const int i1m = tgpig.z; - - const int i12 = i1m%FC_mul_mv_ne12; - const int i13 = i1m/FC_mul_mv_ne12; - - const uint64_t offset0 = i01*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = i11*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const q_t * xq = (i01 < args.ne01) ? (device const q_t *) (src0 + offset0) + tx/chpb : (device const q_t *) src0; - - device const float4 * y4[r1ptg]; - - for (int ir1 = 0; ir1 < r1ptg; ++ir1) { - y4[ir1] = (i11 + ir1 < args.ne11) ? (device const float4 *) (src1 + offset1 + ir1*args.nb11) + tx : (device const float4 *) src1; - } - - float sumf[r1ptg] = { [ 0 ... r1ptg - 1 ] = 0.0f }; - - short cch = tx%chpb; // current chunk index - - for (int ich = tx; 4*ich < args.ne00; ich += chpt*nxpsg) { - float4 lx[chpt]; - -#pragma unroll(chpt) - for (short ch = 0; ch < chpt; ++ch) { - deq_t4(xq, cch, lx[ch]); - - cch += nxpsg; - if (cch >= chpb) { - xq += cch/chpb; - cch %= chpb; - } - } - -#pragma unroll(chpt) - for (short ch = 0; ch < chpt; ++ch) { -#pragma unroll(r1ptg) - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - sumf[ir1] += dot(lx[ch], y4[ir1][ch*nxpsg]); - } - } - -#pragma unroll(r1ptg) - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - y4[ir1] += chpt*nxpsg; - } - } - - // reduce only the threads in each row - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - if (nxpsg >= 32) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 16); - } - if (nxpsg >= 16) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 8); - } - if (nxpsg >= 8) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 4); - } - if (nxpsg >= 4) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 2); - } - if (nxpsg >= 2) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 1); - } - - //sumf[ir1] = simd_sum(sumf[ir1]); - } - - if (tx == 0) { - for (short ir1 = 0; ir1 < r1ptg && i11 + ir1 < args.ne11; ++ir1) { - device float * dst_f32 = (device float *) dst + (uint64_t)i1m*args.ne0*args.ne1 + (uint64_t)(i11 + ir1)*args.ne0; - - if (i01 < args.ne01) { - dst_f32[i01] = sumf[ir1]; - } - } - } -} - -// mat-vec kernel processing in chunks of float4x4 -template -void kernel_mul_mv_ext_q4x4_f32_impl( - constant ggml_metal_kargs_mul_mv_ext & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - const short NSG = FC_mul_mv_nsg; - const short nxpsg = FC_mul_mv_nxpsg; - - const short chpt = 1; - - //const short nxpsg = (32); - const short nypsg = (32/nxpsg); - - const short tx = tiisg%nxpsg; - const short ty = tiisg/nxpsg; - - const int i01 = tgpig.x*(nypsg*NSG) + nypsg*sgitg + ty; - const int i11 = tgpig.y*r1ptg; - const int i1m = tgpig.z; - - const int i12 = i1m%FC_mul_mv_ne12; - const int i13 = i1m/FC_mul_mv_ne12; - - const uint64_t offset0 = i01*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = i11*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const q_t * xq = (i01 < args.ne01) ? (device const q_t *) (src0 + offset0) + tx/chpb : (device const q_t *) src0; - - device const float4x4 * y4x4[r1ptg]; - - for (int ir1 = 0; ir1 < r1ptg; ++ir1) { - y4x4[ir1] = (i11 + ir1 < args.ne11) ? (device const float4x4 *) (src1 + offset1 + ir1*args.nb11) + tx : (device const float4x4 *) src1; - } - - float sumf[r1ptg] = { [ 0 ... r1ptg - 1 ] = 0.0f }; - - short cch = tx%chpb; - - for (int ich = tx; 16*ich < args.ne00; ich += chpt*nxpsg) { - float4x4 lx[chpt]; - -#pragma unroll(chpt) - for (short ch = 0; ch < chpt; ++ch) { - deq_t4x4(xq, cch, lx[ch]); - - cch += nxpsg; - if (cch >= chpb) { - xq += cch/chpb; - cch %= chpb; - } - } - -#pragma unroll(chpt) - for (short ch = 0; ch < chpt; ++ch) { -#pragma unroll(r1ptg) - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - sumf[ir1] += - dot(lx[ch][0], y4x4[ir1][ch*nxpsg][0]) + - dot(lx[ch][1], y4x4[ir1][ch*nxpsg][1]) + - dot(lx[ch][2], y4x4[ir1][ch*nxpsg][2]) + - dot(lx[ch][3], y4x4[ir1][ch*nxpsg][3]); - - } - } - -#pragma unroll(r1ptg) - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - y4x4[ir1] += chpt*nxpsg; - } - } - - for (short ir1 = 0; ir1 < r1ptg; ++ir1) { - if (nxpsg >= 32) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 16); - } - if (nxpsg >= 16) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 8); - } - if (nxpsg >= 8) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 4); - } - if (nxpsg >= 4) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 2); - } - if (nxpsg >= 2) { - sumf[ir1] += simd_shuffle_down(sumf[ir1], 1); - } - - //sumf[ir1] = simd_sum(sumf[ir1]); - } - - if (tx == 0) { - for (short ir1 = 0; ir1 < r1ptg && i11 + ir1 < args.ne11; ++ir1) { - device float * dst_f32 = (device float *) dst + (uint64_t)i1m*args.ne0*args.ne1 + (uint64_t)(i11 + ir1)*args.ne0; - - if (i01 < args.ne01) { - dst_f32[i01] = sumf[ir1]; - } - } - } -} - -// dispatchers needed for compile-time nxpsg -// epb - elements per quantization block -template -kernel void kernel_mul_mv_ext_q4_f32_disp( - constant ggml_metal_kargs_mul_mv_ext & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_ext_q4_f32_impl(args, src0, src1, dst, tgpig, tiisg, sgitg); -} - -template -kernel void kernel_mul_mv_ext_q4x4_f32_disp( - constant ggml_metal_kargs_mul_mv_ext & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_ext_q4x4_f32_impl(args, src0, src1, dst, tgpig, tiisg, sgitg); -} - -typedef decltype(kernel_mul_mv_ext_q4_f32_disp <2, block_q8_0, 32, dequantize_q8_0_t4>) mul_mv_ext_q4_f32_t; -typedef decltype(kernel_mul_mv_ext_q4x4_f32_disp<2, block_q4_K, 256, dequantize_q4_K>) mul_mv_ext_q4x4_f32_t; - -template [[host_name("kernel_mul_mv_ext_f32_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, float4, 4, dequantize_f32_t4>; -template [[host_name("kernel_mul_mv_ext_f32_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, float4, 4, dequantize_f32_t4>; -template [[host_name("kernel_mul_mv_ext_f32_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, float4, 4, dequantize_f32_t4>; -template [[host_name("kernel_mul_mv_ext_f32_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, float4, 4, dequantize_f32_t4>; - -template [[host_name("kernel_mul_mv_ext_f16_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, half4, 4, dequantize_f16_t4>; -template [[host_name("kernel_mul_mv_ext_f16_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, half4, 4, dequantize_f16_t4>; -template [[host_name("kernel_mul_mv_ext_f16_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, half4, 4, dequantize_f16_t4>; -template [[host_name("kernel_mul_mv_ext_f16_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, half4, 4, dequantize_f16_t4>; - -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, bfloat4, 4, dequantize_bf16_t4>; -template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, bfloat4, 4, dequantize_bf16_t4>; -template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, bfloat4, 4, dequantize_bf16_t4>; -template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, bfloat4, 4, dequantize_bf16_t4>; -#endif - -template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q1_0, 128, dequantize_q1_0_t4>; -template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q1_0, 128, dequantize_q1_0_t4>; -template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q1_0, 128, dequantize_q1_0_t4>; -template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q1_0, 128, dequantize_q1_0_t4>; - -template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q2_0, 64, dequantize_q2_0_t4>; -template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q2_0, 64, dequantize_q2_0_t4>; -template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q2_0, 64, dequantize_q2_0_t4>; -template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q2_0, 64, dequantize_q2_0_t4>; - -template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q4_0, 32, dequantize_q4_0_t4>; -template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q4_0, 32, dequantize_q4_0_t4>; -template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q4_0, 32, dequantize_q4_0_t4>; -template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q4_0, 32, dequantize_q4_0_t4>; - -template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q4_1, 32, dequantize_q4_1_t4>; -template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q4_1, 32, dequantize_q4_1_t4>; -template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q4_1, 32, dequantize_q4_1_t4>; -template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q4_1, 32, dequantize_q4_1_t4>; - -template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q5_0, 32, dequantize_q5_0_t4>; -template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q5_0, 32, dequantize_q5_0_t4>; -template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q5_0, 32, dequantize_q5_0_t4>; -template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q5_0, 32, dequantize_q5_0_t4>; - -template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q5_1, 32, dequantize_q5_1_t4>; -template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q5_1, 32, dequantize_q5_1_t4>; -template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q5_1, 32, dequantize_q5_1_t4>; -template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q5_1, 32, dequantize_q5_1_t4>; - -template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q8_0, 32, dequantize_q8_0_t4>; -template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q8_0, 32, dequantize_q8_0_t4>; -template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q8_0, 32, dequantize_q8_0_t4>; -template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q8_0, 32, dequantize_q8_0_t4>; - -template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_mxfp4, 32, dequantize_mxfp4_t4>; -template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_mxfp4, 32, dequantize_mxfp4_t4>; -template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_mxfp4, 32, dequantize_mxfp4_t4>; -template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_mxfp4, 32, dequantize_mxfp4_t4>; - -template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_iq4_nl, 32, dequantize_iq4_nl_t4>; -template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_iq4_nl, 32, dequantize_iq4_nl_t4>; -template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_iq4_nl, 32, dequantize_iq4_nl_t4>; -template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_iq4_nl, 32, dequantize_iq4_nl_t4>; - -template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q4_K, 256, dequantize_q4_K>; -template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q4_K, 256, dequantize_q4_K>; -template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q4_K, 256, dequantize_q4_K>; -template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q4_K, 256, dequantize_q4_K>; - -template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q5_K, 256, dequantize_q5_K>; -template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q5_K, 256, dequantize_q5_K>; -template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q5_K, 256, dequantize_q5_K>; -template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q5_K, 256, dequantize_q5_K>; - -template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q6_K, 256, dequantize_q6_K>; -template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q6_K, 256, dequantize_q6_K>; -template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q6_K, 256, dequantize_q6_K>; -template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q6_K, 256, dequantize_q6_K>; - -template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q2_K, 256, dequantize_q2_K>; -template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q2_K, 256, dequantize_q2_K>; -template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q2_K, 256, dequantize_q2_K>; -template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q2_K, 256, dequantize_q2_K>; - -template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q3_K, 256, dequantize_q3_K>; -template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q3_K, 256, dequantize_q3_K>; -template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q3_K, 256, dequantize_q3_K>; -template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q3_K, 256, dequantize_q3_K>; - -template -void kernel_mul_mv_t_t_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NB = 32; - constexpr short NF = 8; - - const int nb = args.ne00/NB; - - const int r0 = tgpig.x*NR0; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - //device const T0 * x = (device const T0 *) (src0 + offset0); - device const T1 * y = (device const T1 *) (src1 + offset1); - - // pointers to src0 rows - device const T0 * ax [NR0]; - FOR_UNROLL (short row = 0; row < NR0; ++row) { - const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - - ax[row] = (device const T0 *) ((device char *) src0 + offset0); - } - - float sumf[NR0] = { 0.f }; - - const short ix = tiisg/(NW/NF); - const short il = tiisg%(NW/NF); - - const int ib0 = sgitg*NF + ix; - - T1 yl[NF]; - - device const T1 * yb = y + (ib0*NB + il*NF); - - for (int ib = ib0; ib < nb; ib += NSG*NF) { - for (short i = 0; i < NF; ++i) { - yl[i] = yb[i]; - } - - for (short row = 0; row < NR0; row++) { - device const T0 * xb = ax[row] + (ib*NB + il*NF); - - float sumq = 0.f; - FOR_UNROLL (short i = 0; i < NF; ++i) { - sumq += xb[i] * yl[i]; - } - - sumf[row] += sumq; - } - - yb += NSG*NF*NW; - } - - for (int i = nb*NB + sgitg*NW + tiisg; i < args.ne00; i += NW*NSG) { - for (short row = 0; row < NR0; row++) { - sumf[row] += ax[row][i] * y[i]; - } - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); -} - -template -void kernel_mul_mv_t_t_disp( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - switch (args.nr0) { - //case 1: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - case 2: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - //case 3: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - //case 4: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - } -} - -template -kernel void kernel_mul_mv_t_t( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_t_t_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -typedef decltype(kernel_mul_mv_t_t) mul_mv_t_t; - -template [[host_name("kernel_mul_mv_f32_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; -template [[host_name("kernel_mul_mv_f16_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; -template [[host_name("kernel_mul_mv_f16_f16")]] kernel mul_mv_t_t kernel_mul_mv_t_t; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_bf16_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; -template [[host_name("kernel_mul_mv_bf16_bf16")]] kernel mul_mv_t_t kernel_mul_mv_t_t; -#endif - -template -void kernel_mul_mv_t_t_4_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NB = 32; - constexpr short NF = 16; - constexpr short NF4 = NF/4; - - const int nb = args.ne00/NB; - - const int r0 = tgpig.x*NR0; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const T1 * y = (device const T1 *) (src1 + offset1); - device const T14 * y4 = (device const T14 *) (src1 + offset1); - - // pointers to src0 rows - device const T0 * ax [NR0]; - device const T04 * ax4[NR0]; - FOR_UNROLL (short row = 0; row < NR0; ++row) { - const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - - ax [row] = (device const T0 *) ((device char *) src0 + offset0); - ax4[row] = (device const T04 *) ((device char *) src0 + offset0); - } - - float sumf[NR0] = { 0.f }; - - const short ix = tiisg/(NW/NF); - const short il = tiisg%(NW/NF); - - const int ib0 = sgitg*NF + ix; - - T14 yl4[NF4]; - - device const T14 * yb4 = y4 + (ib0*NB + il*NF)/4; - - for (int ib = ib0; ib < nb; ib += NSG*NF) { - for (short i = 0; i < NF4; ++i) { - yl4[i] = yb4[i]; - } - - for (short row = 0; row < NR0; row++) { - device const T04 * xb4 = ax4[row] + (ib*NB + il*NF)/4; - - float sumq = 0.f; - FOR_UNROLL (short i = 0; i < NF4; ++i) { - sumq += dot(float4(xb4[i]), float4(yl4[i])); - } - - sumf[row] += sumq; - } - - yb4 += NSG*NF*NW/4; - } - - for (int i = nb*NB + sgitg*NW + tiisg; i < args.ne00; i += NW*NSG) { - for (short row = 0; row < NR0; row++) { - sumf[row] += ax[row][i] * y[i]; - } - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); -} - -template -void kernel_mul_mv_t_t_4_disp( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - switch (args.nr0) { - //case 1: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - case 2: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - //case 3: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - //case 4: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; - }; -} - -template -kernel void kernel_mul_mv_t_t_4( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_t_t_4_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -typedef decltype(kernel_mul_mv_t_t_4) mul_mv_t_t_4; - -template [[host_name("kernel_mul_mv_f32_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; -template [[host_name("kernel_mul_mv_f16_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; -template [[host_name("kernel_mul_mv_f16_f16_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_bf16_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; -template [[host_name("kernel_mul_mv_bf16_bf16_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; -#endif - -template -void kernel_mul_mv_t_t_short_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig, - ushort tiisg) { - const int r0 = tgpig.x*32 + tiisg; - const int r1 = tgpig.y; - const int im = tgpig.z; - - if (r0 >= args.ne01) { - return; - } - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - - device const T0 * x = (device const T0 *) (src0 + offset0); - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1; - - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const T1 * y = (device const T1 *) (src1 + offset1); - - float res = 0.0f; - - for (int i = 0; i < args.ne00; ++i) { - res += (float) x[i] * (float) y[i]; - } - - dst_f32[(uint64_t)r1*args.ne0 + r0] = res; -} - -template -kernel void kernel_mul_mv_t_t_short( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]]) { - kernel_mul_mv_t_t_short_impl( - args, - src0, - src1, - dst, - tgpig, - tiisg); -} - -typedef decltype(kernel_mul_mv_t_t_short) mul_mv_t_t_short_t; - -template [[host_name("kernel_mul_mv_f32_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; -template [[host_name("kernel_mul_mv_f16_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; -template [[host_name("kernel_mul_mv_f16_f16_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_bf16_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; -template [[host_name("kernel_mul_mv_bf16_bf16_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; -#endif - -constant bool FC_rope_is_imrope [[function_constant(FC_ROPE + 0)]]; -constant bool FC_rope_is_back [[function_constant(FC_ROPE + 1)]]; - -static float rope_yarn_ramp(const float low, const float high, const int i0) { - const float y = (i0 / 2 - low) / max(0.001f, high - low); - return 1.0f - min(1.0f, max(0.0f, y)); -} - -// YaRN algorithm based on LlamaYaRNScaledRotaryEmbedding.py from https://github.com/jquesnelle/yarn -// MIT licensed. Copyright (c) 2023 Jeffrey Quesnelle and Bowen Peng. -static void rope_yarn( - float theta_extrap, float freq_scale, float corr_dims[2], int i0, float ext_factor, float mscale, - thread float * cos_theta, thread float * sin_theta) { - // Get n-d rotational scaling corrected for extrapolation - float theta_interp = freq_scale * theta_extrap; - float theta = theta_interp; - if (ext_factor != 0.0f) { - float ramp_mix = rope_yarn_ramp(corr_dims[0], corr_dims[1], i0) * ext_factor; - theta = theta_interp * (1 - ramp_mix) + theta_extrap * ramp_mix; - - // Get n-d magnitude scaling corrected for interpolation - mscale *= 1.0f + 0.1f * log(1.0f / freq_scale); - } - *cos_theta = cos(theta) * mscale; - *sin_theta = sin(theta) * mscale; - if (FC_rope_is_back) { - *sin_theta *= -1.0f; - } -} - -// Apparently solving `n_rot = 2pi * x * base^((2 * max_pos_emb) / n_dims)` for x, we get -// `corr_fac(n_rot) = n_dims * log(max_pos_emb / (n_rot * 2pi)) / (2 * log(base))` -static float rope_yarn_corr_factor(int n_dims, int n_ctx_orig, float n_rot, float base) { - return n_dims * log(n_ctx_orig / (n_rot * 2 * M_PI_F)) / (2 * log(base)); -} - -static void rope_yarn_corr_dims( - int n_dims, int n_ctx_orig, float freq_base, float beta_fast, float beta_slow, float dims[2] -) { - // start and end correction dims - dims[0] = max(0.0f, floor(rope_yarn_corr_factor(n_dims, n_ctx_orig, beta_fast, freq_base))); - dims[1] = min(n_dims - 1.0f, ceil(rope_yarn_corr_factor(n_dims, n_ctx_orig, beta_slow, freq_base))); -} - -template -kernel void kernel_rope_norm( - constant ggml_metal_kargs_rope & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 tptg [[threads_per_threadgroup]], - uint3 tgpig[[threadgroup_position_in_grid]]) { - const int i3 = tgpig[2]; - const int i2 = tgpig[1]; - const int i1 = tgpig[0]; - - float corr_dims[2]; - rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); - - device const int32_t * pos = (device const int32_t *) src1; - - const float theta_base = (float) pos[i2]; - const float inv_ndims = -1.f/args.n_dims; - - float cos_theta; - float sin_theta; - - for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { - if (i0 < args.n_dims) { - const int ic = i0/2; - - const float theta = theta_base * pow(args.freq_base, inv_ndims*i0); - - const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; - - rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, i0, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); - - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - const float x0 = src[0]; - const float x1 = src[1]; - - dst_data[0] = x0*cos_theta - x1*sin_theta; - dst_data[1] = x0*sin_theta + x1*cos_theta; - } else { - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - dst_data[0] = src[0]; - dst_data[1] = src[1]; - } - } -} - -template -kernel void kernel_rope_neox( - constant ggml_metal_kargs_rope & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 tptg [[threads_per_threadgroup]], - uint3 tgpig[[threadgroup_position_in_grid]]) { - const int i3 = tgpig[2]; - const int i2 = tgpig[1]; - const int i1 = tgpig[0]; - - float corr_dims[2]; - rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); - - device const int32_t * pos = (device const int32_t *) src1; - - const float theta_base = (float) pos[i2]; - const float inv_ndims = -1.f/args.n_dims; - - float cos_theta; - float sin_theta; - - for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { - if (i0 < args.n_dims) { - const int ic = i0/2; - - const float theta = theta_base * pow(args.freq_base, inv_ndims*i0); - - const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; - - rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, i0, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); - - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + ic*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + ic*args.nb0); - - const float x0 = src[0]; - const float x1 = src[args.n_dims/2]; - - dst_data[0] = x0*cos_theta - x1*sin_theta; - dst_data[args.n_dims/2] = x0*sin_theta + x1*cos_theta; - } else { - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - dst_data[0] = src[0]; - dst_data[1] = src[1]; - } - } -} - -template -kernel void kernel_rope_multi( - constant ggml_metal_kargs_rope & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 tptg [[threads_per_threadgroup]], - uint3 tgpig[[threadgroup_position_in_grid]]) { - const int i3 = tgpig[2]; - const int i2 = tgpig[1]; - const int i1 = tgpig[0]; - - float corr_dims[2]; - rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); - - device const int32_t * pos = (device const int32_t *) src1; - - const float inv_ndims = -1.f/args.n_dims; - - float cos_theta; - float sin_theta; - - for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { - if (i0 < args.n_dims) { - const int ic = i0/2; - - // mrope theta calculations - // note: the rest is the same as kernel_rope_neox - const int sect_dims = args.sect_0 + args.sect_1 + args.sect_2 + args.sect_3; - const int sec_w01 = args.sect_0 + args.sect_1; // end of section 1 - const int sec_w012 = args.sect_0 + args.sect_1 + args.sect_2; // end of section 2 - const int sector = ic % sect_dims; - - float theta_base; - if (FC_rope_is_imrope) { - if (sector % 3 == 1 && sector < 3 * args.sect_1) { // h - theta_base = (float) pos[i2 + args.ne02 * 1]; - } else if (sector % 3 == 2 && sector < 3 * args.sect_2) { // w - theta_base = (float) pos[i2 + args.ne02 * 2]; - } else if (sector % 3 == 0 && sector < 3 * args.sect_0) { // t - theta_base = (float) pos[i2 + args.ne02 * 0]; - } else { // e - theta_base = (float) pos[i2 + args.ne02 * 3]; - } - } else { - if (sector < args.sect_0) { - theta_base = (float) pos[i2]; - } else if (sector < sec_w01) { - theta_base = (float) pos[i2 + args.ne02 * 1]; - } else if (sector < sec_w012) { - theta_base = (float) pos[i2 + args.ne02 * 2]; - } else { - theta_base = (float) pos[i2 + args.ne02 * 3]; - } - } - // end of mrope - - const float theta = theta_base * pow(args.freq_base, inv_ndims*i0); - - const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; - - rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, i0, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); - - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + ic*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + ic*args.nb0); - - const float x0 = src[0]; - const float x1 = src[args.n_dims/2]; - - dst_data[0] = x0*cos_theta - x1*sin_theta; - dst_data[args.n_dims/2] = x0*sin_theta + x1*cos_theta; - } else { - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - dst_data[0] = src[0]; - dst_data[1] = src[1]; - } - } -} - -template -kernel void kernel_rope_vision( - constant ggml_metal_kargs_rope & args, - device const char * src0, - device const char * src1, - device const char * src2, - device char * dst, - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 tptg [[threads_per_threadgroup]], - uint3 tgpig[[threadgroup_position_in_grid]]) { - const int i3 = tgpig[2]; - const int i2 = tgpig[1]; - const int i1 = tgpig[0]; - - float corr_dims[2]; - rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); - - device const int32_t * pos = (device const int32_t *) src1; - - const float inv_ndims = -1.f/args.n_dims; - - float cos_theta; - float sin_theta; - - for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { - if (i0 < 2*args.n_dims) { // different from kernel_rope_multi - const int ic = i0/2; - - // mrope theta calculations (only support 2 dimensions) - const int sect_dims = args.sect_0 + args.sect_1; - const int sector = ic % sect_dims; - - float p; - float theta_base; - if (sector < args.sect_1) { - p = (float) sector; - theta_base = (float) pos[i2]; - } else { - p = (float) sector - args.sect_0; - theta_base = (float) pos[i2 + args.ne02]; - } - - const float theta = theta_base * pow(args.freq_base, 2.0f * inv_ndims * p); - // end of mrope - - const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; - - rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, i0, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); - - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + ic*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + ic*args.nb0); - - const float x0 = src[0]; - const float x1 = src[args.n_dims]; // different from kernel_rope_multi - - dst_data[0] = x0*cos_theta - x1*sin_theta; - dst_data[args.n_dims] = x0*sin_theta + x1*cos_theta; // different from kernel_rope_multi - } else { - device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); - device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - dst_data[0] = src[0]; - dst_data[1] = src[1]; - } - } -} - -typedef decltype(kernel_rope_norm) kernel_rope_norm_t; -typedef decltype(kernel_rope_neox) kernel_rope_neox_t; -typedef decltype(kernel_rope_multi) kernel_rope_multi_t; -typedef decltype(kernel_rope_vision) kernel_rope_vision_t; - -template [[host_name("kernel_rope_norm_f32")]] kernel kernel_rope_norm_t kernel_rope_norm; -template [[host_name("kernel_rope_norm_f16")]] kernel kernel_rope_norm_t kernel_rope_norm; - -template [[host_name("kernel_rope_neox_f32")]] kernel kernel_rope_neox_t kernel_rope_neox; -template [[host_name("kernel_rope_neox_f16")]] kernel kernel_rope_neox_t kernel_rope_neox; - -template [[host_name("kernel_rope_multi_f32")]] kernel kernel_rope_multi_t kernel_rope_multi; -template [[host_name("kernel_rope_multi_f16")]] kernel kernel_rope_multi_t kernel_rope_multi; - -template [[host_name("kernel_rope_vision_f32")]] kernel kernel_rope_vision_t kernel_rope_vision; -template [[host_name("kernel_rope_vision_f16")]] kernel kernel_rope_vision_t kernel_rope_vision; - -typedef void (im2col_t)( - constant ggml_metal_kargs_im2col & args, - device const float * x, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template -kernel void kernel_im2col( - constant ggml_metal_kargs_im2col & args, - device const float * x, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { -// const int64_t IC = tgpg[0]; - const int64_t OH = tgpg[1]; - const int64_t OW = tgpg[2]; - - const int64_t KH = ntg[1]; - const int64_t KW = ntg[2]; - - int64_t in = tpitg[0]; - const int64_t ikh = tpitg[1]; - const int64_t ikw = tpitg[2]; - - const int64_t iic = tgpig[0]; - const int64_t ioh = tgpig[1]; - const int64_t iow = tgpig[2]; - - const int64_t iiw = iow*args.s0 + ikw*args.d0 - args.p0; - const int64_t iih = ioh*args.s1 + ikh*args.d1 - args.p1; - - int64_t offset_dst = (in*OH*OW + ioh*OW + iow)*args.CHW + (iic*(KH*KW) + ikh*KW + ikw); - - device T * pdst = (device T *) (dst); - - if (iih < 0 || iih >= args.IH || iiw < 0 || iiw >= args.IW) { - while (in < args.N) { - pdst[offset_dst] = 0.0f; - offset_dst += ntg[0]*args.CHW*OH*OW; - - in += ntg[0]; - } - } else { - int64_t offset_src = in*args.ofs0 + iic*args.ofs1 + iih*args.IW + iiw; - - while (in < args.N) { - pdst[offset_dst] = x[offset_src]; - - offset_dst += ntg[0]*args.CHW*OH*OW; - offset_src += ntg[0]*args.ofs0; - - in += ntg[0]; - } - } -} - -template [[host_name("kernel_im2col_f32")]] kernel im2col_t kernel_im2col; -template [[host_name("kernel_im2col_f16")]] kernel im2col_t kernel_im2col; - -// TODO: optimize -typedef void (im2col_ext_t)( - constant ggml_metal_kargs_im2col & args, - device const float * x, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template -kernel void kernel_im2col_ext( - constant ggml_metal_kargs_im2col & args, - device const float * x, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], // tgpg[0] = D x IC x KH x KW, CHW = IC x KH x KW - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { // [M, 1, 1] - const int64_t KHW = (int64_t)args.KHW; - - const int64_t d = tgpig[0] / args.CHW; - const int64_t chw = tgpig[0] % args.CHW; - const int64_t tgpig_0 = chw / KHW; // 0 ~ (IC - 1) - const int64_t HW = tgpig[0] % KHW; - - const int64_t tpitg_0 = (d * ntg[0]) + tpitg[0]; - if (tpitg_0 >= args.N) { - return; - } - - const int64_t tpitg_1 = HW / args.KW; - const int64_t tpitg_2 = HW % args.KW; - - const int64_t iiw = tgpig[2] * args.s0 + tpitg_2 * args.d0 - args.p0; - const int64_t iih = tgpig[1] * args.s1 + tpitg_1 * args.d1 - args.p1; - - const int64_t offset_dst = - (tpitg_0 * tgpg[1] * tgpg[2] + tgpig[1] * tgpg[2] + tgpig[2]) * args.CHW + - (tgpig_0 * KHW + tpitg_1 * args.KW + tpitg_2); - - device T * pdst = (device T *) (dst); - - if (iih < 0 || iih >= args.IH || iiw < 0 || iiw >= args.IW) { - pdst[offset_dst] = 0.0f; - } else { - const int64_t offset_src = tpitg_0 * args.ofs0 + tgpig_0 * args.ofs1; - pdst[offset_dst] = x[offset_src + iih * args.IW + iiw]; - } -} - -template [[host_name("kernel_im2col_ext_f32")]] kernel im2col_ext_t kernel_im2col_ext; -template [[host_name("kernel_im2col_ext_f16")]] kernel im2col_ext_t kernel_im2col_ext; - -template -kernel void kernel_conv_2d( - constant ggml_metal_kargs_conv_2d & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const uint threads_per_tg = ntg.x * ntg.y * ntg.z; - const uint tg_index = (tgpig.z * tgpg.y + tgpig.y) * tgpg.x + tgpig.x; - const uint local_thread = tpitg.z * (ntg.x * ntg.y) + tpitg.y * ntg.x + tpitg.x; - const uint thread_index = tg_index * threads_per_tg + local_thread; - const uint64_t total_threads = (uint64_t) threads_per_tg * tgpg.x * tgpg.y * tgpg.z; - const uint64_t total_outputs = (uint64_t) args.N * args.OC * args.OH * args.OW; - - for (uint64_t index = thread_index; index < total_outputs; index += total_threads) { - uint64_t tmp = index; - - const int32_t ow = tmp % args.OW; tmp /= args.OW; - const int32_t oh = tmp % args.OH; tmp /= args.OH; - const int32_t oc = tmp % args.OC; tmp /= args.OC; - const int32_t n = tmp; - - float acc = 0.0f; - - const int32_t base_x = ow*args.s0 - args.p0; - const int32_t base_y = oh*args.s1 - args.p1; - - int32_t ky_start = 0; - if (base_y < 0) { - ky_start = (-base_y + args.d1 - 1)/args.d1; - } - int32_t ky_end = args.KH; - const int32_t y_max = args.IH - 1 - base_y; - if (y_max < 0) { - ky_end = ky_start; - } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { - ky_end = min(ky_end, y_max/args.d1 + 1); - } - - int32_t kx_start = 0; - if (base_x < 0) { - kx_start = (-base_x + args.d0 - 1)/args.d0; - } - int32_t kx_end = args.KW; - const int32_t x_max = args.IW - 1 - base_x; - if (x_max < 0) { - kx_end = kx_start; - } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { - kx_end = min(kx_end, x_max/args.d0 + 1); - } - - if (ky_start < ky_end && kx_start < kx_end) { - const uint64_t src_base_n = (uint64_t) n * args.nb13; - const uint64_t w_base_oc = (uint64_t) oc * args.nb03; - - for (int32_t ic = 0; ic < args.IC; ++ic) { - const uint64_t src_base_nc = src_base_n + (uint64_t) ic * args.nb12; - const uint64_t w_base_ocic = w_base_oc + (uint64_t) ic * args.nb02; - - for (int32_t ky = ky_start; ky < ky_end; ++ky) { - const int32_t iy = base_y + ky*args.d1; - const uint64_t src_base_row = src_base_nc + (uint64_t) iy * args.nb11; - const uint64_t w_base_row = w_base_ocic + (uint64_t) ky * args.nb01; - - for (int32_t kx = kx_start; kx < kx_end; ++kx) { - const int32_t ix = base_x + kx*args.d0; - const uint64_t src_offs = src_base_row + (uint64_t) ix * args.nb10; - const uint64_t w_offs = w_base_row + (uint64_t) kx * args.nb00; - - const float x = *(device const float *)(src + src_offs); - const float w = (float) (*(device const TK *)(weights + w_offs)); - - acc += x * w; - } - } - } - } - - const uint64_t dst_offs = - (uint64_t) n * args.nb3 + - (uint64_t) oc * args.nb2 + - (uint64_t) oh * args.nb1 + - (uint64_t) ow * args.nb0; - - *(device float *)(dst + dst_offs) = acc; - } -} - -template [[host_name("kernel_conv_2d_f32_f32")]] -kernel void kernel_conv_2d( - constant ggml_metal_kargs_conv_2d & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template [[host_name("kernel_conv_2d_f16_f32")]] -kernel void kernel_conv_2d( - constant ggml_metal_kargs_conv_2d & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -// grid: x = C tile, y = OH, z = OW * N (for channel-contiguous layouts) -template -kernel void kernel_conv_2d_dw_tiled( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int32_t c = (int32_t)(tgpig.x * ntg.x + tpitg.x); - if (c >= args.C) { - return; - } - - const int32_t oh = tgpig.y; - const int32_t own = tgpig.z; - const int32_t ow = own % args.OW; - const int32_t n = own / args.OW; - - const int32_t base_y = oh*args.s1 - args.p1; - - int32_t ky_start = 0; - if (base_y < 0) { - ky_start = (-base_y + args.d1 - 1)/args.d1; - } - int32_t ky_end = args.KH; - const int32_t y_max = args.IH - 1 - base_y; - if (y_max < 0) { - ky_end = ky_start; - } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { - ky_end = min(ky_end, y_max/args.d1 + 1); - } - - const int32_t base_x = ow*args.s0 - args.p0; - - int32_t kx_start = 0; - if (base_x < 0) { - kx_start = (-base_x + args.d0 - 1)/args.d0; - } - int32_t kx_end = args.KW; - const int32_t x_max = args.IW - 1 - base_x; - if (x_max < 0) { - kx_end = kx_start; - } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { - kx_end = min(kx_end, x_max/args.d0 + 1); - } - - float acc = 0.0f; - - if (ky_start < ky_end && kx_start < kx_end) { - const uint64_t w_base = (uint64_t) c * args.nb02; - const uint64_t src_base = (uint64_t) n * args.nb13 + (uint64_t) c * args.nb12; - - for (int32_t ky = ky_start; ky < ky_end; ++ky) { - const int32_t iy = base_y + ky*args.d1; - const uint64_t src_row = src_base + (uint64_t) iy * args.nb11; - const uint64_t w_row = w_base + (uint64_t) ky * args.nb01; - - for (int32_t kx = kx_start; kx < kx_end; ++kx) { - const int32_t ix = base_x + kx*args.d0; - const float x = *(device const float *)(src + src_row + (uint64_t) ix * args.nb10); - const float w = (float)(*(device const TK *)(weights + w_row + (uint64_t) kx * args.nb00)); - acc += x * w; - } - } - } - - const uint64_t dst_offs = - (uint64_t) n * args.nb3 + - (uint64_t) c * args.nb2 + - (uint64_t) oh * args.nb1 + - (uint64_t) ow * args.nb0; - - *(device float *)(dst + dst_offs) = acc; -} - -// grid: x = OW tile, y = OH, z = C * N (for spatially-contiguous layouts) -template -kernel void kernel_conv_2d_dw( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int32_t oh = tgpig.y; - const int32_t cn = tgpig.z; - const int32_t c = cn % args.C; - const int32_t n = cn / args.C; - - const int32_t base_y = oh*args.s1 - args.p1; - - int32_t ky_start = 0; - if (base_y < 0) { - ky_start = (-base_y + args.d1 - 1)/args.d1; - } - int32_t ky_end = args.KH; - const int32_t y_max = args.IH - 1 - base_y; - if (y_max < 0) { - ky_end = ky_start; - } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { - ky_end = min(ky_end, y_max/args.d1 + 1); - } - - const uint64_t w_base = (uint64_t) c * args.nb02; - const uint64_t src_base = (uint64_t) n * args.nb13 + (uint64_t) c * args.nb12; - - const int32_t ow = (int32_t)(tgpig.x * ntg.x + tpitg.x); - if (ow >= args.OW) { - return; - } - - float acc = 0.0f; - - const int32_t base_x = ow*args.s0 - args.p0; - - int32_t kx_start = 0; - if (base_x < 0) { - kx_start = (-base_x + args.d0 - 1)/args.d0; - } - int32_t kx_end = args.KW; - const int32_t x_max = args.IW - 1 - base_x; - if (x_max < 0) { - kx_end = kx_start; - } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { - kx_end = min(kx_end, x_max/args.d0 + 1); - } - - if (ky_start < ky_end && kx_start < kx_end) { - for (int32_t ky = ky_start; ky < ky_end; ++ky) { - const int32_t iy = base_y + ky*args.d1; - const uint64_t src_row = src_base + (uint64_t) iy * args.nb11; - const uint64_t w_row = w_base + (uint64_t) ky * args.nb01; - - for (int32_t kx = kx_start; kx < kx_end; ++kx) { - const int32_t ix = base_x + kx*args.d0; - const float x = *(device const float *)(src + src_row + (uint64_t) ix * args.nb10); - const float w = (float)(*(device const TK *)(weights + w_row + (uint64_t) kx * args.nb00)); - acc += x * w; - } - } - } - - const uint64_t dst_offs = - (uint64_t) n * args.nb3 + - (uint64_t) c * args.nb2 + - (uint64_t) oh * args.nb1 + - (uint64_t) ow * args.nb0; - - *(device float *)(dst + dst_offs) = acc; -} - -template [[host_name("kernel_conv_2d_dw_f32_f32")]] -kernel void kernel_conv_2d_dw( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template [[host_name("kernel_conv_2d_dw_f16_f32")]] -kernel void kernel_conv_2d_dw( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template [[host_name("kernel_conv_2d_dw_tiled_f32_f32")]] -kernel void kernel_conv_2d_dw_tiled( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template [[host_name("kernel_conv_2d_dw_tiled_f16_f32")]] -kernel void kernel_conv_2d_dw_tiled( - constant ggml_metal_kargs_conv_2d_dw & args, - device const char * weights, - device const char * src, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -typedef void (conv_transpose_1d_t)( - constant ggml_metal_kargs_conv_transpose_1d & args, - device const float * src0, - device const float * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]]); - -template -kernel void kernel_conv_transpose_1d( - constant ggml_metal_kargs_conv_transpose_1d & args, - device const T * src0, - device const float * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]]) { - - // For output position j on the time axis, only input positions - // i such that i*s0 <= j < i*s0 + K - // contribute -- i.e. i in [ceil((j - K + 1)/s0), floor(j/s0)] - // intersected with [0, IL-1]. That's at most ceil(K/s0) values - // (typically 2 for stride==K/2 transposed convs). - const int32_t j = tgpig[0]; - const int32_t s0 = args.s0; - const int32_t K = args.K; - const int32_t IL = args.IL; - - int32_t i_min; - { - int32_t a = j - K + 1; - i_min = a <= 0 ? 0 : (a + s0 - 1) / s0; // ceil(a/s0) for a>0 - } - int32_t i_max = j / s0; - if (i_max > IL - 1) i_max = IL - 1; - - float v = 0.0f; - if (i_min <= i_max) { - for (int64_t c = 0; c < args.IC; c++) { - const int32_t kernel_offset = c * tgpg[1] * K + K * tgpig[1]; - const int32_t input_offset = c * IL; - - for (int32_t i = i_min; i <= i_max; i++) { - v += float(src0[kernel_offset + j - i * s0]) * src1[input_offset + i]; - } - } - } - - device float * dst_ptr = (device float *) (dst + tgpig[0] * args.nb0 + tgpig[1] * args.nb1); - - dst_ptr[0] = v; -} - -template [[host_name("kernel_conv_transpose_1d_f32_f32")]] -kernel void kernel_conv_transpose_1d( - constant ggml_metal_kargs_conv_transpose_1d & args, - device const float * src0, - device const float * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]]); - -template [[host_name("kernel_conv_transpose_1d_f16_f32")]] -kernel void kernel_conv_transpose_1d( - constant ggml_metal_kargs_conv_transpose_1d & args, - device const half * src0, - device const float * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]]); - - -template -kernel void kernel_col2im_1d( - constant ggml_metal_kargs_col2im_1d & args, - device const T * col, - device T * dst, - uint tgpig [[threadgroup_position_in_grid]], - uint tpitg [[thread_position_in_threadgroup]], - uint ntg [[threads_per_threadgroup]]) { - - const int idx = tgpig * ntg + tpitg; - if (idx >= args.T_out * args.OC) { - return; - } - - const int t_out = idx % args.T_out; - const int oc = idx / args.T_out; - const int t_abs = t_out + args.p0; // absolute position in uncropped signal - - int t_in_min = (t_abs - args.K + args.s0) / args.s0; // ceil((t_abs - K + 1) / s0) - if (t_in_min < 0) { - t_in_min = 0; - } - int t_in_max = t_abs / args.s0; - if (t_in_max >= args.T_in) { - t_in_max = args.T_in - 1; - } - - float sum = 0.0f; - for (int t_in = t_in_min; t_in <= t_in_max; t_in++) { - const int k = t_abs - t_in * args.s0; - sum += float(col[(oc * args.K + k) + t_in * args.K_OC]); - } - - dst[t_out + oc * args.T_out] = T(sum); -} - -template [[host_name("kernel_col2im_1d_f32")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const float *, device float *, uint, uint, uint); -template [[host_name("kernel_col2im_1d_f16")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const half *, device half *, uint, uint, uint); -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_col2im_1d_bf16")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const bfloat *, device bfloat *, uint, uint, uint); -#endif - - -template -kernel void kernel_snake( - constant ggml_metal_kargs_snake & args, - device const T * x, - device const float * a, - device const float * inv_b, - device T * dst, - uint tgpig [[threadgroup_position_in_grid]], - uint tpitg [[thread_position_in_threadgroup]], - uint ntg [[threads_per_threadgroup]]) { - - const int idx = tgpig * ntg + tpitg; - if (idx >= args.T * args.C) { - return; - } - - const int c = idx / args.T; // x is [T, C], a / inv_b collapse to [1, C] - const float xi = float(x[idx]); - const float si = sin(a[c] * xi); - dst[idx] = T(xi + si * si * inv_b[c]); -} - -template [[host_name("kernel_snake_f32")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const float *, device const float *, device const float *, device float *, uint, uint, uint); -template [[host_name("kernel_snake_f16")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const half *, device const float *, device const float *, device half *, uint, uint, uint); -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_snake_bf16")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const bfloat *, device const float *, device const float *, device bfloat *, uint, uint, uint); -#endif - - -typedef void (conv_transpose_2d_t)( - constant ggml_metal_kargs_conv_transpose_2d & args, - device const float * src0, - device const float * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]]); - -template -kernel void kernel_conv_transpose_2d( - constant ggml_metal_kargs_conv_transpose_2d & args, - device const T * src0, - device const float * src1, - device char * dst, - threadgroup float * shared_sum [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t out_x = tgpig[0]; - const int64_t out_y = tgpig[1]; - const int64_t out_c = tgpig[2]; - - const int64_t kw = tpitg[0]; - const int64_t kh = tpitg[1]; - - float v = 0.0f; - - for (int64_t in_c = 0; in_c < args.IC; in_c++) { - int64_t in_y = out_y - kh; - - if (in_y < 0 || in_y % args.s0) continue; - - in_y /= args.s0; - - if (in_y >= args.IH) continue; - - int64_t in_x = out_x - kw; - - if (in_x < 0 || in_x % args.s0) continue; - - in_x /= args.s0; - - if (in_x >= args.IW) continue; - - const int64_t input_idx = (args.IW * args.IH) * in_c + (args.IW) * in_y + in_x; - const int64_t kernel_idx = (args.KH * args.KW * args.OC) * in_c + (args.KH * args.KW) * out_c + (args.KW) * kh + kw; - - v += (float)src0[kernel_idx] * src1[input_idx]; - } - - const uint tid = tpitg.y * ntg.x + tpitg.x; - shared_sum[tid] = v; - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (tid == 0) { - float total = 0.0f; - const uint num_threads = ntg.x * ntg.y; - for (uint i = 0; i < num_threads; i++) { - total += shared_sum[i]; - } - - device float * dst_ptr = (device float *) (dst + out_x*args.nb0 + out_y * args.nb1 + out_c*args.nb2); - dst_ptr[0] = total; - } -} - -template [[host_name("kernel_conv_transpose_2d_f32_f32")]] -kernel void kernel_conv_transpose_2d( - constant ggml_metal_kargs_conv_transpose_2d & args, - device const float * src0, - device const float * src1, - device char * dst, - threadgroup float * shared_sum [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -template [[host_name("kernel_conv_transpose_2d_f16_f32")]] -kernel void kernel_conv_transpose_2d( - constant ggml_metal_kargs_conv_transpose_2d & args, - device const half * src0, - device const float * src1, - device char * dst, - threadgroup float * shared_sum [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]); - -constant bool FC_upscale_aa [[function_constant(FC_UPSCALE + 0)]]; - -kernel void kernel_upscale_nearest_f32( - constant ggml_metal_kargs_upscale & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t i3 = tgpig.z; - const int64_t i2 = tgpig.y; - const int64_t i1 = tgpig.x; - - const int64_t i03 = i3/args.sf3; - const int64_t i02 = i2/args.sf2; - const int64_t i01 = i1/args.sf1; - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const int64_t i00 = i0/args.sf0; - - device const float * src0_ptr = (device const float *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + i00*args.nb00); - device float * dst_ptr = (device float *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - dst_ptr[0] = src0_ptr[0]; - } -} - -static inline float bilinear_tri(float x) { - return MAX(0.0f, 1.0f - fabs(x)); -} - -kernel void kernel_upscale_bilinear_f32( - constant ggml_metal_kargs_upscale & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t i3 = tgpig.z; - const int64_t i2 = tgpig.y; - const int64_t i1 = tgpig.x; - - const int64_t i03 = i3 / args.sf3; - const int64_t i02 = i2 / args.sf2; - - const float f01 = ((float)i1 + args.poffs) / args.sf1 - args.poffs; - const int64_t i01 = MAX(0, MIN(args.ne01 - 1, (int64_t)floor(f01))); - const int64_t i01p = MAX(0, MIN(args.ne01 - 1, i01 + 1)); - const float fd1 = MAX(0.0f, MIN(1.0f, f01 - (float)i01)); - - src0 += i03*args.nb03 + i02*args.nb02; - - device float * dst_ptr = (device float *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); - - if (FC_upscale_aa) { - const float support0 = MAX(1.0f, 1.0f / args.sf0); - const float invscale0 = 1.0f / support0; - const float support1 = MAX(1.0f, 1.0f / args.sf1); - const float invscale1 = 1.0f / support1; - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; - - int64_t x_min = MAX((int64_t)0, (int64_t)floor(f00 - support0 + args.poffs)); - int64_t x_max = MIN(args.ne00, (int64_t)ceil (f00 + support0 + args.poffs)); - - int64_t y_min = MAX((int64_t)0, (int64_t)floor(f01 - support1 + args.poffs)); - int64_t y_max = MIN(args.ne01, (int64_t)ceil (f01 + support1 + args.poffs)); - - float sum = 0.0f; - float wsum = 0.0f; - - for (int64_t sy = y_min; sy < y_max; ++sy) { - const float wy = MAX(0.0f, 1.0f - fabs((float)sy - f01) * invscale1); - for (int64_t sx = x_min; sx < x_max; ++sx) { - const float wx = MAX(0.0f, 1.0f - fabs((float)sx - f00) * invscale0); - const float w = wx * wy; - device const float * src_ptr = (device const float *)(src0 + sy*args.nb01 + sx*args.nb00); - sum += (*src_ptr) * w; - wsum += w; - } - } - - const float v = (wsum > 0.0f) ? (sum / wsum) : 0.0f; - dst_ptr[i0] = v; - } - } else { - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; - const int64_t i00 = MAX(0, MIN(args.ne00 - 1, (int64_t)floor(f00))); - const int64_t i00p = MAX(0, MIN(args.ne00 - 1, i00 + 1)); - const float fd0 = MAX(0.0f, MIN(1.0f, f00 - (float)i00)); - - device const float * src00 = (device const float *)(src0 + i01*args.nb01 + i00*args.nb00); - device const float * src10 = (device const float *)(src0 + i01*args.nb01 + i00p*args.nb00); - device const float * src01 = (device const float *)(src0 + i01p*args.nb01 + i00*args.nb00); - device const float * src11 = (device const float *)(src0 + i01p*args.nb01 + i00p*args.nb00); - - const float v = - (*src00) * (1.0f - fd0) * (1.0f - fd1) + - (*src10) * fd0 * (1.0f - fd1) + - (*src01) * (1.0f - fd0) * fd1 + - (*src11) * fd0 * fd1; - - dst_ptr[i0] = v; - } - } -} - -template -kernel void kernel_conv_3d( - constant ggml_metal_kargs_conv_3d & args, - device const char * src0, // Weights [IC * OC, KD, KH, KW] - device const char * src1, // Inputs [IC * N, ID, IH, IW] - device char * dst, // Outputs [OC * N, OD, OH, OW] - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]]) { - - // 1. Un-flatten the spatial dimension from Grid X - int64_t spatial_idx = tgpig.x * 32 + tpitg.x; - - if (spatial_idx >= args.OW * args.OH * args.OD) { - return; // Thread falls outside the spatial volume - } - - int64_t od = spatial_idx / (args.OW * args.OH); - int64_t oh = (spatial_idx / args.OW) % args.OH; - int64_t ow = spatial_idx % args.OW; - - // 2. Map Y to Channels, Z to Batch - int64_t oc = tgpig.y; - int64_t batch_idx = tgpig.z; - - // 3. Calculate anchor coordinates in the Input volume - int64_t i_w_base = ow * args.s0 - args.p0; - int64_t i_h_base = oh * args.s1 - args.p1; - int64_t i_d_base = od * args.s2 - args.p2; - - float sum = 0.0f; - - // 4. Gather Loop (Iterate over Input Channels -> Depth -> Height -> Width) - for (int64_t ic = 0; ic < args.IC; ++ic) { - - // ggml packs batch and channel together in the 4th dimension - int64_t src_cn_idx = batch_idx * args.IC + ic; - int64_t w_cn_idx = oc * args.IC + ic; - - for (int64_t kz = 0; kz < args.KD; ++kz) { - int64_t id = i_d_base + kz * args.d2; - if (id < 0 || id >= args.ID) continue; // Boundary check (Padding) - - for (int64_t ky = 0; ky < args.KH; ++ky) { - int64_t ih = i_h_base + ky * args.d1; - if (ih < 0 || ih >= args.IH) continue; - - for (int64_t kx = 0; kx < args.KW; ++kx) { - int64_t iw = i_w_base + kx * args.d0; - if (iw < 0 || iw >= args.IW) continue; - - // Convert multi-dimensional coordinates to flat byte offsets - int64_t w_idx = kx*args.nb00 + ky*args.nb01 + kz*args.nb02 + w_cn_idx*args.nb03; - int64_t i_idx = iw*args.nb10 + ih*args.nb11 + id*args.nb12 + src_cn_idx*args.nb13; - - // Dereference memory and cast weights to f32 if they were f16 - float w_val = (float)*(device const T*)((device const char*)src0 + w_idx); - float i_val = *(device const float*)((device const char*)src1 + i_idx); - - sum += w_val * i_val; - } - } - } - } - - // 5. Write the accumulated value out to RAM - int64_t dst_cn_idx = batch_idx * args.OC + oc; - int64_t d_idx = ow*args.nb0 + oh*args.nb1 + od*args.nb2 + dst_cn_idx*args.nb3; - - *(device float*)(dst + d_idx) = sum; -} - -// Explicit instantiations so the JIT compiler can find them by name -template [[host_name("kernel_conv_3d_f32_f32")]] -kernel void kernel_conv_3d( - constant ggml_metal_kargs_conv_3d & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]]); - -// Explicit instantiation for f16 weights -template [[host_name("kernel_conv_3d_f16_f32")]] -kernel void kernel_conv_3d( - constant ggml_metal_kargs_conv_3d & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]]); - - -static inline float bicubic_weight1(float x) { - const float a = -0.75f; - return ((a + 2) * x - (a + 3)) * x * x + 1; -} - -static inline float bicubic_weight2(float x) { - const float a = -0.75f; - return ((a * x - 5 * a) * x + 8 * a) * x - 4 * a; -} - -kernel void kernel_upscale_bicubic_f32( - constant ggml_metal_kargs_upscale & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t i3 = tgpig.z; - const int64_t i2 = tgpig.y; - const int64_t i1 = tgpig.x; - - const int64_t i03 = i3 / args.sf3; - const int64_t i02 = i2 / args.sf2; - - const float f01 = ((float)i1 + args.poffs) / args.sf1 - args.poffs; - const int64_t i01 = (int64_t)floor(f01); - const float fd1 = f01 - (float)i01; - - const float w_y0 = bicubic_weight2(fd1 + 1.0f); - const float w_y1 = bicubic_weight1(fd1); - const float w_y2 = bicubic_weight1(1.0f - fd1); - const float w_y3 = bicubic_weight2(2.0f - fd1); - - const device char * src_slice = src0 + i03 * args.nb03 + i02 * args.nb02; - - device float * dst_ptr = (device float *)(dst + i3 * args.nb3 + i2 * args.nb2 + i1 * args.nb1); - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; - const int64_t i00 = (int64_t)floor(f00); - const float fd0 = f00 - (float)i00; - - const float w_x0 = bicubic_weight2(fd0 + 1.0f); - const float w_x1 = bicubic_weight1(fd0); - const float w_x2 = bicubic_weight1(1.0f - fd0); - const float w_x3 = bicubic_weight2(2.0f - fd0); - - float sum = 0.0f; - - for (int dy = -1; dy <= 2; ++dy) { - const int64_t iy = MAX(0, MIN(args.ne01 - 1, i01 + dy)); - const float wy = (dy == -1) ? w_y0 : (dy == 0) ? w_y1 : (dy == 1) ? w_y2 : w_y3; - - for (int dx = -1; dx <= 2; ++dx) { - const int64_t ix = MAX(0, MIN(args.ne00 - 1, i00 + dx)); - const float wx = (dx == -1) ? w_x0 : (dx == 0) ? w_x1 : (dx == 1) ? w_x2 : w_x3; - - device const float * src_ptr = (device const float *)(src_slice + iy * args.nb01 + ix * args.nb00); - sum += (*src_ptr) * wx * wy; - } - } - - dst_ptr[i0] = sum; - } -} - -kernel void kernel_roll_f32( - constant ggml_metal_kargs_roll & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t i3 = tgpig.z; - const int64_t i2 = tgpig.y; - const int64_t i1 = tgpig.x; - - device const float * src0_ptr = (device const float *) src0; - device float * dst_ptr = (device float *) dst; - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - // apply shifts and wrap around - int64_t i00 = i0 - args.s0; - int64_t i01 = i1 - args.s1; - int64_t i02 = i2 - args.s2; - int64_t i03 = i3 - args.s3; - - if (i00 < 0) { i00 += args.ne00; } else if (i00 >= args.ne00) { i00 -= args.ne00; } - if (i01 < 0) { i01 += args.ne01; } else if (i01 >= args.ne01) { i01 -= args.ne01; } - if (i02 < 0) { i02 += args.ne02; } else if (i02 >= args.ne02) { i02 -= args.ne02; } - if (i03 < 0) { i03 += args.ne03; } else if (i03 >= args.ne03) { i03 -= args.ne03; } - - int64_t src_idx = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00 + i00; - int64_t dst_idx = i3 *args.ne2 *args.ne1 *args.ne0 + i2 *args.ne1 *args.ne0 + i1 *args.ne0 + i0; - - dst_ptr[dst_idx] = src0_ptr[src_idx]; - } -} - -template -kernel void kernel_pad_impl( - constant ggml_metal_kargs_pad & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - const int32_t i3 = tgpig.z; - const int32_t i2 = tgpig.y; - const int32_t k0 = tgpig.x/args.ne1; - const int32_t i1 = tgpig.x - k0*args.ne1; - - const int32_t i03 = i3; - const int32_t i02 = i2; - const int32_t i01 = i1; - - device const T * src0_ptr = (device const T *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); - device T * dst_ptr = (device T *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); - - for (int32_t l0 = 0; l0 < 1024; l0 += ntg.x) { - const int32_t i0 = k0*1024 + tpitg.x + l0; - if (i0 >= args.ne0) { - break; - } - - if (i0 < args.ne00 && i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { - dst_ptr[i0] = src0_ptr[i0]; - } else { - dst_ptr[i0] = 0.0f; - } - } -} - -typedef decltype(kernel_pad_impl) kernel_pad_t; - -template [[host_name("kernel_pad_f32")]] kernel kernel_pad_t kernel_pad_impl; -template [[host_name("kernel_pad_f32_4")]] kernel kernel_pad_t kernel_pad_impl; - -// TODO: this is slow - optimize -kernel void kernel_pad_reflect_1d_f32( - constant ggml_metal_kargs_pad_reflect_1d & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tgpg[[threadgroups_per_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - const int64_t i3 = tgpig.z; - const int64_t i2 = tgpig.y; - const int64_t i1 = tgpig.x; - - const int64_t i03 = i3; - const int64_t i02 = i2; - const int64_t i01 = i1; - - device const float * src0_ptr = (device const float *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); - device float * dst_ptr = (device float *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); - - if (i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - if (i0 < args.p0) { - dst_ptr[i0] = src0_ptr[args.p0 - i0]; - } else if (i0 < args.ne0 - args.p1) { - dst_ptr[i0] = src0_ptr[i0 - args.p0]; - } else { - dst_ptr[i0] = src0_ptr[(args.ne0 - args.p1 - args.p0) - (args.p1 + 1 - (args.ne0 - i0)) - 1]; - } - } - } -} - -kernel void kernel_arange_f32( - constant ggml_metal_kargs_arange & args, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - device float * dst_ptr = (device float *) dst; - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - dst_ptr[i0] = args.start + args.step * i0; - } -} - -kernel void kernel_timestep_embedding_f32( - constant ggml_metal_kargs_timestep_embedding & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint3 tpitg[[thread_position_in_threadgroup]], - uint3 ntg[[threads_per_threadgroup]]) { - - int i = tgpig.x; - device float * embed_data = (device float *)(dst + i*args.nb1); - - int half_ = args.dim / 2; - for (int j = tpitg.x; j < half_; j += ntg.x) { - float timestep = ((device float *)src0)[i]; - float freq = (float)exp(-log((float)args.max_period) * j / half_); - float arg = timestep * freq; - embed_data[j ] = cos(arg); - embed_data[j + half_] = sin(arg); - } - - if (args.dim % 2 != 0 && tpitg.x == 0) { - embed_data[2 * half_] = 0.f; - } -} - -// bitonic sort implementation following the CUDA kernels as reference -typedef void (argsort_t)( - constant ggml_metal_kargs_argsort & args, - device const char * src0, - device int32_t * dst, - threadgroup int32_t * shmem_i32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]); - -template -kernel void kernel_argsort_f32_i32( - constant ggml_metal_kargs_argsort & args, - device const char * src0, - device int32_t * dst, - threadgroup int32_t * shmem_i32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - // bitonic sort - const int col = tpitg[0]; - const int ib = tgpig[0] / args.ne01; - - const int i00 = ib*ntg.x; - const int i01 = tgpig[0] % args.ne01; - const int i02 = tgpig[1]; - const int i03 = tgpig[2]; - - device const float * src0_row = (device const float *) (src0 + args.nb01*i01 + args.nb02*i02 + args.nb03*i03); - - // initialize indices - shmem_i32[col] = i00 + col; - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (int k = 2; k <= ntg.x; k *= 2) { - for (int j = k / 2; j > 0; j /= 2) { - int ixj = col ^ j; - if (ixj > col) { - if ((col & k) == 0) { - if (shmem_i32[col] >= args.ne00 || - (shmem_i32[ixj] < args.ne00 && (order == GGML_SORT_ORDER_ASC ? - src0_row[shmem_i32[col]] > src0_row[shmem_i32[ixj]] : - src0_row[shmem_i32[col]] < src0_row[shmem_i32[ixj]])) - ) { - SWAP(shmem_i32[col], shmem_i32[ixj]); - } - } else { - if (shmem_i32[ixj] >= args.ne00 || - (shmem_i32[col] < args.ne00 && (order == GGML_SORT_ORDER_ASC ? - src0_row[shmem_i32[col]] < src0_row[shmem_i32[ixj]] : - src0_row[shmem_i32[col]] > src0_row[shmem_i32[ixj]])) - ) { - SWAP(shmem_i32[col], shmem_i32[ixj]); - } - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - } - - const int64_t i0 = ib*args.top_k; - - // copy the result to dst without the padding - if (i0 + col < args.ne0 && col < args.top_k) { - dst += i0 + args.ne0*i01 + args.ne0*args.ne1*i02 + args.ne0*args.ne1*args.ne2*i03; - - dst[col] = shmem_i32[col]; - } -} - -template [[host_name("kernel_argsort_f32_i32_asc")]] kernel argsort_t kernel_argsort_f32_i32; -template [[host_name("kernel_argsort_f32_i32_desc")]] kernel argsort_t kernel_argsort_f32_i32; - -typedef void (argsort_merge_t)( - constant ggml_metal_kargs_argsort_merge & args, - device const char * src0, - device const int32_t * tmp, - device int32_t * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]); - -template -kernel void kernel_argsort_merge_f32_i32( - constant ggml_metal_kargs_argsort_merge & args, - device const char * src0, - device const int32_t * tmp, - device int32_t * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - - const int im = tgpig[0] / args.ne01; - const int i01 = tgpig[0] % args.ne01; - const int i02 = tgpig[1]; - const int i03 = tgpig[2]; - - const int start = im * (2 * args.len); - - const int len0 = MIN(args.len, MAX(0, args.ne0 - (int)(start))); - const int len1 = MIN(args.len, MAX(0, args.ne0 - (int)(start + args.len))); - - const int total = len0 + len1; - - device const int32_t * tmp0 = tmp + start - + i01*args.ne0 - + i02*args.ne0*args.ne01 - + i03*args.ne0*args.ne01*args.ne02; - - device const int32_t * tmp1 = tmp0 + args.len; - - dst += start - + i01*args.top_k - + i02*args.top_k*args.ne01 - + i03*args.top_k*args.ne01*args.ne02; - - device const float * src0_row = (device const float *)(src0 - + args.nb01*i01 - + args.nb02*i02 - + args.nb03*i03); - - if (total == 0) { - return; - } - - const int chunk = (total + ntg.x - 1) / ntg.x; - - const int k0 = tpitg.x * chunk; - const int k1 = MIN(MIN(k0 + chunk, total), args.top_k); - - if (k0 >= args.top_k) { - return; - } - - if (k0 >= total) { - return; - } - - int low = k0 > len1 ? k0 - len1 : 0; - int high = MIN(k0, len0); - - // binary-search partition (i, j) such that i + j = k - while (low < high) { - const int mid = (low + high) >> 1; - - const int32_t idx0 = tmp0[mid]; - const int32_t idx1 = tmp1[k0 - mid - 1]; - - const float val0 = src0_row[idx0]; - const float val1 = src0_row[idx1]; - - bool take_left; - if (order == GGML_SORT_ORDER_ASC) { - take_left = (val0 <= val1); - } else { - take_left = (val0 >= val1); - } - - if (take_left) { - low = mid + 1; - } else { - high = mid; - } - } - - int i = low; - int j = k0 - i; - - // keep the merge fronts into registers - int32_t idx0 = 0; - float val0 = 0.0f; - if (i < len0) { - idx0 = tmp0[i]; - val0 = src0_row[idx0]; - } - - int32_t idx1 = 0; - float val1 = 0.0f; - if (j < len1) { - idx1 = tmp1[j]; - val1 = src0_row[idx1]; - } - - for (int k = k0; k < k1; ++k) { - int32_t out_idx; - - if (i >= len0) { - while (k < k1) { - dst[k++] = tmp1[j++]; - } - break; - } else if (j >= len1) { - while (k < k1) { - dst[k++] = tmp0[i++]; - } - break; - } else { - bool take_left; - - if (order == GGML_SORT_ORDER_ASC) { - take_left = (val0 <= val1); - } else { - take_left = (val0 >= val1); - } - - if (take_left) { - out_idx = idx0; - ++i; - if (i < len0) { - idx0 = tmp0[i]; - val0 = src0_row[idx0]; - } - } else { - out_idx = idx1; - ++j; - if (j < len1) { - idx1 = tmp1[j]; - val1 = src0_row[idx1]; - } - } - } - - dst[k] = out_idx; - } -} - -template [[host_name("kernel_argsort_merge_f32_i32_asc")]] kernel argsort_merge_t kernel_argsort_merge_f32_i32; -template [[host_name("kernel_argsort_merge_f32_i32_desc")]] kernel argsort_merge_t kernel_argsort_merge_f32_i32; - -template -kernel void kernel_fwht_f32( - constant ggml_metal_kargs_fwht & args, - device const float * src, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - - constexpr int NW = N_SIMDWIDTH; - constexpr int NE = N / NW; - - const float scale = 1.0f / sqrt((float) N); - - const int sg_per_tg = ntg.x / NW; - const int64_t r = tgpig.x * sg_per_tg + sgitg; - if (r >= args.nrows) { - return; - } - - src += r * N; - dst += r * N; - - const int lane = tiisg; - - float reg[NE]; - for (int i = 0; i < NE; i++) { - reg[i] = src[i*NW + lane]*scale; - } - for (int i = 1; i < NW; i *= 2) { - for (int j = 0; j < NE; j++) { - const float val = reg[j]; - const float val2 = simd_shuffle_xor(val, i); - reg[j] = (lane & i) == 0 ? val2 + val : val2 - val; - } - } - - for (int i = NW; i < N; i *= 2) { - const int step = i / NW; - for (int j = 0; j < NE; j += (2 * step)) { - for (int k = 0; k < step; k++) { - const float x = reg[j + k ]; - const float y = reg[j + k + step]; - reg[j + k] = x + y; - reg[j + k + step] = x - y; - } - } - } - - for (int i = 0; i < NE; i++) { - dst[i*NW + lane] = reg[i]; - } -} - -typedef decltype(kernel_fwht_f32<64>) kernel_fwht_t; - -template [[host_name("kernel_fwht_f32_64")]] kernel kernel_fwht_t kernel_fwht_f32<64>; -template [[host_name("kernel_fwht_f32_128")]] kernel kernel_fwht_t kernel_fwht_f32<128>; -template [[host_name("kernel_fwht_f32_256")]] kernel kernel_fwht_t kernel_fwht_f32<256>; -template [[host_name("kernel_fwht_f32_512")]] kernel kernel_fwht_t kernel_fwht_f32<512>; - -constant bool FC_flash_attn_ext_pad_has_mask [[function_constant(FC_FLASH_ATTN_EXT_PAD + 0)]]; - -constant int32_t FC_flash_attn_ext_pad_ncpsg [[function_constant(FC_FLASH_ATTN_EXT_PAD + 25)]]; - -// pad the last chunk of C elements of k and v into a an extra pad buffer -kernel void kernel_flash_attn_ext_pad( - constant ggml_metal_kargs_flash_attn_ext_pad & args, - device const char * k, - device const char * v, - device const char * mask, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int32_t C = FC_flash_attn_ext_pad_ncpsg; - - device char * k_pad = dst; - device char * v_pad = k_pad + args.nb11*C*args.ne_12_2*args.ne_12_3; - device char * mask_pad = v_pad + args.nb21*C*args.ne_12_2*args.ne_12_3; - - const int32_t icp = args.ne11 % C; - const int32_t ic0 = args.ne11 - icp; - - const int32_t i1 = tgpig[0]; - const int32_t i2 = tgpig[1]; - const int32_t i3 = tgpig[2]; - - if (i2 < args.ne_12_2 && i3 < args.ne_12_3) { - device const char * k_src = k + args.nb11*(ic0 + i1) + args.nb12*i2 + args.nb13*i3; - device const char * v_src = v + args.nb21*(ic0 + i1) + args.nb22*i2 + args.nb23*i3; - - device char * k_dst = k_pad + args.nb11*i1 + args.nb11*C*i2 + args.nb11*C*args.ne_12_2*i3; - device char * v_dst = v_pad + args.nb21*i1 + args.nb21*C*i2 + args.nb21*C*args.ne_12_2*i3; - - if (i1 >= icp) { - // here it is not important the exact value that will be used as we rely on masking out the scores in the attention - for (uint64_t i = tiitg; i < args.nb11; i += ntg.x) { - k_dst[i] = 0; - } - for (uint64_t i = tiitg; i < args.nb21; i += ntg.x) { - v_dst[i] = 0; - } - } else { - for (uint64_t i = tiitg; i < args.nb11; i += ntg.x) { - k_dst[i] = k_src[i]; - } - for (uint64_t i = tiitg; i < args.nb21; i += ntg.x) { - v_dst[i] = v_src[i]; - } - } - } - - if (FC_flash_attn_ext_pad_has_mask) { - if (i2 < args.ne32 && i3 < args.ne33) { - for (int ib = i1; ib < args.ne31; ib += C) { - device const half * mask_src = (device const half *)(mask + args.nb31*ib + args.nb32*i2 + args.nb33*i3) + ic0; - device half * mask_dst = (device half *)(mask_pad) + C*ib + C*args.ne31*i2 + C*args.ne31*args.ne32*i3; - - for (int i = tiitg; i < C; i += ntg.x) { - if (i >= icp) { - mask_dst[i] = -MAXHALF; - } else { - mask_dst[i] = mask_src[i]; - } - } - } - } - } -} - -constant int32_t FC_flash_attn_ext_blk_nqptg [[function_constant(FC_FLASH_ATTN_EXT_BLK + 24)]]; -constant int32_t FC_flash_attn_ext_blk_ncpsg [[function_constant(FC_FLASH_ATTN_EXT_BLK + 25)]]; - -// scan the blocks of the mask that are not masked -// 0 - masked (i.e. full of -INF, skip) -// 1 - not masked (i.e. at least one element of the mask is not -INF) -// 2 - all zero -kernel void kernel_flash_attn_ext_blk( - constant ggml_metal_kargs_flash_attn_ext_blk & args, - device const char * mask, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]]) { - // block size C x Q - const int32_t Q = FC_flash_attn_ext_blk_nqptg; - const int32_t C = FC_flash_attn_ext_blk_ncpsg; - - constexpr short NW = N_SIMDWIDTH; - - const int32_t i3 = tgpig[2]/args.ne32; - const int32_t i2 = tgpig[2]%args.ne32; - const int32_t i1 = tgpig[1]; - const int32_t i0 = tgpig[0]; - - char res = i0*C + C > args.ne30 ? 1 : 0; - - device const half * mask_src = (device const half *) (mask + (i1*Q)*args.nb31 + i2*args.nb32 + i3*args.nb33) + i0*C + tiisg; - - // detailed check of the elements of the block - if ((C > NW || Q > 1) && res == 0) { - half mmin = MAXHALF; - half mmax = -MAXHALF; - - FOR_UNROLL (short j = 0; j < Q; ++j) { - FOR_UNROLL (short ii = 0; ii < C/NW; ++ii) { - mmin = min(mmin, mask_src[ii*NW]); - mmax = max(mmax, mask_src[ii*NW]); - } - - mask_src += args.nb31/2; - } - - mmin = simd_min(mmin); - mmax = simd_max(mmax); - - if (mmax > -MAXHALF) { - if (mmin == 0.0 && mmax == 0.0) { - res = 2; - } else { - res = 1; - } - } - } - - const int32_t nblk1 = ((args.ne01 + Q - 1)/Q); - const int32_t nblk0 = ((args.ne30 + C - 1)/C); - - if (tiisg == 0) { - dst[((i3*args.ne32 + i2)*nblk1 + i1)*nblk0 + i0] = res; - } -} - -constant bool FC_flash_attn_ext_has_mask [[function_constant(FC_FLASH_ATTN_EXT + 0)]]; -constant bool FC_flash_attn_ext_has_sinks [[function_constant(FC_FLASH_ATTN_EXT + 1)]]; -constant bool FC_flash_attn_ext_has_bias [[function_constant(FC_FLASH_ATTN_EXT + 2)]]; -constant bool FC_flash_attn_ext_has_scap [[function_constant(FC_FLASH_ATTN_EXT + 3)]]; -constant bool FC_flash_attn_ext_has_kvpad [[function_constant(FC_FLASH_ATTN_EXT + 4)]]; - -constant bool FC_flash_attn_ext_bc_mask [[function_constant(FC_FLASH_ATTN_EXT + 10)]]; - -//constant float FC_flash_attn_ext_scale [[function_constant(FC_FLASH_ATTN_EXT + 10)]]; -//constant float FC_flash_attn_ext_max_bias [[function_constant(FC_FLASH_ATTN_EXT + 11)]]; -//constant float FC_flash_attn_ext_logit_softcap [[function_constant(FC_FLASH_ATTN_EXT + 12)]]; - -constant int32_t FC_flash_attn_ext_ns10 [[function_constant(FC_FLASH_ATTN_EXT + 20)]]; -constant int32_t FC_flash_attn_ext_ns20 [[function_constant(FC_FLASH_ATTN_EXT + 21)]]; -constant int32_t FC_flash_attn_ext_nsg [[function_constant(FC_FLASH_ATTN_EXT + 22)]]; - -// ref: https://arxiv.org/pdf/2307.08691.pdf -template< - typename q_t, // query types in shared memory - typename q4_t, - typename q8x8_t, - typename k_t, // key types in shared memory - typename k4x4_t, - typename k8x8_t, - typename v_t, // value types in shared memory - typename v4x4_t, - typename v8x8_t, - typename qk_t, // Q*K types - typename qk8x8_t, - typename s_t, // soft-max types - typename s2_t, - typename s8x8_t, - typename o_t, // attention accumulation types - typename o4_t, - typename o8x8_t, - typename kd4x4_t, // key type in device memory - short nl_k, - void (*deq_k)(device const kd4x4_t *, short, thread k4x4_t &), - typename vd4x4_t, // value type in device memory - short nl_v, - void (*deq_v)(device const vd4x4_t *, short, thread v4x4_t &), - short DK, // K head size - short DV, // V head size - short Q, // queries per threadgroup - short C, // cache items per threadgroup - short NSG> // number of simd groups -void kernel_flash_attn_ext_impl( - constant ggml_metal_kargs_flash_attn_ext & args, - device const char * q, - device const char * k, - device const char * v, - device const char * mask, - device const char * sinks, - device const char * pad, - device const char * blk, - device char * dst, - threadgroup half * shmem_f16, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const ushort iq3 = tgpig[2]; - const ushort iq2 = tgpig[1]; - const ushort iq1 = tgpig[0]*Q; - -#define NS10 (FC_flash_attn_ext_ns10) -#define NS20 (FC_flash_attn_ext_ns20) - - // note: I had some concerns that using this instead of the ugly macros above was affecting performance - // need to re-check carefully and if no regressions are observerd - remove the macros - // the concerns is that maybe using const variables requires extra registers? but not sure if the compiler - // is clever enough to avoid this. unfortunately, using constexpr is not possible with FC - //const short NS10 = FC_flash_attn_ext_ns10; - //const short NS20 = FC_flash_attn_ext_ns20; - - constexpr short KV = 8; - - constexpr short DK4 = DK/4; - constexpr short DK8 = DK/8; - constexpr short DK16 = DK/16; - constexpr short DV4 = DV/4; - //constexpr short DV8 = DV/8; - constexpr short DV16 = DV/16; - - constexpr short PV = PAD2(DV, 64); - constexpr short PV4 = PV/4; - constexpr short PV8 = PV/8; - //constexpr short PV16 = PV/16; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NQ = Q/NSG; - constexpr short SH = 2*C; // shared memory per simdgroup (s_t == float) - - constexpr short TS = 2*SH; - constexpr short T = DK + 2*PV; // shared memory size per query in (half) - - threadgroup q_t * sq = (threadgroup q_t *) (shmem_f16 + 0*T); // holds the query data - threadgroup q4_t * sq4 = (threadgroup q4_t *) (shmem_f16 + 0*T); // same as above but in q4_t - threadgroup o_t * so = (threadgroup o_t *) (shmem_f16 + 0*T + Q*DK); // the result for all queries in 8x8 matrices (the O matrix from the paper) - threadgroup o4_t * so4 = (threadgroup o4_t *) (shmem_f16 + 0*T + Q*DK); - threadgroup s_t * ss = (threadgroup s_t *) (shmem_f16 + Q*T); // scratch buffer for attention, mask and diagonal matrix - threadgroup s2_t * ss2 = (threadgroup s2_t *) (shmem_f16 + Q*T); // same as above but in s2_t - - threadgroup k_t * sk = (threadgroup k_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // scratch buffer to load K in shared memory - threadgroup k4x4_t * sk4x4 = (threadgroup k4x4_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // same as above but in k4x4_t - - threadgroup v_t * sv = (threadgroup v_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // scratch buffer to load V in shared memory - threadgroup v4x4_t * sv4x4 = (threadgroup v4x4_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // same as above but in v4x4_t - - // mask storage in shared mem - threadgroup half2 * sm2 = (threadgroup half2 *) (shmem_f16 + Q*T + 2*C); - - // per-query mask pointers - device const half2 * pm2[NQ]; - - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - pm2[jj] = (device const half2 *) ((device const char *) mask + (iq1 + j)*args.nb31 + (iq2%args.ne32)*args.nb32 + (iq3%args.ne33)*args.nb33); - } - - { - const int32_t nblk1 = ((args.ne01 + Q - 1)/Q); - const int32_t nblk0 = ((args.ne11 + C - 1)/C); - - blk += (((iq3%args.ne33)*args.ne32 + (iq2%args.ne32))*nblk1 + iq1/Q)*nblk0; - } - - { - q += iq1*args.nb01 + iq2*args.nb02 + iq3*args.nb03; - - const short ikv2 = iq2/(args.ne02/args.ne_12_2); - const short ikv3 = iq3/(args.ne03/args.ne_12_3); - - k += ikv2*args.nb12 + ikv3*args.nb13; - v += ikv2*args.nb22 + ikv3*args.nb23; - } - - // load heads from Q to shared memory - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - device const float4 * q4 = (device const float4 *) ((device const char *) q + j*args.nb01); - - for (short i = tiisg; i < DK4; i += NW) { - if (iq1 + j < args.ne01) { - sq4[j*DK4 + i] = (q4_t) q4[i]; - } else { - sq4[j*DK4 + i] = 0; - } - } - } - - // zero out - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - for (short i = tiisg; i < DV4; i += NW) { - so4[j*PV4 + i] = 0; - } - - for (short i = tiisg; i < SH; i += NW) { - ss[j*SH + i] = 0.0f; - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - float S[NQ] = { [0 ... NQ-1] = 0.0f }; - - { - float M[NQ] = { [0 ... NQ-1] = -FLT_MAX/2 }; - - float slope = 1.0f; - - // ALiBi - if (FC_flash_attn_ext_has_bias) { - const short h = iq2; - - const float base = h < args.n_head_log2 ? args.m0 : args.m1; - const short exph = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; - - slope = pow(base, exph); - } - - // loop over the KV cache - // each simdgroup handles blocks of Q rows and C columns - for (int ic0 = 0; ; ++ic0) { - int ic = ic0*C; - if (ic >= args.ne11) { - break; - } - - // the last partial chunk uses the pad buffer as source - if (FC_flash_attn_ext_has_kvpad && ic + C > args.ne11) { - k = pad; - v = k + args.nb11*C*args.ne_12_2*args.ne_12_3; - mask = v + args.nb21*C*args.ne_12_2*args.ne_12_3; - - const short ikv2 = iq2/(args.ne02/args.ne_12_2); - const short ikv3 = iq3/(args.ne03/args.ne_12_3); - - k += (ikv2 + ikv3*args.ne_12_2)*args.nb11*C; - v += (ikv2 + ikv3*args.ne_12_2)*args.nb21*C; - - if (!FC_flash_attn_ext_has_mask) { - threadgroup half * sm = (threadgroup half *) (sm2); - - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - for (short i = tiisg; i < C; i += NW) { - if (ic + i >= args.ne11) { - sm[2*j*SH + i] = -MAXHALF; - } - } - } - } else { - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - pm2[jj] = (device const half2 *) ((device const half *) mask + - (iq1 + j)*C + - (iq2%args.ne32)*(C*args.ne31) + - (iq3%args.ne33)*(C*args.ne31*args.ne32)); - } - } - - ic = 0; - } - - char blk_cur = 1; - - // read the mask into shared mem - if (FC_flash_attn_ext_has_mask) { - blk_cur = blk[ic0]; - - if (blk_cur == 0) { - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - pm2[jj] += NW; - } - - continue; - } - - if (blk_cur == 1) { - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - if (FC_flash_attn_ext_bc_mask) { - sm2[j*SH + tiisg] = (iq1 + j) < args.ne31 ? pm2[jj][tiisg] : half2(-MAXHALF, -MAXHALF); - } else { - sm2[j*SH + tiisg] = pm2[jj][tiisg]; - } - - pm2[jj] += NW; - } - } else if (blk_cur == 2) { - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - pm2[jj] += NW; - } - } - -#if 0 - // note: old -INF block optimization - obsoleted by pre-computing non-masked blocks - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // used to detect blocks full of -INF - // skip only when the entire threadgroup is masked - half2 smax2(-MAXHALF/2, -MAXHALF/2); - - FOR_UNROLL (short j = 0; j < Q; ++j) { - smax2 = max(smax2, sm2[j*SH + tiisg]); - } - - smax2 = simd_max(smax2); - - if (max(smax2[0], smax2[1]) <= -MAXHALF/2) { - // this barrier is important - threadgroup_barrier(mem_flags::mem_threadgroup); - - continue; - } -#endif - } - - // Q*K^T - // this is compile-time check, so it does not have runtime overhead - if (is_same::value) { - // we can read directly from global memory - device const k_t * pk = (device const k_t *) (k + ic*args.nb11); - threadgroup const q_t * pq = sq; - threadgroup s_t * ps = ss; - - pk += sgitg*(8*NS10); - ps += sgitg*(8*1); - - static_assert((C/8) % NSG == 0, ""); - - constexpr short NC = (C/8)/NSG; - - FOR_UNROLL (short cc = 0; cc < NC; ++cc) { - qk8x8_t mqk = make_filled_simdgroup_matrix((qk_t) 0.0f); - - if (DK % 16 != 0) { - k8x8_t mk; - q8x8_t mq; - - FOR_UNROLL (short i = 0; i < DK8; ++i) { - simdgroup_barrier(mem_flags::mem_none); - - simdgroup_load(mk, pk + 8*i, NS10, 0, true); - simdgroup_load(mq, pq + 8*i, DK); - - simdgroup_barrier(mem_flags::mem_none); - - simdgroup_multiply_accumulate(mqk, mq, mk, mqk); - } - } else { - k8x8_t mk[2]; - q8x8_t mq[2]; - - // note: too much unroll can tank the performance for large heads - #pragma unroll (MIN(DK8/2, 4*NSG)) - for (short i = 0; i < DK8/2; ++i) { - simdgroup_barrier(mem_flags::mem_none); - - simdgroup_load(mq[0], pq + 0*8 + 16*i, DK); - simdgroup_load(mq[1], pq + 1*8 + 16*i, DK); - - simdgroup_load(mk[0], pk + 0*8 + 16*i, NS10, 0, true); - simdgroup_load(mk[1], pk + 1*8 + 16*i, NS10, 0, true); - - simdgroup_barrier(mem_flags::mem_none); - - simdgroup_multiply_accumulate(mqk, mq[0], mk[0], mqk); - simdgroup_multiply_accumulate(mqk, mq[1], mk[1], mqk); - } - } - - simdgroup_store(mqk, ps, SH, 0, false); - - pk += 8*(NSG*NS10); - ps += 8*(NSG); - } - } else { - // TODO: this is the quantized K cache branch - not optimized yet - for (short ccc = 0; ccc < (C/8)/NSG; ++ccc) { - const short cc = ccc*NSG + sgitg; - - const short tx = tiisg%4; - const short ty = tiisg/4; - - qk8x8_t mqk = make_filled_simdgroup_matrix((qk_t) 0.0f); - - for (short ii = 0; ii < DK16; ii += 4) { - device const kd4x4_t * pk4x4 = (device const kd4x4_t *) (k + ((ic + 8*cc + ty)*args.nb11)); - - if (DK16%4 == 0) { - // the head is evenly divisible by 4*16 = 64, so no need for bound checks - { - k4x4_t tmp; - deq_k(pk4x4 + (ii + tx)/nl_k, (ii + tx)%nl_k, tmp); - sk4x4[4*ty + tx] = tmp; - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - FOR_UNROLL (short k = 0; k < 4; ++k) { - k8x8_t mk; - q8x8_t mq; - - simdgroup_load(mk, sk + 16*k + 0*8, 4*16, 0, true); // transpose - simdgroup_load(mq, sq + (2*(ii + k) + 0)*8, DK); - simdgroup_multiply_accumulate(mqk, mq, mk, mqk); - - simdgroup_load(mk, sk + 16*k + 1*8, 4*16, 0, true); // transpose - simdgroup_load(mq, sq + (2*(ii + k) + 1)*8, DK); - simdgroup_multiply_accumulate(mqk, mq, mk, mqk); - } - } else { - if (ii + tx < DK16) { - k4x4_t tmp; - deq_k(pk4x4 + (ii + tx)/nl_k, (ii + tx)%nl_k, tmp); - sk4x4[4*ty + tx] = tmp; - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - for (short k = 0; k < 4 && ii + k < DK16; ++k) { - k8x8_t mk; - q8x8_t mq; - - simdgroup_load(mk, sk + 16*k + 0*8, 4*16, 0, true); // transpose - simdgroup_load(mq, sq + (2*(ii + k) + 0)*8, DK); - simdgroup_multiply_accumulate(mqk, mq, mk, mqk); - - simdgroup_load(mk, sk + 16*k + 1*8, 4*16, 0, true); // transpose - simdgroup_load(mq, sq + (2*(ii + k) + 1)*8, DK); - simdgroup_multiply_accumulate(mqk, mq, mk, mqk); - } - } - } - - simdgroup_store(mqk, ss + 8*cc, SH, 0, false); - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // online softmax - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - const float m = M[jj]; - - // scale and apply the logitcap / mask - float2 s2 = ss2[j*SH/2 + tiisg]*args.scale; - - if (FC_flash_attn_ext_has_scap) { - s2 = args.logit_softcap*precise::tanh(s2); - } - - // mqk = mqk + slope*mask - if (blk_cur != 2) { - if (FC_flash_attn_ext_has_bias) { - s2 += s2_t(sm2[j*SH + tiisg])*slope; - } else { - s2 += s2_t(sm2[j*SH + tiisg]); - } - } - - M[jj] = simd_max(max(M[jj], max(s2[0], s2[1]))); - - const float ms = exp(m - M[jj]); - const float2 vs2 = exp(s2 - M[jj]); - - S[jj] = S[jj]*ms + simd_sum(vs2[0] + vs2[1]); - - // the P matrix from the paper (Q rows, C columns) - ss2[j*SH/2 + tiisg] = vs2; - - if (DV4 % NW == 0) { - FOR_UNROLL (short ii = 0; ii < DV4/NW; ++ii) { - const short i = ii*NW + tiisg; - - so4[j*PV4 + i] *= ms; - } - } else { - for (short i = tiisg; i < DV4; i += NW) { - so4[j*PV4 + i] *= ms; - } - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // O = O + (Q*K^T)*V - { - // we can read directly from global memory - if (is_same::value) { - static_assert(PV8 % NSG == 0, ""); - - constexpr short NO = PV8/NSG; - - o8x8_t lo[NO]; - - { - auto sot = so + 8*sgitg; - - FOR_UNROLL (short ii = 0; ii < NO; ++ii) { - simdgroup_load(lo[ii], sot, PV, 0, false); - - sot += 8*NSG; - } - } - - { - device const v_t * pv = (device const v_t *) (v + ic*args.nb21); - - pv += 8*sgitg; - - if (DV <= 64) { - FOR_UNROLL (short cc = 0; cc < C/8; ++cc) { - s8x8_t vs; - simdgroup_load(vs, ss + 8*cc, SH, 0, false); - - FOR_UNROLL (short ii = 0; ii < NO/2; ++ii) { - v8x8_t mv[2]; - - simdgroup_load(mv[0], pv + 0*NSG + 16*ii*NSG, NS20, 0, false); - simdgroup_load(mv[1], pv + 8*NSG + 16*ii*NSG, NS20, 0, false); - - simdgroup_multiply_accumulate(lo[2*ii + 0], vs, mv[0], lo[2*ii + 0]); - simdgroup_multiply_accumulate(lo[2*ii + 1], vs, mv[1], lo[2*ii + 1]); - } - - pv += 8*NS20; - } - } else { - constexpr short NC = (C/8)/2; - - FOR_UNROLL (short cc = 0; cc < NC; ++cc) { - s8x8_t vs[2]; - - simdgroup_load(vs[0], ss + 16*cc + 0, SH, 0, false); - simdgroup_load(vs[1], ss + 16*cc + 8, SH, 0, false); - - FOR_UNROLL (short ii = 0; ii < NO/2; ++ii) { - v8x8_t mv[4]; - - simdgroup_load(mv[0], pv + 0*NSG + 16*ii*NSG + 0*8*NS20, NS20, 0, false); - simdgroup_load(mv[1], pv + 8*NSG + 16*ii*NSG + 0*8*NS20, NS20, 0, false); - simdgroup_load(mv[2], pv + 0*NSG + 16*ii*NSG + 1*8*NS20, NS20, 0, false); - simdgroup_load(mv[3], pv + 8*NSG + 16*ii*NSG + 1*8*NS20, NS20, 0, false); - - simdgroup_multiply_accumulate(lo[2*ii + 0], vs[0], mv[0], lo[2*ii + 0]); - simdgroup_multiply_accumulate(lo[2*ii + 1], vs[0], mv[1], lo[2*ii + 1]); - simdgroup_multiply_accumulate(lo[2*ii + 0], vs[1], mv[2], lo[2*ii + 0]); - simdgroup_multiply_accumulate(lo[2*ii + 1], vs[1], mv[3], lo[2*ii + 1]); - } - - pv += 2*8*NS20; - } - } - } - - { - auto sot = so + 8*sgitg; - - FOR_UNROLL (short ii = 0; ii < NO; ++ii) { - simdgroup_store(lo[ii], sot, PV, 0, false); - - sot += 8*NSG; - } - } - } else { - // TODO: this is the quantized V cache branch - not optimized yet - - const short tx = tiisg%4; - const short ty = tiisg/4; - - for (short cc = 0; cc < C/8; ++cc) { - s8x8_t vs; - simdgroup_load(vs, ss + 8*cc, SH, 0, false); - - for (short ii = 4*sgitg; ii < DV16; ii += 4*NSG) { - device const vd4x4_t * pv4x4 = (device const vd4x4_t *) (v + ((ic + 8*cc + ty)*args.nb21)); - - if (DV16%4 == 0) { - // no need for bound checks - { - v4x4_t tmp; - deq_v(pv4x4 + (ii + tx)/nl_v, (ii + tx)%nl_v, tmp); - sv4x4[4*ty + tx] = tmp; - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - FOR_UNROLL (short k = 0; k < 4; ++k) { - v8x8_t mv[2]; - o8x8_t lo[2]; - - simdgroup_load(mv[0], sv + 16*k + 0*8, 4*16, 0, false); - simdgroup_load(mv[1], sv + 16*k + 1*8, 4*16, 0, false); - simdgroup_load(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); - simdgroup_load(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); - - simdgroup_multiply_accumulate(lo[0], vs, mv[0], lo[0]); - simdgroup_multiply_accumulate(lo[1], vs, mv[1], lo[1]); - - simdgroup_store(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); - simdgroup_store(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); - } - } else { - if (ii + tx < DV16) { - v4x4_t tmp; - deq_v(pv4x4 + (ii + tx)/nl_v, (ii + tx)%nl_v, tmp); - sv4x4[4*ty + tx] = tmp; - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - for (short k = 0; k < 4 && ii + k < DV16; ++k) { - v8x8_t mv[2]; - o8x8_t lo[2]; - - simdgroup_load(mv[0], sv + 16*k + 0*8, 4*16, 0, false); - simdgroup_load(mv[1], sv + 16*k + 1*8, 4*16, 0, false); - simdgroup_load(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); - simdgroup_load(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); - - simdgroup_multiply_accumulate(lo[0], vs, mv[0], lo[0]); - simdgroup_multiply_accumulate(lo[1], vs, mv[1], lo[1]); - - simdgroup_store(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); - simdgroup_store(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); - } - } - } - } - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - if (FC_flash_attn_ext_has_sinks) { - FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - - const float m = M[jj]; - const float s = tiisg == 0 ? ((device const float *) sinks)[iq2] : -FLT_MAX/2; - - M[jj] = simd_max(max(M[jj], s)); - - const float ms = exp(m - M[jj]); - const float vs = exp(s - M[jj]); - - S[jj] = S[jj]*ms + simd_sum(vs); - - for (short i = tiisg; i < DV4; i += NW) { - so4[j*PV4 + i] *= ms; - } - } - } - } - - // store to global memory - for (short jj = 0; jj < NQ; ++jj) { - const short j = jj*NSG + sgitg; - if (iq1 + j >= args.ne01) { - break; - } - - device float4 * dst4 = (device float4 *) dst + ((uint64_t)iq3*args.ne2*args.ne1 + iq2 + (uint64_t)(iq1 + j)*args.ne1)*DV4; - - const float scale = S[jj] == 0.0 ? 0.0f : 1.0f/S[jj]; - - if (DV4 % NW == 0) { - FOR_UNROLL (short ii = 0; ii < DV4/NW; ++ii) { - const short i = ii*NW + tiisg; - - dst4[i] = (float4) so4[j*PV4 + i]*scale; - } - } else { - for (short i = tiisg; i < DV4; i += NW) { - dst4[i] = (float4) so4[j*PV4 + i]*scale; - } - } - } - -#undef NS10 -#undef NS20 -} - -template< - typename q_t, // query types in shared memory - typename q4_t, - typename q8x8_t, - typename k_t, // key types in shared memory - typename k4x4_t, - typename k8x8_t, - typename v_t, // value types in shared memory - typename v4x4_t, - typename v8x8_t, - typename qk_t, // Q*K types - typename qk8x8_t, - typename s_t, // soft-max types - typename s2_t, - typename s8x8_t, - typename o_t, // attention accumulation types - typename o4_t, - typename o8x8_t, - typename kd4x4_t, // key type in device memory - short nl_k, - void (*deq_k)(device const kd4x4_t *, short, thread k4x4_t &), - typename vd4x4_t, // value type in device memory - short nl_v, - void (*deq_v)(device const vd4x4_t *, short, thread v4x4_t &), - short DK, // K head size - short DV, // V head size - short Q = OP_FLASH_ATTN_EXT_NQPSG, // queries per threadgroup - short C = OP_FLASH_ATTN_EXT_NCPSG> // cache items per threadgroup -kernel void kernel_flash_attn_ext( - constant ggml_metal_kargs_flash_attn_ext & args, - device const char * q, - device const char * k, - device const char * v, - device const char * mask, - device const char * sinks, - device const char * pad, - device const char * blk, - device char * dst, - threadgroup half * shmem_f16 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { -#define FWD_TMPL q_t, q4_t, q8x8_t, k_t, k4x4_t, k8x8_t, v_t, v4x4_t, v8x8_t, qk_t, qk8x8_t, s_t, s2_t, s8x8_t, o_t, o4_t, o8x8_t, kd4x4_t, nl_k, deq_k, vd4x4_t, nl_v, deq_v, DK, DV, Q, C -#define FWD_ARGS args, q, k, v, mask, sinks, pad, blk, dst, shmem_f16, tgpig, tiisg, sgitg - switch (FC_flash_attn_ext_nsg) { - // note: disabled cases to reduce library load time - //case 1: kernel_flash_attn_ext_impl(FWD_ARGS); break; - //case 2: kernel_flash_attn_ext_impl(FWD_ARGS); break; - case 4: kernel_flash_attn_ext_impl(FWD_ARGS); break; - case 8: kernel_flash_attn_ext_impl(FWD_ARGS); break; - } -#undef FWD_TMPL -#undef FWD_ARGS -} - -// TODO: this is quite ugly. in the future these types will be hardcoded in the kernel, but for now keep them as -// template to be able to explore different combinations -// -#define FA_TYPES \ - half, half4, simdgroup_half8x8, \ - half, half4x4, simdgroup_half8x8, \ - half, half4x4, simdgroup_half8x8, \ - float, simdgroup_float8x8, \ - float, float2, simdgroup_float8x8, \ - float, float4, simdgroup_float8x8 - //half, half4, simdgroup_half8x8 - -#define FA_TYPES_BF \ - bfloat, bfloat4, simdgroup_bfloat8x8, \ - bfloat, bfloat4x4, simdgroup_bfloat8x8, \ - bfloat, bfloat4x4, simdgroup_bfloat8x8, \ - float, simdgroup_float8x8, \ - float, float2, simdgroup_float8x8, \ - half, half4, simdgroup_half8x8 - //float, float4, simdgroup_float8x8 - -#define FA_TYPES_F32 \ - half, half4, simdgroup_half8x8, \ - float, float4x4, simdgroup_float8x8, \ - float, float4x4, simdgroup_float8x8, \ - float, simdgroup_float8x8, \ - float, float2, simdgroup_float8x8, \ - float, float4, simdgroup_float8x8 - //half, half4, simdgroup_half8x8 - -typedef decltype(kernel_flash_attn_ext) flash_attn_ext_t; - -template [[host_name("kernel_flash_attn_ext_f32_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f32_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -template [[host_name("kernel_flash_attn_ext_f16_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_f16_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_bf16_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_bf16_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -#endif - -template [[host_name("kernel_flash_attn_ext_q4_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -template [[host_name("kernel_flash_attn_ext_q4_1_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q4_1_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -template [[host_name("kernel_flash_attn_ext_q5_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -template [[host_name("kernel_flash_attn_ext_q5_1_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q5_1_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -template [[host_name("kernel_flash_attn_ext_q8_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; -template [[host_name("kernel_flash_attn_ext_q8_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; - -#undef FA_TYPES -#undef FA_TYPES_BF -#undef FA_TYPES_F32 - -constant bool FC_flash_attn_ext_vec_has_mask [[function_constant(FC_FLASH_ATTN_EXT_VEC + 0)]]; -constant bool FC_flash_attn_ext_vec_has_sinks [[function_constant(FC_FLASH_ATTN_EXT_VEC + 1)]]; -constant bool FC_flash_attn_ext_vec_has_bias [[function_constant(FC_FLASH_ATTN_EXT_VEC + 2)]]; -constant bool FC_flash_attn_ext_vec_has_scap [[function_constant(FC_FLASH_ATTN_EXT_VEC + 3)]]; -constant bool FC_flash_attn_ext_vec_has_kvpad [[function_constant(FC_FLASH_ATTN_EXT_VEC + 4)]]; - -//constant float FC_flash_attn_ext_vec_scale [[function_constant(FC_FLASH_ATTN_EXT_VEC + 10)]]; -//constant float FC_flash_attn_ext_vec_max_bias [[function_constant(FC_FLASH_ATTN_EXT_VEC + 11)]]; -//constant float FC_flash_attn_ext_vec_logit_softcap [[function_constant(FC_FLASH_ATTN_EXT_VEC + 12)]]; - -constant int32_t FC_flash_attn_ext_vec_ns10 [[function_constant(FC_FLASH_ATTN_EXT_VEC + 20)]]; -constant int32_t FC_flash_attn_ext_vec_ns20 [[function_constant(FC_FLASH_ATTN_EXT_VEC + 21)]]; -constant int32_t FC_flash_attn_ext_vec_nsg [[function_constant(FC_FLASH_ATTN_EXT_VEC + 22)]]; -constant int32_t FC_flash_attn_ext_vec_nwg [[function_constant(FC_FLASH_ATTN_EXT_VEC + 23)]]; - -template< - typename q4_t, // query types in shared memory - typename k4_t, // key types in shared memory - typename v4_t, // value types in shared memory - typename qk_t, // Q*K types - typename s_t, // soft-max types - typename s4_t, - typename o4_t, // attention accumulation types - typename kd4_t, // key type in device memory - short nl_k, - void (*deq_k_t4)(device const kd4_t *, short, thread k4_t &), - typename vd4_t, // value type in device memory - short nl_v, - void (*deq_v_t4)(device const vd4_t *, short, thread v4_t &), - short DK, // K head size - short DV, // V head size - short NE = 4, // head elements per thread - short Q = OP_FLASH_ATTN_EXT_VEC_NQPSG, // queries per threadgroup - short C = OP_FLASH_ATTN_EXT_VEC_NCPSG> // cache items per threadgroup -kernel void kernel_flash_attn_ext_vec( - constant ggml_metal_kargs_flash_attn_ext_vec & args, - device const char * q, - device const char * k, - device const char * v, - device const char * mask, - device const char * sinks, - device const char * pad, - device char * dst, - threadgroup half * shmem_f16 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - static_assert(DK % 32 == 0, "DK must be divisible by 32"); - static_assert(DV % 32 == 0, "DV must be divisible by 32"); - -#define NWG (FC_flash_attn_ext_vec_nwg) -#define NSG (FC_flash_attn_ext_vec_nsg) - -#define NS10 (FC_flash_attn_ext_vec_ns10) -#define NS20 (FC_flash_attn_ext_vec_ns20) - - const short iwg = tgpig[2]%NWG; - - const ushort iq3 = tgpig[2]/NWG; - const ushort iq2 = tgpig[1]; - const ushort iq1 = tgpig[0]; - - constexpr short DK4 = DK/4; - constexpr short DV4 = DV/4; - - constexpr short PK = PAD2(DK, 128); - constexpr short PK4 = PK/4; - - constexpr short PV = PAD2(DV, 128); - constexpr short PV4 = PV/4; - - constexpr short NW = N_SIMDWIDTH; - constexpr short NL = NW/NE; // note: this can be adjusted to support different head sizes and simdgroup work loads - constexpr short SH = 4*C; // shared memory per simdgroup - - static_assert(DK4 % NL == 0, "DK4 must be divisible by NL"); - static_assert(DV4 % NL == 0, "DV4 must be divisible by NL"); - - //const short T = PK + NSG*SH; // shared memory size per query in (half) - - //threadgroup q_t * sq = (threadgroup q_t *) (shmem_f16 + 0*PK); // holds the query data - threadgroup q4_t * sq4 = (threadgroup q4_t *) (shmem_f16 + 0*PK); // same as above but in q4_t - threadgroup s_t * ss = (threadgroup s_t *) (shmem_f16 + sgitg*SH + NSG*PK); // scratch buffer for attention - threadgroup s4_t * ss4 = (threadgroup s4_t *) (shmem_f16 + sgitg*SH + NSG*PK); // same as above but in s4_t - threadgroup half * sm = (threadgroup half *) (shmem_f16 + sgitg*SH + 2*C + NSG*PK); // scratch buffer for mask - threadgroup o4_t * so4 = (threadgroup o4_t *) (shmem_f16 + 2*sgitg*PV + NSG*PK + NSG*SH); // scratch buffer for the results - - // store the result for all queries in shared memory (the O matrix from the paper) - so4 += tiisg; - - { - q += iq1*args.nb01 + iq2*args.nb02 + iq3*args.nb03; - - const short ikv2 = iq2/(args.ne02/args.ne_12_2); - const short ikv3 = iq3/(args.ne03/args.ne_12_3); - - k += ikv2*args.nb12 + ikv3*args.nb13; - v += ikv2*args.nb22 + ikv3*args.nb23; - } - - // load heads from Q to shared memory - device const float4 * q4 = (device const float4 *) ((device const char *) q); - - if (iq1 < args.ne01) { - for (short i = tiisg; i < PK4; i += NW) { - if (i < DK4) { - sq4[i] = (q4_t) q4[i]; - } else { - sq4[i] = (q4_t) 0.0f; - } - } - } - - // zero out so - for (short i = 0; i < DV4/NL; ++i) { - so4[i*NL] = (o4_t) 0.0f; - } - - // zero out shared memory SH - for (short i = tiisg; i < SH/4; i += NW) { - ss4[i] = (s4_t) 0.0f; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - { - float S = 0.0f; - float M = -FLT_MAX/2; - - // thread indices inside the simdgroup - const short tx = tiisg%NL; - const short ty = tiisg/NL; - - // pointer to the mask - device const half * pm = (device const half *) (mask + iq1*args.nb31 + (iq2%args.ne32)*args.nb32 + (iq3%args.ne33)*args.nb33); - - float slope = 1.0f; - - // ALiBi - if (FC_flash_attn_ext_vec_has_bias) { - const short h = iq2; - - const float base = h < args.n_head_log2 ? args.m0 : args.m1; - const short exph = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; - - slope = pow(base, exph); - } - - // loop over the KV cache - // each simdgroup handles blocks of Q rows and C columns - for (int ic0 = iwg*NSG + sgitg; ; ic0 += NWG*NSG) { - int ic = ic0*C; - if (ic >= args.ne11) { - break; - } - - // the last partial chunk uses the pad buffer as source - if (FC_flash_attn_ext_vec_has_kvpad && ic + C > args.ne11) { - k = pad; - v = k + args.nb11*C*args.ne_12_2*args.ne_12_3; - mask = v + args.nb21*C*args.ne_12_2*args.ne_12_3; - - const short ikv2 = iq2/(args.ne02/args.ne_12_2); - const short ikv3 = iq3/(args.ne03/args.ne_12_3); - - k += (ikv2 + ikv3*args.ne_12_2)*args.nb11*C; - v += (ikv2 + ikv3*args.ne_12_2)*args.nb21*C; - - if (!FC_flash_attn_ext_vec_has_mask) { - if (ic + tiisg >= args.ne11) { - sm[tiisg] = -MAXHALF; - } - } else { - pm = (device const half *) (mask) + - iq1*C + - (iq2%args.ne32)*(C*args.ne31) + - (iq3%args.ne33)*(C*args.ne31*args.ne32); - } - - ic = 0; - } - - if (FC_flash_attn_ext_vec_has_mask) { - sm[tiisg] = pm[ic + tiisg]; - } - - // skip -INF blocks - if (simd_max(sm[tiisg]) <= -MAXHALF) { - continue; - } - - // Q*K^T - { - device const k4_t * pk4 = (device const k4_t *) (k + ic*args.nb11); - threadgroup const q4_t * pq4 = sq4; - - pk4 += ty*NS10/4 + tx; - pq4 += tx; - - qk_t mqk[C/NE] = { [ 0 ... C/NE - 1] = 0.0f }; - - // each simdgroup processes 1 query and NE (NW/NL) cache elements - FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { - if (is_same::value) { - FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { - mqk[cc] += dot((float4) pk4[cc*NE*NS10/4 + ii*NL], (float4) pq4[ii*NL]); - } - } else { - device const kd4_t * pk = (device const kd4_t *) (k + ((ic + NE*cc + ty)*args.nb11)); - - k4_t mk; - - FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { - const short i = ii*NL + tx; - - deq_k_t4(pk + i/nl_k, i%nl_k, mk); - - mqk[cc] += dot((float4) mk, (float4) sq4[i]); - } - } - - if (NE == 1) { - mqk[cc] = simd_sum(mqk[cc]); - } else { - // simdgroup reduce (NE = 4) - // [ 0 .. 7] -> [ 0] - // [ 8 .. 15] -> [ 8] - // [16 .. 23] -> [16] - // [24 .. 31] -> [24] - if (NE <= 1) { - mqk[cc] += simd_shuffle_down(mqk[cc], 16); - } - if (NE <= 2) { - mqk[cc] += simd_shuffle_down(mqk[cc], 8); - } - if (NE <= 4) { - mqk[cc] += simd_shuffle_down(mqk[cc], 4); - } - if (NE <= 8) { - mqk[cc] += simd_shuffle_down(mqk[cc], 2); - } - if (NE <= 16) { - mqk[cc] += simd_shuffle_down(mqk[cc], 1); - } - - // broadcast - mqk[cc] = simd_shuffle(mqk[cc], NL*ty); - } - } - - if (FC_flash_attn_ext_vec_has_mask && - !FC_flash_attn_ext_vec_has_scap && - !FC_flash_attn_ext_vec_has_bias) { - ss[NE*tx + ty] = fma(mqk[tx], args.scale, (qk_t) sm[NE*tx + ty]); - } else { - mqk[tx] *= args.scale; - - if (FC_flash_attn_ext_vec_has_scap) { - mqk[tx] = args.logit_softcap*precise::tanh(mqk[tx]); - } - - if (FC_flash_attn_ext_vec_has_bias) { - mqk[tx] += (qk_t) sm[NE*tx + ty]*slope; - } else { - mqk[tx] += (qk_t) sm[NE*tx + ty]; - } - - ss[NE*tx + ty] = mqk[tx]; - } - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - // online softmax - { - const float m = M; - const float s = ss[tiisg]; - - M = simd_max(max(M, s)); - - const float ms = exp(m - M); - const float vs = exp(s - M); - - S = S*ms + simd_sum(vs); - - // the P matrix from the paper (Q rows, C columns) - ss[tiisg] = vs; - - // O = diag(ms)*O - if ((DV4/NL % NW == 0) || ty == 0) { - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - so4[ii*NL] *= ms; - } - } - } - - simdgroup_barrier(mem_flags::mem_threadgroup); - - // O = O + (Q*K^T)*V - { - o4_t lo[DV4/NL]; - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - lo[ii] = 0.0f; - } - - if (is_same::value) { - device const v4_t * pv4 = (device const v4_t *) (v + ic*args.nb21); - - pv4 += ty*NS20/4 + tx; - - const auto sst = ss + ty; - - FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - lo[ii] += o4_t(float4(pv4[cc*NE*NS20/4 + ii*NL])*float4(sst[cc*NE])); - } - } - } else { - FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { - device const vd4_t * pv4 = (device const vd4_t *) (v + ((ic + NE*cc + ty)*args.nb21)); - - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - const short i = ii*NL + tx; - - v4_t mv; - deq_v_t4(pv4 + i/nl_v, i%nl_v, mv); - - lo[ii] += o4_t(float4(mv)*float4(ss[NE*cc + ty])); - } - } - } - - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - if (NE > 1) { - lo[ii][0] += simd_shuffle_down(lo[ii][0], 16); - lo[ii][1] += simd_shuffle_down(lo[ii][1], 16); - lo[ii][2] += simd_shuffle_down(lo[ii][2], 16); - lo[ii][3] += simd_shuffle_down(lo[ii][3], 16); - } - - if (NE > 2) { - lo[ii][0] += simd_shuffle_down(lo[ii][0], 8); - lo[ii][1] += simd_shuffle_down(lo[ii][1], 8); - lo[ii][2] += simd_shuffle_down(lo[ii][2], 8); - lo[ii][3] += simd_shuffle_down(lo[ii][3], 8); - } - - if (NE > 4) { - lo[ii][0] += simd_shuffle_down(lo[ii][0], 4); - lo[ii][1] += simd_shuffle_down(lo[ii][1], 4); - lo[ii][2] += simd_shuffle_down(lo[ii][2], 4); - lo[ii][3] += simd_shuffle_down(lo[ii][3], 4); - } - - if (NE > 8) { - lo[ii][0] += simd_shuffle_down(lo[ii][0], 2); - lo[ii][1] += simd_shuffle_down(lo[ii][1], 2); - lo[ii][2] += simd_shuffle_down(lo[ii][2], 2); - lo[ii][3] += simd_shuffle_down(lo[ii][3], 2); - } - - if (NE > 16) { - lo[ii][0] += simd_shuffle_down(lo[ii][0], 1); - lo[ii][1] += simd_shuffle_down(lo[ii][1], 1); - lo[ii][2] += simd_shuffle_down(lo[ii][2], 1); - lo[ii][3] += simd_shuffle_down(lo[ii][3], 1); - } - } - - if ((DV4/NL % NW == 0) || ty == 0) { - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - so4[ii*NL] += lo[ii]; - } - } - } - } - - if (FC_flash_attn_ext_vec_has_sinks && sgitg == 0 && iwg == 0) { - const float m = M; - const float s = tiisg == 0 ? ((device const float *) sinks)[iq2] : -FLT_MAX/2; - - M = simd_max(max(M, s)); - - const float ms = exp(m - M); - const float vs = exp(s - M); - - S = S*ms + simd_sum(vs); - - if ((DV4/NL % NW == 0) || ty == 0) { - FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { - so4[ii*NL] *= ms; - } - } - } - - // these are needed for reducing the results from the simdgroups (reuse the ss buffer) - if (tiisg == 0) { - ss[0] = (s_t) S; - ss[1] = (s_t) M; - } - } - - so4 -= tiisg; - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // parallel reduce - for (short r = NSG/2; r > 0; r >>= 1) { - if (sgitg < r) { - const float S0 = ss[ 0]; - const float S1 = ss[r*(SH/2) + 0]; - - const float M0 = ss[ 1]; - const float M1 = ss[r*(SH/2) + 1]; - - const float M = max(M0, M1); - - const float ms0 = exp(M0 - M); - const float ms1 = exp(M1 - M); - - const float S = S0*ms0 + S1*ms1; - - if (tiisg == 0) { - ss[0] = S; - ss[1] = M; - } - - // O_0 = diag(ms0)*O_0 + diag(ms1)*O_1 - for (short i = tiisg; i < DV4; i += NW) { - so4[i] = so4[i]*ms0 + so4[i + r*PV4]*ms1; - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - // final rescale with 1/S and store to global memory - if (sgitg == 0) { - const int64_t nrows = args.ne3*args.ne2*args.ne1; - const int64_t rid = iq3*args.ne2*args.ne1 + iq2 + iq1*args.ne1; - - device float4 * dst4 = (device float4 *) dst; - device float * dst1 = (device float *) dst + nrows*DV*NWG; // the S and M are stored after the results - - const float S = NWG == 1 ? (ss[0] == 0.0f ? 0.0f : 1.0f/ss[0]) : 1.0f; - - // interleave the workgroup data - for (short i = tiisg; i < DV4; i += NW) { - dst4[rid*DV4*NWG + NWG*i + iwg] = (float4) so4[i]*S; - } - - // store S and M - if (NWG > 1) { - if (tiisg == 0) { - dst1[rid*(2*NWG) + 2*iwg + 0] = ss[0]; - dst1[rid*(2*NWG) + 2*iwg + 1] = ss[1]; - } - } - } - -#undef NWG -#undef NSG -#undef NS10 -#undef NS20 -} - -// note: I think the s_t can be half instead of float, because the Q*K scaling is done before storing to shared mem -// in the other (non-vec) kernel, we need s_t to also be float because we scale during the soft_max -// -#define FA_TYPES \ - half4, \ - half4, \ - half4, \ - float, \ - float, float4, \ - float4 - -#define FA_TYPES_F32 \ - half4, \ - float4, \ - float4, \ - float, \ - float, float4, \ - float4 - -typedef decltype(kernel_flash_attn_ext_vec) flash_attn_ext_vec_t; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -template [[host_name("kernel_flash_attn_ext_vec_f32_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_flash_attn_ext_vec_bf16_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -#endif -template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; -template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; - -#undef FA_TYPES -#undef FA_TYPES_F32 - -constant int32_t FC_flash_attn_ext_vec_reduce_DV [[function_constant(FC_FLASH_ATTN_EXT_VEC_REDUCE + 0)]]; -constant int32_t FC_flash_attn_ext_vec_reduce_NWG [[function_constant(FC_FLASH_ATTN_EXT_VEC_REDUCE + 1)]]; - -kernel void kernel_flash_attn_ext_vec_reduce( - constant ggml_metal_kargs_flash_attn_ext_vec_reduce & args, - device const char * htmp, - device char * dst, - uint tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { -#define NWG (FC_flash_attn_ext_vec_reduce_NWG) -#define DV (FC_flash_attn_ext_vec_reduce_DV) - - const uint64_t rid = tgpig; - - const short iwg = tiisg; - - device const float * ss = (device const float *) htmp + (uint64_t)args.nrows*DV*NWG; - - float S = ss[rid*(2*NWG) + 2*iwg + 0]; - float M = ss[rid*(2*NWG) + 2*iwg + 1]; - - const float m = simd_max(M); - const float ms = exp(M - m); - - S = simd_sum(S*ms); - S = S == 0.0f ? 0.0f : 1.0f/S; - - const short DV4 = DV/4; - - device const float4 * htmp4 = (device const float4 *) htmp + rid*DV4*NWG; - device float4 * dst4 = (device float4 *) dst + rid*DV4; - - for (short i = sgitg; i < DV4; i += NWG) { - const float4 v = simd_sum(htmp4[i*NWG + iwg]*ms); - - if (iwg == 0) { - dst4[i] = v*S; - } - } - -#undef NWG -#undef DV -} - -template -kernel void kernel_cpy_t_t( - constant ggml_metal_kargs_cpy & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int32_t i03 = tgpig[2]; - const int32_t i02 = tgpig[1]; - const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; - const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; - - if (i01 >= args.ne01) { - return; - } - - const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; - - const int32_t i3 = n/(args.ne2*args.ne1*args.ne0); - const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0)/(args.ne1*args.ne0); - const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0)/args.ne0; - const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0); - - device T1 * dst_data = (device T1 *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.ne00;) { - device const T0 * src = (device T0 *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + i00*args.nb00); - dst_data[i00] = (T1) src[0]; - break; - } -} - -typedef decltype(kernel_cpy_t_t) kernel_cpy_t; - -template [[host_name("kernel_cpy_f32_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_f32_f16")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_f32_i32")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_i32_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_i32_i32")]] kernel kernel_cpy_t kernel_cpy_t_t; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_cpy_f32_bf16")]] kernel kernel_cpy_t kernel_cpy_t_t; -#endif -template [[host_name("kernel_cpy_f16_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_f16_f16")]] kernel kernel_cpy_t kernel_cpy_t_t; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_cpy_bf16_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; -template [[host_name("kernel_cpy_bf16_bf16")]] kernel kernel_cpy_t kernel_cpy_t_t; -#endif - -template -kernel void kernel_cpy_f32_q( - constant ggml_metal_kargs_cpy & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int32_t i03 = tgpig[2]; - const int32_t i02 = tgpig[1]; - const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; - const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; - - if (i01 >= args.ne01) { - return; - } - - const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; - - const int32_t i3 = n / (args.ne2*args.ne1*args.ne0); - const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0) / (args.ne1*args.ne0); - const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0) / args.ne0; - const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0)/QK; - - device block_q * dst_data = (device block_q *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.nk0;) { - device const float * src = (device const float *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + (i00*QK)*args.nb00); - - quantize_func(src, dst_data[i00]); - - break; - } -} - -typedef decltype(kernel_cpy_f32_q) cpy_f_q_t; - -template [[host_name("kernel_cpy_f32_q8_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q1_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q2_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q4_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q4_1")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q5_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_q5_1")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_iq4_nl")]] kernel cpy_f_q_t kernel_cpy_f32_q; -template [[host_name("kernel_cpy_f32_tq2_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; - -template -kernel void kernel_cpy_q_f32( - constant ggml_metal_kargs_cpy & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const int32_t i03 = tgpig[2]; - const int32_t i02 = tgpig[1]; - const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; - const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; - - if (i01 >= args.ne01) { - return; - } - - const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; - - const int32_t i3 = n/(args.ne2*args.ne1*args.ne0); - const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0)/(args.ne1*args.ne0); - const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0)/args.ne0; - const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0); - - device const block_q * src_data = (device const block_q *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); - device T4x4 * dst_data = (device T4x4 *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.nk0;) { - T4x4 temp; - dequantize_func(src_data + i00/nl, i00%nl, temp); - dst_data[i00] = temp; - - break; - } -} - -typedef decltype(kernel_cpy_q_f32) cpy_q_f_t; - -template [[host_name("kernel_cpy_q1_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q2_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q4_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q4_1_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q5_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q5_1_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q8_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; - -template [[host_name("kernel_cpy_tq2_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; - -template [[host_name("kernel_cpy_q1_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q2_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q4_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q4_1_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q5_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q5_1_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; -template [[host_name("kernel_cpy_q8_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; - -template [[host_name("kernel_cpy_tq2_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; - -template -kernel void kernel_concat( - constant ggml_metal_kargs_concat & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - - const int i3 = tgpig.z; - const int i2 = tgpig.y; - const int i1 = ntg.y == 1 ? tgpig.x : tgpig.x*ntg.y + tpitg.y; - - if (i1 >= args.ne1) { - return; - } - - int o[4] = {0, 0, 0, 0}; - o[args.dim] = args.dim == 0 ? args.ne00 : (args.dim == 1 ? args.ne01 : (args.dim == 2 ? args.ne02 : args.ne03)); - - for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { - device const T * x; - - if (i0 < args.ne00 && i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { - x = (device const T *)(src0 + (i3 )*args.nb03 + (i2 )*args.nb02 + (i1 )*args.nb01 + (i0 )*args.nb00); - } else { - x = (device const T *)(src1 + (i3 - o[3])*args.nb13 + (i2 - o[2])*args.nb12 + (i1 - o[1])*args.nb11 + (i0 - o[0])*args.nb10); - } - - device T * y = (device T *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); - - *y = *x; - } -} - -typedef decltype(kernel_concat) kernel_concat_t; - -template [[host_name("kernel_concat_f32")]] kernel kernel_concat_t kernel_concat; -template [[host_name("kernel_concat_f16")]] kernel kernel_concat_t kernel_concat; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_concat_bf16")]] kernel kernel_concat_t kernel_concat; -#endif -template [[host_name("kernel_concat_i8")]] kernel kernel_concat_t kernel_concat; -template [[host_name("kernel_concat_i16")]] kernel kernel_concat_t kernel_concat; -template [[host_name("kernel_concat_i32")]] kernel kernel_concat_t kernel_concat; -template [[host_name("kernel_concat_i64")]] kernel kernel_concat_t kernel_concat; - -template -void kernel_mul_mv_q2_K_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_q2_K * x = (device const block_q2_K *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const short ix = tiisg/8; // 0...3 - const short it = tiisg%8; // 0...7 - const short iq = it/4; // 0 or 1 - const short ir = it%4; // 0...3 - const short is = (8*ir)/16;// 0 or 1 - - device const float * y4 = y + ix * QK_K + 128 * iq + 8 * ir; - - for (int ib = ix; ib < nb; ib += 4) { - float4 sumy = {0.f, 0.f, 0.f, 0.f}; - for (short i = 0; i < 8; ++i) { - yl[i+ 0] = y4[i+ 0]; sumy[0] += yl[i+ 0]; - yl[i+ 8] = y4[i+32]; sumy[1] += yl[i+ 8]; - yl[i+16] = y4[i+64]; sumy[2] += yl[i+16]; - yl[i+24] = y4[i+96]; sumy[3] += yl[i+24]; - } - - device const uint8_t * sc = (device const uint8_t *)x[ib].scales + 8*iq + is; - device const uint16_t * qs = (device const uint16_t *)x[ib].qs + 16 * iq + 4 * ir; - device const half * dh = &x[ib].d; - - for (short row = 0; row < nr0; row++) { - float4 acc1 = {0.f, 0.f, 0.f, 0.f}; - float4 acc2 = {0.f, 0.f, 0.f, 0.f}; - for (int i = 0; i < 8; i += 2) { - acc1[0] += yl[i+ 0] * (qs[i/2] & 0x0003); - acc2[0] += yl[i+ 1] * (qs[i/2] & 0x0300); - acc1[1] += yl[i+ 8] * (qs[i/2] & 0x000c); - acc2[1] += yl[i+ 9] * (qs[i/2] & 0x0c00); - acc1[2] += yl[i+16] * (qs[i/2] & 0x0030); - acc2[2] += yl[i+17] * (qs[i/2] & 0x3000); - acc1[3] += yl[i+24] * (qs[i/2] & 0x00c0); - acc2[3] += yl[i+25] * (qs[i/2] & 0xc000); - } - float dall = dh[0]; - float dmin = dh[1] * 1.f/16.f; - sumf[row] += dall * ((acc1[0] + 1.f/256.f * acc2[0]) * (sc[0] & 0xF) * 1.f/ 1.f + - (acc1[1] + 1.f/256.f * acc2[1]) * (sc[2] & 0xF) * 1.f/ 4.f + - (acc1[2] + 1.f/256.f * acc2[2]) * (sc[4] & 0xF) * 1.f/16.f + - (acc1[3] + 1.f/256.f * acc2[3]) * (sc[6] & 0xF) * 1.f/64.f) - - dmin * (sumy[0] * (sc[0] & 0xF0) + sumy[1] * (sc[2] & 0xF0) + sumy[2] * (sc[4] & 0xF0) + sumy[3] * (sc[6] & 0xF0)); - - qs += args.nb01/2; - sc += args.nb01; - dh += args.nb01/2; - } - - y4 += 4 * QK_K; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_q2_K_f32")]] -kernel void kernel_mul_mv_q2_K_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_q2_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q3_K_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_q3_K * x = (device const block_q3_K *) (src0 + offset0); - device const float * yy = (device const float *) (src1 + offset1); - - float yl[32]; - - //const uint16_t kmask1 = 0x3030; - //const uint16_t kmask2 = 0x0f0f; - - const short tid = tiisg/4; - const short ix = tiisg%4; - const short ip = tid/4; // 0 or 1 - const short il = 2*((tid%4)/2); // 0 or 2 - const short ir = tid%2; - const short l0 = 8*ir; - - // One would think that the Metal compiler would figure out that ip and il can only have - // 4 possible states, and optimize accordingly. Well, no. It needs help, and we do it - // with these two tales. - // - // Possible masks for the high bit - const ushort4 mm[4] = {{0x0001, 0x0100, 0x0002, 0x0200}, // ip = 0, il = 0 - {0x0004, 0x0400, 0x0008, 0x0800}, // ip = 0, il = 2 - {0x0010, 0x1000, 0x0020, 0x2000}, // ip = 1, il = 0 - {0x0040, 0x4000, 0x0080, 0x8000}}; // ip = 1, il = 2 - - // Possible masks for the low 2 bits - const int4 qm[2] = {{0x0003, 0x0300, 0x000c, 0x0c00}, {0x0030, 0x3000, 0x00c0, 0xc000}}; - - const ushort4 hm = mm[2*ip + il/2]; - - const short shift = 2*il; - - const float v1 = il == 0 ? 4.f : 64.f; - const float v2 = 4.f * v1; - - const uint16_t s_shift1 = 4*ip; - const uint16_t s_shift2 = s_shift1 + il; - - const short q_offset = 32*ip + l0; - const short y_offset = 128*ip + 32*il + l0; - - device const float * y1 = yy + ix*QK_K + y_offset; - - uint32_t scales32, aux32; - thread uint16_t * scales16 = (thread uint16_t *)&scales32; - thread const int8_t * scales = (thread const int8_t *)&scales32; - - float sumf1[nr0] = {0.f}; - float sumf2[nr0] = {0.f}; - - for (int i = ix; i < nb; i += 4) { - for (short l = 0; l < 8; ++l) { - yl[l+ 0] = y1[l+ 0]; - yl[l+ 8] = y1[l+16]; - yl[l+16] = y1[l+32]; - yl[l+24] = y1[l+48]; - } - - device const uint16_t * q = (device const uint16_t *)(x[i].qs + q_offset); - device const uint16_t * h = (device const uint16_t *)(x[i].hmask + l0); - device const uint16_t * a = (device const uint16_t *)(x[i].scales); - device const half * dh = &x[i].d; - - for (short row = 0; row < nr0; ++row) { - const float d_all = (float)dh[0]; - - scales16[0] = a[4]; - scales16[1] = a[5]; - aux32 = ((scales32 >> s_shift2) << 4) & 0x30303030; - scales16[0] = a[il+0]; - scales16[1] = a[il+1]; - scales32 = ((scales32 >> s_shift1) & 0x0f0f0f0f) | aux32; - - float s1 = 0, s2 = 0, s3 = 0, s4 = 0, s5 = 0, s6 = 0; - for (short l = 0; l < 8; l += 2) { - const int32_t qs = q[l/2]; - s1 += yl[l+0] * (qs & qm[il/2][0]); - s2 += yl[l+1] * (qs & qm[il/2][1]); - s3 += ((h[l/2] & hm[0]) ? 0.f : yl[l+0]) + ((h[l/2] & hm[1]) ? 0.f : yl[l+1]); - s4 += yl[l+16] * (qs & qm[il/2][2]); - s5 += yl[l+17] * (qs & qm[il/2][3]); - s6 += ((h[l/2] & hm[2]) ? 0.f : yl[l+16]) + ((h[l/2] & hm[3]) ? 0.f : yl[l+17]); - } - float d1 = d_all * (s1 + 1.f/256.f * s2 - s3*v1); - float d2 = d_all * (s4 + 1.f/256.f * s5 - s6*v2); - sumf1[row] += d1 * (scales[0] - 32); - sumf2[row] += d2 * (scales[2] - 32); - - s1 = s2 = s3 = s4 = s5 = s6 = 0; - for (short l = 0; l < 8; l += 2) { - const int32_t qs = q[l/2+8]; - s1 += yl[l+8] * (qs & qm[il/2][0]); - s2 += yl[l+9] * (qs & qm[il/2][1]); - s3 += ((h[l/2+8] & hm[0]) ? 0.f : yl[l+8]) + ((h[l/2+8] & hm[1]) ? 0.f : yl[l+9]); - s4 += yl[l+24] * (qs & qm[il/2][2]); - s5 += yl[l+25] * (qs & qm[il/2][3]); - s6 += ((h[l/2+8] & hm[2]) ? 0.f : yl[l+24]) + ((h[l/2+8] & hm[3]) ? 0.f : yl[l+25]); - } - d1 = d_all * (s1 + 1.f/256.f * s2 - s3*v1); - d2 = d_all * (s4 + 1.f/256.f * s5 - s6*v2); - sumf1[row] += d1 * (scales[1] - 32); - sumf2[row] += d2 * (scales[3] - 32); - - q += args.nb01/2; - h += args.nb01/2; - a += args.nb01/2; - dh += args.nb01/2; - } - - y1 += 4 * QK_K; - } - - for (int row = 0; row < nr0; ++row) { - const float sumf = (sumf1[row] + 0.25f * sumf2[row]) / (1 << shift); - sumf1[row] = simd_sum(sumf); - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - if (tiisg == 0) { - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - dst_f32[first_row + row] = sumf1[row]; - } - } -} - -[[host_name("kernel_mul_mv_q3_K_f32")]] -kernel void kernel_mul_mv_q3_K_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_q3_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q4_K_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr uint16_t kmask1 = 0x3f3f; - constexpr uint16_t kmask2 = 0x0f0f; - constexpr uint16_t kmask3 = 0xc0c0; - - const short ix = tiisg/8; // 0...3 - const short it = tiisg%8; // 0...7 - const short iq = it/4; // 0 or 1 - const short ir = it%4; // 0...3 - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_q4_K * x = (device const block_q4_K *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[16]; - float yh[16]; - - float sumf[nr0]={0.f}; - - device const float * y4 = y + ix * QK_K + 64 * iq + 8 * ir; - - uint16_t sc16[4]; - thread const uint8_t * sc8 = (thread const uint8_t *)sc16; - - for (int ib = ix; ib < nb; ib += 4) { - float4 sumy = {0.f, 0.f, 0.f, 0.f}; - - for (short i = 0; i < 8; ++i) { - yl[i+0] = y4[i+ 0]; sumy[0] += yl[i+0]; - yl[i+8] = y4[i+ 32]; sumy[1] += yl[i+8]; - yh[i+0] = y4[i+128]; sumy[2] += yh[i+0]; - yh[i+8] = y4[i+160]; sumy[3] += yh[i+8]; - } - - device const uint16_t * sc = (device const uint16_t *)x[ib].scales + iq; - device const uint16_t * q1 = (device const uint16_t *)x[ib].qs + 16 * iq + 4 * ir; - device const half * dh = &x[ib].d; - - for (short row = 0; row < nr0; row++) { - sc16[0] = sc[0] & kmask1; - sc16[1] = sc[2] & kmask1; - sc16[2] = ((sc[4] >> 0) & kmask2) | ((sc[0] & kmask3) >> 2); - sc16[3] = ((sc[4] >> 4) & kmask2) | ((sc[2] & kmask3) >> 2); - - device const uint16_t * q2 = q1 + 32; - - float4 acc1 = {0.f, 0.f, 0.f, 0.f}; - float4 acc2 = {0.f, 0.f, 0.f, 0.f}; - - FOR_UNROLL (short i = 0; i < 4; ++i) { - acc1[0] += yl[2*i + 0] * (q1[i] & 0x000F); - acc1[1] += yl[2*i + 1] * (q1[i] & 0x0F00); - acc1[2] += yl[2*i + 8] * (q1[i] & 0x00F0); - acc1[3] += yl[2*i + 9] * (q1[i] & 0xF000); - acc2[0] += yh[2*i + 0] * (q2[i] & 0x000F); - acc2[1] += yh[2*i + 1] * (q2[i] & 0x0F00); - acc2[2] += yh[2*i + 8] * (q2[i] & 0x00F0); - acc2[3] += yh[2*i + 9] * (q2[i] & 0xF000); - } - - sumf[row] += dh[0] * ((acc1[0] + 1.f/256.f * acc1[1]) * sc8[0] + - (acc1[2] + 1.f/256.f * acc1[3]) * sc8[1] * 1.f/16.f + - (acc2[0] + 1.f/256.f * acc2[1]) * sc8[4] + - (acc2[2] + 1.f/256.f * acc2[3]) * sc8[5] * 1.f/16.f) - - dh[1] * (sumy[0] * sc8[2] + sumy[1] * sc8[3] + sumy[2] * sc8[6] + sumy[3] * sc8[7]); - - q1 += args.nb01/2; - sc += args.nb01/2; - dh += args.nb01/2; - } - - y4 += 4 * QK_K; - } - - device float * dst_f32 = (device float *) dst + (int64_t)im*args.ne0*args.ne1 + (int64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_q4_K_f32")]] -kernel void kernel_mul_mv_q4_K_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_q4_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q5_K_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_q5_K * x = (device const block_q5_K *) (src0 + offset0); - device const float * yy = (device const float *) (src1 + offset1); - - float sumf[nr0]={0.f}; - - float yl[16], yh[16]; - - constexpr uint16_t kmask1 = 0x3f3f; - constexpr uint16_t kmask2 = 0x0f0f; - constexpr uint16_t kmask3 = 0xc0c0; - - const short tid = tiisg/4; - const short ix = tiisg%4; - const short iq = tid/4; - const short ir = tid%4; - - const short l0 = 8*ir; - const short q_offset = 32*iq + l0; - const short y_offset = 64*iq + l0; - - const uint8_t hm1 = 1u << (2*iq); - const uint8_t hm2 = hm1 << 1; - const uint8_t hm3 = hm1 << 4; - const uint8_t hm4 = hm2 << 4; - - uint16_t sc16[4]; - thread const uint8_t * sc8 = (thread const uint8_t *)sc16; - - device const float * y1 = yy + ix*QK_K + y_offset; - - for (int i = ix; i < nb; i += 4) { - device const uint8_t * q1 = x[i].qs + q_offset; - device const uint8_t * qh = x[i].qh + l0; - device const half * dh = &x[i].d; - device const uint16_t * a = (device const uint16_t *)x[i].scales + iq; - - device const float * y2 = y1 + 128; - float4 sumy = {0.f, 0.f, 0.f, 0.f}; - for (short l = 0; l < 8; ++l) { - yl[l+0] = y1[l+ 0]; sumy[0] += yl[l+0]; - yl[l+8] = y1[l+32]; sumy[1] += yl[l+8]; - yh[l+0] = y2[l+ 0]; sumy[2] += yh[l+0]; - yh[l+8] = y2[l+32]; sumy[3] += yh[l+8]; - } - - for (short row = 0; row < nr0; ++row) { - device const uint8_t * q2 = q1 + 64; - - sc16[0] = a[0] & kmask1; - sc16[1] = a[2] & kmask1; - sc16[2] = ((a[4] >> 0) & kmask2) | ((a[0] & kmask3) >> 2); - sc16[3] = ((a[4] >> 4) & kmask2) | ((a[2] & kmask3) >> 2); - - float4 acc1 = {0.f}; - float4 acc2 = {0.f}; - FOR_UNROLL (short l = 0; l < 8; ++l) { - uint8_t h = qh[l]; - acc1[0] += yl[l+0] * (q1[l] & 0x0F); - acc1[1] += yl[l+8] * (q1[l] & 0xF0); - acc1[2] += yh[l+0] * (q2[l] & 0x0F); - acc1[3] += yh[l+8] * (q2[l] & 0xF0); - acc2[0] += h & hm1 ? yl[l+0] : 0.f; - acc2[1] += h & hm2 ? yl[l+8] : 0.f; - acc2[2] += h & hm3 ? yh[l+0] : 0.f; - acc2[3] += h & hm4 ? yh[l+8] : 0.f; - } - - sumf[row] += dh[0] * (sc8[0] * (acc1[0] + 16.f*acc2[0]) + - sc8[1] * (acc1[1]/16.f + 16.f*acc2[1]) + - sc8[4] * (acc1[2] + 16.f*acc2[2]) + - sc8[5] * (acc1[3]/16.f + 16.f*acc2[3])) - - dh[1] * (sumy[0] * sc8[2] + sumy[1] * sc8[3] + sumy[2] * sc8[6] + sumy[3] * sc8[7]); - - q1 += args.nb01; - qh += args.nb01; - dh += args.nb01/2; - a += args.nb01/2; - } - - y1 += 4 * QK_K; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - const float tot = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = tot; - } - } -} - -[[host_name("kernel_mul_mv_q5_K_f32")]] -kernel void kernel_mul_mv_q5_K_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_q5_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_q6_K_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - constexpr uint8_t kmask1 = 0x03; - constexpr uint8_t kmask2 = 0x0C; - constexpr uint8_t kmask3 = 0x30; - constexpr uint8_t kmask4 = 0xC0; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_q6_K * x = (device const block_q6_K *) (src0 + offset0); - device const float * yy = (device const float *) (src1 + offset1); - - float sumf[nr0] = { 0.f }; - - float yl[16]; - - const short tid = tiisg/2; - const short ix = tiisg%2; - const short ip = tid/8; // 0 or 1 - const short il = tid%8; - const short l0 = 4*il; - const short is = 8*ip + l0/16; - - const short y_offset = 128*ip + l0; - const short q_offset_l = 64*ip + l0; - const short q_offset_h = 32*ip + l0; - - for (int i = ix; i < nb; i += 2) { - device const uint8_t * q1 = x[i].ql + q_offset_l; - device const uint8_t * q2 = q1 + 32; - device const uint8_t * qh = x[i].qh + q_offset_h; - device const int8_t * sc = x[i].scales + is; - device const half * dh = &x[i].d; - - device const float * y = yy + i * QK_K + y_offset; - - for (short l = 0; l < 4; ++l) { - yl[4*l + 0] = y[l + 0]; - yl[4*l + 1] = y[l + 32]; - yl[4*l + 2] = y[l + 64]; - yl[4*l + 3] = y[l + 96]; - } - - for (short row = 0; row < nr0; ++row) { - float4 sums = {0.f, 0.f, 0.f, 0.f}; - - FOR_UNROLL (short l = 0; l < 4; ++l) { - sums[0] += yl[4*l + 0] * ((int8_t)((q1[l] & 0xF) | ((qh[l] & kmask1) << 4)) - 32); - sums[1] += yl[4*l + 1] * ((int8_t)((q2[l] & 0xF) | ((qh[l] & kmask2) << 2)) - 32); - sums[2] += yl[4*l + 2] * ((int8_t)((q1[l] >> 4) | ((qh[l] & kmask3) << 0)) - 32); - sums[3] += yl[4*l + 3] * ((int8_t)((q2[l] >> 4) | ((qh[l] & kmask4) >> 2)) - 32); - } - - sumf[row] += dh[0] * (sums[0] * sc[0] + sums[1] * sc[2] + sums[2] * sc[4] + sums[3] * sc[6]); - - q1 += args.nb01; - q2 += args.nb01; - qh += args.nb01; - sc += args.nb01; - dh += args.nb01/2; - } - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_q6_K_f32")]] -kernel void kernel_mul_mv_q6_K_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_q6_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -// ======================= "True" 2-bit - -template -void kernel_mul_mv_iq2_xxs_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq2_xxs * x = (device const block_iq2_xxs *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - threadgroup uint64_t * svalues = (threadgroup uint64_t *)(shmem); - threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 256); - { - int nval = 4; - int pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2xxs_grid[pos + i]; - nval = 2; - pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - const int ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq2_xxs * xr = x + ibl; - device const uint16_t * q2 = xr->qs + 4 * ib; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - const float db = dh[0]; - device const uint8_t * aux8 = (device const uint8_t *)q2; - const uint32_t aux32 = q2[2] | (q2[3] << 16); - const float d = db * (0.5f + (aux32 >> 28)); - - float sum = 0; - for (short l = 0; l < 4; ++l) { - const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + aux8[l]); - const uint8_t signs = ssigns[(aux32 >> 7*l) & 127]; - for (short j = 0; j < 8; ++j) { - sum += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); - } - } - sumf[row] += d * sum; - - dh += args.nb01/2; - q2 += args.nb01/2; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all * 0.25f; - } - } -} - -[[host_name("kernel_mul_mv_iq2_xxs_f32")]] -kernel void kernel_mul_mv_iq2_xxs_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - kernel_mul_mv_iq2_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq2_xs_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq2_xs * x = (device const block_iq2_xs *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - threadgroup uint64_t * svalues = (threadgroup uint64_t *)(shmem); - threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 512); - { - int nval = 8; - int pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2xs_grid[pos + i]; - nval = 2; - pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - const int ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq2_xs * xr = x + ibl; - device const uint16_t * q2 = xr->qs + 4 * ib; - device const uint8_t * sc = xr->scales + ib; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - const float db = dh[0]; - const uint8_t ls1 = sc[0] & 0xf; - const uint8_t ls2 = sc[0] >> 4; - const float d1 = db * (0.5f + ls1); - const float d2 = db * (0.5f + ls2); - - float sum1 = 0, sum2 = 0; - for (short l = 0; l < 2; ++l) { - const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + (q2[l] & 511)); - const uint8_t signs = ssigns[(q2[l] >> 9)]; - for (short j = 0; j < 8; ++j) { - sum1 += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); - } - } - for (short l = 2; l < 4; ++l) { - const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + (q2[l] & 511)); - const uint8_t signs = ssigns[(q2[l] >> 9)]; - for (short j = 0; j < 8; ++j) { - sum2 += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); - } - } - sumf[row] += d1 * sum1 + d2 * sum2; - - dh += args.nb01/2; - q2 += args.nb01/2; - sc += args.nb01; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all * 0.25f; - } - } -} - -[[host_name("kernel_mul_mv_iq2_xs_f32")]] -kernel void kernel_mul_mv_iq2_xs_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq2_xs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq3_xxs_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq3_xxs * x = (device const block_iq3_xxs *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - threadgroup uint32_t * svalues = (threadgroup uint32_t *)(shmem); - threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 256); - { - int nval = 4; - int pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) svalues[pos + i] = iq3xxs_grid[pos + i]; - nval = 2; - pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - const int ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq3_xxs * xr = x + ibl; - device const uint8_t * q3 = xr->qs + 8 * ib; - device const uint16_t * gas = (device const uint16_t *)(xr->qs + QK_K/4) + 2 * ib; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - const float db = dh[0]; - const uint32_t aux32 = gas[0] | (gas[1] << 16); - const float d = db * (0.5f + (aux32 >> 28)); - - float2 sum = {0}; - for (short l = 0; l < 4; ++l) { - const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(svalues + q3[2*l+0]); - const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(svalues + q3[2*l+1]); - const uint8_t signs = ssigns[(aux32 >> 7*l) & 127]; - for (short j = 0; j < 4; ++j) { - sum[0] += yl[8*l + j + 0] * grid1[j] * (signs & kmask_iq2xs[j+0] ? -1.f : 1.f); - sum[1] += yl[8*l + j + 4] * grid2[j] * (signs & kmask_iq2xs[j+4] ? -1.f : 1.f); - } - } - sumf[row] += d * (sum[0] + sum[1]); - - dh += args.nb01/2; - q3 += args.nb01; - gas += args.nb01/2; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all * 0.5f; - } - } -} - -[[host_name("kernel_mul_mv_iq3_xxs_f32")]] -kernel void kernel_mul_mv_iq3_xxs_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq3_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq3_s_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq3_s * x = (device const block_iq3_s *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - threadgroup uint32_t * svalues = (threadgroup uint32_t *) shmem; - { - int nval = 8; - int pos = (32*sgitg + tiisg)*nval; - for (int i = 0; i < nval; ++i) svalues[pos + i] = iq3s_grid[pos + i]; - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - const int ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq3_s * xr = x + ibl; - device const uint8_t * qs = xr->qs + 8 * ib; - device const uint8_t * qh = xr->qh + ib; - device const uint8_t * sc = xr->scales + (ib/2); - device const uint8_t * signs = xr->signs + 4 * ib; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - const float db = dh[0]; - const float d = db * (1 + 2*((sc[0] >> 4*(ib%2)) & 0xf)); - - float2 sum = {0}; - for (short l = 0; l < 4; ++l) { - const threadgroup uint32_t * table1 = qh[0] & kmask_iq2xs[2*l+0] ? svalues + 256 : svalues; - const threadgroup uint32_t * table2 = qh[0] & kmask_iq2xs[2*l+1] ? svalues + 256 : svalues; - const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(table1 + qs[2*l+0]); - const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(table2 + qs[2*l+1]); - for (short j = 0; j < 4; ++j) { - sum[0] += yl[8*l + j + 0] * grid1[j] * select(1, -1, signs[l] & kmask_iq2xs[j+0]); - sum[1] += yl[8*l + j + 4] * grid2[j] * select(1, -1, signs[l] & kmask_iq2xs[j+4]); - } - } - sumf[row] += d * (sum[0] + sum[1]); - - dh += args.nb01/2; - qs += args.nb01; - qh += args.nb01; - sc += args.nb01; - signs += args.nb01; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_iq3_s_f32")]] -kernel void kernel_mul_mv_iq3_s_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq3_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq2_s_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq2_s * x = (device const block_iq2_s *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - //threadgroup uint64_t * svalues = (threadgroup uint64_t *) shmem; - //{ - // int nval = 32; - // int pos = (32*sgitg + tiisg)*nval; - // for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2s_grid[pos + i]; - // threadgroup_barrier(mem_flags::mem_threadgroup); - //} - - const short ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq2_s * xr = x + ibl; - device const uint8_t * qs = xr->qs + 4 * ib; - device const uint8_t * qh = xr->qh + ib; - device const uint8_t * sc = xr->scales + ib; - device const uint8_t * signs = qs + QK_K/8; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - const float db = dh[0]; - const float d1 = db * (0.5f + (sc[0] & 0xf)); - const float d2 = db * (0.5f + (sc[0] >> 4)); - - float2 sum = {0}; - for (short l = 0; l < 2; ++l) { - //const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(svalues + (qs[l+0] | ((qh[0] << (8-2*l)) & 0x300))); - //const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(svalues + (qs[l+2] | ((qh[0] << (4-2*l)) & 0x300))); - constant uint8_t * grid1 = (constant uint8_t *)(iq2s_grid + (qs[l+0] | ((qh[0] << (8-2*l)) & 0x300))); - constant uint8_t * grid2 = (constant uint8_t *)(iq2s_grid + (qs[l+2] | ((qh[0] << (4-2*l)) & 0x300))); - for (short j = 0; j < 8; ++j) { - sum[0] += yl[8*l + j + 0] * grid1[j] * select(1, -1, signs[l+0] & kmask_iq2xs[j]); - sum[1] += yl[8*l + j + 16] * grid2[j] * select(1, -1, signs[l+2] & kmask_iq2xs[j]); - } - } - sumf[row] += d1 * sum[0] + d2 * sum[1]; - - dh += args.nb01/2; - qs += args.nb01; - qh += args.nb01; - sc += args.nb01; - signs += args.nb01; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all * 0.25f; - } - } -} - -[[host_name("kernel_mul_mv_iq2_s_f32")]] -kernel void kernel_mul_mv_iq2_s_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq2_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq1_s_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq1_s * x = (device const block_iq1_s *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - const short ix = tiisg; - - device const float * y4 = y + 32 * ix; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - float sumy = 0; - for (short i = 0; i < 32; ++i) { - yl[i] = y4[i]; - sumy += yl[i]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq1_s * xr = x + ibl; - device const uint8_t * qs = xr->qs + 4 * ib; - device const uint16_t * qh = xr->qh + ib; - device const half * dh = &xr->d; - - for (short row = 0; row < nr0; row++) { - constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); - constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 5) & 0x700))); - constant uint8_t * grid3 = (constant uint8_t *)(iq1s_grid_gpu + (qs[2] | ((qh[0] << 2) & 0x700))); - constant uint8_t * grid4 = (constant uint8_t *)(iq1s_grid_gpu + (qs[3] | ((qh[0] >> 1) & 0x700))); - - float sum = 0; - for (short j = 0; j < 4; ++j) { - sum += yl[j+ 0] * (grid1[j] & 0xf) + yl[j+ 4] * (grid1[j] >> 4) - + yl[j+ 8] * (grid2[j] & 0xf) + yl[j+12] * (grid2[j] >> 4) - + yl[j+16] * (grid3[j] & 0xf) + yl[j+20] * (grid3[j] >> 4) - + yl[j+24] * (grid4[j] & 0xf) + yl[j+28] * (grid4[j] >> 4); - } - sumf[row] += (float)dh[0] * (sum + sumy * (qh[0] & 0x8000 ? -1 - IQ1S_DELTA : -1 + IQ1S_DELTA)) * (2*((qh[0] >> 12) & 7) + 1); - - dh += args.nb01/2; - qs += args.nb01; - qh += args.nb01/2; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_iq1_s_f32")]] -kernel void kernel_mul_mv_iq1_s_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq1_s_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq1_m_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq1_m * x = (device const block_iq1_m *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - float yl[32]; - float sumf[nr0]={0.f}; - - const int nb32 = nb * (QK_K / 32); - - const short ix = tiisg; - - device const float * y4 = y + 32 * ix; - - iq1m_scale_t scale; - - for (int ib32 = ix; ib32 < nb32; ib32 += 32) { - float4 sumy = {0.f}; - for (short i = 0; i < 8; ++i) { - yl[i+ 0] = y4[i+ 0]; sumy[0] += yl[i+ 0]; - yl[i+ 8] = y4[i+ 8]; sumy[1] += yl[i+ 8]; - yl[i+16] = y4[i+16]; sumy[2] += yl[i+16]; - yl[i+24] = y4[i+24]; sumy[3] += yl[i+24]; - } - - const int ibl = ib32 / (QK_K / 32); - const int ib = ib32 % (QK_K / 32); - - device const block_iq1_m * xr = x + ibl; - device const uint8_t * qs = xr->qs + 4 * ib; - device const uint8_t * qh = xr->qh + 2 * ib; - device const uint16_t * sc = (device const uint16_t *)xr->scales; - - for (short row = 0; row < nr0; row++) { - scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000); - - constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); - constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 4) & 0x700))); - constant uint8_t * grid3 = (constant uint8_t *)(iq1s_grid_gpu + (qs[2] | ((qh[1] << 8) & 0x700))); - constant uint8_t * grid4 = (constant uint8_t *)(iq1s_grid_gpu + (qs[3] | ((qh[1] << 4) & 0x700))); - - float2 sum = {0.f}; - for (short j = 0; j < 4; ++j) { - sum[0] += yl[j+ 0] * (grid1[j] & 0xf) + yl[j+ 4] * (grid1[j] >> 4) - + yl[j+ 8] * (grid2[j] & 0xf) + yl[j+12] * (grid2[j] >> 4); - sum[1] += yl[j+16] * (grid3[j] & 0xf) + yl[j+20] * (grid3[j] >> 4) - + yl[j+24] * (grid4[j] & 0xf) + yl[j+28] * (grid4[j] >> 4); - } - const float delta1 = sumy[0] * (qh[0] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA) + sumy[1] * (qh[0] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); - const float delta2 = sumy[2] * (qh[1] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA) + sumy[3] * (qh[1] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); - - sumf[row] += (float)scale.f16 * ((sum[0] + delta1) * (2*((sc[ib/2] >> (6*(ib%2)+0)) & 7) + 1) + - (sum[1] + delta2) * (2*((sc[ib/2] >> (6*(ib%2)+3)) & 7) + 1)); - - sc += args.nb01/2; - qs += args.nb01; - qh += args.nb01; - } - - y4 += 32 * 32; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_iq1_m_f32")]] -kernel void kernel_mul_mv_iq1_m_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq1_m_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq4_nl_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - threadgroup float * shmem_f32 = (threadgroup float *) shmem; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * NR0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq4_nl * x = (device const block_iq4_nl *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - const int nb = args.ne00/QK4_NL; - const int ns01 = args.nb01/args.nb00; - - const short ix = tiisg/2; // 0...15 - const short it = tiisg%2; // 0 or 1 - - shmem_f32[tiisg] = kvalues_iq4nl_f[tiisg%16]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - float4 yl[4]; - float sumf[NR0]={0.f}; - - device const float * yb = y + ix*QK4_NL + it*8; - - uint32_t aux32[2]; - thread const uint8_t * q8 = (thread const uint8_t *)aux32; - - float4 qf1, qf2; - - // [TAG_MUL_MV_WEIRD] - for (int ib = ix; ib < nb && ib < ns01; ib += 16) { - device const float4 * y4 = (device const float4 *)yb; - yl[0] = y4[0]; - yl[1] = y4[4]; - yl[2] = y4[1]; - yl[3] = y4[5]; - - for (short row = 0; row < NR0; row++) { - device const block_iq4_nl & xb = x[row*ns01 + ib]; - device const uint16_t * q4 = (device const uint16_t *)(xb.qs + 8*it); - - float4 acc1 = {0.f}, acc2 = {0.f}; - - aux32[0] = q4[0] | (q4[1] << 16); - aux32[1] = (aux32[0] >> 4) & 0x0f0f0f0f; - aux32[0] &= 0x0f0f0f0f; - qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; - qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; - acc1 += yl[0] * qf1; - acc2 += yl[1] * qf2; - - aux32[0] = q4[2] | (q4[3] << 16); - aux32[1] = (aux32[0] >> 4) & 0x0f0f0f0f; - aux32[0] &= 0x0f0f0f0f; - qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; - qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; - acc1 += yl[2] * qf1; - acc2 += yl[3] * qf2; - - acc1 += acc2; - - sumf[row] += (float)xb.d * (acc1[0] + acc1[1] + acc1[2] + acc1[3]); - } - - yb += 16 * QK4_NL; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_iq4_nl_f32")]] -kernel void kernel_mul_mv_iq4_nl_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq4_nl_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_iq4_xs_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - threadgroup float * shmem_f32 = (threadgroup float *) shmem; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - const int first_row = (r0 * NSG + sgitg) * NR0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_iq4_xs * x = (device const block_iq4_xs *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - const int nb = args.ne00/QK_K; - const int ns01 = args.nb01/args.nb00; - - const short ix = tiisg/16; // 0 or 1 - const short it = tiisg%16; // 0...15 - const short ib = it/2; - const short il = it%2; - - shmem_f32[tiisg] = kvalues_iq4nl_f[tiisg%16]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - float4 yl[4]; - float sumf[NR0]={0.f}; - - device const float * yb = y + ix * QK_K + ib * 32 + il * 8; - - uint32_t aux32[2]; - thread const uint8_t * q8 = (thread const uint8_t *)aux32; - - float4 qf1, qf2; - - // [TAG_MUL_MV_WEIRD] - for (int ibl = ix; ibl < nb && ibl < ns01; ibl += 2) { - device const float4 * y4 = (device const float4 *)yb; - yl[0] = y4[0]; - yl[1] = y4[4]; - yl[2] = y4[1]; - yl[3] = y4[5]; - - for (short row = 0; row < NR0; ++row) { - device const block_iq4_xs & xb = x[row*ns01 + ibl]; - device const uint32_t * q4 = (device const uint32_t *)(xb.qs + 16*ib + 8*il); - - float4 acc1 = {0.f}, acc2 = {0.f}; - - aux32[0] = (q4[0] ) & 0x0f0f0f0f; - aux32[1] = (q4[0] >> 4) & 0x0f0f0f0f; - qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; - qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; - acc1 += yl[0] * qf1; - acc2 += yl[1] * qf2; - - aux32[0] = (q4[1] ) & 0x0f0f0f0f; - aux32[1] = (q4[1] >> 4) & 0x0f0f0f0f; - qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; - qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; - acc1 += yl[2] * qf1; - acc2 += yl[3] * qf2; - - acc1 += acc2; - - const int ls = (((xb.scales_l[ib/2] >> 4*(ib%2)) & 0xf) | (((xb.scales_h >> 2*ib) & 3) << 4)) - 32; - sumf[row] += (float)xb.d * ls * (acc1[0] + acc1[1] + acc1[2] + acc1[3]); - } - - yb += 2 * QK_K; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_iq4_xs_f32")]] -kernel void kernel_mul_mv_iq4_xs_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_iq4_xs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_mxfp4_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - threadgroup float * shmem_f32 = (threadgroup float *) shmem; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * NR0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const block_mxfp4 * x = (device const block_mxfp4 *) (src0 + offset0); - device const float * y = (device const float *) (src1 + offset1); - - const int nb = args.ne00/QK_MXFP4; - const int ns01 = args.nb01/args.nb00; // this can be larger than nb for permuted src0 tensors - - const short ix = tiisg/2; // 0...15 - const short it = tiisg%2; // 0 or 1 - - shmem_f32[tiisg] = kvalues_mxfp4_f[tiisg%16]; - threadgroup_barrier(mem_flags::mem_threadgroup); - - float4 yl[4]; - float sumf[NR0]={0.f}; - - device const float * yb = y + ix*QK_MXFP4 + it*8; - - // note: just the check `ib < nb` is enough, but adding the redundant `&& ib < ns01` check makes the kernel a bit faster - // no idea why that is - needs some deeper investigation [TAG_MUL_MV_WEIRD] - for (int ib = ix; ib < nb && ib < ns01; ib += 16) { - device const float4 * y4 = (device const float4 *) yb; - - yl[0] = y4[0]; - yl[1] = y4[4]; - yl[2] = y4[1]; - yl[3] = y4[5]; - - FOR_UNROLL (short row = 0; row < NR0; row++) { - device const block_mxfp4 & xb = x[row*ns01 + ib]; - device const uint8_t * q2 = (device const uint8_t *)(xb.qs + 8*it); - - float4 acc1 = yl[0]*float4(shmem_f32[q2[0] & 0x0F], shmem_f32[q2[1] & 0x0F], shmem_f32[q2[2] & 0x0F], shmem_f32[q2[3] & 0x0F]); - float4 acc2 = yl[1]*float4(shmem_f32[q2[0] >> 4 ], shmem_f32[q2[1] >> 4 ], shmem_f32[q2[2] >> 4 ], shmem_f32[q2[3] >> 4 ]); - float4 acc3 = yl[2]*float4(shmem_f32[q2[4] & 0x0F], shmem_f32[q2[5] & 0x0F], shmem_f32[q2[6] & 0x0F], shmem_f32[q2[7] & 0x0F]); - float4 acc4 = yl[3]*float4(shmem_f32[q2[4] >> 4 ], shmem_f32[q2[5] >> 4 ], shmem_f32[q2[6] >> 4 ], shmem_f32[q2[7] >> 4 ]); - - acc1 = (acc1 + acc3) + (acc2 + acc4); - - sumf[row] += e8m0_to_fp32(xb.e) * ((acc1[0] + acc1[1]) + (acc1[2] + acc1[3])); - } - - yb += 16 * QK_MXFP4; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { - float sum_all = simd_sum(sumf[row]); - if (tiisg == 0) { - dst_f32[first_row + row] = sum_all; - } - } -} - -[[host_name("kernel_mul_mv_mxfp4_f32")]] -kernel void kernel_mul_mv_mxfp4_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_mxfp4_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -template -void kernel_mul_mv_tq2_0_f32_impl( - args_t args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg) { - const short NSG = FC_mul_mv_nsg; - - const int nb = args.ne00/QK_K; - - const int r0 = tgpig.x; - const int r1 = tgpig.y; - const int im = tgpig.z; - - const int first_row = (r0 * NSG + sgitg) * nr0; - - const uint i12 = im%FC_mul_mv_ne12; - const uint i13 = im/FC_mul_mv_ne12; - - const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; - - device const float * y = (device const float *) (src1 + offset1); - - device const block_tq2_0 * ax[nr0]; - for (int row = 0; row < nr0; ++row) { - const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; - ax[row] = (device const block_tq2_0 *) ((device char *) src0 + offset0); - } - - float sumf[nr0] = {0.f}; - - // 8 threads per block, NBLOCK blocks per pass, 2 halves per block per pass - constexpr short NBLOCK = 4; - - constexpr short NB = N_SIMDWIDTH/NBLOCK; // threads per block - - const short blk = tiisg / NB; // 0..NBLOCK-1, block handled by this thread - const short htg = tiisg % NB; // 0..NB-1, thread within block (0..7) - - // byte and y base offsets within the block (32 elements per thread, 4 per byte) - device const float4 * yb4 = (device const float4 *)(y + 4*htg + blk*QK_K); - - // hoisted per-byte coefficients (from y) and total y-sum, shared across rows - // ref: https://github.com/ggml-org/llama.cpp/pull/26980 - float4 coef[4]; - - for (int ib = blk; ib < nb; ib += NBLOCK) { - FOR_UNROLL (short h0 = 0; h0 < 2; ++h0) { - const float4 y0 = yb4[ 0 + 32*h0]; - const float4 y1 = yb4[ 8 + 32*h0]; - const float4 y2 = yb4[16 + 32*h0]; - const float4 y3 = yb4[24 + 32*h0]; - - float sumy = 0.f; - FOR_UNROLL (short j = 0; j < 4; ++j) { - coef[j] = float4( - y0[j], - y1[j] - 4.0f*y0[j], - y2[j] - 4.0f*y1[j], - y3[j] - 4.0f*y2[j]); - - sumy += (y0[j] + y1[j]) + (y2[j] + y3[j]); - } - - FOR_UNROLL (short row = 0; row < nr0; ++row) { - device const block_tq2_0 & xb = ax[row][ib]; - device const uchar * qs = xb.qs + 4*htg + 32*h0; - - float sum = -sumy; - FOR_UNROLL (short j = 0; j < 4; ++j) { - // express the 2-bit field shifts (v>>2, v>>4, v>>6) as float floor ops - const float v = (float)qs[j]; - - const float f0 = v; - const float f1 = floor(v*0.25f); // v>>2 - const float f2 = floor(v*0.0625); // v>>4 - const float f3 = floor(v*0.015625); // v>>6 - - sum += coef[j][0]*f0 + coef[j][1]*f1 + coef[j][2]*f2 + coef[j][3]*f3; - } - - sumf[row] += xb.d * sum; - } - } - - yb4 += QK_K * NBLOCK / 4; - } - - device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; - - for (int row = 0; row < nr0; ++row) { - const float tot = simd_sum(sumf[row]); - if (tiisg == 0 && first_row + row < args.ne01) { - dst_f32[first_row + row] = tot; - } - } -} - -[[host_name("kernel_mul_mv_tq2_0_f32")]] -kernel void kernel_mul_mv_tq2_0_f32( - constant ggml_metal_kargs_mul_mv & args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - kernel_mul_mv_tq2_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); -} - -template -kernel void kernel_get_rows_q( - constant ggml_metal_kargs_get_rows & args, - device const void * src0, - device const void * src1, - device void * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 ntg [[threads_per_threadgroup]]) { - const int32_t iw0 = tgpig.x/args.ne10; - const int32_t i10 = tgpig.x%args.ne10; - const int32_t i11 = tgpig.y; - const int32_t i12 = tgpig.z; - - const int32_t r = ((const device int32_t *) ((const device char *) src1 + i12*args.nb12 + i11*args.nb11 + i10*args.nb10))[0]; - - const int32_t i02 = i11; - const int32_t i03 = i12; - - auto psrc = (device const block_q *) ((const device char *) src0 + i03*args.nb03 + i02*args.nb02 + r*args.nb01); - auto pdst = (device float4x4 *) (( device char *) dst + i12*args.nb3 + i11*args.nb2 + i10*args.nb1); - - for (int ind = iw0*ntg.x + tiitg; ind < args.ne00t;) { - float4x4 temp; - dequantize_func(psrc + ind/nl, ind%nl, temp); - pdst[ind] = temp; - - break; - } -} - -template -kernel void kernel_get_rows_f( - constant ggml_metal_kargs_get_rows & args, - device const void * src0, - device const void * src1, - device void * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort3 ntg [[threads_per_threadgroup]]) { - const int32_t iw0 = tgpig.x/args.ne10; - const int32_t i10 = tgpig.x%args.ne10; - const int32_t i11 = tgpig.y; - const int32_t i12 = tgpig.z; - - const int32_t r = ((const device int32_t *) ((const device char *) src1 + i12*args.nb12 + i11*args.nb11 + i10*args.nb10))[0]; - - const int32_t i02 = i11; - const int32_t i03 = i12; - - auto psrc = (const device T0 *) ((const device char *) src0 + i03*args.nb03 + i02*args.nb02 + r*args.nb01); - auto pdst = ( device T *) (( device char *) dst + i12*args.nb3 + i11*args.nb2 + i10*args.nb1); - - for (int ind = iw0*ntg.x + tiitg; ind < args.ne00t;) { - pdst[ind] = psrc[ind]; - - break; - } -} - -typedef decltype(kernel_get_rows_f) get_rows_f_t; - -template [[host_name("kernel_get_rows_f32")]] kernel get_rows_f_t kernel_get_rows_f; -template [[host_name("kernel_get_rows_f16")]] kernel get_rows_f_t kernel_get_rows_f; -template [[host_name("kernel_get_rows_i32")]] kernel get_rows_f_t kernel_get_rows_f; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_get_rows_bf16")]] kernel get_rows_f_t kernel_get_rows_f; -#endif - -typedef decltype(kernel_get_rows_q) get_rows_q_t; - -template [[host_name("kernel_get_rows_q1_0")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q2_0")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q4_0")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q4_1")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q5_0")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q5_1")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q8_0")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_mxfp4")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q2_K")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q3_K")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q4_K")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q5_K")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_q6_K")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq2_xxs")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq2_xs")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq3_xxs")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq3_s")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq2_s")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq1_s")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq1_m")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq4_nl")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_iq4_xs")]] kernel get_rows_q_t kernel_get_rows_q; -template [[host_name("kernel_get_rows_tq2_0")]] kernel get_rows_q_t kernel_get_rows_q; - -template -kernel void kernel_set_rows_q( - constant ggml_metal_kargs_set_rows & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint tiitg[[thread_index_in_threadgroup]], - uint3 tptg [[threads_per_threadgroup]]) { - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - - const int32_t i12 = i03%args.ne12; - const int32_t i11 = i02%args.ne11; - - const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; - if (i01 >= args.ne01) { - return; - } - - const int32_t i10 = i01; - const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; - - device block_q * dst_row = ( device block_q *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); - const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); - - for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { - quantize_func(src_row + QK*ind, dst_row[ind]); - } -} - -template -kernel void kernel_set_rows_q32( - constant ggml_metal_kargs_set_rows & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint tiitg[[thread_index_in_threadgroup]], - uint3 tptg [[threads_per_threadgroup]]) { - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - - const int32_t i12 = i03%args.ne12; - const int32_t i11 = i02%args.ne11; - - const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; - if (i01 >= args.ne01) { - return; - } - - const int32_t i10 = i01; - const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; - - device block_q * dst_row = ( device block_q *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); - const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); - - for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { - quantize_func(src_row + 32*ind, dst_row[ind]); - } -} - -template -kernel void kernel_set_rows_f( - constant ggml_metal_kargs_set_rows & args, - device const void * src0, - device const void * src1, - device float * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - uint tiitg[[thread_index_in_threadgroup]], - uint3 tptg [[threads_per_threadgroup]]) { - const int32_t i03 = tgpig.z; - const int32_t i02 = tgpig.y; - - const int32_t i12 = i03%args.ne12; - const int32_t i11 = i02%args.ne11; - - const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; - if (i01 >= args.ne01) { - return; - } - - const int32_t i10 = i01; - const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; - - device TD * dst_row = ( device TD *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); - const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); - - for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { - dst_row[ind] = (TD) src_row[ind]; - } -} - -typedef decltype(kernel_set_rows_f) set_rows_f_t; - -template [[host_name("kernel_set_rows_f32_i64_f32")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_f32_i32_f32")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_f32_i64_f16")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_f32_i32_f16")]] kernel set_rows_f_t kernel_set_rows_f; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_set_rows_f32_i64_bf16")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_f32_i32_bf16")]] kernel set_rows_f_t kernel_set_rows_f; -#endif - -template [[host_name("kernel_set_rows_f16_i64_f16")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_f16_i32_f16")]] kernel set_rows_f_t kernel_set_rows_f; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_set_rows_bf16_i64_bf16")]] kernel set_rows_f_t kernel_set_rows_f; -template [[host_name("kernel_set_rows_bf16_i32_bf16")]] kernel set_rows_f_t kernel_set_rows_f; -#endif - -typedef decltype(kernel_set_rows_q32) set_rows_q32_t; - -template [[host_name("kernel_set_rows_f32_i64_q8_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_q8_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i64_q4_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_q4_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i64_q4_1")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_q4_1")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i64_q5_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_q5_0")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i64_q5_1")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_q5_1")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i64_iq4_nl")]] kernel set_rows_q32_t kernel_set_rows_q32; -template [[host_name("kernel_set_rows_f32_i32_iq4_nl")]] kernel set_rows_q32_t kernel_set_rows_q32; - -typedef decltype(kernel_set_rows_q) set_rows_qK_t; - -template [[host_name("kernel_set_rows_f32_i64_tq2_0")]] kernel set_rows_qK_t kernel_set_rows_q; -template [[host_name("kernel_set_rows_f32_i32_tq2_0")]] kernel set_rows_qK_t kernel_set_rows_q; - -kernel void kernel_diag_f32( - constant ggml_metal_kargs_diag & args, - device const char * src0, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]]) { - constexpr short NW = N_SIMDWIDTH; - - const int32_t i3 = tgpig.z; - const int32_t i2 = tgpig.y; - const int32_t i1 = tgpig.x; - - device const float * src0_ptr = (device const float *)(src0 + i2*args.nb02 + i3*args.nb03); - device float * dst_ptr = (device float *)(dst + i1*args.nb01 + i2*args.nb2 + i3*args.nb3); - - for (int i0 = tiitg; i0 < args.ne0; i0 += NW) { - dst_ptr[i0] = i0 == i1 ? src0_ptr[i0] : 0.0f; - } -} - -constant bool FC_mul_mm_bc_inp [[function_constant(FC_MUL_MM + 0)]]; -constant bool FC_mul_mm_bc_out [[function_constant(FC_MUL_MM + 1)]]; -constant short FC_mul_mm_ne12 [[function_constant(FC_MUL_MM + 2)]]; -constant short FC_mul_mm_ne13 [[function_constant(FC_MUL_MM + 3)]]; -constant short FC_mul_mm_r2 [[function_constant(FC_MUL_MM + 4)]]; -constant short FC_mul_mm_r3 [[function_constant(FC_MUL_MM + 5)]]; - -// each block_q contains 16*nl weights -#ifdef GGML_METAL_HAS_TENSOR -template< - typename SA, typename SA_4x4, typename SA_8x8, - typename SB, typename SB_2x4, typename SB_8x8, - typename block_q, short nl, void (*dequantize_func)(device const block_q *, short, thread SA_4x4 &), - typename T0, typename T0_4x4, typename T1, typename T1_2x4> -kernel void kernel_mul_mm( - constant ggml_metal_kargs_mul_mm & args, - device const char * srcA, - device const char * srcB, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig [[threadgroup_position_in_grid]], - ushort tiitg [[thread_index_in_threadgroup]], - ushort sgitg [[simdgroup_index_in_threadgroup]]) { - (void) sgitg; - - // Matrix dimensions: A(M,K) x B(K,N) -> C(M,N) - const int K = args.ne00; - const int M = args.ne0; - const int N = args.ne1; - - // Batch dimension handling - const int im = tgpig.z; - const int i12 = im % FC_mul_mm_ne12; - const int i13 = im / FC_mul_mm_ne12; - - // Batch offsets for srcA and srcB - const uint64_t offset0 = (i12/FC_mul_mm_r2)*args.nb02 + (i13/FC_mul_mm_r3)*args.nb03; - - // Tile dimensions - constexpr int NRB = SZ_SIMDGROUP * N_MM_BLOCK_X * N_MM_SIMD_GROUP_X; - constexpr int NRA = SZ_SIMDGROUP * N_MM_BLOCK_Y * N_MM_SIMD_GROUP_Y; - - // Tile offsets in output matrix - const int ra = tgpig.y * NRA; - const int rb = tgpig.x * NRB; - - // Threadgroup memory for dequantized A tile only - threadgroup SA * sa = (threadgroup SA *)(shmem); - - // Work-item count for A loading - constexpr int A_WORK_ITEMS = NRA * N_MM_NK; - constexpr int NUM_THREADS = N_SIMDWIDTH * N_MM_SIMD_GROUP_X * N_MM_SIMD_GROUP_Y; - - // tA wraps threadgroup memory - auto tA = tensor(sa, dextents(N_MM_NK_TOTAL, NRA)); - - // tB wraps device memory directly - device T1 * ptrB = (device T1 *)(srcB + args.nb12*i12 + args.nb13*i13); - const int strideB = args.nb11 / sizeof(T1); - auto tB = tensor(ptrB, dextents(K, N), array({1, strideB})); - - // Configure matmul operation - mpp::tensor_ops::matmul2d< - mpp::tensor_ops::matmul2d_descriptor( - NRB, NRA, N_MM_NK_TOTAL, false, true, true, - mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate), - execution_simdgroups> mm; - - auto cT = mm.get_destination_cooperative_tensor(); - - // Accumulate partial results over K dimension - for (int loop_k = 0; loop_k < K; loop_k += N_MM_NK_TOTAL) { - // === PHASE 1: Dequantization of A into threadgroup memory === - for (int work = tiitg; work < A_WORK_ITEMS; work += NUM_THREADS) { - const int row = work / N_MM_NK; - const int k_chunk = work % N_MM_NK; - const int k_pos = loop_k + k_chunk * 16; - const short k_base = k_chunk * 16; - - // Bounds check: skip device read if row is out of matrix bounds - if (ra + row < M) { - if (is_same::value && FC_mul_mm_bc_inp) { - // Element-wise reads when K is not aligned (nb01 not aligned for half4x4/float4x4). - // MSL spec Table 2.5: half4x4 requires 8-byte alignment. When K is odd, - // nb01 = K*2 is not 8-byte aligned, so odd-row pointers are misaligned. - // Mirrors the legacy kernel's existing guard. - device const T0 * row_ptr = (device const T0 *)(srcA + args.nb01 * (ra + row) + offset0); - - FOR_UNROLL (short i = 0; i < 16; i++) { - sa[row * N_MM_NK_TOTAL + (k_base + i)] = (k_pos + i < K) ? (SA) row_ptr[k_pos + i] : (SA)0; - } - } else { - const int block_idx = k_pos / (16 * nl); - const short il = (k_pos / 16) % nl; - - device const block_q * row_ptr = (device const block_q *)(srcA + args.nb01 * (ra + row) + offset0); - - SA_4x4 temp_a; - dequantize_func(row_ptr + block_idx, il, temp_a); - - FOR_UNROLL (short i = 0; i < 16; i++) { - // Zero-pad A for K positions beyond valid range (handles partial K iterations) - sa[row * N_MM_NK_TOTAL + (k_base + i)] = (k_pos + i < K) ? temp_a[i/4][i%4] : (SA)0; - } - } - } else { - // Zero-pad rows beyond matrix bounds - FOR_UNROLL (short i = 0; i < 16; i++) { - sa[row * N_MM_NK_TOTAL + (k_base + i)] = (SA)0; - } - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // === PHASE 2: Tensor matmul === - auto mA = tA.slice(0, 0); - auto mB = tB.slice(loop_k, rb); - - mm.run(mB, mA, cT); - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - // Store result tile to output matrix (with batch offset) - // cT.store handles bounds checking via tD's extents (M, N) - device float * dstBatch = (device float *)dst + im * N * M; - - auto tD = tensor(dstBatch, dextents(M, N), array({1, M})); - cT.store(tD.slice(ra, rb)); -} - -#else - -template< - typename S0, typename S0_4x4, typename S0_8x8, - typename S1, typename S1_2x4, typename S1_8x8, - typename block_q, short nl, void (*dequantize_func)(device const block_q *, short, thread S0_4x4 &), - typename T0, typename T0_4x4, typename T1, typename T1_2x4> -kernel void kernel_mul_mm( - constant ggml_metal_kargs_mul_mm & args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - - threadgroup S0 * sa = (threadgroup S0 *)(shmem); - threadgroup S1 * sb = (threadgroup S1 *)(shmem + 4096); - - constexpr int NR0 = 64; - constexpr int NR1 = 32; - - constexpr int NK = 32; - constexpr int NL0 = NK/16; - constexpr int NL1 = NK/8; - - const int im = tgpig.z; - const int r0 = tgpig.y*NR0; - const int r1 = tgpig.x*NR1; - - // if this block is of 64x32 shape or smaller - const short nr0 = (args.ne0 - r0 < NR0) ? (args.ne0 - r0) : NR0; - const short nr1 = (args.ne1 - r1 < NR1) ? (args.ne1 - r1) : NR1; - - // a thread shouldn't load data outside of the matrix - const short lr0 = ((short)tiitg/NL0) < nr0 ? ((short)tiitg/NL0) : nr0 - 1; // 0 .. 63 - const short lr1 = ((short)tiitg/NL1) < nr1 ? ((short)tiitg/NL1) : nr1 - 1; // 0 .. 31 - - const short il0 = (tiitg % NL0); - - short il = il0; - - const int i12 = im % FC_mul_mm_ne12; - const int i13 = im / FC_mul_mm_ne12; - - const uint64_t offset0 = (i12/FC_mul_mm_r2)*args.nb02 + (i13/FC_mul_mm_r3)*args.nb03; - const short offset1 = il0/nl; - - device const block_q * x = (device const block_q *)(src0 + args.nb01*(r0 + lr0) + offset0) + offset1; - - const short iy = 8*(tiitg % NL1); - - device const T1 * y = (device const T1 *)(src1 - + args.nb13*i13 - + args.nb12*i12 - + args.nb11*(r1 + lr1) - + args.nb10*iy); - - S0_8x8 ma[4]; - S1_8x8 mb[2]; - - simdgroup_float8x8 mc[8]; - - for (short i = 0; i < 8; i++){ - mc[i] = make_filled_simdgroup_matrix(0.f); - } - - for (int loop_k = 0; loop_k < args.ne00; loop_k += NK) { - // load data and store to threadgroup memory - if (is_same::value && FC_mul_mm_bc_inp) { - threadgroup_barrier(mem_flags::mem_threadgroup); - - // no need for dequantization - for (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - //const short lx = i%8; - //const short ly = (tiitg/NL0)%8; - const short lx = (tiitg/NL0)%8; - const short ly = i%8; - - const short ib = 8*sx + sy; - - *(sa + 64*ib + 8*ly + lx) = loop_k + 16*il + i < args.ne00 ? *((device T0 *) x + i) : 0; - } - } else { - S0_4x4 temp_a; - dequantize_func(x, il, temp_a); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - FOR_UNROLL (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - //const short lx = i%8; - //const short ly = (tiitg/NL0)%8; - const short lx = (tiitg/NL0)%8; - const short ly = i%8; - - const short ib = 8*sx + sy; - - // NOTE: this is massively slower.. WTF? - //sa[64*ib + 8*ly + lx] = temp_a[i/4][i%4]; - - *(sa + 64*ib + 8*ly + lx) = temp_a[i/4][i%4]; - } - } - - if (FC_mul_mm_bc_inp) { - for (short i = 0; i < 8; ++i) { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - const short lx = i; - const short ly = (tiitg/NL1)%8; - //const short lx = (tiitg/NL1)%8; - //const short ly = i; - - const short ib = 4*sx + sy; - - *(sb + 64*ib + 8*ly + lx) = loop_k + iy + i < args.ne00 ? (S1) *((device T1 *) y + i) : 0; - } - } else { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - //const short dx = sx; - //const short dy = sy; - - const short ly = (tiitg/NL1)%8; - - const short ib = 4*sx + sy; - - *(threadgroup S1_2x4 *)(sb + 64*ib + 8*ly) = (S1_2x4)(*((device T1_2x4 *) y)); - } - - il = (il + 2 < nl) ? il + 2 : il % 2; - x = (il < 2) ? x + (2 + nl - 1)/nl : x; - - y += NK; - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // load matrices from threadgroup memory and conduct outer products - threadgroup const S0 * lsma = (sa + 4*64*(sgitg%2)); - threadgroup const S1 * lsmb = (sb + 2*64*(sgitg/2)); - - FOR_UNROLL (short ik = 0; ik < NK/8; ik++) { - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 4; i++) { - simdgroup_load(ma[i], lsma + 64*i, 8, 0, false); - } - - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 2; i++) { - simdgroup_load(mb[i], lsmb + 64*i, 8, 0, false); - } - - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 8; i++){ - simdgroup_multiply_accumulate(mc[i], mb[i/4], ma[i%4], mc[i]); - } - - lsma += 8*64; - lsmb += 4*64; - } - } - - if (!FC_mul_mm_bc_out || (r0 + NR0 <= args.ne0 && r1 + NR1 <= args.ne1)) { - // if no bounds checks on the output are needed, we can directly write to device memory - device float * C = (device float *) dst + - (r0 + 32*(sgitg & 1)) + \ - (r1 + 16*(sgitg >> 1)) * args.ne0 + im*args.ne1*args.ne0; - - for (short i = 0; i < 8; i++) { - simdgroup_store(mc[i], C + 8*(i%4) + 8*args.ne0*(i/4), args.ne0, 0, false); - } - } else { - // block is smaller than 64x32, we should avoid writing data outside of the matrix - threadgroup_barrier(mem_flags::mem_threadgroup); - - threadgroup float * temp_str = ((threadgroup float *) shmem) + 32*(sgitg&1) + (16*(sgitg >> 1))*NR0; - - for (short i = 0; i < 8; i++) { - simdgroup_store(mc[i], temp_str + 8*(i%4) + 8*NR0*(i/4), NR0, 0, false); - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (sgitg == 0) { - for (int j = tiitg; j < nr1; j += NR1) { - device float * D = (device float *) dst + r0 + (r1 + j)*args.ne0 + im*args.ne1*args.ne0; - device float4 * D4 = (device float4 *) D; - - threadgroup float * C = temp_str + (j*NR0); - threadgroup float4 * C4 = (threadgroup float4 *) C; - - int i = 0; - for (; i < nr0/4; i++) { - *(D4 + i) = *(C4 + i); - } - - i *= 4; - for (; i < nr0; i++) { - *(D + i) = *(C + i); - } - } - } - } -} - -#endif // GGML_METAL_HAS_TENSOR - -template // n_expert_used -kernel void kernel_mul_mm_id_map0( - constant ggml_metal_kargs_mul_mm_id_map0 & args, - device const char * src2, - device char * htpe, - device char * hids, - threadgroup char * shmem [[threadgroup(0)]], - ushort tpitg[[thread_position_in_threadgroup]], - ushort ntg[[threads_per_threadgroup]]) { - const short ide = tpitg; // expert id - - uint32_t n_all = 0; - - device int32_t * ids_i32 = (device int32_t *) hids + ide*args.ne21; - - for (int i21 = 0; i21 < args.ne21; i21 += ntg) { // n_tokens - if (i21 + tpitg < args.ne21) { - device const int32_t * src2_i32 = (device const int32_t *) (src2 + (i21 + tpitg)*args.nb21); - - threadgroup uint16_t * sids = (threadgroup uint16_t *) shmem + tpitg*ne20; - - #pragma unroll(ne20) - for (short i20 = 0; i20 < ne20; i20++) { - sids[i20] = src2_i32[i20]; - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (short t = 0; t < ntg; t++) { - if (i21 + t >= args.ne21) { - break; - } - - threadgroup const uint16_t * sids = (threadgroup const uint16_t *) shmem + t*ne20; - - short sel = 0; - #pragma unroll(ne20) - for (short i20 = 0; i20 < ne20; i20++) { - sel += (sids[i20] == ide)*(i20 + 1); - } - - ids_i32[n_all] = (i21 + t)*ne20 + sel - 1; - - n_all += sel > 0; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - device uint32_t * tpe_u32 = (device uint32_t *) (htpe); - tpe_u32[ide] = n_all; -} - -typedef decltype(kernel_mul_mm_id_map0<1>) kernel_mul_mm_id_map0_t; - -template [[host_name("kernel_mul_mm_id_map0_ne20_1" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<1>; -template [[host_name("kernel_mul_mm_id_map0_ne20_2" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<2>; -template [[host_name("kernel_mul_mm_id_map0_ne20_4" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<4>; -template [[host_name("kernel_mul_mm_id_map0_ne20_5" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<5>; -template [[host_name("kernel_mul_mm_id_map0_ne20_6" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<6>; -template [[host_name("kernel_mul_mm_id_map0_ne20_8" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<8>; -template [[host_name("kernel_mul_mm_id_map0_ne20_10")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<10>; -template [[host_name("kernel_mul_mm_id_map0_ne20_16")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<16>; -template [[host_name("kernel_mul_mm_id_map0_ne20_22")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<22>; - -template -kernel void kernel_mul_mm_id( - constant ggml_metal_kargs_mul_mm_id & args, - device const char * src0, - device const char * src1, - device const char * htpe, - device const char * hids, - device char * dst, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - threadgroup S0 * sa = (threadgroup S0 *)(shmem); - threadgroup S1 * sb = (threadgroup S1 *)(shmem + 4096); - -#ifdef GGML_METAL_HAS_TENSOR - threadgroup float * sc = (threadgroup float *)(shmem); -#endif - - constexpr int NR0 = 64; - constexpr int NR1 = 32; - - constexpr int NK = 32; - constexpr int NL0 = NK/16; - constexpr int NL1 = NK/8; - - const int im = tgpig.z; // expert - const int r0 = tgpig.y*NR0; - const int r1 = tgpig.x*NR1; - - device const uint32_t * tpe_u32 = (device const uint32_t *) (htpe); - device const int32_t * ids_i32 = (device const int32_t *) (hids); - - const int32_t neh1 = tpe_u32[im]; - - if (r1 >= neh1) { - return; - } - - // if this block is of 64x32 shape or smaller - const short nr0 = (args.ne0 - r0 < NR0) ? (args.ne0 - r0) : NR0; - const short nr1 = ( neh1 - r1 < NR1) ? ( neh1 - r1) : NR1; - - // a thread shouldn't load data outside of the matrix - const short lr0 = ((short)tiitg/NL0) < nr0 ? ((short)tiitg/NL0) : nr0 - 1; // 0 .. 63 - const short lr1 = ((short)tiitg/NL1) < nr1 ? ((short)tiitg/NL1) : nr1 - 1; // 0 .. 31 - - const short il0 = (tiitg % NL0); - - short il = il0; - - const int id = ids_i32[im*args.ne21 + r1 + lr1]; - - const short i11 = (id % args.ne20) % args.ne11; - const short i12 = (id / args.ne20); - const short i13 = 0; - - const uint64_t offset0 = im*args.nb02 + i13*args.nb03; - const short offset1 = il0/nl; - - device const block_q * x = (device const block_q *)(src0 + args.nb01*(r0 + lr0) + offset0) + offset1; - - const short iy = 8*(tiitg % NL1); - - device const T1 * y = (device const T1 *)(src1 - + args.nb13*i13 - + args.nb12*i12 - + args.nb11*i11 - + args.nb10*iy); - -#ifndef GGML_METAL_HAS_TENSOR - S0_8x8 ma[4]; - S1_8x8 mb[2]; - - simdgroup_float8x8 mc[8]; - - for (short i = 0; i < 8; i++){ - mc[i] = make_filled_simdgroup_matrix(0.f); - } -#else - auto tA = tensor, tensor_inline>(sa, dextents(NK, NR0)); - auto tB = tensor, tensor_inline>(sb, dextents(NR1, NK )); - - mpp::tensor_ops::matmul2d< - mpp::tensor_ops::matmul2d_descriptor(NR1, NR0, NK, false, true, false, mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate), - execution_simdgroups<4>> mm; - - auto cT = mm.get_destination_cooperative_tensor(); -#endif - - for (int loop_k = 0; loop_k < args.ne00; loop_k += NK) { -#ifndef GGML_METAL_HAS_TENSOR - // load data and store to threadgroup memory - if (is_same::value && FC_mul_mm_bc_inp) { - threadgroup_barrier(mem_flags::mem_threadgroup); - - // no need for dequantization - for (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - //const short lx = i%8; - //const short ly = (tiitg/NL0)%8; - const short lx = (tiitg/NL0)%8; - const short ly = i%8; - - const short ib = 8*sx + sy; - - *(sa + 64*ib + 8*ly + lx) = loop_k + 16*il + i < args.ne00 ? (S0) *((device T0 *) x + i) : (S0) 0; - } - } else { - S0_4x4 temp_a; - dequantize_func(x, il, temp_a); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - FOR_UNROLL (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - //const short lx = i%8; - //const short ly = (tiitg/NL0)%8; - const short lx = (tiitg/NL0)%8; - const short ly = i%8; - - const short ib = 8*sx + sy; - - // NOTE: this is massively slower.. WTF? - //sa[64*ib + 8*ly + lx] = temp_a[i/4][i%4]; - - *(sa + 64*ib + 8*ly + lx) = temp_a[i/4][i%4]; - } - } - - if (FC_mul_mm_bc_inp) { - for (short i = 0; i < 8; ++i) { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - const short lx = i; - const short ly = (tiitg/NL1)%8; - //const short lx = (tiitg/NL1)%8; - //const short ly = i; - - const short ib = 4*sx + sy; - - *(sb + 64*ib + 8*ly + lx) = loop_k + iy + i < args.ne00 ? (S1) *((device T1 *) y + i) : 0; - } - } else { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - //const short dx = sx; - //const short dy = sy; - - const short ly = (tiitg/NL1)%8; - - const short ib = 4*sx + sy; - - *(threadgroup S1_2x4 *)(sb + 64*ib + 8*ly) = (S1_2x4)(*((device T1_2x4 *) y)); - } -#else - // load data and store to threadgroup memory - if (is_same::value && FC_mul_mm_bc_inp) { - threadgroup_barrier(mem_flags::mem_threadgroup); - - // no need for dequantization - for (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - const short lx = i%8; - const short ly = (tiitg/NL0)%8; - //const short lx = (tiitg/NL0)%8; - //const short ly = i%8; - - *(sa + NK*(8*sy + ly) + 8*sx + lx) = loop_k + 16*il + i < args.ne00 ? *((device T0 *) x + i) : 0; - } - } else { - S0_4x4 temp_a; - dequantize_func(x, il, temp_a); - - threadgroup_barrier(mem_flags::mem_threadgroup); - - FOR_UNROLL (short i = 0; i < 16; i++) { - const short sx = 2*il0 + i/8; - const short sy = (tiitg/NL0)/8; - - const short lx = i%8; - const short ly = (tiitg/NL0)%8; - //const short lx = (tiitg/NL0)%8; - //const short ly = i%8; - - *(sa + NK*(8*sy + ly) + 8*sx + lx) = temp_a[i/4][i%4]; - } - } - - if (FC_mul_mm_bc_inp) { - for (short i = 0; i < 8; ++i) { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - const short lx = i; - const short ly = (tiitg/NL1)%8; - //const short lx = (tiitg/NL1)%8; - //const short ly = i; - - *(sb + NK*(8*sy + ly) + 8*sx + lx) = loop_k + iy + i < args.ne00 ? (S1) *((device T1 *) y + i) : 0; - } - } else { - const short sx = (tiitg%NL1); - const short sy = (tiitg/NL1)/8; - - //const short lx = i; - const short ly = (tiitg/NL1)%8; - //const short lx = (tiitg/NL1)%8; - //const short ly = i; - - *(threadgroup S1_2x4 *)(sb + NK*(8*sy + ly) + 8*sx) = (S1_2x4)(*((device T1_2x4 *) y)); - } -#endif - - il = (il + 2 < nl) ? il + 2 : il % 2; - x = (il < 2) ? x + (2 + nl - 1)/nl : x; - - y += NK; - - threadgroup_barrier(mem_flags::mem_threadgroup); - -#ifndef GGML_METAL_HAS_TENSOR - // load matrices from threadgroup memory and conduct outer products - threadgroup const S0 * lsma = (sa + 4*64*(sgitg%2)); - threadgroup const S1 * lsmb = (sb + 2*64*(sgitg/2)); - - FOR_UNROLL (short ik = 0; ik < NK/8; ik++) { - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 4; i++) { - simdgroup_load(ma[i], lsma + 64*i, 8, 0, false); - } - - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 2; i++) { - simdgroup_load(mb[i], lsmb + 64*i, 8, 0, false); - } - - simdgroup_barrier(mem_flags::mem_none); - - FOR_UNROLL (short i = 0; i < 8; i++){ - simdgroup_multiply_accumulate(mc[i], mb[i/4], ma[i%4], mc[i]); - } - - lsma += 8*64; - lsmb += 4*64; - } -#else - auto sA = tA.slice(0, 0); - auto sB = tB.slice(0, 0); - - mm.run(sB, sA, cT); -#endif - } - - // block is smaller than 64x32, we should avoid writing data outside of the matrix - threadgroup_barrier(mem_flags::mem_threadgroup); - -#ifdef GGML_METAL_HAS_TENSOR - auto tC = tensor, tensor_inline>(sc, dextents(NR0, NR1)); - cT.store(tC); -#else - threadgroup float * temp_str = ((threadgroup float *) shmem) + 32*(sgitg&1) + (16*(sgitg >> 1))*NR0; - - for (short i = 0; i < 8; i++) { - simdgroup_store(mc[i], temp_str + 8*(i%4) + 8*NR0*(i/4), NR0, 0, false); - } -#endif - - threadgroup_barrier(mem_flags::mem_threadgroup); - - for (short j = sgitg; j < nr1; j += 4) { - const int id = ids_i32[im*args.ne21 + r1 + j]; - - const short ide = id % args.ne20; - const short idt = id / args.ne20; - - device float * D = (device float *) dst + r0 + ide*args.ne0 + idt*args.ne1*args.ne0; - device float4 * D4 = (device float4 *) D; - - threadgroup float * C = (threadgroup float *) shmem + j*NR0; - threadgroup float4 * C4 = (threadgroup float4 *) C; - - int i = tiisg; - for (; i < nr0/4; i += 32) { - *(D4 + i) = *(C4 + i); - } - - i = (4*(nr0/4)) + tiisg; - for (; i < nr0; i += 32) { - *(D + i) = *(C + i); - } - } -} - -// -// matrix-matrix multiplication -// - -typedef decltype(kernel_mul_mm) mul_mm_t; - -template [[host_name("kernel_mul_mm_f32_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_f16_f32")]] kernel mul_mm_t kernel_mul_mm; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mm_bf16_f32")]] kernel mul_mm_t kernel_mul_mm; -#endif -template [[host_name("kernel_mul_mm_q1_0_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q2_0_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_0_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_1_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_0_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_1_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q8_0_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_mxfp4_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q2_K_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q3_K_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_K_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_K_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q6_K_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_xxs_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_xs_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq3_xxs_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq3_s_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_s_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq1_s_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq1_m_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq4_nl_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq4_xs_f32")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_tq2_0_f32")]] kernel mul_mm_t kernel_mul_mm; - -template [[host_name("kernel_mul_mm_f32_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_f16_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q1_0_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q2_0_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_0_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_1_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_0_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_1_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q8_0_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_mxfp4_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q2_K_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q3_K_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q4_K_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q5_K_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_q6_K_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_xxs_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_xs_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq3_xxs_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq3_s_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq2_s_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq1_s_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq1_m_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq4_nl_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_iq4_xs_f16")]] kernel mul_mm_t kernel_mul_mm; -template [[host_name("kernel_mul_mm_tq2_0_f16")]] kernel mul_mm_t kernel_mul_mm; - -// -// indirect matrix-matrix multiplication -// - -typedef decltype(kernel_mul_mm_id) mul_mm_id; - -template [[host_name("kernel_mul_mm_id_f32_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_f16_f32")]] kernel mul_mm_id kernel_mul_mm_id; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mm_id_bf16_f32")]] kernel mul_mm_id kernel_mul_mm_id; -#endif -template [[host_name("kernel_mul_mm_id_q1_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q2_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_1_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_1_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q8_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_mxfp4_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q2_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q3_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q6_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_xxs_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_xs_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq3_xxs_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq3_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq1_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq1_m_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq4_nl_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq4_xs_f32")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_tq2_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; - -template [[host_name("kernel_mul_mm_id_f32_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_f16_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q1_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q2_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_1_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_1_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q8_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_mxfp4_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q2_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q3_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q4_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q5_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_q6_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_xxs_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_xs_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq3_xxs_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq3_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq2_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq1_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq1_m_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq4_nl_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_iq4_xs_f16")]] kernel mul_mm_id kernel_mul_mm_id; -template [[host_name("kernel_mul_mm_id_tq2_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; - -// -// matrix-vector multiplication -// - -typedef void (kernel_mul_mv_disp_t)( - ggml_metal_kargs_mul_mv args, - device const char * src0, - device const char * src1, - device char * dst, - uint3 tgpig, - ushort tiisg); - -typedef void (kernel_mul_mv2_disp_t)( - ggml_metal_kargs_mul_mv args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiisg, - ushort sgitg); - -template -void mmv_fn( - ggml_metal_kargs_mul_mv args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiitg, - ushort tiisg, - ushort sgitg) { - disp_fn(args, src0, src1, dst, tgpig, tiisg); -} - -template -void mmv_fn( - ggml_metal_kargs_mul_mv args, - device const char * src0, - device const char * src1, - device char * dst, - threadgroup char * shmem, - uint3 tgpig, - ushort tiitg, - ushort tiisg, - ushort sgitg) { - disp_fn(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); -} - -typedef decltype(mmv_fn>) mul_mv_disp_fn_t; - -template -kernel void kernel_mul_mv_id( - constant ggml_metal_kargs_mul_mv_id & args, - device const char * src0s, - device const char * src1, - device char * dst, - device const char * ids, - threadgroup char * shmem [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - const int iid1 = tgpig.z/args.nei0; - const int idx = tgpig.z%args.nei0; - - tgpig.z = 0; - - const int32_t i02 = ((device const int32_t *) (ids + iid1*args.nbi1))[idx]; - - const int64_t i11 = idx % args.ne11; - const int64_t i12 = iid1; - - const int64_t i1 = idx; - const int64_t i2 = i12; - - device const char * src0_cur = src0s + i02*args.nb02; - device const char * src1_cur = src1 + i11*args.nb11 + i12*args.nb12; - - device char * dst_cur = dst + (i1*args.ne0 + i2*args.ne1*args.ne0)*sizeof(float); - - ggml_metal_kargs_mul_mv args0 = { - /*.ne00 =*/ args.ne00, - /*.ne01 =*/ args.ne01, - /*.ne02 =*/ 1, // args.ne02, - /*.nb00 =*/ args.nb00, - /*.nb01 =*/ args.nb01, - /*.nb02 =*/ args.nb02, - /*.nb03 =*/ args.nb02, // args.ne02 == 1 - /*.ne10 =*/ args.ne10, - /*.ne11 =*/ 1, // args.ne11, - /*.ne12 =*/ 1, // args.ne12, - /*.nb10 =*/ args.nb10, - /*.nb11 =*/ args.nb11, - /*.nb12 =*/ args.nb12, - /*.nb13 =*/ args.nb12, // ne12 == 1 - /*.ne0 =*/ args.ne0, - /*.ne1 =*/ 1, // args.ne1, - /*.nr0 =*/ args.nr0, - /*.r2 =*/ 1, - /*.r3 =*/ 1, - }; - - disp_fn( - args0, - /* src0 */ src0_cur, - /* src1 */ src1_cur, - /* dst */ dst_cur, - shmem, - tgpig, - tiitg, - tiisg, - sgitg); -} - -typedef decltype(kernel_mul_mv_id>>) kernel_mul_mv_id_t; - -typedef decltype(kernel_mul_mv_id>>) kernel_mul_mv_id_4_t; - -template [[host_name("kernel_mul_mv_id_f32_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_f16_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_id_bf16_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -#endif -template [[host_name("kernel_mul_mv_id_f32_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_f16_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_mul_mv_id_bf16_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; -#endif - -template [[host_name("kernel_mul_mv_id_q8_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; - -template [[host_name("kernel_mul_mv_id_q1_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q2_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q4_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q4_1_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q5_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q5_1_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; - -template [[host_name("kernel_mul_mv_id_mxfp4_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; - -template [[host_name("kernel_mul_mv_id_q2_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q3_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q4_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q5_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_q6_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq1_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq1_m_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq2_xxs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq2_xs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq3_xxs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq3_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq2_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq4_nl_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_iq4_xs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; -template [[host_name("kernel_mul_mv_id_tq2_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; - -kernel void kernel_pool_2d_max_f32( - constant ggml_metal_kargs_pool_2d & args, - device const float * src0, - device float * dst, - uint gid[[thread_position_in_grid]]) { - - if (gid >= args.np) { - return; - } - - const int idx = gid; - const int I_HW = args.IH * args.IW; - const int O_HW = args.OH * args.OW; - const int nc = idx / O_HW; - const int cur_oh = idx % O_HW / args.OW; - const int cur_ow = idx % O_HW % args.OW; - - device const float * i_ptr = src0 + nc * I_HW; - device float * o_ptr = dst + nc * O_HW; - - const int start_h = cur_oh * args.s1 - args.p1; - const int bh = MAX(0, start_h); - const int eh = MIN(args.IH, start_h + args.k1); - const int start_w = cur_ow * args.s0 - args.p0; - const int bw = MAX(0, start_w); - const int ew = MIN(args.IW, start_w + args.k0); - - float res = -INFINITY; - - for (int i = bh; i < eh; i += 1) { - for (int j = bw; j < ew; j += 1) { - res = MAX(res, i_ptr[i * args.IW + j]); - } - } - - o_ptr[cur_oh * args.OW + cur_ow] = res; -} - -kernel void kernel_pool_2d_avg_f32( - constant ggml_metal_kargs_pool_2d & args, - device const float * src0, - device float * dst, - uint gid[[thread_position_in_grid]]) { - - if (gid >= args.np) { - return; - } - - const int idx = gid; - const int I_HW = args.IH * args.IW; - const int O_HW = args.OH * args.OW; - const int nc = idx / O_HW; - const int cur_oh = idx % O_HW / args.OW; - const int cur_ow = idx % O_HW % args.OW; - - device const float * i_ptr = src0 + nc * I_HW; - device float * o_ptr = dst + nc * O_HW; - - const int start_h = cur_oh * args.s1 - args.p1; - const int bh = MAX(0, start_h); - const int eh = MIN(args.IH, start_h + args.k1); - const int start_w = cur_ow * args.s0 - args.p0; - const int bw = MAX(0, start_w); - const int ew = MIN(args.IW, start_w + args.k0); - // const float scale = 1. / ((eh - bh) * (ew - bw)); - const float scale = 1. / (args.k0 * args.k1); - - float res = 0; - - for (int i = bh; i < eh; i += 1) { - for (int j = bw; j < ew; j += 1) { - float cur = i_ptr[i * args.IW + j]; - res += cur * scale; - } - } - - o_ptr[cur_oh * args.OW + cur_ow] = res; -} - - -kernel void kernel_pool_1d_max_f32( - constant ggml_metal_kargs_pool_1d & args, - device const float * src, - device float * dst, - uint gid [[thread_position_in_grid]] -) { - - if (gid >= args.np) { - return; - } - - const int ow = (int)gid % args.OW; - const int row = (int)gid / args.OW; - - const int base = ow * args.s0 - args.p0; - - float acc = -INFINITY; - - const int src_off = row * args.IW; - const int dst_off = row * args.OW; - - for (int ki = 0; ki < args.k0; ++ki) { - int j = base + ki; - if (j < 0 || j >= args.IW){ - continue; - } - float v = src[src_off + j]; - acc = max(acc, v); - } - - dst[dst_off + ow] = acc; -} - -kernel void kernel_pool_1d_avg_f32( - constant ggml_metal_kargs_pool_1d & args, - device const float * src, - device float * dst, - uint gid [[thread_position_in_grid]] -) { - - if (gid >= args.np) { - return; - } - - const int ow = (int)gid % args.OW; - const int row = (int)gid / args.OW; - - const int base = ow * args.s0 - args.p0; - - float acc = 0.0f; - int cnt = 0; - - const int src_off = row * args.IW; - const int dst_off = row * args.OW; - - for (int ki = 0; ki < args.k0; ++ki) { - const int j = base + ki; - if (j < 0 || j >= args.IW) { - continue; - } - acc += src[src_off + j]; - cnt += 1; - } - - dst[dst_off + ow] = (cnt > 0) ? (acc / (float)cnt) : 0.0f; -} - -kernel void kernel_opt_step_adamw_f32( - constant ggml_metal_kargs_opt_step_adamw & args, - device float * x, - device const float * g, - device float * g_m, - device float * g_v, - device const float * pars, - uint gid[[thread_position_in_grid]]) { - - if (gid >= args.np) { - return; - } - - const float alpha = pars[0]; - const float beta1 = pars[1]; - const float beta2 = pars[2]; - const float eps = pars[3]; - const float wd = pars[4]; - const float beta1h = pars[5]; - const float beta2h = pars[6]; - - const float gi = g[gid]; - const float gmi = g_m[gid] * beta1 + gi * (1.0f - beta1); - const float gvi = g_v[gid] * beta2 + gi * gi * (1.0f - beta2); - - g_m[gid] = gmi; - g_v[gid] = gvi; - - const float mh = gmi * beta1h; - const float vh = sqrt(gvi * beta2h) + eps; - - x[gid] = x[gid] * (1.0f - alpha * wd) - alpha * mh / vh; -} - -kernel void kernel_opt_step_sgd_f32( - constant ggml_metal_kargs_opt_step_sgd & args, - device float * x, - device const float * g, - device const float * pars, - uint gid[[thread_position_in_grid]]) { - - if (gid >= args.np) { - return; - } - - x[gid] = x[gid] * (1.0f - pars[0] * pars[1]) - pars[0] * g[gid]; -} - -template -kernel void kernel_memset( - constant ggml_metal_kargs_memset & args, - device T * dst, - uint tpig[[thread_position_in_grid]]) { - dst[tpig] = args.val; -} - -typedef decltype(kernel_memset) kernel_memset_t; - -template [[host_name("kernel_memset_i64")]] kernel kernel_memset_t kernel_memset; - -constant short FC_count_equal_nsg [[function_constant(FC_COUNT_EQUAL + 0)]]; - -template -kernel void kernel_count_equal( - constant ggml_metal_kargs_count_equal & args, - device const char * src0, - device const char * src1, - device atomic_int * dst, - threadgroup int32_t * shmem_i32 [[threadgroup(0)]], - uint3 tgpig[[threadgroup_position_in_grid]], - ushort3 tpitg[[thread_position_in_threadgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - const short NSG = FC_count_equal_nsg; - - const int i3 = tgpig.z; - const int i2 = tgpig.y; - const int i1 = tgpig.x; - - if (i3 >= args.ne03 || i2 >= args.ne02 || i1 >= args.ne01) { - return; - } - - int sum = 0; - - device const char * base0 = src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03; - device const char * base1 = src1 + i1*args.nb11 + i2*args.nb12 + i3*args.nb13; - - for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { - const T v0 = *(device const T *)(base0 + i0*args.nb00); - const T v1 = *(device const T *)(base1 + i0*args.nb10); - sum += (v0 == v1); - } - - sum = simd_sum(sum); - - if (tiisg == 0) { - shmem_i32[sgitg] = sum; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - if (sgitg == 0) { - float v = 0.0f; - if (tpitg.x < NSG) { - v = shmem_i32[tpitg.x]; - } - - float total = simd_sum(v); - if (tpitg.x == 0) { - atomic_fetch_add_explicit(dst, (int32_t) total, memory_order_relaxed); - } - } -} - -typedef decltype(kernel_count_equal) kernel_count_equal_t; - -template [[host_name("kernel_count_equal_i32")]] kernel kernel_count_equal_t kernel_count_equal; - -template< - typename kd4x4_t, - short nl_k, - void (*deq_k)(device const kd4x4_t *, short, thread half4x4 &)> -kernel void kernel_lightning_indexer( - constant ggml_metal_kargs_lightning_indexer & args, - device const char * q, - device const char * k, - device const char * w, - device const char * m, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiitg[[thread_index_in_threadgroup]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]]) { - constexpr short DK = OP_LIGHTNING_INDEXER_DK; - constexpr short NH = OP_LIGHTNING_INDEXER_NH; - constexpr short NHPTG = OP_LIGHTNING_INDEXER_NHPTG; - constexpr short NKPSG = OP_LIGHTNING_INDEXER_NKPSG; - constexpr short NSG = OP_LIGHTNING_INDEXER_NSG; - constexpr short NBPTG = OP_LIGHTNING_INDEXER_NBPTG; - - constexpr short DK4 = DK/4; - constexpr short DK8 = DK/8; - constexpr short DK16 = DK/16; - - constexpr short NK = NKPSG*NSG; // keys per threadgroup - constexpr short NTG = 32*NSG; // threads per threadgroup - - const int i_stream = tgpig.z; - const int i_kv_0 = tgpig.x*NK; // first key of this threadgroup - const int i_kv = i_kv_0 + sgitg*NKPSG; // first key of this simdgroup - - threadgroup half sk[NK * DK16 * 16]; - threadgroup half4x4 * sk4x4 = (threadgroup half4x4 *) sk; - - for (short i = tiitg; i < NK*DK16; i += NTG) { - const short ik = i/DK16; - const short i16 = i%DK16; - - half4x4 tmp; - - if (i_kv_0 + ik < args.n_kv) { - device const kd4x4_t * kr = (device const kd4x4_t *) (k + (i_kv_0 + ik)*args.nbk2 + i_stream*args.nbk3); - - deq_k(kr + i16/nl_k, i16%nl_k, tmp); - } else { - FOR_UNROLL (short j = 0; j < 4; ++j) { - tmp[j] = half4(0.0h); - } - } - - sk4x4[i] = tmp; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - // K tile of this simdgroup, transposed to [DK, NKPSG] - simdgroup_half8x8 mk[DK8]; - - FOR_UNROLL (short i = 0; i < DK8; ++i) { - simdgroup_load(mk[i], sk + sgitg*NKPSG*DK + 8*i, DK, 0, true); - } - - threadgroup half4 sq4[NHPTG*DK4]; - threadgroup half * sq = (threadgroup half *) sq4; - - threadgroup float sw [NHPTG]; - threadgroup float sqk[NSG*NHPTG*NKPSG]; - - const int i_batch_0 = tgpig.y*NBPTG; - const int n_batch = min((int) NBPTG, args.n_batch - i_batch_0); - - for (short ib = 0; ib < n_batch; ++ib) { - const int i_batch = i_batch_0 + ib; - - device const char * pq = q + i_batch*args.nbq2 + i_stream*args.nbq3; - device const char * pw = w + i_batch*args.nbw1 + i_stream*args.nbw3; - - float score = 0.0f; - - FOR_UNROLL (short i_head = 0; i_head < NH; i_head += NHPTG) { - // stage the Q tile [DK, NHPTG] and the (prescaled) head weights - for (short i = tiitg; i < NHPTG*DK4; i += NTG) { - const short ih = i/DK4; - const short i4 = i%DK4; - - device const float4 * q4 = (device const float4 *) (pq + (i_head + ih)*args.nbq1); - - sq4[ih*DK4 + i4] = half4(q4[i4]); - } - - if (tiitg < NHPTG) { - sw[tiitg] = ((device const float *) pw)[i_head + tiitg]; - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - - simdgroup_float8x8 mqk = make_filled_simdgroup_matrix(0.0f); - - FOR_UNROLL (short i = 0; i < DK8; ++i) { - simdgroup_half8x8 mq; - - simdgroup_load(mq, sq + 8*i, DK, 0, false); - simdgroup_multiply_accumulate(mqk, mq, mk[i], mqk); - } - - threadgroup float * pqk = sqk + sgitg*NHPTG*NKPSG; - - simdgroup_store(mqk, pqk, NKPSG, 0, false); - simdgroup_barrier(mem_flags::mem_threadgroup); - - // one lane per key: ReLU, apply the head weight and accumulate over the head tile - if (tiisg < NKPSG) { - FOR_UNROLL (short ih = 0; ih < NHPTG; ++ih) { - score += max(pqk[ih*NKPSG + tiisg], 0.0f)*sw[ih]; - } - } - - threadgroup_barrier(mem_flags::mem_threadgroup); - } - - if (tiisg < NKPSG) { - const int ik = i_kv + tiisg; - if (ik < args.n_kv) { - device const half * pm = (device const half *) (m + i_batch*args.nbm1 + (i_stream % args.mask_ne3)*args.nbm3); - device float * pd = (device float *) (dst + i_batch*args.nb1 + i_stream*args.nb3); - - pd[ik] = score + (float) pm[ik]; - } - } - } -} - -typedef decltype(kernel_lightning_indexer) kernel_lightning_indexer_t; - -template [[host_name("kernel_lightning_indexer_f32")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -template [[host_name("kernel_lightning_indexer_f16")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; - -#if defined(GGML_METAL_HAS_BF16) -template [[host_name("kernel_lightning_indexer_bf16")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -#endif - -template [[host_name("kernel_lightning_indexer_q4_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -template [[host_name("kernel_lightning_indexer_q4_1")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -template [[host_name("kernel_lightning_indexer_q5_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -template [[host_name("kernel_lightning_indexer_q5_1")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; -template [[host_name("kernel_lightning_indexer_q8_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; - -kernel void kernel_dsv4_hc_comb_f32( - constant ggml_metal_kargs_dsv4_hc_comb & args, - device const char * mixes, - device const char * scale, - device const char * base, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - constexpr ushort hc = 4; - constexpr ushort comb_offset = 2*hc; - - const int it = tgpig.x*ntg.y + sgitg; - if (it >= args.n_tokens) { - return; - } - - float scale_lane = 0.0f; - if (tiisg == 0) { - scale_lane = *(device const float *) (scale + 2*args.nb_s0); - } - const float scale_comb = simd_shuffle(scale_lane, 0); - - float v = 0.0f; - if (tiisg < hc*hc) { - v = *(device const float *) (mixes + (comb_offset + tiisg)*args.nb_m0 + it*args.nb_m1)*scale_comb - + *(device const float *) (base + (comb_offset + tiisg)*args.nb_b0); - } - - // Softmax across destinations (the four contiguous lanes for each source). - float vmax = max(v, simd_shuffle_xor(v, 1)); - vmax = max(vmax, simd_shuffle_xor(vmax, 2)); - v = exp(v - vmax); - - float sum = v + simd_shuffle_xor(v, 1); - sum += simd_shuffle_xor(sum, 2); - v = v/sum + args.eps; - - // Normalize columns: equal destination indices are four lanes apart. - sum = v + simd_shuffle_xor(v, 4); - sum += simd_shuffle_xor(sum, 8); - v /= sum + args.eps; - - for (int i = 1; i < args.n_iter; ++i) { - sum = v + simd_shuffle_xor(v, 1); - sum += simd_shuffle_xor(sum, 2); - v /= sum + args.eps; - - sum = v + simd_shuffle_xor(v, 4); - sum += simd_shuffle_xor(sum, 8); - v /= sum + args.eps; - } - - if (tiisg < hc*hc) { - const ushort idst = tiisg & 3; - const ushort isrc = tiisg >> 2; - *(device float *) (dst + idst*args.nb_d0 + isrc*args.nb_d1 + it*args.nb_d2) = v; - } -} - -kernel void kernel_dsv4_hc_pre_f32( - constant ggml_metal_kargs_dsv4_hc_pre & args, - device const char * x, - device const char * weights, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - constexpr ushort hc = 4; - - const int it = tgpig.y; - const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; - - float weight_lane = 0.0f; - if (tiisg < hc) { - weight_lane = *(device const float *) (weights + tiisg*args.nb_w0 + it*args.nb_w1); - } - - float w[hc]; - FOR_UNROLL (ushort ih = 0; ih < hc; ++ih) { - w[ih] = simd_shuffle(weight_lane, ih); - } - - if (i0 >= args.n_embd) { - return; - } - - device const char * xb = x + i0*args.nb_x0 + it*args.nb_x2; - float result = 0.0f; - FOR_UNROLL (ushort ih = 0; ih < hc; ++ih) { - result = fma(*(device const float *) (xb + ih*args.nb_x1), w[ih], result); - } - - *(device float *) (dst + i0*args.nb_d0 + it*args.nb_d1) = result; -} - -kernel void kernel_dsv4_hc_post_f32( - constant ggml_metal_kargs_dsv4_hc_post & args, - device const char * x, - device const char * residual, - device const char * post, - device const char * comb, - device char * dst, - uint3 tgpig[[threadgroup_position_in_grid]], - ushort tiisg[[thread_index_in_simdgroup]], - ushort sgitg[[simdgroup_index_in_threadgroup]], - ushort3 ntg[[threads_per_threadgroup]]) { - constexpr ushort hc = 4; - - const int it = tgpig.y; - const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; - - float coeff_lane = 0.0f; - if (tiisg < hc) { - coeff_lane = *(device const float *) (post + tiisg*args.nb_p0 + it*args.nb_p1); - } else if (tiisg < hc + hc*hc) { - const ushort idx = tiisg - hc; - const ushort idst = idx & 3; - const ushort isrc = idx >> 2; - coeff_lane = *(device const float *) (comb + idst*args.nb_c0 + isrc*args.nb_c1 + it*args.nb_c2); - } - - float post_reg[hc]; - float comb_reg[hc][hc]; - FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { - post_reg[idst] = simd_shuffle(coeff_lane, idst); - } - FOR_UNROLL (ushort isrc = 0; isrc < hc; ++isrc) { - FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { - comb_reg[isrc][idst] = simd_shuffle(coeff_lane, hc + idst + hc*isrc); - } - } - - if (i0 >= args.n_embd) { - return; - } - - const float xv = *(device const float *) (x + i0*args.nb_x0 + it*args.nb_x1); - float result[hc]; - FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { - result[idst] = xv*post_reg[idst]; - } - - device const char * rb = residual + i0*args.nb_r0 + it*args.nb_r2; - FOR_UNROLL (ushort isrc = 0; isrc < hc; ++isrc) { - const float rv = *(device const float *) (rb + isrc*args.nb_r1); - FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { - result[idst] = fma(rv, comb_reg[isrc][idst], result[idst]); - } - } - - FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { - *(device float *) (dst + i0*args.nb_d0 + idst*args.nb_d1 + it*args.nb_d2) = result[idst]; - } -} diff --git a/ggml/src/ggml-metal/kernels/argsort.metal b/ggml/src/ggml-metal/kernels/argsort.metal new file mode 100644 index 00000000..5231b839 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/argsort.metal @@ -0,0 +1,479 @@ +#include "common.h" + +constant bool FC_topk_moe_with_norm [[function_constant(FC_TOPK_MOE + 0)]]; +constant int FC_topk_moe_n_expert [[function_constant(FC_TOPK_MOE + 1)]]; +constant int FC_topk_moe_top_k [[function_constant(FC_TOPK_MOE + 2)]]; + +constant int FC_moe_reduce_n_expert_used [[function_constant(FC_MOE_REDUCE + 0)]]; + +// bitonic sort implementation following the CUDA kernels as reference +typedef void (argsort_t)( + constant ggml_metal_kargs_argsort & args, + device const char * src0, + device int32_t * dst, + threadgroup int32_t * shmem_i32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]); + +template +kernel void kernel_argsort_f32_i32( + constant ggml_metal_kargs_argsort & args, + device const char * src0, + device int32_t * dst, + threadgroup int32_t * shmem_i32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + // bitonic sort + const int col = tpitg[0]; + const int ib = tgpig[0] / args.ne01; + + const int i00 = ib*ntg.x; + const int i01 = tgpig[0] % args.ne01; + const int i02 = tgpig[1]; + const int i03 = tgpig[2]; + + device const float * src0_row = (device const float *) (src0 + args.nb01*i01 + args.nb02*i02 + args.nb03*i03); + + // initialize indices + shmem_i32[col] = i00 + col; + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (int k = 2; k <= ntg.x; k *= 2) { + for (int j = k / 2; j > 0; j /= 2) { + int ixj = col ^ j; + if (ixj > col) { + if ((col & k) == 0) { + if (shmem_i32[col] >= args.ne00 || + (shmem_i32[ixj] < args.ne00 && (order == GGML_SORT_ORDER_ASC ? + src0_row[shmem_i32[col]] > src0_row[shmem_i32[ixj]] : + src0_row[shmem_i32[col]] < src0_row[shmem_i32[ixj]])) + ) { + SWAP(shmem_i32[col], shmem_i32[ixj]); + } + } else { + if (shmem_i32[ixj] >= args.ne00 || + (shmem_i32[col] < args.ne00 && (order == GGML_SORT_ORDER_ASC ? + src0_row[shmem_i32[col]] < src0_row[shmem_i32[ixj]] : + src0_row[shmem_i32[col]] > src0_row[shmem_i32[ixj]])) + ) { + SWAP(shmem_i32[col], shmem_i32[ixj]); + } + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + } + + const int64_t i0 = ib*args.top_k; + + // copy the result to dst without the padding + if (i0 + col < args.ne0 && col < args.top_k) { + dst += i0 + args.ne0*i01 + args.ne0*args.ne1*i02 + args.ne0*args.ne1*args.ne2*i03; + + dst[col] = shmem_i32[col]; + } +} + +template [[host_name("kernel_argsort_f32_i32_asc")]] kernel argsort_t kernel_argsort_f32_i32; +template [[host_name("kernel_argsort_f32_i32_desc")]] kernel argsort_t kernel_argsort_f32_i32; + +typedef void (argsort_merge_t)( + constant ggml_metal_kargs_argsort_merge & args, + device const char * src0, + device const int32_t * tmp, + device int32_t * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]); + +template +kernel void kernel_argsort_merge_f32_i32( + constant ggml_metal_kargs_argsort_merge & args, + device const char * src0, + device const int32_t * tmp, + device int32_t * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + const int im = tgpig[0] / args.ne01; + const int i01 = tgpig[0] % args.ne01; + const int i02 = tgpig[1]; + const int i03 = tgpig[2]; + + const int start = im * (2 * args.len); + + const int len0 = MIN(args.len, MAX(0, args.ne0 - (int)(start))); + const int len1 = MIN(args.len, MAX(0, args.ne0 - (int)(start + args.len))); + + const int total = len0 + len1; + + device const int32_t * tmp0 = tmp + start + + i01*args.ne0 + + i02*args.ne0*args.ne01 + + i03*args.ne0*args.ne01*args.ne02; + + device const int32_t * tmp1 = tmp0 + args.len; + + dst += start + + i01*args.top_k + + i02*args.top_k*args.ne01 + + i03*args.top_k*args.ne01*args.ne02; + + device const float * src0_row = (device const float *)(src0 + + args.nb01*i01 + + args.nb02*i02 + + args.nb03*i03); + + if (total == 0) { + return; + } + + const int chunk = (total + ntg.x - 1) / ntg.x; + + const int k0 = tpitg.x * chunk; + const int k1 = MIN(MIN(k0 + chunk, total), args.top_k); + + if (k0 >= args.top_k) { + return; + } + + if (k0 >= total) { + return; + } + + int low = k0 > len1 ? k0 - len1 : 0; + int high = MIN(k0, len0); + + // binary-search partition (i, j) such that i + j = k + while (low < high) { + const int mid = (low + high) >> 1; + + const int32_t idx0 = tmp0[mid]; + const int32_t idx1 = tmp1[k0 - mid - 1]; + + const float val0 = src0_row[idx0]; + const float val1 = src0_row[idx1]; + + bool take_left; + if (order == GGML_SORT_ORDER_ASC) { + take_left = (val0 <= val1); + } else { + take_left = (val0 >= val1); + } + + if (take_left) { + low = mid + 1; + } else { + high = mid; + } + } + + int i = low; + int j = k0 - i; + + // keep the merge fronts into registers + int32_t idx0 = 0; + float val0 = 0.0f; + if (i < len0) { + idx0 = tmp0[i]; + val0 = src0_row[idx0]; + } + + int32_t idx1 = 0; + float val1 = 0.0f; + if (j < len1) { + idx1 = tmp1[j]; + val1 = src0_row[idx1]; + } + + for (int k = k0; k < k1; ++k) { + int32_t out_idx; + + if (i >= len0) { + while (k < k1) { + dst[k++] = tmp1[j++]; + } + break; + } else if (j >= len1) { + while (k < k1) { + dst[k++] = tmp0[i++]; + } + break; + } else { + bool take_left; + + if (order == GGML_SORT_ORDER_ASC) { + take_left = (val0 <= val1); + } else { + take_left = (val0 >= val1); + } + + if (take_left) { + out_idx = idx0; + ++i; + if (i < len0) { + idx0 = tmp0[i]; + val0 = src0_row[idx0]; + } + } else { + out_idx = idx1; + ++j; + if (j < len1) { + idx1 = tmp1[j]; + val1 = src0_row[idx1]; + } + } + } + + dst[k] = out_idx; + } +} + +template [[host_name("kernel_argsort_merge_f32_i32_asc")]] kernel argsort_merge_t kernel_argsort_merge_f32_i32; +template [[host_name("kernel_argsort_merge_f32_i32_desc")]] kernel argsort_merge_t kernel_argsort_merge_f32_i32; + +static inline uint ggml_top_k_f2ui(float x) { + uint y = as_type(x); + if ((y & 0x80000000u) != 0u) { + y ^= 0xFFFFFFFFu; // negative floats: flip all bits + } else { + y |= 0x80000000u; // positive floats: set the sign bit + } + return y; +} + +kernel void kernel_top_k_f32_i32( + constant ggml_metal_kargs_top_k & args, + device const char * src0, + device int32_t * dst, + threadgroup atomic_uint * histo [[threadgroup(0)]], + threadgroup uint * sh_bucket [[threadgroup(1)]], + threadgroup uint * sh_above [[threadgroup(2)]], + threadgroup atomic_uint * out_count [[threadgroup(3)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + const uint ncols = args.ne00; + const uint top_k = args.top_k; + const uint i01 = tgpig[0]; + const uint i02 = tgpig[1]; + const uint i03 = tgpig[2]; + + device const float * src0_row = (device const float *) (src0 + args.nb01*i01 + args.nb02*i02 + args.nb03*i03); + + device int32_t * dst_row = dst + top_k*(i01 + args.ne01*i02 + args.ne01*args.ne02*i03); + + const uint tid = tpitg.x; + const uint ntg_x = ntg.x; + + uint prefix = 0; // fixed high bits of the threshold key + uint desired = top_k; // count still needed from the candidate range + + for (int shift = 24; shift >= 0; shift -= 8) { + for (uint i = tid; i < 256; i += ntg_x) { + atomic_store_explicit(&histo[i], 0u, memory_order_relaxed); + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + const uint hi_mask = (shift + 8 >= 32) ? 0u : (0xFFFFFFFFu << uint(shift + 8)); + const uint prefix_hi = prefix & hi_mask; + + for (uint i = tid; i < ncols; i += ntg_x) { + const uint key = ggml_top_k_f2ui(src0_row[i]); + if ((key & hi_mask) == prefix_hi) { + atomic_fetch_add_explicit(&histo[(key >> uint(shift)) & 0xFFu], 1u, memory_order_relaxed); + } + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + // top-down scan for the bucket holding the k-th value + if (tid == 0) { + uint acc = 0; + uint b = 0; + for (int bb = 255; bb >= 0; --bb) { + const uint c = atomic_load_explicit(&histo[bb], memory_order_relaxed); + if (acc + c >= desired) { + b = uint(bb); + break; + } + acc += c; + } + *sh_bucket = b; + *sh_above = acc; + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + prefix |= *sh_bucket << uint(shift); + desired -= *sh_above; + + // ensure every thread has consumed sh_bucket/sh_above before the next pass + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + if (tid == 0) { + atomic_store_explicit(out_count, 0u, memory_order_relaxed); + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + // emit everything above the threshold, then fill the rest from ties + const uint threshold = prefix; + + for (uint i = tid; i < ncols; i += ntg_x) { + if (ggml_top_k_f2ui(src0_row[i]) > threshold) { + const uint pos = atomic_fetch_add_explicit(out_count, 1u, memory_order_relaxed); + dst_row[pos] = (int32_t) i; + } + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (uint i = tid; i < ncols; i += ntg_x) { + if (ggml_top_k_f2ui(src0_row[i]) == threshold) { + const uint pos = atomic_fetch_add_explicit(out_count, 1u, memory_order_relaxed); + if (pos < top_k) { + dst_row[pos] = (int32_t) i; + } + } + } +} + +// fused SOFT_MAX + top-k + GET_ROWS (+ optional norm/scale) for MoE routing. +// One SIMDgroup handles one token row; n_expert is limited to 1024 by the host. +kernel void kernel_topk_moe_f32( + constant ggml_metal_kargs_topk_moe & args, + device const char * src0, + device float * weights, + device int32_t * ids, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]]) { + const int row = (int) tgpig.x; + if (row >= args.ne01) { + return; + } + + const int n_expert = FC_topk_moe_n_expert; + const int top_k = FC_topk_moe_top_k; + const int lane = (int) tiisg; + const int n_per_lane = (n_expert + 31) / 32; + + device const float * logits_row = (device const float *) (src0 + row * args.nb01); + device float * weights_row = weights + row * top_k; + device int32_t * ids_row = ids + row * (args.nb1_ids / sizeof(int32_t)); + + float wt[32]; + float output_weights[32]; + FOR_UNROLL (int i = 0; i < 32; ++i) { + wt[i] = -INFINITY; + output_weights[i] = 0.0f; + } + + for (int i = lane; i < n_expert; i += 32) { + const float v = logits_row[i]; + wt[i / 32] = isnan(v) ? -FLT_MAX : v; + } + + // softmax over the expert logits + float max_val = -INFINITY; + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + max_val = max(max_val, wt[i]); + } + max_val = simd_max(max_val); + + float sum_val = 0.0f; + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + wt[i] = exp(wt[i] - max_val); + sum_val += wt[i]; + } + sum_val = simd_sum(sum_val); + + const float inv_sum = 1.0f / sum_val; + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + wt[i] *= inv_sum; + } + + float wt_sum = 0.0f; + + for (int k = 0; k < top_k; ++k) { + float best_val = -INFINITY; + int best_expert = -1; + + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + const int expert = lane + i * 32; + if (expert < n_expert && (wt[i] > best_val || (wt[i] == best_val && expert < best_expert))) { + best_val = wt[i]; + best_expert = expert; + } + } + + FOR_UNROLL (int mask = 16; mask > 0; mask >>= 1) { + const float val = simd_shuffle_xor(best_val, mask); + const int expert = simd_shuffle_xor(best_expert, mask); + if (val > best_val || (val == best_val && expert < best_expert)) { + best_val = val; + best_expert = expert; + } + } + + if ((best_expert & 31) == lane) { + wt[best_expert / 32] = -INFINITY; + } + + if ((k & 31) == lane) { + output_weights[k / 32] = best_val; + } + + if ((best_expert & 31) == lane) { + ids_row[k] = best_expert; + if (FC_topk_moe_with_norm) { + wt_sum += best_val; + } + } + } + + if (FC_topk_moe_with_norm) { + wt_sum = simd_sum(wt_sum); + wt_sum = max(wt_sum, args.clamp); + const float inv = 1.0f / wt_sum; + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + output_weights[i] *= inv; + } + } + + FOR_UNROLL (int i = 0; i < n_per_lane; ++i) { + const int idx = i * 32 + lane; + if (idx < top_k) { + weights_row[idx] = output_weights[i] * args.scale; + } + } +} + +// fused MoE expert weighting + reduction: weighted = sum(experts[e] * weights[e]). +// The host guarantees all tensors are contiguous F32. +kernel void kernel_moe_reduce_f32( + constant ggml_metal_kargs_moe_reduce & args, + device const float * experts, + device const float * weights, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int64_t token = tgpig.x; + const int64_t col = (int64_t) tgpig.y * ntg.x + tpitg.x; + if (token >= args.ne02 || col >= args.ne00) { + return; + } + + const int n_expert_used = FC_moe_reduce_n_expert_used; + + const int64_t base = token * (int64_t) n_expert_used * args.ne00 + col; + float sum = 0.0f; + FOR_UNROLL (int e = 0; e < n_expert_used; ++e) { + sum += experts[base + e * args.ne00] * weights[token * n_expert_used + e]; + } + dst[token * args.ne00 + col] = sum; +} diff --git a/ggml/src/ggml-metal/kernels/binbcast.metal b/ggml/src/ggml-metal/kernels/binbcast.metal new file mode 100644 index 00000000..7c7ab9b5 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/binbcast.metal @@ -0,0 +1,228 @@ +#include "common.h" + +// OP: 0 - add, 1 - sub, 2 - mul, 3 - div +constant short FC_bin_op [[function_constant(FC_BIN + 0)]]; +constant short FC_bin_f [[function_constant(FC_BIN + 1)]]; +constant bool FC_bin_rb [[function_constant(FC_BIN + 2)]]; +constant bool FC_bin_cb [[function_constant(FC_BIN + 3)]]; + +template +kernel void kernel_bin_fuse_impl( + constant ggml_metal_kargs_bin & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { +#define FC_OP FC_bin_op +#define FC_F FC_bin_f +#define FC_RB FC_bin_rb +#define FC_CB FC_bin_cb + + if (FC_RB) { + // row broadcast + const uint i0 = tgpig.y*args.ne00 + tgpig.x; + const uint i1 = FC_CB ? tgpig.x%args.ne10 : tgpig.x; + + device const T0 * src0_row = (device const T0 *) (src0); + device T * dst_row = (device T *) (dst); + + if (FC_F == 1) { + device const T1 * src1_row = (device const T1 *) (src1 + args.o1[0]); + + if (FC_OP == 0) { + dst_row[i0] = src0_row[i0] + src1_row[i1]; + } + + if (FC_OP == 1) { + dst_row[i0] = src0_row[i0] - src1_row[i1]; + } + + if (FC_OP == 2) { + dst_row[i0] = src0_row[i0] * src1_row[i1]; + } + + if (FC_OP == 3) { + dst_row[i0] = src0_row[i0] / src1_row[i1]; + } + } else { + T0 res = src0_row[i0]; + + if (FC_OP == 0) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res += ((device const T1 *) (src1 + args.o1[j]))[i1]; + } + } + + if (FC_OP == 1) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res -= ((device const T1 *) (src1 + args.o1[j]))[i1]; + } + } + + if (FC_OP == 2) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res *= ((device const T1 *) (src1 + args.o1[j]))[i1]; + } + } + + if (FC_OP == 3) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res /= ((device const T1 *) (src1 + args.o1[j]))[i1]; + } + } + + dst_row[i0] = res; + } + } else { + const int i03 = tgpig.z; + const int i02 = tgpig.y; + const int i01 = tgpig.x; + + if (i01 >= args.ne01) { + return; + } + + const int i13 = i03%args.ne13; + const int i12 = i02%args.ne12; + const int i11 = i01%args.ne11; + + device const T0 * src0_ptr = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + args.offs); + device T * dst_ptr = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1 + args.offs); + + if (FC_F == 1) { + device const T1 * src1_ptr = (device const T1 *) (src1 + args.o1[0] + i13*args.nb13 + i12*args.nb12 + i11*args.nb11); + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const int i10 = FC_CB ? i0%args.ne10 : i0; + + if (FC_OP == 0) { + dst_ptr[i0] = src0_ptr[i0] + src1_ptr[i10]; + } + + if (FC_OP == 1) { + dst_ptr[i0] = src0_ptr[i0] - src1_ptr[i10]; + } + + if (FC_OP == 2) { + dst_ptr[i0] = src0_ptr[i0] * src1_ptr[i10]; + } + + if (FC_OP == 3) { + dst_ptr[i0] = src0_ptr[i0] / src1_ptr[i10]; + } + } + } else { + device const T1 * src1_ptr[8]; + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + src1_ptr[j] = (device const T1 *) (src1 + args.o1[j] + i13*args.nb13 + i12*args.nb12 + i11*args.nb11); + } + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const int i10 = FC_CB ? i0%args.ne10 : i0; + + T res = src0_ptr[i0]; + + if (FC_OP == 0) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res += src1_ptr[j][i10]; + } + } + + if (FC_OP == 1) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res -= src1_ptr[j][i10]; + } + } + + if (FC_OP == 2) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res *= src1_ptr[j][i10]; + } + } + + if (FC_OP == 3) { + FOR_UNROLL (short j = 0; j < FC_F; ++j) { + res /= src1_ptr[j][i10]; + } + } + + dst_ptr[i0] = res; + } + } + } + +#undef FC_OP +#undef FC_F +#undef FC_RB +#undef FC_CB +} + +typedef decltype(kernel_bin_fuse_impl) kernel_bin_fuse_t; + +template [[host_name("kernel_bin_fuse_f32_f32_f32")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; +template [[host_name("kernel_bin_fuse_f32_f32_f32_4")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; +template [[host_name("kernel_bin_fuse_f16_f16_f16")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; +template [[host_name("kernel_bin_fuse_f16_f16_f16_4")]] kernel kernel_bin_fuse_t kernel_bin_fuse_impl; + +kernel void kernel_add_id( + constant ggml_metal_kargs_add_id & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int i1 = tgpig.x; + const int i2 = tgpig.y; + + const int i11 = *((device const int32_t *) (src2 + i1*sizeof(int32_t) + i2*args.nb21)); + + const size_t nb1 = args.ne0 * sizeof(float); + const size_t nb2 = args.ne1 * nb1; + + device float * dst_row = (device float *)((device char *)dst + i1*nb1 + i2*nb2); + device const float * src0_row = (device const float *)((device char *)src0 + i1*args.nb01 + i2*args.nb02); + device const float * src1_row = (device const float *)((device char *)src1 + i11*args.nb11); + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + dst_row[i0] = src0_row[i0] + src1_row[i0]; + } +} + +template +kernel void kernel_repeat( + constant ggml_metal_kargs_repeat & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = tgpig.x; + + const int i03 = i3%args.ne03; + const int i02 = i2%args.ne02; + const int i01 = i1%args.ne01; + + device const char * src0_ptr = src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01; + device char * dst_ptr = dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1; + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const int i00 = i0%args.ne00; + *((device T *)(dst_ptr + i0*args.nb0)) = *((device T *)(src0_ptr + i00*args.nb00)); + } +} + +typedef decltype(kernel_repeat) kernel_repeat_t; + +template [[host_name("kernel_repeat_f32")]] kernel kernel_repeat_t kernel_repeat; +template [[host_name("kernel_repeat_f16")]] kernel kernel_repeat_t kernel_repeat; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_repeat_bf16")]] kernel kernel_repeat_t kernel_repeat; +#endif +template [[host_name("kernel_repeat_i32")]] kernel kernel_repeat_t kernel_repeat; +template [[host_name("kernel_repeat_i16")]] kernel kernel_repeat_t kernel_repeat; diff --git a/ggml/src/ggml-metal/kernels/common.h b/ggml/src/ggml-metal/kernels/common.h new file mode 100644 index 00000000..c4d67439 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/common.h @@ -0,0 +1,126 @@ +#pragma once + +#include "ggml-metal-impl.h" + +#include + +#ifdef GGML_METAL_HAS_TENSOR +#include + +#include +#endif + +using namespace metal; + +#define MAX(x, y) ((x) > (y) ? (x) : (y)) +#define MIN(x, y) ((x) < (y) ? (x) : (y)) +#define SWAP(x, y) { auto tmp = (x); (x) = (y); (y) = tmp; } + +#define PAD2(x, n) (((x) + (n) - 1) & ~((n) - 1)) + +#define FOR_UNROLL(x) _Pragma("clang loop unroll(full)") for (x) + +#define N_SIMDWIDTH 32 // assuming SIMD group size is 32 + +// ref: https://developer.apple.com/metal/Metal-Shading-Language-Specification.pdf +// +// cmd: +// .../usr/bin/metal -dM -E -c ggml/src/ggml-metal/kernels/.metal +// .../usr/bin/metal -dM -E -c -target air64-apple-ios14.0 ggml/src/ggml-metal/kernels/.metal +// +#if __METAL_VERSION__ < 310 && defined(GGML_METAL_HAS_BF16) +#undef GGML_METAL_HAS_BF16 +#endif + +#if defined(GGML_METAL_HAS_BF16) +typedef matrix bfloat4x4; +typedef matrix bfloat2x4; +#endif + +constexpr constant static float kvalues_iq4nl_f[16] = { + -127.f, -104.f, -83.f, -65.f, -49.f, -35.f, -22.f, -10.f, 1.f, 13.f, 25.f, 38.f, 53.f, 69.f, 89.f, 113.f +}; + +constexpr constant static float kvalues_mxfp4_f[16] = { + 0, .5f, 1.f, 1.5f, 2.f, 3.f, 4.f, 6.f, -0, -.5f, -1.f, -1.5f, -2.f, -3.f, -4.f, -6.f +}; + +static inline int best_index_int8(int n, constant float * val, float x) { + if (x <= val[0]) return 0; + if (x >= val[n-1]) return n-1; + int ml = 0, mu = n-1; + while (mu-ml > 1) { + int mav = (ml+mu)/2; + if (x < val[mav]) mu = mav; else ml = mav; + } + return x - val[mu-1] < val[mu] - x ? mu-1 : mu; +} + +static inline float e8m0_to_fp32(uint8_t x) { + uint32_t bits; + + if (x == 0) { + bits = 0x00400000; + } else { + bits = (uint32_t) x << 23; + } + + return as_type(bits); +} + +static inline float dot(float x, float y) { + return x*y; +} + +static inline float sum(float x) { + return x; +} + +static inline float sum(float4 x) { + return x[0] + x[1] + x[2] + x[3]; +} + +enum ggml_sort_order { + GGML_SORT_ORDER_ASC, + GGML_SORT_ORDER_DESC, +}; + +constant float GELU_COEF_A = 0.044715f; +constant float GELU_QUICK_COEF = -1.702f; +constant float SQRT_2_OVER_PI = 0.79788456080286535587989211986876f; +constant float SQRT_2_INV = 0.70710678118654752440084436210484f; + +// based on Abramowitz and Stegun formula 7.1.26 or similar Hastings' approximation +// ref: https://www.johndcook.com/blog/python_erf/ +constant float p_erf = 0.3275911f; +constant float a1_erf = 0.254829592f; +constant float a2_erf = -0.284496736f; +constant float a3_erf = 1.421413741f; +constant float a4_erf = -1.453152027f; +constant float a5_erf = 1.061405429f; + +template +inline T erf_approx(T x) { + T sign_x = sign(x); + x = fabs(x); + T t = 1.0f / (1.0f + p_erf * x); + T y = 1.0f - (((((a5_erf * t + a4_erf) * t) + a3_erf) * t + a2_erf) * t + a1_erf) * t * exp(-x * x); + return sign_x * y; +} + +template T elu_approx(T x); + +template<> inline float elu_approx(float x) { + return (x > 0.f) ? x : (exp(x) - 1); +} + +template<> inline float4 elu_approx(float4 x) { + float4 res; + + res[0] = (x[0] > 0.0f) ? x[0] : (exp(x[0]) - 1.0f); + res[1] = (x[1] > 0.0f) ? x[1] : (exp(x[1]) - 1.0f); + res[2] = (x[2] > 0.0f) ? x[2] : (exp(x[2]) - 1.0f); + res[3] = (x[3] > 0.0f) ? x[3] : (exp(x[3]) - 1.0f); + + return res; +} diff --git a/ggml/src/ggml-metal/kernels/conv.metal b/ggml/src/ggml-metal/kernels/conv.metal new file mode 100644 index 00000000..a5d5aa9d --- /dev/null +++ b/ggml/src/ggml-metal/kernels/conv.metal @@ -0,0 +1,724 @@ +#include "common.h" + +typedef void (im2col_t)( + constant ggml_metal_kargs_im2col & args, + device const float * x, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template +kernel void kernel_im2col( + constant ggml_metal_kargs_im2col & args, + device const float * x, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { +// const int64_t IC = tgpg[0]; + const int64_t OH = tgpg[1]; + const int64_t OW = tgpg[2]; + + const int64_t KH = ntg[1]; + const int64_t KW = ntg[2]; + + int64_t in = tpitg[0]; + const int64_t ikh = tpitg[1]; + const int64_t ikw = tpitg[2]; + + const int64_t iic = tgpig[0]; + const int64_t ioh = tgpig[1]; + const int64_t iow = tgpig[2]; + + const int64_t iiw = iow*args.s0 + ikw*args.d0 - args.p0; + const int64_t iih = ioh*args.s1 + ikh*args.d1 - args.p1; + + int64_t offset_dst = (in*OH*OW + ioh*OW + iow)*args.CHW + (iic*(KH*KW) + ikh*KW + ikw); + + device T * pdst = (device T *) (dst); + + if (iih < 0 || iih >= args.IH || iiw < 0 || iiw >= args.IW) { + while (in < args.N) { + pdst[offset_dst] = 0.0f; + offset_dst += ntg[0]*args.CHW*OH*OW; + + in += ntg[0]; + } + } else { + int64_t offset_src = in*args.ofs0 + iic*args.ofs1 + iih*args.IW + iiw; + + while (in < args.N) { + pdst[offset_dst] = x[offset_src]; + + offset_dst += ntg[0]*args.CHW*OH*OW; + offset_src += ntg[0]*args.ofs0; + + in += ntg[0]; + } + } +} + +template [[host_name("kernel_im2col_f32")]] kernel im2col_t kernel_im2col; +template [[host_name("kernel_im2col_f16")]] kernel im2col_t kernel_im2col; + +// TODO: optimize +typedef void (im2col_ext_t)( + constant ggml_metal_kargs_im2col & args, + device const float * x, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template +kernel void kernel_im2col_ext( + constant ggml_metal_kargs_im2col & args, + device const float * x, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], // tgpg[0] = D x IC x KH x KW, CHW = IC x KH x KW + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { // [M, 1, 1] + const int64_t KHW = (int64_t)args.KHW; + + const int64_t d = tgpig[0] / args.CHW; + const int64_t chw = tgpig[0] % args.CHW; + const int64_t tgpig_0 = chw / KHW; // 0 ~ (IC - 1) + const int64_t HW = tgpig[0] % KHW; + + const int64_t tpitg_0 = (d * ntg[0]) + tpitg[0]; + if (tpitg_0 >= args.N) { + return; + } + + const int64_t tpitg_1 = HW / args.KW; + const int64_t tpitg_2 = HW % args.KW; + + const int64_t iiw = tgpig[2] * args.s0 + tpitg_2 * args.d0 - args.p0; + const int64_t iih = tgpig[1] * args.s1 + tpitg_1 * args.d1 - args.p1; + + const int64_t offset_dst = + (tpitg_0 * tgpg[1] * tgpg[2] + tgpig[1] * tgpg[2] + tgpig[2]) * args.CHW + + (tgpig_0 * KHW + tpitg_1 * args.KW + tpitg_2); + + device T * pdst = (device T *) (dst); + + if (iih < 0 || iih >= args.IH || iiw < 0 || iiw >= args.IW) { + pdst[offset_dst] = 0.0f; + } else { + const int64_t offset_src = tpitg_0 * args.ofs0 + tgpig_0 * args.ofs1; + pdst[offset_dst] = x[offset_src + iih * args.IW + iiw]; + } +} + +template [[host_name("kernel_im2col_ext_f32")]] kernel im2col_ext_t kernel_im2col_ext; +template [[host_name("kernel_im2col_ext_f16")]] kernel im2col_ext_t kernel_im2col_ext; + +template +kernel void kernel_col2im_1d( + constant ggml_metal_kargs_col2im_1d & args, + device const T * col, + device T * dst, + uint tgpig [[threadgroup_position_in_grid]], + uint tpitg [[thread_position_in_threadgroup]], + uint ntg [[threads_per_threadgroup]]) { + + const int idx = tgpig * ntg + tpitg; + if (idx >= args.T_out * args.OC) { + return; + } + + const int t_out = idx % args.T_out; + const int oc = idx / args.T_out; + const int t_abs = t_out + args.p0; // absolute position in uncropped signal + + int t_in_min = (t_abs - args.K + args.s0) / args.s0; // ceil((t_abs - K + 1) / s0) + if (t_in_min < 0) { + t_in_min = 0; + } + int t_in_max = t_abs / args.s0; + if (t_in_max >= args.T_in) { + t_in_max = args.T_in - 1; + } + + float sum = 0.0f; + for (int t_in = t_in_min; t_in <= t_in_max; t_in++) { + const int k = t_abs - t_in * args.s0; + sum += float(col[(oc * args.K + k) + t_in * args.K_OC]); + } + + dst[t_out + oc * args.T_out] = T(sum); +} + +template [[host_name("kernel_col2im_1d_f32")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const float *, device float *, uint, uint, uint); +template [[host_name("kernel_col2im_1d_f16")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const half *, device half *, uint, uint, uint); +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_col2im_1d_bf16")]] kernel void kernel_col2im_1d(constant ggml_metal_kargs_col2im_1d &, device const bfloat *, device bfloat *, uint, uint, uint); +#endif + +template +kernel void kernel_conv_2d( + constant ggml_metal_kargs_conv_2d & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const uint threads_per_tg = ntg.x * ntg.y * ntg.z; + const uint tg_index = (tgpig.z * tgpg.y + tgpig.y) * tgpg.x + tgpig.x; + const uint local_thread = tpitg.z * (ntg.x * ntg.y) + tpitg.y * ntg.x + tpitg.x; + const uint thread_index = tg_index * threads_per_tg + local_thread; + const uint64_t total_threads = (uint64_t) threads_per_tg * tgpg.x * tgpg.y * tgpg.z; + const uint64_t total_outputs = (uint64_t) args.N * args.OC * args.OH * args.OW; + + for (uint64_t index = thread_index; index < total_outputs; index += total_threads) { + uint64_t tmp = index; + + const int32_t ow = tmp % args.OW; tmp /= args.OW; + const int32_t oh = tmp % args.OH; tmp /= args.OH; + const int32_t oc = tmp % args.OC; tmp /= args.OC; + const int32_t n = tmp; + + float acc = 0.0f; + + const int32_t base_x = ow*args.s0 - args.p0; + const int32_t base_y = oh*args.s1 - args.p1; + + int32_t ky_start = 0; + if (base_y < 0) { + ky_start = (-base_y + args.d1 - 1)/args.d1; + } + int32_t ky_end = args.KH; + const int32_t y_max = args.IH - 1 - base_y; + if (y_max < 0) { + ky_end = ky_start; + } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { + ky_end = min(ky_end, y_max/args.d1 + 1); + } + + int32_t kx_start = 0; + if (base_x < 0) { + kx_start = (-base_x + args.d0 - 1)/args.d0; + } + int32_t kx_end = args.KW; + const int32_t x_max = args.IW - 1 - base_x; + if (x_max < 0) { + kx_end = kx_start; + } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { + kx_end = min(kx_end, x_max/args.d0 + 1); + } + + if (ky_start < ky_end && kx_start < kx_end) { + const uint64_t src_base_n = (uint64_t) n * args.nb13; + const uint64_t w_base_oc = (uint64_t) oc * args.nb03; + + for (int32_t ic = 0; ic < args.IC; ++ic) { + const uint64_t src_base_nc = src_base_n + (uint64_t) ic * args.nb12; + const uint64_t w_base_ocic = w_base_oc + (uint64_t) ic * args.nb02; + + for (int32_t ky = ky_start; ky < ky_end; ++ky) { + const int32_t iy = base_y + ky*args.d1; + const uint64_t src_base_row = src_base_nc + (uint64_t) iy * args.nb11; + const uint64_t w_base_row = w_base_ocic + (uint64_t) ky * args.nb01; + + for (int32_t kx = kx_start; kx < kx_end; ++kx) { + const int32_t ix = base_x + kx*args.d0; + const uint64_t src_offs = src_base_row + (uint64_t) ix * args.nb10; + const uint64_t w_offs = w_base_row + (uint64_t) kx * args.nb00; + + const float x = *(device const float *)(src + src_offs); + const float w = (float) (*(device const TK *)(weights + w_offs)); + + acc += x * w; + } + } + } + } + + const uint64_t dst_offs = + (uint64_t) n * args.nb3 + + (uint64_t) oc * args.nb2 + + (uint64_t) oh * args.nb1 + + (uint64_t) ow * args.nb0; + + *(device float *)(dst + dst_offs) = acc; + } +} + +template [[host_name("kernel_conv_2d_f32_f32")]] +kernel void kernel_conv_2d( + constant ggml_metal_kargs_conv_2d & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template [[host_name("kernel_conv_2d_f16_f32")]] +kernel void kernel_conv_2d( + constant ggml_metal_kargs_conv_2d & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +typedef void (conv_transpose_1d_t)( + constant ggml_metal_kargs_conv_transpose_1d & args, + device const float * src0, + device const float * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]]); + +template +kernel void kernel_conv_transpose_1d( + constant ggml_metal_kargs_conv_transpose_1d & args, + device const T * src0, + device const float * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]]) { + + // For output position j on the time axis, only input positions + // i such that i*s0 <= j < i*s0 + K + // contribute -- i.e. i in [ceil((j - K + 1)/s0), floor(j/s0)] + // intersected with [0, IL-1]. That's at most ceil(K/s0) values + // (typically 2 for stride==K/2 transposed convs). + const int32_t j = tgpig[0]; + const int32_t s0 = args.s0; + const int32_t K = args.K; + const int32_t IL = args.IL; + + int32_t i_min; + { + int32_t a = j - K + 1; + i_min = a <= 0 ? 0 : (a + s0 - 1) / s0; // ceil(a/s0) for a>0 + } + int32_t i_max = j / s0; + if (i_max > IL - 1) i_max = IL - 1; + + float v = 0.0f; + if (i_min <= i_max) { + for (int64_t c = 0; c < args.IC; c++) { + const int32_t kernel_offset = c * tgpg[1] * K + K * tgpig[1]; + const int32_t input_offset = c * IL; + + for (int32_t i = i_min; i <= i_max; i++) { + v += float(src0[kernel_offset + j - i * s0]) * src1[input_offset + i]; + } + } + } + + device float * dst_ptr = (device float *) (dst + tgpig[0] * args.nb0 + tgpig[1] * args.nb1); + + dst_ptr[0] = v; +} + +template [[host_name("kernel_conv_transpose_1d_f32_f32")]] +kernel void kernel_conv_transpose_1d( + constant ggml_metal_kargs_conv_transpose_1d & args, + device const float * src0, + device const float * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]]); + +template [[host_name("kernel_conv_transpose_1d_f16_f32")]] +kernel void kernel_conv_transpose_1d( + constant ggml_metal_kargs_conv_transpose_1d & args, + device const half * src0, + device const float * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]]); + + +typedef void (conv_transpose_2d_t)( + constant ggml_metal_kargs_conv_transpose_2d & args, + device const float * src0, + device const float * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]]); + +template +kernel void kernel_conv_transpose_2d( + constant ggml_metal_kargs_conv_transpose_2d & args, + device const T * src0, + device const float * src1, + device char * dst, + threadgroup float * shared_sum [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t out_x = tgpig[0]; + const int64_t out_y = tgpig[1]; + const int64_t batch = tgpig[2] / args.OC; + const int64_t out_c = tgpig[2] % args.OC; + + const int64_t kw = tpitg[0]; + const int64_t kh = tpitg[1]; + + float v = 0.0f; + + for (int64_t in_c = 0; in_c < args.IC; in_c++) { + int64_t in_y = out_y - kh; + + if (in_y < 0 || in_y % args.s0) continue; + + in_y /= args.s0; + + if (in_y >= args.IH) continue; + + int64_t in_x = out_x - kw; + + if (in_x < 0 || in_x % args.s0) continue; + + in_x /= args.s0; + + if (in_x >= args.IW) continue; + + const int64_t input_idx = (args.IW * args.IH) * (args.IC * batch + in_c) + (args.IW) * in_y + in_x; + const int64_t kernel_idx = (args.KH * args.KW * args.OC) * in_c + (args.KH * args.KW) * out_c + (args.KW) * kh + kw; + + v += (float)src0[kernel_idx] * src1[input_idx]; + } + + const uint tid = tpitg.y * ntg.x + tpitg.x; + shared_sum[tid] = v; + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tid == 0) { + float total = 0.0f; + const uint num_threads = ntg.x * ntg.y; + for (uint i = 0; i < num_threads; i++) { + total += shared_sum[i]; + } + + device float * dst_ptr = (device float *) (dst + batch*args.nb3 + out_c*args.nb2 + out_y * args.nb1 + out_x*args.nb0); + dst_ptr[0] = total; + } +} + +template [[host_name("kernel_conv_transpose_2d_f32_f32")]] +kernel void kernel_conv_transpose_2d( + constant ggml_metal_kargs_conv_transpose_2d & args, + device const float * src0, + device const float * src1, + device char * dst, + threadgroup float * shared_sum [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template [[host_name("kernel_conv_transpose_2d_f16_f32")]] +kernel void kernel_conv_transpose_2d( + constant ggml_metal_kargs_conv_transpose_2d & args, + device const half * src0, + device const float * src1, + device char * dst, + threadgroup float * shared_sum [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +// grid: x = C tile, y = OH, z = OW * N (for channel-contiguous layouts) +template +kernel void kernel_conv_2d_dw_tiled( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int32_t c = (int32_t)(tgpig.x * ntg.x + tpitg.x); + if (c >= args.C) { + return; + } + + const int32_t oh = tgpig.y; + const int32_t own = tgpig.z; + const int32_t ow = own % args.OW; + const int32_t n = own / args.OW; + + const int32_t base_y = oh*args.s1 - args.p1; + + int32_t ky_start = 0; + if (base_y < 0) { + ky_start = (-base_y + args.d1 - 1)/args.d1; + } + int32_t ky_end = args.KH; + const int32_t y_max = args.IH - 1 - base_y; + if (y_max < 0) { + ky_end = ky_start; + } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { + ky_end = min(ky_end, y_max/args.d1 + 1); + } + + const int32_t base_x = ow*args.s0 - args.p0; + + int32_t kx_start = 0; + if (base_x < 0) { + kx_start = (-base_x + args.d0 - 1)/args.d0; + } + int32_t kx_end = args.KW; + const int32_t x_max = args.IW - 1 - base_x; + if (x_max < 0) { + kx_end = kx_start; + } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { + kx_end = min(kx_end, x_max/args.d0 + 1); + } + + float acc = 0.0f; + + if (ky_start < ky_end && kx_start < kx_end) { + const uint64_t w_base = (uint64_t) c * args.nb02; + const uint64_t src_base = (uint64_t) n * args.nb13 + (uint64_t) c * args.nb12; + + for (int32_t ky = ky_start; ky < ky_end; ++ky) { + const int32_t iy = base_y + ky*args.d1; + const uint64_t src_row = src_base + (uint64_t) iy * args.nb11; + const uint64_t w_row = w_base + (uint64_t) ky * args.nb01; + + for (int32_t kx = kx_start; kx < kx_end; ++kx) { + const int32_t ix = base_x + kx*args.d0; + const float x = *(device const float *)(src + src_row + (uint64_t) ix * args.nb10); + const float w = (float)(*(device const TK *)(weights + w_row + (uint64_t) kx * args.nb00)); + acc += x * w; + } + } + } + + const uint64_t dst_offs = + (uint64_t) n * args.nb3 + + (uint64_t) c * args.nb2 + + (uint64_t) oh * args.nb1 + + (uint64_t) ow * args.nb0; + + *(device float *)(dst + dst_offs) = acc; +} + +// grid: x = OW tile, y = OH, z = C * N (for spatially-contiguous layouts) +template +kernel void kernel_conv_2d_dw( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int32_t oh = tgpig.y; + const int32_t cn = tgpig.z; + const int32_t c = cn % args.C; + const int32_t n = cn / args.C; + + const int32_t base_y = oh*args.s1 - args.p1; + + int32_t ky_start = 0; + if (base_y < 0) { + ky_start = (-base_y + args.d1 - 1)/args.d1; + } + int32_t ky_end = args.KH; + const int32_t y_max = args.IH - 1 - base_y; + if (y_max < 0) { + ky_end = ky_start; + } else if (base_y + (args.KH - 1)*args.d1 >= args.IH) { + ky_end = min(ky_end, y_max/args.d1 + 1); + } + + const uint64_t w_base = (uint64_t) c * args.nb02; + const uint64_t src_base = (uint64_t) n * args.nb13 + (uint64_t) c * args.nb12; + + const int32_t ow = (int32_t)(tgpig.x * ntg.x + tpitg.x); + if (ow >= args.OW) { + return; + } + + float acc = 0.0f; + + const int32_t base_x = ow*args.s0 - args.p0; + + int32_t kx_start = 0; + if (base_x < 0) { + kx_start = (-base_x + args.d0 - 1)/args.d0; + } + int32_t kx_end = args.KW; + const int32_t x_max = args.IW - 1 - base_x; + if (x_max < 0) { + kx_end = kx_start; + } else if (base_x + (args.KW - 1)*args.d0 >= args.IW) { + kx_end = min(kx_end, x_max/args.d0 + 1); + } + + if (ky_start < ky_end && kx_start < kx_end) { + for (int32_t ky = ky_start; ky < ky_end; ++ky) { + const int32_t iy = base_y + ky*args.d1; + const uint64_t src_row = src_base + (uint64_t) iy * args.nb11; + const uint64_t w_row = w_base + (uint64_t) ky * args.nb01; + + for (int32_t kx = kx_start; kx < kx_end; ++kx) { + const int32_t ix = base_x + kx*args.d0; + const float x = *(device const float *)(src + src_row + (uint64_t) ix * args.nb10); + const float w = (float)(*(device const TK *)(weights + w_row + (uint64_t) kx * args.nb00)); + acc += x * w; + } + } + } + + const uint64_t dst_offs = + (uint64_t) n * args.nb3 + + (uint64_t) c * args.nb2 + + (uint64_t) oh * args.nb1 + + (uint64_t) ow * args.nb0; + + *(device float *)(dst + dst_offs) = acc; +} + +template [[host_name("kernel_conv_2d_dw_f32_f32")]] +kernel void kernel_conv_2d_dw( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template [[host_name("kernel_conv_2d_dw_f16_f32")]] +kernel void kernel_conv_2d_dw( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template [[host_name("kernel_conv_2d_dw_tiled_f32_f32")]] +kernel void kernel_conv_2d_dw_tiled( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template [[host_name("kernel_conv_2d_dw_tiled_f16_f32")]] +kernel void kernel_conv_2d_dw_tiled( + constant ggml_metal_kargs_conv_2d_dw & args, + device const char * weights, + device const char * src, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]); + +template +kernel void kernel_conv_3d( + constant ggml_metal_kargs_conv_3d & args, + device const char * src0, // Weights [IC * OC, KD, KH, KW] + device const char * src1, // Inputs [IC * N, ID, IH, IW] + device char * dst, // Outputs [OC * N, OD, OH, OW] + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]]) { + + // 1. Un-flatten the spatial dimension from Grid X + int64_t spatial_idx = tgpig.x * 32 + tpitg.x; + + if (spatial_idx >= args.OW * args.OH * args.OD) { + return; // Thread falls outside the spatial volume + } + + int64_t od = spatial_idx / (args.OW * args.OH); + int64_t oh = (spatial_idx / args.OW) % args.OH; + int64_t ow = spatial_idx % args.OW; + + // 2. Map Y to Channels, Z to Batch + int64_t oc = tgpig.y; + int64_t batch_idx = tgpig.z; + + // 3. Calculate anchor coordinates in the Input volume + int64_t i_w_base = ow * args.s0 - args.p0; + int64_t i_h_base = oh * args.s1 - args.p1; + int64_t i_d_base = od * args.s2 - args.p2; + + float sum = 0.0f; + + // 4. Gather Loop (Iterate over Input Channels -> Depth -> Height -> Width) + for (int64_t ic = 0; ic < args.IC; ++ic) { + + // ggml packs batch and channel together in the 4th dimension + int64_t src_cn_idx = batch_idx * args.IC + ic; + int64_t w_cn_idx = oc * args.IC + ic; + + for (int64_t kz = 0; kz < args.KD; ++kz) { + int64_t id = i_d_base + kz * args.d2; + if (id < 0 || id >= args.ID) continue; // Boundary check (Padding) + + for (int64_t ky = 0; ky < args.KH; ++ky) { + int64_t ih = i_h_base + ky * args.d1; + if (ih < 0 || ih >= args.IH) continue; + + for (int64_t kx = 0; kx < args.KW; ++kx) { + int64_t iw = i_w_base + kx * args.d0; + if (iw < 0 || iw >= args.IW) continue; + + // Convert multi-dimensional coordinates to flat byte offsets + int64_t w_idx = kx*args.nb00 + ky*args.nb01 + kz*args.nb02 + w_cn_idx*args.nb03; + int64_t i_idx = iw*args.nb10 + ih*args.nb11 + id*args.nb12 + src_cn_idx*args.nb13; + + // Dereference memory and cast weights to f32 if they were f16 + float w_val = (float)*(device const T*)((device const char*)src0 + w_idx); + float i_val = *(device const float*)((device const char*)src1 + i_idx); + + sum += w_val * i_val; + } + } + } + } + + // 5. Write the accumulated value out to RAM + int64_t dst_cn_idx = batch_idx * args.OC + oc; + int64_t d_idx = ow*args.nb0 + oh*args.nb1 + od*args.nb2 + dst_cn_idx*args.nb3; + + *(device float*)(dst + d_idx) = sum; +} + +// Explicit instantiations so the JIT compiler can find them by name +template [[host_name("kernel_conv_3d_f32_f32")]] +kernel void kernel_conv_3d( + constant ggml_metal_kargs_conv_3d & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]]); + +// Explicit instantiation for f16 weights +template [[host_name("kernel_conv_3d_f16_f32")]] +kernel void kernel_conv_3d( + constant ggml_metal_kargs_conv_3d & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]]); diff --git a/ggml/src/ggml-metal/kernels/dequantize.h b/ggml/src/ggml-metal/kernels/dequantize.h new file mode 100644 index 00000000..0d1429d9 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/dequantize.h @@ -0,0 +1,735 @@ +#pragma once + +#include "common.h" + +#define GGML_COMMON_DECL_METAL +#define GGML_COMMON_IMPL_METAL +#if defined(GGML_METAL_EMBED_LIBRARY) +__embed_ggml-common.h__ +#else +#include "ggml-common.h" +#endif + +#define QK_NL 16 // shared by mul_mm and get_rows_q instantiations + +// NOTE: this is not dequantizing - we are simply fitting the template +template +void dequantize_f32(device const float4x4 * src, short il, thread type4x4 & reg) { + reg = (type4x4)(*src); +} + +template +void dequantize_f32_t4(device const float4 * src, short il, thread type4 & reg) { + reg = (type4)(*src); +} + +template +void dequantize_f16(device const half4x4 * src, short il, thread type4x4 & reg) { + reg = (type4x4)(*src); +} + +template +void dequantize_f16_t4(device const half4 * src, short il, thread type4 & reg) { + reg = (type4)(*(src)); +} + +#if defined(GGML_METAL_HAS_BF16) +template +void dequantize_bf16(device const bfloat4x4 * src, short il, thread type4x4 & reg) { + reg = (type4x4)(*src); +} + +template +void dequantize_bf16_t4(device const bfloat4 * src, short il, thread type4 & reg) { + reg = (type4)(*(src)); +} +#endif + +template +void dequantize_q1_0(device const block_q1_0 * xb, short il, thread type4x4 & reg) { + device const uint8_t * qs = xb->qs; + const float d = xb->d; + const float neg_d = -d; + + const int byte_offset = il * 2; // il*16 bits = il*2 bytes + const uint8_t b0 = qs[byte_offset]; + const uint8_t b1 = qs[byte_offset + 1]; + + float4x4 reg_f; + + reg_f[0][0] = select(neg_d, d, bool(b0 & 0x01)); + reg_f[0][1] = select(neg_d, d, bool(b0 & 0x02)); + reg_f[0][2] = select(neg_d, d, bool(b0 & 0x04)); + reg_f[0][3] = select(neg_d, d, bool(b0 & 0x08)); + reg_f[1][0] = select(neg_d, d, bool(b0 & 0x10)); + reg_f[1][1] = select(neg_d, d, bool(b0 & 0x20)); + reg_f[1][2] = select(neg_d, d, bool(b0 & 0x40)); + reg_f[1][3] = select(neg_d, d, bool(b0 & 0x80)); + + reg_f[2][0] = select(neg_d, d, bool(b1 & 0x01)); + reg_f[2][1] = select(neg_d, d, bool(b1 & 0x02)); + reg_f[2][2] = select(neg_d, d, bool(b1 & 0x04)); + reg_f[2][3] = select(neg_d, d, bool(b1 & 0x08)); + reg_f[3][0] = select(neg_d, d, bool(b1 & 0x10)); + reg_f[3][1] = select(neg_d, d, bool(b1 & 0x20)); + reg_f[3][2] = select(neg_d, d, bool(b1 & 0x40)); + reg_f[3][3] = select(neg_d, d, bool(b1 & 0x80)); + + reg = (type4x4) reg_f; +} + +template +void dequantize_q1_0_t4(device const block_q1_0 * xb, short il, thread type4 & reg) { + const float d = xb->d; + const float neg_d = -d; + const int base = il * 4; + const uint8_t byte = xb->qs[base / 8]; + const int s = base % 8; + + float4 reg_f; + reg_f[0] = select(neg_d, d, bool((byte >> (s )) & 1)); + reg_f[1] = select(neg_d, d, bool((byte >> (s + 1)) & 1)); + reg_f[2] = select(neg_d, d, bool((byte >> (s + 2)) & 1)); + reg_f[3] = select(neg_d, d, bool((byte >> (s + 3)) & 1)); + + reg = (type4) reg_f; +} + +template +void dequantize_q2_0(device const block_q2_0 * xb, short il, thread type4x4 & reg) { + device const uint8_t * qs = xb->qs; + const float d = xb->d; + + const int byte_offset = il * 4; // il*16 elements = il*4 bytes (4 elements per byte) + float4x4 reg_f; + + for (int i = 0; i < 4; i++) { + const uint8_t b = qs[byte_offset + i]; + reg_f[i][0] = ((float)((b >> 0) & 3) - 1.0f) * d; + reg_f[i][1] = ((float)((b >> 2) & 3) - 1.0f) * d; + reg_f[i][2] = ((float)((b >> 4) & 3) - 1.0f) * d; + reg_f[i][3] = ((float)((b >> 6) & 3) - 1.0f) * d; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q2_0_t4(device const block_q2_0 * xb, short il, thread type4 & reg) { + const float d = xb->d; + const uint8_t b = xb->qs[il]; + + float4 reg_f; + reg_f[0] = ((float)((b >> 0) & 3) - 1.0f) * d; + reg_f[1] = ((float)((b >> 2) & 3) - 1.0f) * d; + reg_f[2] = ((float)((b >> 4) & 3) - 1.0f) * d; + reg_f[3] = ((float)((b >> 6) & 3) - 1.0f) * d; + + reg = (type4) reg_f; +} + +template +void dequantize_q4_0(device const block_q4_0 * xb, short il, thread type4x4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 1); + const float d1 = il ? (xb->d / 16.h) : xb->d; + const float d2 = d1 / 256.f; + const float md = -8.h * xb->d; + const ushort mask0 = il ? 0x00F0 : 0x000F; + const ushort mask1 = mask0 << 8; + + float4x4 reg_f; + + for (int i = 0; i < 8; i++) { + reg_f[i/2][2*(i%2) + 0] = d1 * (qs[i] & mask0) + md; + reg_f[i/2][2*(i%2) + 1] = d2 * (qs[i] & mask1) + md; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q4_0_t4(device const block_q4_0 * xb, short il, thread type4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 1); + const float d1 = (il/4) ? (xb->d / 16.h) : xb->d; + const float d2 = d1 / 256.f; + const float md = -8.h * xb->d; + const ushort mask0 = (il/4) ? 0x00F0 : 0x000F; + const ushort mask1 = mask0 << 8; + + for (int i = 0; i < 2; i++) { + reg[2*i + 0] = d1 * (qs[2*(il%4) + i] & mask0) + md; + reg[2*i + 1] = d2 * (qs[2*(il%4) + i] & mask1) + md; + } +} + + + +template +void dequantize_q4_1(device const block_q4_1 * xb, short il, thread type4x4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 2); + const float d1 = il ? (xb->d / 16.h) : xb->d; + const float d2 = d1 / 256.f; + const float m = xb->m; + const ushort mask0 = il ? 0x00F0 : 0x000F; + const ushort mask1 = mask0 << 8; + + float4x4 reg_f; + + for (int i = 0; i < 8; i++) { + reg_f[i/2][2*(i%2) + 0] = ((qs[i] & mask0) * d1) + m; + reg_f[i/2][2*(i%2) + 1] = ((qs[i] & mask1) * d2) + m; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q4_1_t4(device const block_q4_1 * xb, short il, thread type4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 2); + const float d1 = (il/4) ? (xb->d / 16.h) : xb->d; + const float d2 = d1 / 256.f; + const float m = xb->m; + const ushort mask0 = (il/4) ? 0x00F0 : 0x000F; + const ushort mask1 = mask0 << 8; + + for (int i = 0; i < 2; i++) { + reg[2*i + 0] = d1 * (qs[2*(il%4) + i] & mask0) + m; + reg[2*i + 1] = d2 * (qs[2*(il%4) + i] & mask1) + m; + } +} + +template +void dequantize_q5_0(device const block_q5_0 * xb, short il, thread type4x4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 3); + const float d = xb->d; + const float md = -16.h * xb->d; + const ushort mask = il ? 0x00F0 : 0x000F; + + const uint32_t qh = *((device const uint32_t *)xb->qh); + + const int x_mv = il ? 4 : 0; + + const int gh_mv = il ? 12 : 0; + const int gh_bk = il ? 0 : 4; + + float4x4 reg_f; + + for (int i = 0; i < 8; i++) { + // extract the 5-th bits for x0 and x1 + const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; + const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; + + // combine the 4-bits from qs with the 5th bit + const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); + const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); + + reg_f[i/2][2*(i%2) + 0] = d * x0 + md; + reg_f[i/2][2*(i%2) + 1] = d * x1 + md; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q5_0_t4(device const block_q5_0 * xb, short il, thread type4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 3); + const float d = xb->d; + const float md = -16.h * xb->d; + const ushort mask = (il/4) ? 0x00F0 : 0x000F; + + const uint32_t qh = *((device const uint32_t *)xb->qh); + + const int x_mv = (il/4) ? 4 : 0; + + const int gh_mv = (il/4) ? 12 : 0; + const int gh_bk = (il/4) ? 0 : 4; + + for (int ii = 0; ii < 2; ii++) { + int i = 2*(il%4) + ii; + + // extract the 5-th bits for x0 and x1 + const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; + const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; + + // combine the 4-bits from qs with the 5th bit + const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); + const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); + + reg[2*ii + 0] = d * x0 + md; + reg[2*ii + 1] = d * x1 + md; + } +} + +template +void dequantize_q5_1(device const block_q5_1 * xb, short il, thread type4x4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 4); + const float d = xb->d; + const float m = xb->m; + const ushort mask = il ? 0x00F0 : 0x000F; + + const uint32_t qh = *((device const uint32_t *)xb->qh); + + const int x_mv = il ? 4 : 0; + + const int gh_mv = il ? 12 : 0; + const int gh_bk = il ? 0 : 4; + + float4x4 reg_f; + + for (int i = 0; i < 8; i++) { + // extract the 5-th bits for x0 and x1 + const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; + const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; + + // combine the 4-bits from qs with the 5th bit + const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); + const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); + + reg_f[i/2][2*(i%2) + 0] = d * x0 + m; + reg_f[i/2][2*(i%2) + 1] = d * x1 + m; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q5_1_t4(device const block_q5_1 * xb, short il, thread type4 & reg) { + device const uint16_t * qs = ((device const uint16_t *)xb + 4); + const float d = xb->d; + const float m = xb->m; + const ushort mask = (il/4) ? 0x00F0 : 0x000F; + + const uint32_t qh = *((device const uint32_t *)xb->qh); + + const int x_mv = (il/4) ? 4 : 0; + + const int gh_mv = (il/4) ? 12 : 0; + const int gh_bk = (il/4) ? 0 : 4; + + for (int ii = 0; ii < 2; ii++) { + int i = 2*(il%4) + ii; + + // extract the 5-th bits for x0 and x1 + const uint8_t xh_0 = ((qh >> (gh_mv + 2*i )) << gh_bk) & 0x10; + const uint8_t xh_1 = ((qh >> (gh_mv + 2*i+1)) << gh_bk) & 0x10; + + // combine the 4-bits from qs with the 5th bit + const int32_t x0 = ((((qs[i] ) & mask) >> x_mv) | xh_0); + const int32_t x1 = ((((qs[i] >> 8) & mask) >> x_mv) | xh_1); + + reg[2*ii + 0] = d * x0 + m; + reg[2*ii + 1] = d * x1 + m; + } +} + +template +void dequantize_q8_0(device const block_q8_0 *xb, short il, thread type4x4 & reg) { + device const packed_char4 * qs = (device const packed_char4 *) xb->qs; + const float d = xb->d; + + float4x4 reg_f; + + for (int i = 0; i < 4; ++i) { + reg_f[i] = float4(qs[4*il + i]) * d; + } + + reg = (type4x4) reg_f; +} + +template +void dequantize_q8_0_t4(device const block_q8_0 *xb, short il, thread type4 & reg) { + device const packed_char4 * qs = (device const packed_char4 *) xb->qs; + const float d = xb->d; + + reg = (type4) (float4(qs[il]) * d); +} + +template +void dequantize_mxfp4(device const block_mxfp4 * xb, short il, thread type4x4 & reg) { + device const uint8_t * q2 = (device const uint8_t *)xb->qs; + + const float d = e8m0_to_fp32(xb->e); + const uint8_t shr = il >= 1 ? 4 : 0; + + for (int i = 0; i < 4; ++i) { + reg[i][0] = d * kvalues_mxfp4_f[(q2[4*i + 0] >> shr) & 0x0F]; + reg[i][1] = d * kvalues_mxfp4_f[(q2[4*i + 1] >> shr) & 0x0F]; + reg[i][2] = d * kvalues_mxfp4_f[(q2[4*i + 2] >> shr) & 0x0F]; + reg[i][3] = d * kvalues_mxfp4_f[(q2[4*i + 3] >> shr) & 0x0F]; + } +} + +template +void dequantize_mxfp4_t4(device const block_mxfp4 * xb, short il, thread type4 & reg) { + device const uint8_t * q2 = (device const uint8_t *)xb->qs; + + const float d = e8m0_to_fp32(xb->e); + const short il4 = il%4; + + const uint8_t shr = il >= 4 ? 4 : 0; + + reg[0] = d * kvalues_mxfp4_f[(q2[4*il4 + 0] >> shr) & 0x0F]; + reg[1] = d * kvalues_mxfp4_f[(q2[4*il4 + 1] >> shr) & 0x0F]; + reg[2] = d * kvalues_mxfp4_f[(q2[4*il4 + 2] >> shr) & 0x0F]; + reg[3] = d * kvalues_mxfp4_f[(q2[4*il4 + 3] >> shr) & 0x0F]; +} + +template +void dequantize_q2_K(device const block_q2_K *xb, short il, thread type4x4 & reg) { + const float d = xb->d; + const float min = xb->dmin; + device const uint8_t * q = (device const uint8_t *)xb->qs; + float dl, ml; + uint8_t sc = xb->scales[il]; + + q = q + 32*(il/8) + 16*(il&1); + il = (il/2)%4; + + half coef = il>1 ? (il>2 ? 1/64.h : 1/16.h) : (il>0 ? 1/4.h : 1.h); + uchar mask = il>1 ? (il>2 ? 192 : 48) : (il>0 ? 12 : 3); + dl = d * (sc & 0xF) * coef, ml = min * (sc >> 4); + for (int i = 0; i < 16; ++i) { + reg[i/4][i%4] = dl * (q[i] & mask) - ml; + } +} + +template +void dequantize_q3_K(device const block_q3_K *xb, short il, thread type4x4 & reg) { + const half d_all = xb->d; + device const uint8_t * q = (device const uint8_t *)xb->qs; + device const uint8_t * h = (device const uint8_t *)xb->hmask; + device const int8_t * scales = (device const int8_t *)xb->scales; + + q = q + 32 * (il/8) + 16 * (il&1); + h = h + 16 * (il&1); + uint8_t m = 1 << (il/2); + uint16_t kmask1 = (il/4)>1 ? ((il/4)>2 ? 192 : 48) : \ + ((il/4)>0 ? 12 : 3); + uint16_t kmask2 = il/8 ? 0xF0 : 0x0F; + uint16_t scale_2 = scales[il%8], scale_1 = scales[8 + il%4]; + int16_t dl_int = (il/4)&1 ? (scale_2&kmask2) | ((scale_1&kmask1) << 2) + : (scale_2&kmask2) | ((scale_1&kmask1) << 4); + float dl = il<8 ? d_all * (dl_int - 32.f) : d_all * (dl_int / 16.f - 32.f); + const float ml = 4.f * dl; + + il = (il/2) & 3; + const half coef = il>1 ? (il>2 ? 1/64.h : 1/16.h) : (il>0 ? 1/4.h : 1.h); + const uint8_t mask = il>1 ? (il>2 ? 192 : 48) : (il>0 ? 12 : 3); + dl *= coef; + + for (int i = 0; i < 16; ++i) { + reg[i/4][i%4] = dl * (q[i] & mask) - (h[i] & m ? 0 : ml); + } +} + +static inline uchar2 get_scale_min_k4_just2(int j, int k, device const uchar * q) { + return j < 4 ? uchar2{uchar(q[j+0+k] & 63), uchar(q[j+4+k] & 63)} + : uchar2{uchar((q[j+4+k] & 0xF) | ((q[j-4+k] & 0xc0) >> 2)), uchar((q[j+4+k] >> 4) | ((q[j-0+k] & 0xc0) >> 2))}; +} + +template +void dequantize_q4_K(device const block_q4_K * xb, short il, thread type4x4 & reg) { + device const uchar * q = xb->qs; + + short is = (il/4) * 2; + q = q + (il/4) * 32 + 16 * (il&1); + il = il & 3; + const uchar2 sc = get_scale_min_k4_just2(is, il/2, xb->scales); + const float d = il < 2 ? xb->d : xb->d / 16.h; + const float min = xb->dmin; + const float dl = d * sc[0]; + const float ml = min * sc[1]; + + const ushort mask = il < 2 ? 0x0F : 0xF0; + for (int i = 0; i < 16; ++i) { + reg[i/4][i%4] = dl * (q[i] & mask) - ml; + } +} + +template +void dequantize_q5_K(device const block_q5_K *xb, short il, thread type4x4 & reg) { + device const uint8_t * q = xb->qs; + device const uint8_t * qh = xb->qh; + + short is = (il/4) * 2; + q = q + 32 * (il/4) + 16 * (il&1); + qh = qh + 16 * (il&1); + uint8_t ul = 1 << (il/2); + il = il & 3; + const uchar2 sc = get_scale_min_k4_just2(is, il/2, xb->scales); + const float d = il < 2 ? xb->d : xb->d / 16.f; + const float min = xb->dmin; + const float dl = d * sc[0]; + const float ml = min * sc[1]; + + const ushort mask = il<2 ? 0x0F : 0xF0; + const float qh_val = il<2 ? 16.f : 256.f; + for (int i = 0; i < 16; ++i) { + reg[i/4][i%4] = dl * ((q[i] & mask) + (qh[i] & ul ? qh_val : 0)) - ml; + } +} + +template +void dequantize_q6_K(device const block_q6_K *xb, short il, thread type4x4 & reg) { + const half d_all = xb->d; + device const uint16_t * ql = (device const uint16_t *)xb->ql; + device const uint16_t * qh = (device const uint16_t *)xb->qh; + device const int8_t * scales = (device const int8_t *)xb->scales; + + ql = ql + 32*(il/8) + 16*((il/2)&1) + 8*(il&1); + qh = qh + 16*(il/8) + 8*(il&1); + float sc = scales[(il%2) + 2 * ((il/2))]; + il = (il/2) & 3; + + const uint32_t kmask1 = il>1 ? (il>2 ? 0xC0C0C0C0 : 0x30303030) : (il>0 ? 0x0C0C0C0C : 0x03030303); + const uint32_t kmask2 = il>1 ? 0xF0F0F0F0 : 0x0F0F0F0F; + const float ml = d_all * sc * 32.f; + const float dl0 = d_all * sc; + const float dl1 = dl0 / 256.f; + const float dl2 = dl0 / (256.f * 256.f); + const float dl3 = dl0 / (256.f * 256.f * 256.f); + const uint8_t shr_h = il>2 ? 2 : 0; + const uint8_t shl_h = il>1 ? 0 : (il>0 ? 2 : 4); + const uint8_t shr_l = il>1 ? 4 : 0; + for (int i = 0; i < 4; ++i) { + const uint32_t low = (ql[2*i] | (uint32_t)(ql[2*i+1] << 16)) & kmask2; + const uint32_t high = (qh[2*i] | (uint32_t)(qh[2*i+1] << 16)) & kmask1; + const uint32_t q = ((high << shl_h) >> shr_h) | (low >> shr_l); + reg[i][0] = dl0 * ((half)(q & 0xFF)) - ml; + reg[i][1] = dl1 * ((float)(q & 0xFF00)) - ml; + reg[i][2] = dl2 * ((float)(q & 0xFF0000)) - ml; + reg[i][3] = dl3 * ((float)(q & 0xFF000000)) - ml; + } +} + +template +void dequantize_iq2_xxs(device const block_iq2_xxs * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const float d = xb->d; + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + // each block of 32 needs 2 uint32_t's for the quants & scale, so 4 uint16_t's. + device const uint16_t * q2 = xb->qs + 4*ib32; + const uint32_t aux32_g = q2[0] | (q2[1] << 16); + const uint32_t aux32_s = q2[2] | (q2[3] << 16); + thread const uint8_t * aux8 = (thread const uint8_t *)&aux32_g; + const float dl = d * (0.5f + (aux32_s >> 28)) * 0.25f; + constant uint8_t * grid = (constant uint8_t *)(iq2xxs_grid + aux8[2*il+0]); + uint8_t signs = ksigns_iq2xs[(aux32_s >> 14*il) & 127]; + for (int i = 0; i < 8; ++i) { + reg[i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); + } + grid = (constant uint8_t *)(iq2xxs_grid + aux8[2*il+1]); + signs = ksigns_iq2xs[(aux32_s >> (14*il+7)) & 127]; + for (int i = 0; i < 8; ++i) { + reg[2+i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); + } +} + +template +void dequantize_iq2_xs(device const block_iq2_xs * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const float d = xb->d; + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + device const uint16_t * q2 = xb->qs + 4*ib32; + const float dl = d * (0.5f + ((xb->scales[ib32] >> 4*il) & 0xf)) * 0.25f; + constant uint8_t * grid = (constant uint8_t *)(iq2xs_grid + (q2[2*il+0] & 511)); + uint8_t signs = ksigns_iq2xs[q2[2*il+0] >> 9]; + for (int i = 0; i < 8; ++i) { + reg[i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); + } + grid = (constant uint8_t *)(iq2xs_grid + (q2[2*il+1] & 511)); + signs = ksigns_iq2xs[q2[2*il+1] >> 9]; + for (int i = 0; i < 8; ++i) { + reg[2+i/4][i%4] = dl * grid[i] * (signs & kmask_iq2xs[i] ? -1.f : 1.f); + } +} + +template +void dequantize_iq3_xxs(device const block_iq3_xxs * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const float d = xb->d; + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + device const uint8_t * q3 = xb->qs + 8*ib32; + device const uint16_t * gas = (device const uint16_t *)(xb->qs + QK_K/4) + 2*ib32; + const uint32_t aux32 = gas[0] | (gas[1] << 16); + const float dl = d * (0.5f + (aux32 >> 28)) * 0.5f; + constant uint8_t * grid1 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+0]); + constant uint8_t * grid2 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+1]); + uint8_t signs = ksigns_iq2xs[(aux32 >> 14*il) & 127]; + for (int i = 0; i < 4; ++i) { + reg[0][i] = dl * grid1[i] * (signs & kmask_iq2xs[i+0] ? -1.f : 1.f); + reg[1][i] = dl * grid2[i] * (signs & kmask_iq2xs[i+4] ? -1.f : 1.f); + } + grid1 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+2]); + grid2 = (constant uint8_t *)(iq3xxs_grid + q3[4*il+3]); + signs = ksigns_iq2xs[(aux32 >> (14*il+7)) & 127]; + for (int i = 0; i < 4; ++i) { + reg[2][i] = dl * grid1[i] * (signs & kmask_iq2xs[i+0] ? -1.f : 1.f); + reg[3][i] = dl * grid2[i] * (signs & kmask_iq2xs[i+4] ? -1.f : 1.f); + } +} + +template +void dequantize_iq3_s(device const block_iq3_s * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const float d = xb->d; + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + device const uint8_t * qs = xb->qs + 8*ib32; + device const uint8_t * signs = xb->signs + 4*ib32 + 2*il; + const uint8_t qh = xb->qh[ib32] >> 4*il; + const float dl = d * (1 + 2*((xb->scales[ib32/2] >> 4*(ib32%2)) & 0xf)); + constant uint8_t * grid1 = (constant uint8_t *)(iq3s_grid + (qs[4*il+0] | ((qh << 8) & 256))); + constant uint8_t * grid2 = (constant uint8_t *)(iq3s_grid + (qs[4*il+1] | ((qh << 7) & 256))); + for (int i = 0; i < 4; ++i) { + reg[0][i] = dl * grid1[i] * select(1, -1, signs[0] & kmask_iq2xs[i+0]); + reg[1][i] = dl * grid2[i] * select(1, -1, signs[0] & kmask_iq2xs[i+4]); + } + grid1 = (constant uint8_t *)(iq3s_grid + (qs[4*il+2] | ((qh << 6) & 256))); + grid2 = (constant uint8_t *)(iq3s_grid + (qs[4*il+3] | ((qh << 5) & 256))); + for (int i = 0; i < 4; ++i) { + reg[2][i] = dl * grid1[i] * select(1, -1, signs[1] & kmask_iq2xs[i+0]); + reg[3][i] = dl * grid2[i] * select(1, -1, signs[1] & kmask_iq2xs[i+4]); + } +} + +template +void dequantize_iq2_s(device const block_iq2_s * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const float d = xb->d; + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; + device const uint8_t * signs = qs + QK_K/8; + const uint8_t qh = xb->qh[ib32] >> 4*il; + const float dl = d * (0.5f + ((xb->scales[ib32] >> 4*il) & 0xf)) * 0.25f; + constant uint8_t * grid1 = (constant uint8_t *)(iq2s_grid + (qs[0] | ((qh << 8) & 0x300))); + constant uint8_t * grid2 = (constant uint8_t *)(iq2s_grid + (qs[1] | ((qh << 6) & 0x300))); + for (int i = 0; i < 8; ++i) { + reg[i/4+0][i%4] = dl * grid1[i] * select(1, -1, signs[0] & kmask_iq2xs[i]); + reg[i/4+2][i%4] = dl * grid2[i] * select(1, -1, signs[1] & kmask_iq2xs[i]); + } +} + +template +void dequantize_iq1_s(device const block_iq1_s * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const int ib32 = il/2; + il = il%2; + const float d = xb->d; + device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; + device const uint16_t * qh = xb->qh; + const float dl = d * (2*((qh[ib32] >> 12) & 7) + 1); + const float ml = dl * (qh[ib32] & 0x8000 ? -1 - IQ1S_DELTA : -1 + IQ1S_DELTA); + const uint16_t h = qh[ib32] >> 6*il; + constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((h << 8) & 0x700))); + constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((h << 5) & 0x700))); + for (int i = 0; i < 4; ++i) { + reg[0][i] = dl * (grid1[i] & 0xf) + ml; + reg[1][i] = dl * (grid1[i] >> 4) + ml; + reg[2][i] = dl * (grid2[i] & 0xf) + ml; + reg[3][i] = dl * (grid2[i] >> 4) + ml; + } +} + +template +void dequantize_iq1_m(device const block_iq1_m * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const int ib32 = il/2; + il = il%2; + device const uint16_t * sc = (device const uint16_t *)xb->scales; + + iq1m_scale_t scale; + scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000); + const float d = scale.f16; + + device const uint8_t * qs = xb->qs + 4*ib32 + 2*il; + device const uint8_t * qh = xb->qh + 2*ib32 + il; + + const float dl = d * (2*((sc[ib32/2] >> (6*(ib32%2)+3*il)) & 7) + 1); + const float ml1 = dl * (qh[0] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); + const float ml2 = dl * (qh[0] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); + constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); + constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 4) & 0x700))); + for (int i = 0; i < 4; ++i) { + reg[0][i] = dl * (grid1[i] & 0xf) + ml1; + reg[1][i] = dl * (grid1[i] >> 4) + ml1; + reg[2][i] = dl * (grid2[i] & 0xf) + ml2; + reg[3][i] = dl * (grid2[i] >> 4) + ml2; + } +} + +template +void dequantize_iq4_nl(device const block_iq4_nl * xb, short il, thread type4x4 & reg) { + device const uint16_t * q4 = (device const uint16_t *)xb->qs; + const float d = xb->d; + uint32_t aux32; + thread const uint8_t * q8 = (thread const uint8_t *)&aux32; + for (int i = 0; i < 4; ++i) { + aux32 = ((q4[2*i] | (q4[2*i+1] << 16)) >> 4*il) & 0x0f0f0f0f; + reg[i][0] = d * kvalues_iq4nl_f[q8[0]]; + reg[i][1] = d * kvalues_iq4nl_f[q8[1]]; + reg[i][2] = d * kvalues_iq4nl_f[q8[2]]; + reg[i][3] = d * kvalues_iq4nl_f[q8[3]]; + } +} + +template +void dequantize_iq4_nl_t4(device const block_iq4_nl * xb, short il, thread type4 & reg) { + device const uint16_t * q4 = (device const uint16_t *)xb->qs; + const float d = xb->d; + uint32_t aux32; + thread const uint8_t * q8 = (thread const uint8_t *)&aux32; + aux32 = ((q4[2*(il%4)] | (q4[2*(il%4)+1] << 16)) >> 4*(il/4)) & 0x0f0f0f0f; + reg[0] = d * kvalues_iq4nl_f[q8[0]]; + reg[1] = d * kvalues_iq4nl_f[q8[1]]; + reg[2] = d * kvalues_iq4nl_f[q8[2]]; + reg[3] = d * kvalues_iq4nl_f[q8[3]]; +} + +template +void dequantize_iq4_xs(device const block_iq4_xs * xb, short il, thread type4x4 & reg) { + // il is 0...15 for QK_K = 256 => index of block of 32 is il/2 + const int ib32 = il/2; + il = il%2; + // il = 0 or 1. il = 0 processes the first 16 quants in a block of 32, il = 1 the second 16 + device const uint32_t * q4 = (device const uint32_t *)xb->qs + 4*ib32; + const int ls = ((xb->scales_l[ib32/2] >> 4*(ib32%2)) & 0xf) | (((xb->scales_h >> 2*ib32) & 3) << 4); + const float d = (float)xb->d * (ls - 32); + uint32_t aux32; + thread const uint8_t * q8 = (thread const uint8_t *)&aux32; + for (int i = 0; i < 4; ++i) { + aux32 = (q4[i] >> 4*il) & 0x0f0f0f0f; + reg[i][0] = d * kvalues_iq4nl_f[q8[0]]; + reg[i][1] = d * kvalues_iq4nl_f[q8[1]]; + reg[i][2] = d * kvalues_iq4nl_f[q8[2]]; + reg[i][3] = d * kvalues_iq4nl_f[q8[3]]; + } +} + +template +void dequantize_tq2_0(device const block_tq2_0 * xb, short il, thread type4x4 & reg) { + device const uint8_t * qs = xb->qs; + const float d = xb->d; + + float4x4 reg_f; + + // 2 bits per element, 4 elements per byte, 128 elements per 32-byte group + const short base = il * 16; + for (int k = 0; k < 16; k++) { + const int i = base + k; + const int byte = ((i >> 7) & 1) * 32 + (i & 31); + const int l = (i >> 5) & 3; + reg_f[k/4][k%4] = d * (float)(((qs[byte] >> (2*l)) & 3) - 1); + } + + reg = (type4x4) reg_f; +} diff --git a/ggml/src/ggml-metal/kernels/fa.metal b/ggml/src/ggml-metal/kernels/fa.metal new file mode 100644 index 00000000..f26d493d --- /dev/null +++ b/ggml/src/ggml-metal/kernels/fa.metal @@ -0,0 +1,2476 @@ +#include "common.h" +#include "dequantize.h" + +// dequantize a quantized KV cache tensor to contiguous F16 before running the F16 flash attention kernels +// - one thread per block; dispatched separately for K and V +// - ref: https://github.com/ggml-org/llama.cpp/pull/27390 +template < + typename block_t, + short QK, + void (*deq_t4x4)(device const block_t *, short, thread float4x4 &)> +kernel void kernel_flash_attn_ext_kv_f16( + constant ggml_metal_kargs_flash_attn_ext_kv_f16 & args, + device const char * x, + device half * x_dst, + uint gid [[thread_position_in_grid]]) { + if (gid >= (uint) args.nblocks) { + return; + } + + const uint nb = args.ne0/QK; + const uint i0 = gid%nb; + uint ib = gid/nb; + const uint i1 = ib%args.ne1; + ib /= args.ne1; + const uint i2 = ib%args.ne2; + const uint i3 = ib/args.ne2; + + const uint64_t offs = i0*args.nb0 + i1*args.nb1 + i2*args.nb2 + i3*args.nb3; + + device const block_t * src = (device const block_t *) (x + offs); + device half4 * dst = (device half4 *) x_dst + (QK/4)*gid; + + for (short i = 0; i < QK/16; ++i) { + float4x4 reg; + deq_t4x4(src, i, reg); + dst[4*i + 0] = (half4) reg[0]; + dst[4*i + 1] = (half4) reg[1]; + dst[4*i + 2] = (half4) reg[2]; + dst[4*i + 3] = (half4) reg[3]; + } +} + +typedef decltype(kernel_flash_attn_ext_kv_f16) kernel_flash_attn_ext_kv_f16_t; + +template [[host_name("kernel_flash_attn_ext_kv_q4_0_f16")]] kernel kernel_flash_attn_ext_kv_f16_t kernel_flash_attn_ext_kv_f16; +template [[host_name("kernel_flash_attn_ext_kv_q4_1_f16")]] kernel kernel_flash_attn_ext_kv_f16_t kernel_flash_attn_ext_kv_f16; +template [[host_name("kernel_flash_attn_ext_kv_q5_0_f16")]] kernel kernel_flash_attn_ext_kv_f16_t kernel_flash_attn_ext_kv_f16; +template [[host_name("kernel_flash_attn_ext_kv_q5_1_f16")]] kernel kernel_flash_attn_ext_kv_f16_t kernel_flash_attn_ext_kv_f16; +template [[host_name("kernel_flash_attn_ext_kv_q8_0_f16")]] kernel kernel_flash_attn_ext_kv_f16_t kernel_flash_attn_ext_kv_f16; + +constant bool FC_flash_attn_ext_pad_has_mask [[function_constant(FC_FLASH_ATTN_EXT_PAD + 0)]]; + +constant int32_t FC_flash_attn_ext_pad_ncpsg [[function_constant(FC_FLASH_ATTN_EXT_PAD + 25)]]; + +// pad the last chunk of C elements of k and v into a an extra pad buffer +kernel void kernel_flash_attn_ext_pad( + constant ggml_metal_kargs_flash_attn_ext_pad & args, + device const char * k, + device const char * v, + device const char * mask, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int32_t C = FC_flash_attn_ext_pad_ncpsg; + + device char * k_pad = dst; + device char * v_pad = k_pad + args.nb11*C*args.ne_12_2*args.ne_12_3; + device char * mask_pad = v_pad + args.nb21*C*args.ne_12_2*args.ne_12_3; + + const int32_t icp = args.ne11 % C; + const int32_t ic0 = args.ne11 - icp; + + const int32_t i1 = tgpig[0]; + const int32_t i2 = tgpig[1]; + const int32_t i3 = tgpig[2]; + + if (i2 < args.ne_12_2 && i3 < args.ne_12_3) { + device const char * k_src = k + args.nb11*(ic0 + i1) + args.nb12*i2 + args.nb13*i3; + device const char * v_src = v + args.nb21*(ic0 + i1) + args.nb22*i2 + args.nb23*i3; + + device char * k_dst = k_pad + args.nb11*i1 + args.nb11*C*i2 + args.nb11*C*args.ne_12_2*i3; + device char * v_dst = v_pad + args.nb21*i1 + args.nb21*C*i2 + args.nb21*C*args.ne_12_2*i3; + + if (i1 >= icp) { + // here it is not important the exact value that will be used as we rely on masking out the scores in the attention + for (uint64_t i = tiitg; i < args.nb11; i += ntg.x) { + k_dst[i] = 0; + } + for (uint64_t i = tiitg; i < args.nb21; i += ntg.x) { + v_dst[i] = 0; + } + } else { + for (uint64_t i = tiitg; i < args.nb11; i += ntg.x) { + k_dst[i] = k_src[i]; + } + for (uint64_t i = tiitg; i < args.nb21; i += ntg.x) { + v_dst[i] = v_src[i]; + } + } + } + + if (FC_flash_attn_ext_pad_has_mask) { + if (i2 < args.ne32 && i3 < args.ne33) { + for (int ib = i1; ib < args.ne31; ib += C) { + device const half * mask_src = (device const half *)(mask + args.nb31*ib + args.nb32*i2 + args.nb33*i3) + ic0; + device half * mask_dst = (device half *)(mask_pad) + C*ib + C*args.ne31*i2 + C*args.ne31*args.ne32*i3; + + for (int i = tiitg; i < C; i += ntg.x) { + if (i >= icp) { + mask_dst[i] = -MAXHALF; + } else { + mask_dst[i] = mask_src[i]; + } + } + } + } + } +} + +constant int32_t FC_flash_attn_ext_blk_nqptg [[function_constant(FC_FLASH_ATTN_EXT_BLK + 24)]]; +constant int32_t FC_flash_attn_ext_blk_ncpsg [[function_constant(FC_FLASH_ATTN_EXT_BLK + 25)]]; + +// scan the blocks of the mask that are not masked +// 0 - masked (i.e. full of -INF, skip) +// 1 - not masked (i.e. at least one element of the mask is not -INF) +// 2 - all zero +kernel void kernel_flash_attn_ext_blk( + constant ggml_metal_kargs_flash_attn_ext_blk & args, + device const char * mask, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]]) { + // block size C x Q + const int32_t Q = FC_flash_attn_ext_blk_nqptg; + const int32_t C = FC_flash_attn_ext_blk_ncpsg; + + constexpr short NW = N_SIMDWIDTH; + + const int32_t i3 = tgpig[2]/args.ne32; + const int32_t i2 = tgpig[2]%args.ne32; + const int32_t i1 = tgpig[1]; + const int32_t i0 = tgpig[0]; + + char res = i0*C + C > args.ne30 || i1*Q + Q > args.ne31 ? 1 : 0; + + device const half * mask_src = (device const half *) (mask + (i1*Q)*args.nb31 + i2*args.nb32 + i3*args.nb33) + i0*C + tiisg; + + // detailed check of the elements of the block + if ((C > NW || Q > 1) && res == 0) { + half mmin = MAXHALF; + half mmax = -MAXHALF; + + FOR_UNROLL (short j = 0; j < Q; ++j) { + FOR_UNROLL (short ii = 0; ii < C/NW; ++ii) { + mmin = min(mmin, mask_src[ii*NW]); + mmax = max(mmax, mask_src[ii*NW]); + } + + mask_src += args.nb31/2; + } + + mmin = simd_min(mmin); + mmax = simd_max(mmax); + + if (mmax > -MAXHALF) { + if (mmin == 0.0 && mmax == 0.0) { + res = 2; + } else { + res = 1; + } + } + } + + const int32_t nblk1 = ((args.ne01 + Q - 1)/Q); + const int32_t nblk0 = ((args.ne30 + C - 1)/C); + + if (tiisg == 0) { + dst[((i3*args.ne32 + i2)*nblk1 + i1)*nblk0 + i0] = res; + } +} + +constant bool FC_flash_attn_ext_has_mask [[function_constant(FC_FLASH_ATTN_EXT + 0)]]; +constant bool FC_flash_attn_ext_has_sinks [[function_constant(FC_FLASH_ATTN_EXT + 1)]]; +constant bool FC_flash_attn_ext_has_bias [[function_constant(FC_FLASH_ATTN_EXT + 2)]]; +constant bool FC_flash_attn_ext_has_scap [[function_constant(FC_FLASH_ATTN_EXT + 3)]]; +constant bool FC_flash_attn_ext_has_kvpad [[function_constant(FC_FLASH_ATTN_EXT + 4)]]; + +constant bool FC_flash_attn_ext_bc_mask [[function_constant(FC_FLASH_ATTN_EXT + 10)]]; + +//constant float FC_flash_attn_ext_scale [[function_constant(FC_FLASH_ATTN_EXT + 10)]]; +//constant float FC_flash_attn_ext_max_bias [[function_constant(FC_FLASH_ATTN_EXT + 11)]]; +//constant float FC_flash_attn_ext_logit_softcap [[function_constant(FC_FLASH_ATTN_EXT + 12)]]; + +constant int32_t FC_flash_attn_ext_ns10 [[function_constant(FC_FLASH_ATTN_EXT + 20)]]; +constant int32_t FC_flash_attn_ext_ns20 [[function_constant(FC_FLASH_ATTN_EXT + 21)]]; +constant int32_t FC_flash_attn_ext_nsg [[function_constant(FC_FLASH_ATTN_EXT + 22)]]; + +// ref: https://arxiv.org/pdf/2307.08691.pdf +template< + typename q_t, // query types in shared memory + typename q4_t, + typename q8x8_t, + typename k_t, // key types in shared memory + typename k4x4_t, + typename k8x8_t, + typename v_t, // value types in shared memory + typename v4x4_t, + typename v8x8_t, + typename qk_t, // Q*K types + typename qk8x8_t, + typename s_t, // soft-max types + typename s2_t, + typename s8x8_t, + typename o_t, // attention accumulation types + typename o4_t, + typename o8x8_t, + typename kd4x4_t, // key type in device memory + short nl_k, + void (*deq_k)(device const kd4x4_t *, short, thread k4x4_t &), + typename vd4x4_t, // value type in device memory + short nl_v, + void (*deq_v)(device const vd4x4_t *, short, thread v4x4_t &), + short DK, // K head size + short DV, // V head size + short Q, // queries per threadgroup + short C, // cache items per threadgroup + short NSG> // number of simd groups +void kernel_flash_attn_ext_impl( + constant ggml_metal_kargs_flash_attn_ext & args, + device const char * q, + device const char * k, + device const char * v, + device const char * mask, + device const char * sinks, + device const char * pad, + device const char * blk, + device char * dst, + threadgroup half * shmem_f16, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const ushort iq3 = tgpig[2]; + const ushort iq2 = tgpig[1]; + const ushort iq1 = tgpig[0]*Q; + +#define NS10 (FC_flash_attn_ext_ns10) +#define NS20 (FC_flash_attn_ext_ns20) + + // note: I had some concerns that using this instead of the ugly macros above was affecting performance + // need to re-check carefully and if no regressions are observerd - remove the macros + // the concerns is that maybe using const variables requires extra registers? but not sure if the compiler + // is clever enough to avoid this. unfortunately, using constexpr is not possible with FC + //const short NS10 = FC_flash_attn_ext_ns10; + //const short NS20 = FC_flash_attn_ext_ns20; + + constexpr short KV = 8; + + constexpr short DK4 = DK/4; + constexpr short DK8 = DK/8; + constexpr short DK16 = DK/16; + constexpr short DV4 = DV/4; + //constexpr short DV8 = DV/8; + constexpr short DV16 = DV/16; + + constexpr short PV = PAD2(DV, 64); + constexpr short PV4 = PV/4; + constexpr short PV8 = PV/8; + //constexpr short PV16 = PV/16; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NQ = Q/NSG; + constexpr short SH = 2*C; // shared memory per simdgroup (s_t == float) + + constexpr short TS = 2*SH; + constexpr short T = DK + 2*PV; // shared memory size per query in (half) + + threadgroup q_t * sq = (threadgroup q_t *) (shmem_f16 + 0*T); // holds the query data + threadgroup q4_t * sq4 = (threadgroup q4_t *) (shmem_f16 + 0*T); // same as above but in q4_t + threadgroup o_t * so = (threadgroup o_t *) (shmem_f16 + 0*T + Q*DK); // the result for all queries in 8x8 matrices (the O matrix from the paper) + threadgroup o4_t * so4 = (threadgroup o4_t *) (shmem_f16 + 0*T + Q*DK); + threadgroup s_t * ss = (threadgroup s_t *) (shmem_f16 + Q*T); // scratch buffer for attention, mask and diagonal matrix + threadgroup s2_t * ss2 = (threadgroup s2_t *) (shmem_f16 + Q*T); // same as above but in s2_t + + threadgroup k_t * sk = (threadgroup k_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // scratch buffer to load K in shared memory + threadgroup k4x4_t * sk4x4 = (threadgroup k4x4_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // same as above but in k4x4_t + + threadgroup v_t * sv = (threadgroup v_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // scratch buffer to load V in shared memory + threadgroup v4x4_t * sv4x4 = (threadgroup v4x4_t *) (shmem_f16 + sgitg*(4*16*KV) + Q*T + Q*TS); // same as above but in v4x4_t + + // mask storage in shared mem + threadgroup half2 * sm2 = (threadgroup half2 *) (shmem_f16 + Q*T + 2*C); + + // per-query mask pointers + device const half2 * pm2[NQ]; + + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + pm2[jj] = (device const half2 *) ((device const char *) mask + (iq1 + j)*args.nb31 + (iq2%args.ne32)*args.nb32 + (iq3%args.ne33)*args.nb33); + } + + { + const int32_t nblk1 = ((args.ne01 + Q - 1)/Q); + const int32_t nblk0 = ((args.ne11 + C - 1)/C); + + blk += (((iq3%args.ne33)*args.ne32 + (iq2%args.ne32))*nblk1 + iq1/Q)*nblk0; + } + + { + q += iq1*args.nb01 + iq2*args.nb02 + iq3*args.nb03; + + const short ikv2 = iq2/(args.ne02/args.ne_12_2); + const short ikv3 = iq3/(args.ne03/args.ne_12_3); + + k += ikv2*args.nb12 + ikv3*args.nb13; + v += ikv2*args.nb22 + ikv3*args.nb23; + } + + // load heads from Q to shared memory + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + device const float4 * q4 = (device const float4 *) ((device const char *) q + j*args.nb01); + + for (short i = tiisg; i < DK4; i += NW) { + if (iq1 + j < args.ne01) { + sq4[j*DK4 + i] = (q4_t) q4[i]; + } else { + sq4[j*DK4 + i] = 0; + } + } + } + + // zero out + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + for (short i = tiisg; i < DV4; i += NW) { + so4[j*PV4 + i] = 0; + } + + for (short i = tiisg; i < SH; i += NW) { + ss[j*SH + i] = 0.0f; + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + float S[NQ] = { [0 ... NQ-1] = 0.0f }; + + { + float M[NQ] = { [0 ... NQ-1] = -FLT_MAX/2 }; + + float slope = 1.0f; + + // ALiBi + if (FC_flash_attn_ext_has_bias) { + const short h = iq2; + + const float base = h < args.n_head_log2 ? args.m0 : args.m1; + const short exph = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; + + slope = pow(base, exph); + } + + // loop over the KV cache + // each simdgroup handles blocks of Q rows and C columns + for (int ic0 = 0; ; ++ic0) { + int ic = ic0*C; + if (ic >= args.ne11) { + break; + } + + // the last partial chunk uses the pad buffer as source + if (FC_flash_attn_ext_has_kvpad && ic + C > args.ne11) { + k = pad; + v = k + args.nb11*C*args.ne_12_2*args.ne_12_3; + mask = v + args.nb21*C*args.ne_12_2*args.ne_12_3; + + const short ikv2 = iq2/(args.ne02/args.ne_12_2); + const short ikv3 = iq3/(args.ne03/args.ne_12_3); + + k += (ikv2 + ikv3*args.ne_12_2)*args.nb11*C; + v += (ikv2 + ikv3*args.ne_12_2)*args.nb21*C; + + if (!FC_flash_attn_ext_has_mask) { + threadgroup half * sm = (threadgroup half *) (sm2); + + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + for (short i = tiisg; i < C; i += NW) { + if (ic + i >= args.ne11) { + sm[2*j*SH + i] = -MAXHALF; + } + } + } + } else { + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + pm2[jj] = (device const half2 *) ((device const half *) mask + + (iq1 + j)*C + + (iq2%args.ne32)*(C*args.ne31) + + (iq3%args.ne33)*(C*args.ne31*args.ne32)); + } + } + + ic = 0; + } + + char blk_cur = 1; + + // read the mask into shared mem + if (FC_flash_attn_ext_has_mask) { + blk_cur = blk[ic0]; + + if (blk_cur == 0) { + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + pm2[jj] += NW; + } + + continue; + } + + if (blk_cur == 1) { + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + if (FC_flash_attn_ext_bc_mask) { + sm2[j*SH + tiisg] = (iq1 + j) < args.ne31 ? pm2[jj][tiisg] : half2(-MAXHALF, -MAXHALF); + } else { + sm2[j*SH + tiisg] = pm2[jj][tiisg]; + } + + pm2[jj] += NW; + } + } else if (blk_cur == 2) { + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + pm2[jj] += NW; + } + } + +#if 0 + // note: old -INF block optimization - obsoleted by pre-computing non-masked blocks + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // used to detect blocks full of -INF + // skip only when the entire threadgroup is masked + half2 smax2(-MAXHALF/2, -MAXHALF/2); + + FOR_UNROLL (short j = 0; j < Q; ++j) { + smax2 = max(smax2, sm2[j*SH + tiisg]); + } + + smax2 = simd_max(smax2); + + if (max(smax2[0], smax2[1]) <= -MAXHALF/2) { + // this barrier is important + threadgroup_barrier(mem_flags::mem_threadgroup); + + continue; + } +#endif + } + + // Q*K^T + // this is compile-time check, so it does not have runtime overhead + if (is_same::value) { + // we can read directly from global memory + device const k_t * pk = (device const k_t *) (k + ic*args.nb11); + threadgroup const q_t * pq = sq; + threadgroup s_t * ps = ss; + + pk += sgitg*(8*NS10); + ps += sgitg*(8*1); + + static_assert((C/8) % NSG == 0, ""); + + constexpr short NC = (C/8)/NSG; + + FOR_UNROLL (short cc = 0; cc < NC; ++cc) { + qk8x8_t mqk = make_filled_simdgroup_matrix((qk_t) 0.0f); + + if (DK % 16 != 0) { + k8x8_t mk; + q8x8_t mq; + + FOR_UNROLL (short i = 0; i < DK8; ++i) { + simdgroup_barrier(mem_flags::mem_none); + + simdgroup_load(mk, pk + 8*i, NS10, 0, true); + simdgroup_load(mq, pq + 8*i, DK); + + simdgroup_barrier(mem_flags::mem_none); + + simdgroup_multiply_accumulate(mqk, mq, mk, mqk); + } + } else { + k8x8_t mk[2]; + q8x8_t mq[2]; + + // note: too much unroll can tank the performance for large heads + #pragma unroll (MIN(DK8/2, 4*NSG)) + for (short i = 0; i < DK8/2; ++i) { + simdgroup_barrier(mem_flags::mem_none); + + simdgroup_load(mq[0], pq + 0*8 + 16*i, DK); + simdgroup_load(mq[1], pq + 1*8 + 16*i, DK); + + simdgroup_load(mk[0], pk + 0*8 + 16*i, NS10, 0, true); + simdgroup_load(mk[1], pk + 1*8 + 16*i, NS10, 0, true); + + simdgroup_barrier(mem_flags::mem_none); + + simdgroup_multiply_accumulate(mqk, mq[0], mk[0], mqk); + simdgroup_multiply_accumulate(mqk, mq[1], mk[1], mqk); + } + } + + simdgroup_store(mqk, ps, SH, 0, false); + + pk += 8*(NSG*NS10); + ps += 8*(NSG); + } + } else { + // TODO: this is the quantized K cache branch - not optimized yet + for (short ccc = 0; ccc < (C/8)/NSG; ++ccc) { + const short cc = ccc*NSG + sgitg; + + const short tx = tiisg%4; + const short ty = tiisg/4; + + qk8x8_t mqk = make_filled_simdgroup_matrix((qk_t) 0.0f); + + for (short ii = 0; ii < DK16; ii += 4) { + device const kd4x4_t * pk4x4 = (device const kd4x4_t *) (k + ((ic + 8*cc + ty)*args.nb11)); + + if (DK16%4 == 0) { + // the head is evenly divisible by 4*16 = 64, so no need for bound checks + { + k4x4_t tmp; + deq_k(pk4x4 + (ii + tx)/nl_k, (ii + tx)%nl_k, tmp); + sk4x4[4*ty + tx] = tmp; + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + FOR_UNROLL (short k = 0; k < 4; ++k) { + k8x8_t mk; + q8x8_t mq; + + simdgroup_load(mk, sk + 16*k + 0*8, 4*16, 0, true); // transpose + simdgroup_load(mq, sq + (2*(ii + k) + 0)*8, DK); + simdgroup_multiply_accumulate(mqk, mq, mk, mqk); + + simdgroup_load(mk, sk + 16*k + 1*8, 4*16, 0, true); // transpose + simdgroup_load(mq, sq + (2*(ii + k) + 1)*8, DK); + simdgroup_multiply_accumulate(mqk, mq, mk, mqk); + } + } else { + if (ii + tx < DK16) { + k4x4_t tmp; + deq_k(pk4x4 + (ii + tx)/nl_k, (ii + tx)%nl_k, tmp); + sk4x4[4*ty + tx] = tmp; + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + for (short k = 0; k < 4 && ii + k < DK16; ++k) { + k8x8_t mk; + q8x8_t mq; + + simdgroup_load(mk, sk + 16*k + 0*8, 4*16, 0, true); // transpose + simdgroup_load(mq, sq + (2*(ii + k) + 0)*8, DK); + simdgroup_multiply_accumulate(mqk, mq, mk, mqk); + + simdgroup_load(mk, sk + 16*k + 1*8, 4*16, 0, true); // transpose + simdgroup_load(mq, sq + (2*(ii + k) + 1)*8, DK); + simdgroup_multiply_accumulate(mqk, mq, mk, mqk); + } + } + } + + simdgroup_store(mqk, ss + 8*cc, SH, 0, false); + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // online softmax + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + const float m = M[jj]; + + // scale and apply the logitcap / mask + float2 s2 = ss2[j*SH/2 + tiisg]*args.scale; + + if (FC_flash_attn_ext_has_scap) { + s2 = args.logit_softcap*precise::tanh(s2); + } + + // mqk = mqk + slope*mask + if (blk_cur != 2) { + if (FC_flash_attn_ext_has_bias) { + s2 += s2_t(sm2[j*SH + tiisg])*slope; + } else { + s2 += s2_t(sm2[j*SH + tiisg]); + } + } + + M[jj] = simd_max(max(M[jj], max(s2[0], s2[1]))); + + const float ms = exp(m - M[jj]); + const float2 vs2 = exp(s2 - M[jj]); + + S[jj] = S[jj]*ms + simd_sum(vs2[0] + vs2[1]); + + // the P matrix from the paper (Q rows, C columns) + ss2[j*SH/2 + tiisg] = vs2; + + if (DV4 % NW == 0) { + FOR_UNROLL (short ii = 0; ii < DV4/NW; ++ii) { + const short i = ii*NW + tiisg; + + so4[j*PV4 + i] *= ms; + } + } else { + for (short i = tiisg; i < DV4; i += NW) { + so4[j*PV4 + i] *= ms; + } + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // O = O + (Q*K^T)*V + { + // we can read directly from global memory + if (is_same::value) { + static_assert(PV8 % NSG == 0, ""); + + constexpr short NO = PV8/NSG; + + o8x8_t lo[NO]; + + { + auto sot = so + 8*sgitg; + + FOR_UNROLL (short ii = 0; ii < NO; ++ii) { + simdgroup_load(lo[ii], sot, PV, 0, false); + + sot += 8*NSG; + } + } + + { + device const v_t * pv = (device const v_t *) (v + ic*args.nb21); + + pv += 8*sgitg; + + if (DV <= 64) { + FOR_UNROLL (short cc = 0; cc < C/8; ++cc) { + s8x8_t vs; + simdgroup_load(vs, ss + 8*cc, SH, 0, false); + + FOR_UNROLL (short ii = 0; ii < NO/2; ++ii) { + v8x8_t mv[2]; + + simdgroup_load(mv[0], pv + 0*NSG + 16*ii*NSG, NS20, 0, false); + simdgroup_load(mv[1], pv + 8*NSG + 16*ii*NSG, NS20, 0, false); + + simdgroup_multiply_accumulate(lo[2*ii + 0], vs, mv[0], lo[2*ii + 0]); + simdgroup_multiply_accumulate(lo[2*ii + 1], vs, mv[1], lo[2*ii + 1]); + } + + pv += 8*NS20; + } + } else { + constexpr short NC = (C/8)/2; + + FOR_UNROLL (short cc = 0; cc < NC; ++cc) { + s8x8_t vs[2]; + + simdgroup_load(vs[0], ss + 16*cc + 0, SH, 0, false); + simdgroup_load(vs[1], ss + 16*cc + 8, SH, 0, false); + + FOR_UNROLL (short ii = 0; ii < NO/2; ++ii) { + v8x8_t mv[4]; + + simdgroup_load(mv[0], pv + 0*NSG + 16*ii*NSG + 0*8*NS20, NS20, 0, false); + simdgroup_load(mv[1], pv + 8*NSG + 16*ii*NSG + 0*8*NS20, NS20, 0, false); + simdgroup_load(mv[2], pv + 0*NSG + 16*ii*NSG + 1*8*NS20, NS20, 0, false); + simdgroup_load(mv[3], pv + 8*NSG + 16*ii*NSG + 1*8*NS20, NS20, 0, false); + + simdgroup_multiply_accumulate(lo[2*ii + 0], vs[0], mv[0], lo[2*ii + 0]); + simdgroup_multiply_accumulate(lo[2*ii + 1], vs[0], mv[1], lo[2*ii + 1]); + simdgroup_multiply_accumulate(lo[2*ii + 0], vs[1], mv[2], lo[2*ii + 0]); + simdgroup_multiply_accumulate(lo[2*ii + 1], vs[1], mv[3], lo[2*ii + 1]); + } + + pv += 2*8*NS20; + } + } + } + + { + auto sot = so + 8*sgitg; + + FOR_UNROLL (short ii = 0; ii < NO; ++ii) { + simdgroup_store(lo[ii], sot, PV, 0, false); + + sot += 8*NSG; + } + } + } else { + // TODO: this is the quantized V cache branch - not optimized yet + + const short tx = tiisg%4; + const short ty = tiisg/4; + + for (short cc = 0; cc < C/8; ++cc) { + s8x8_t vs; + simdgroup_load(vs, ss + 8*cc, SH, 0, false); + + for (short ii = 4*sgitg; ii < DV16; ii += 4*NSG) { + device const vd4x4_t * pv4x4 = (device const vd4x4_t *) (v + ((ic + 8*cc + ty)*args.nb21)); + + if (DV16%4 == 0) { + // no need for bound checks + { + v4x4_t tmp; + deq_v(pv4x4 + (ii + tx)/nl_v, (ii + tx)%nl_v, tmp); + sv4x4[4*ty + tx] = tmp; + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + FOR_UNROLL (short k = 0; k < 4; ++k) { + v8x8_t mv[2]; + o8x8_t lo[2]; + + simdgroup_load(mv[0], sv + 16*k + 0*8, 4*16, 0, false); + simdgroup_load(mv[1], sv + 16*k + 1*8, 4*16, 0, false); + simdgroup_load(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); + simdgroup_load(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); + + simdgroup_multiply_accumulate(lo[0], vs, mv[0], lo[0]); + simdgroup_multiply_accumulate(lo[1], vs, mv[1], lo[1]); + + simdgroup_store(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); + simdgroup_store(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); + } + } else { + if (ii + tx < DV16) { + v4x4_t tmp; + deq_v(pv4x4 + (ii + tx)/nl_v, (ii + tx)%nl_v, tmp); + sv4x4[4*ty + tx] = tmp; + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + for (short k = 0; k < 4 && ii + k < DV16; ++k) { + v8x8_t mv[2]; + o8x8_t lo[2]; + + simdgroup_load(mv[0], sv + 16*k + 0*8, 4*16, 0, false); + simdgroup_load(mv[1], sv + 16*k + 1*8, 4*16, 0, false); + simdgroup_load(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); + simdgroup_load(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); + + simdgroup_multiply_accumulate(lo[0], vs, mv[0], lo[0]); + simdgroup_multiply_accumulate(lo[1], vs, mv[1], lo[1]); + + simdgroup_store(lo[0], so + 8*(2*(ii + k) + 0), PV, 0, false); + simdgroup_store(lo[1], so + 8*(2*(ii + k) + 1), PV, 0, false); + } + } + } + } + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + if (FC_flash_attn_ext_has_sinks) { + FOR_UNROLL (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + + const float m = M[jj]; + const float s = tiisg == 0 ? ((device const float *) sinks)[iq2] : -FLT_MAX/2; + + M[jj] = simd_max(max(M[jj], s)); + + const float ms = exp(m - M[jj]); + const float vs = exp(s - M[jj]); + + S[jj] = S[jj]*ms + simd_sum(vs); + + for (short i = tiisg; i < DV4; i += NW) { + so4[j*PV4 + i] *= ms; + } + } + } + } + + // store to global memory + for (short jj = 0; jj < NQ; ++jj) { + const short j = jj*NSG + sgitg; + if (iq1 + j >= args.ne01) { + break; + } + + device float4 * dst4 = (device float4 *) dst + ((uint64_t)iq3*args.ne2*args.ne1 + iq2 + (uint64_t)(iq1 + j)*args.ne1)*DV4; + + const float scale = S[jj] == 0.0 ? 0.0f : 1.0f/S[jj]; + + if (DV4 % NW == 0) { + FOR_UNROLL (short ii = 0; ii < DV4/NW; ++ii) { + const short i = ii*NW + tiisg; + + dst4[i] = (float4) so4[j*PV4 + i]*scale; + } + } else { + for (short i = tiisg; i < DV4; i += NW) { + dst4[i] = (float4) so4[j*PV4 + i]*scale; + } + } + } + +#undef NS10 +#undef NS20 +} + +template< + typename q_t, // query types in shared memory + typename q4_t, + typename q8x8_t, + typename k_t, // key types in shared memory + typename k4x4_t, + typename k8x8_t, + typename v_t, // value types in shared memory + typename v4x4_t, + typename v8x8_t, + typename qk_t, // Q*K types + typename qk8x8_t, + typename s_t, // soft-max types + typename s2_t, + typename s8x8_t, + typename o_t, // attention accumulation types + typename o4_t, + typename o8x8_t, + typename kd4x4_t, // key type in device memory + short nl_k, + void (*deq_k)(device const kd4x4_t *, short, thread k4x4_t &), + typename vd4x4_t, // value type in device memory + short nl_v, + void (*deq_v)(device const vd4x4_t *, short, thread v4x4_t &), + short DK, // K head size + short DV, // V head size + short Q = OP_FLASH_ATTN_EXT_NQPSG, // queries per threadgroup + short C = OP_FLASH_ATTN_EXT_NCPSG> // cache items per threadgroup +kernel void kernel_flash_attn_ext( + constant ggml_metal_kargs_flash_attn_ext & args, + device const char * q, + device const char * k, + device const char * v, + device const char * mask, + device const char * sinks, + device const char * pad, + device const char * blk, + device char * dst, + threadgroup half * shmem_f16 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { +#define FWD_TMPL q_t, q4_t, q8x8_t, k_t, k4x4_t, k8x8_t, v_t, v4x4_t, v8x8_t, qk_t, qk8x8_t, s_t, s2_t, s8x8_t, o_t, o4_t, o8x8_t, kd4x4_t, nl_k, deq_k, vd4x4_t, nl_v, deq_v, DK, DV, Q, C +#define FWD_ARGS args, q, k, v, mask, sinks, pad, blk, dst, shmem_f16, tgpig, tiisg, sgitg + switch (FC_flash_attn_ext_nsg) { + // note: disabled cases to reduce library load time + //case 1: kernel_flash_attn_ext_impl(FWD_ARGS); break; + //case 2: kernel_flash_attn_ext_impl(FWD_ARGS); break; + case 4: kernel_flash_attn_ext_impl(FWD_ARGS); break; + case 8: kernel_flash_attn_ext_impl(FWD_ARGS); break; + } +#undef FWD_TMPL +#undef FWD_ARGS +} + +// TODO: this is quite ugly. in the future these types will be hardcoded in the kernel, but for now keep them as +// template to be able to explore different combinations +// +#define FA_TYPES \ + half, half4, simdgroup_half8x8, \ + half, half4x4, simdgroup_half8x8, \ + half, half4x4, simdgroup_half8x8, \ + float, simdgroup_float8x8, \ + float, float2, simdgroup_float8x8, \ + float, float4, simdgroup_float8x8 + //half, half4, simdgroup_half8x8 + +#define FA_TYPES_BF \ + bfloat, bfloat4, simdgroup_bfloat8x8, \ + bfloat, bfloat4x4, simdgroup_bfloat8x8, \ + bfloat, bfloat4x4, simdgroup_bfloat8x8, \ + float, simdgroup_float8x8, \ + float, float2, simdgroup_float8x8, \ + half, half4, simdgroup_half8x8 + //float, float4, simdgroup_float8x8 + +#define FA_TYPES_F32 \ + half, half4, simdgroup_half8x8, \ + float, float4x4, simdgroup_float8x8, \ + float, float4x4, simdgroup_float8x8, \ + float, simdgroup_float8x8, \ + float, float2, simdgroup_float8x8, \ + float, float4, simdgroup_float8x8 + //half, half4, simdgroup_half8x8 + +typedef decltype(kernel_flash_attn_ext) flash_attn_ext_t; + +template [[host_name("kernel_flash_attn_ext_f32_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f32_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +template [[host_name("kernel_flash_attn_ext_f16_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_f16_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_bf16_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_bf16_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +#endif + +template [[host_name("kernel_flash_attn_ext_q4_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +template [[host_name("kernel_flash_attn_ext_q4_1_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q4_1_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +template [[host_name("kernel_flash_attn_ext_q5_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +template [[host_name("kernel_flash_attn_ext_q5_1_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q5_1_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +template [[host_name("kernel_flash_attn_ext_q8_0_dk32_dv32" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk40_dv40" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk48_dv48" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk64_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk72_dv72" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk80_dv80" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk96_dv96" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk96_dv64" )]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk112_dv112")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk128_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk192_dv192")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk192_dv128")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk256_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk320_dv256")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk512_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; +template [[host_name("kernel_flash_attn_ext_q8_0_dk576_dv512")]] kernel flash_attn_ext_t kernel_flash_attn_ext; + +#undef FA_TYPES +#undef FA_TYPES_BF +#undef FA_TYPES_F32 + +constant bool FC_flash_attn_ext_vec_has_mask [[function_constant(FC_FLASH_ATTN_EXT_VEC + 0)]]; +constant bool FC_flash_attn_ext_vec_has_sinks [[function_constant(FC_FLASH_ATTN_EXT_VEC + 1)]]; +constant bool FC_flash_attn_ext_vec_has_bias [[function_constant(FC_FLASH_ATTN_EXT_VEC + 2)]]; +constant bool FC_flash_attn_ext_vec_has_scap [[function_constant(FC_FLASH_ATTN_EXT_VEC + 3)]]; +constant bool FC_flash_attn_ext_vec_has_kvpad [[function_constant(FC_FLASH_ATTN_EXT_VEC + 4)]]; + +//constant float FC_flash_attn_ext_vec_scale [[function_constant(FC_FLASH_ATTN_EXT_VEC + 10)]]; +//constant float FC_flash_attn_ext_vec_max_bias [[function_constant(FC_FLASH_ATTN_EXT_VEC + 11)]]; +//constant float FC_flash_attn_ext_vec_logit_softcap [[function_constant(FC_FLASH_ATTN_EXT_VEC + 12)]]; + +constant int32_t FC_flash_attn_ext_vec_ns10 [[function_constant(FC_FLASH_ATTN_EXT_VEC + 20)]]; +constant int32_t FC_flash_attn_ext_vec_ns20 [[function_constant(FC_FLASH_ATTN_EXT_VEC + 21)]]; +constant int32_t FC_flash_attn_ext_vec_nsg [[function_constant(FC_FLASH_ATTN_EXT_VEC + 22)]]; +constant int32_t FC_flash_attn_ext_vec_nwg [[function_constant(FC_FLASH_ATTN_EXT_VEC + 23)]]; +constant bool FC_flash_attn_ext_vec_has_sparse [[function_constant(FC_FLASH_ATTN_EXT_VEC + 5)]]; + +// compress the finite entries of each KQ mask row into a list of KV indices (ascending order), +// padded with -1 up to n_kv_max_padded (a multiple of OP_FLASH_ATTN_EXT_VEC_NCPSG) +// one threadgroup per mask row; the mask remains the single source of truth for the values +kernel void kernel_flash_attn_ext_vec_idx( + constant ggml_metal_kargs_flash_attn_ext_vec_idx & args, + device const half * mask, + device int * idx, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + constexpr short NW = N_SIMDWIDTH; + constexpr short NLOCAL = 32; // max finite positions kept in registers per thread + + const int i1 = tgpig[0]; + const int i2 = tgpig[1]; + const int i3 = tgpig[2]; + + device const half * pm = (device const half *) ((device const char *) mask + i1*args.nb31 + i2*args.nb32 + i3*args.nb33); + device int * pidx = idx + (((int64_t)i3*args.ne32 + i2)*args.ne31 + i1)*args.n_kv_max_padded; + + const int n = args.ne30; + const int q = n/ntg.x; + const int r = n%ntg.x; + + // each thread handles a contiguous slice of the mask row + const int r0 = q*tiitg + min((int) tiitg, r); + const int r1 = r0 + q + (tiitg < r ? 1 : 0); + + // count the finite entries in the slice and keep their positions in registers (single mask read) + int cnt = 0; // total finite entries in the slice + int nloc = 0; // finite entries kept in registers + int local[NLOCAL]; + for (int i = r0; i < r1; ++i) { + if (isfinite((float) pm[i])) { + if (nloc < NLOCAL) { + local[nloc] = i; + nloc++; + } + cnt++; + } + } + + const short sgitg = tiitg/NW; + const short tiisg = tiitg%NW; + + threadgroup int tcount[8]; + + // simd_sum is a collective: all lanes must evaluate it + const int sg_sum = simd_sum(cnt); + if (tiisg == 0) { + tcount[sgitg] = sg_sum; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + int total = 0; + for (short s = 0; s < ntg.x/NW; ++s) { + total += tcount[s]; + } + + // base offset of this thread's slice in the output list (exclusive scan within the simdgroup) + int sg_base = 0; + for (short s = 0; s < sgitg; ++s) { + sg_base += tcount[s]; + } + + // exclusive prefix scan of the per-thread counts within the simdgroup + int incl = cnt; + for (int d = 1; d < NW; d <<= 1) { + const int v = simd_shuffle_up(incl, d); + if (tiisg >= d) { + incl += v; + } + } + const int base = sg_base + (incl - cnt); + + // write the finite positions in order; if the hint is violated, keep only the first n_kv_max entries + int j = 0; + for (; j < nloc && base + j < args.n_kv_max; ++j) { + pidx[base + j] = local[j]; + } + + // a dense mask may have more than NLOCAL finite entries in a slice; re-read the mask to write the rest + if (cnt > nloc && base + nloc < args.n_kv_max) { + int j2 = 0; + for (int i = r0; i < r1; ++i) { + if (isfinite((float) pm[i])) { + if (j2 >= nloc) { + pidx[base + j2] = i; + } + j2++; + if (base + j2 >= args.n_kv_max) { + break; + } + } + } + } + + // pad the tail of the list with -1 + const int count = min(total, args.n_kv_max); + for (int i = count + tiitg; i < args.n_kv_max_padded; i += ntg.x) { + pidx[i] = -1; + } +} + +template< + typename q4_t, // query types in shared memory + typename k4_t, // key types in shared memory + typename v4_t, // value types in shared memory + typename qk_t, // Q*K types + typename s_t, // soft-max types + typename s4_t, + typename o4_t, // attention accumulation types + typename kd4_t, // key type in device memory + short nl_k, + void (*deq_k_t4)(device const kd4_t *, short, thread k4_t &), + typename vd4_t, // value type in device memory + short nl_v, + void (*deq_v_t4)(device const vd4_t *, short, thread v4_t &), + short DK, // K head size + short DV, // V head size + short NE = 4, // head elements per thread + short Q = OP_FLASH_ATTN_EXT_VEC_NQPSG, // queries per threadgroup + short C = OP_FLASH_ATTN_EXT_VEC_NCPSG> // cache items per threadgroup + +kernel void kernel_flash_attn_ext_vec( + constant ggml_metal_kargs_flash_attn_ext_vec & args, + device const char * q, + device const char * k, + device const char * v, + device const char * mask, + device const char * sinks, + device const char * pad, + device char * dst, + device const char * idx, + threadgroup half * shmem_f16 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + static_assert(DK % 32 == 0, "DK must be divisible by 32"); + static_assert(DV % 32 == 0, "DV must be divisible by 32"); + +#define NWG (FC_flash_attn_ext_vec_nwg) +#define NSG (FC_flash_attn_ext_vec_nsg) + +#define NS10 (FC_flash_attn_ext_vec_ns10) +#define NS20 (FC_flash_attn_ext_vec_ns20) + + const short iwg = tgpig[2]%NWG; + + const ushort iq3 = tgpig[2]/NWG; + const ushort iq2 = tgpig[1]; + const ushort iq1 = tgpig[0]; + + constexpr short DK4 = DK/4; + constexpr short DV4 = DV/4; + + constexpr short PK = PAD2(DK, 128); + constexpr short PK4 = PK/4; + + constexpr short PV = PAD2(DV, 128); + constexpr short PV4 = PV/4; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NL = NW/NE; // note: this can be adjusted to support different head sizes and simdgroup work loads + constexpr short SH = 4*Q*C; // shared memory per simdgroup + + static_assert(DK4 % NL == 0, "DK4 must be divisible by NL"); + static_assert(DV4 % NL == 0, "DV4 must be divisible by NL"); + + //const short T = PK + NSG*SH; // shared memory size per query in (half) + + //threadgroup q_t * sq = (threadgroup q_t *) (shmem_f16 + 0*PK); // holds the query data + threadgroup q4_t * sq4 = (threadgroup q4_t *) (shmem_f16 + 0*PK); // same as above but in q4_t + threadgroup s_t * ss = (threadgroup s_t *) (shmem_f16 + sgitg*SH + Q*NSG*PK); // scratch buffer for attention + threadgroup s4_t * ss4 = (threadgroup s4_t *) (shmem_f16 + sgitg*SH + Q*NSG*PK); // same as above but in s4_t + threadgroup half * sm = (threadgroup half *) (shmem_f16 + sgitg*SH + 2*Q*C + Q*NSG*PK); // scratch buffer for mask + threadgroup o4_t * so4 = (threadgroup o4_t *) (shmem_f16 + 2*sgitg*Q*PV + Q*NSG*PK + NSG*SH); // scratch buffer for the results + + // store the result for all queries in shared memory (the O matrix from the paper) + so4 += tiisg; + + { + q += iq1*Q*args.nb01 + iq2*args.nb02 + iq3*args.nb03; + + const short ikv2 = iq2/(args.ne02/args.ne_12_2); + const short ikv3 = iq3/(args.ne03/args.ne_12_3); + + k += ikv2*args.nb12 + ikv3*args.nb13; + v += ikv2*args.nb22 + ikv3*args.nb23; + } + + // load Q query rows to shared memory + { + for (short qq = 0; qq < Q; ++qq) { + const int iq1_q = iq1*Q + qq; + device const float4 * q4 = (device const float4 *) ((device const char *) q + qq*args.nb01); + if (iq1_q < args.ne01) { + for (short i = tiisg; i < PK4; i += NW) { + if (i < DK4) { + sq4[qq*PK4 + i] = (q4_t) q4[i]; + } else { + sq4[qq*PK4 + i] = (q4_t) 0.0f; + } + } + } else { + for (short i = tiisg; i < PK4; i += NW) { + sq4[qq*PK4 + i] = (q4_t) 0.0f; + } + } + } + } + + // zero out so + for (short qq = 0; qq < Q; ++qq) { + for (short i = 0; i < DV4/NL; ++i) { + so4[qq*DV4 + i*NL] = (o4_t) 0.0f; + } + } + + // zero out shared memory SH + for (short i = tiisg; i < SH/4; i += NW) { + ss4[i] = (s4_t) 0.0f; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + { + float S[Q]; + float M[Q]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + S[qq] = 0.0f; + M[qq] = -FLT_MAX/2; + } + + // thread indices inside the simdgroup + const short tx = tiisg%NL; + const short ty = tiisg/NL; + + // pointer to the mask + device const half * pm_base = (device const half *) (mask + iq1*Q*args.nb31 + (iq2%args.ne32)*args.nb32 + (iq3%args.ne33)*args.nb33); + + // sparse indices: the list of finite mask entries per query row + // the sparse path requires Q == 1 (enforced by the host) + device const int * pidx = nullptr; + if (FC_flash_attn_ext_vec_has_sparse) { + pidx = (device const int *) idx + + ((int64_t)(iq3%args.ne33)*args.ne32 + (iq2%args.ne32))*args.ne31*args.n_kv_max_padded + (iq1%args.ne31)*args.n_kv_max_padded; + } + + float slope = 1.0f; + + // ALiBi + if (FC_flash_attn_ext_vec_has_bias) { + const short h = iq2; + + const float base = h < args.n_head_log2 ? args.m0 : args.m1; + const short exph = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; + + slope = pow(base, exph); + } + + // loop over the KV cache + // each simdgroup handles blocks of Q rows and C columns + for (int ic0 = iwg*NSG + sgitg; ; ic0 += NWG*NSG) { + int ic = ic0*C; + if (ic >= args.ne11) { + break; + } + + device const half * pm[Q]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + // padded query rows clamp to row 0 of the mask to avoid OOB; their scores + // are forced to -inf below, so the values never affect the result. + pm[qq] = pm_base + ((iq1*Q + qq) < args.ne01 ? qq*(args.nb31/sizeof(half)) : -iq1*Q*(args.nb31/sizeof(half))); + } + + // the last partial chunk uses the pad buffer as source + if (FC_flash_attn_ext_vec_has_kvpad && ic + C > args.ne11) { + k = pad; + v = k + args.nb11*C*args.ne_12_2*args.ne_12_3; + mask = v + args.nb21*C*args.ne_12_2*args.ne_12_3; + + const short ikv2 = iq2/(args.ne02/args.ne_12_2); + const short ikv3 = iq3/(args.ne03/args.ne_12_3); + + k += (ikv2 + ikv3*args.ne_12_2)*args.nb11*C; + v += (ikv2 + ikv3*args.ne_12_2)*args.nb21*C; + + if (!FC_flash_attn_ext_vec_has_mask) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if (ic + tiisg >= args.ne11) { + sm[qq*C + tiisg] = -MAXHALF; + } + } + } else { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + pm[qq] = (device const half *) (mask) + + (iq1*Q + qq)*C + + (iq2%args.ne32)*(C*args.ne31) + + (iq3%args.ne33)*(C*args.ne31*args.ne32); + } + } + + ic = 0; + } + + if (FC_flash_attn_ext_vec_has_mask) { + if (FC_flash_attn_ext_vec_has_sparse) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + const int i11 = pidx[ic + tiisg]; + if ((iq1*Q + qq) < args.ne01 && i11 >= 0) { + sm[qq*C + tiisg] = pm[qq][i11]; + } else { + sm[qq*C + tiisg] = -MAXHALF; + } + } + } else { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if ((iq1*Q + qq) < args.ne01) { + sm[qq*C + tiisg] = pm[qq][ic + tiisg]; + } else { + sm[qq*C + tiisg] = -MAXHALF; + } + } + } + } else { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if ((iq1*Q + qq) >= args.ne01) { + sm[qq*C + tiisg] = -MAXHALF; + } + } + } + + // skip -INF mask + { + bool any_finite = false; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if (simd_max(sm[qq*C + tiisg]) > -MAXHALF) { + any_finite = true; + } + } + if (!any_finite) { + continue; + } + } + + // Q*K^T + { + device const k4_t * pk4 = nullptr; + + if (!FC_flash_attn_ext_vec_has_sparse) { + pk4 = (device const k4_t *) (k + ic*args.nb11); + + pk4 += ty*NS10/4 + tx; + } + + qk_t mqk[Q][C/NE]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { + mqk[qq][cc] = 0.0f; + } + } + + // each simdgroup processes Q queries and NE (NW/NL) cache elements + FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { + if (FC_flash_attn_ext_vec_has_sparse) { + // the KV rows are gathered from the index list; -1 entries are padding + const int i11 = pidx[ic + NE*cc + ty]; + if (i11 >= 0) { + if (is_same::value) { + device const k4_t * pk4s = (device const k4_t *) (k + i11*args.nb11) + tx; + FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { + const k4_t k_elem = pk4s[ii*NL]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + mqk[qq][cc] += dot((float4) k_elem, (float4) sq4[qq*PK4 + ii*NL + tx]); + } + } + } else { + device const kd4_t * pk = (device const kd4_t *) (k + i11*args.nb11); + + k4_t mk; + + FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { + const short i = ii*NL + tx; + + deq_k_t4(pk + i/nl_k, i%nl_k, mk); + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + mqk[qq][cc] += dot((float4) mk, (float4) sq4[qq*PK4 + i]); + } + } + } + } + } else if (is_same::value) { + FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { + const k4_t k_elem = pk4[cc*NE*NS10/4 + ii*NL]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + mqk[qq][cc] += dot((float4) k_elem, (float4) sq4[qq*PK4 + ii*NL + tx]); + } + } + } else { + device const kd4_t * pk = (device const kd4_t *) (k + ((ic + NE*cc + ty)*args.nb11)); + + k4_t mk; + + FOR_UNROLL (short ii = 0; ii < DK4/NL; ++ii) { + const short i = ii*NL + tx; + + deq_k_t4(pk + i/nl_k, i%nl_k, mk); + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + mqk[qq][cc] += dot((float4) mk, (float4) sq4[qq*PK4 + i]); + } + } + } + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if (NE == 1) { + mqk[qq][cc] = simd_sum(mqk[qq][cc]); + } else { + // simdgroup reduce (NE = 4) + // [ 0 .. 7] -> [ 0] + // [ 8 .. 15] -> [ 8] + // [16 .. 23] -> [16] + // [24 .. 31] -> [24] + if (NE <= 1) { + mqk[qq][cc] += simd_shuffle_down(mqk[qq][cc], 16); + } + if (NE <= 2) { + mqk[qq][cc] += simd_shuffle_down(mqk[qq][cc], 8); + } + if (NE <= 4) { + mqk[qq][cc] += simd_shuffle_down(mqk[qq][cc], 4); + } + if (NE <= 8) { + mqk[qq][cc] += simd_shuffle_down(mqk[qq][cc], 2); + } + if (NE <= 16) { + mqk[qq][cc] += simd_shuffle_down(mqk[qq][cc], 1); + } + + // broadcast + mqk[qq][cc] = simd_shuffle(mqk[qq][cc], NL*ty); + } + } + } + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + if (FC_flash_attn_ext_vec_has_mask && + !FC_flash_attn_ext_vec_has_scap && + !FC_flash_attn_ext_vec_has_bias) { + ss[qq*C + NE*tx + ty] = fma(mqk[qq][tx], args.scale, (qk_t) sm[qq*C + NE*tx + ty]); + } else { + mqk[qq][tx] *= args.scale; + + if (FC_flash_attn_ext_vec_has_scap) { + mqk[qq][tx] = args.logit_softcap*precise::tanh(mqk[qq][tx]); + } + + if (FC_flash_attn_ext_vec_has_bias) { + mqk[qq][tx] += (qk_t) sm[qq*C + NE*tx + ty]*slope; + } else { + mqk[qq][tx] += (qk_t) sm[qq*C + NE*tx + ty]; + } + + ss[qq*C + NE*tx + ty] = mqk[qq][tx]; + } + } + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + // online softmax + { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + const float m = M[qq]; + const float s = ss[qq*C + tiisg]; + + M[qq] = simd_max(max(M[qq], s)); + + const float ms = exp(m - M[qq]); + const float vs = exp(s - M[qq]); + + S[qq] = S[qq]*ms + simd_sum(vs); + + // the P matrix from the paper (Q rows, C columns) + ss[qq*C + tiisg] = vs; + + // O = diag(ms)*O + if ((DV4/NL % NW == 0) || ty == 0) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + so4[qq*DV4 + ii*NL] *= ms; + } + } + } + } + + simdgroup_barrier(mem_flags::mem_threadgroup); + + // O = O + (Q*K^T)*V + { + o4_t lo[Q][DV4/NL]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + lo[qq][ii] = 0.0f; + } + } + + if (FC_flash_attn_ext_vec_has_sparse) { + FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { + // the KV rows are gathered from the index list; -1 entries are padding + const int i11 = pidx[ic + NE*cc + ty]; + if (i11 >= 0) { + if (is_same::value) { + device const v4_t * pv4 = (device const v4_t *) (v + i11*args.nb21); + + pv4 += tx; + + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + const v4_t v_elem = pv4[ii*NL]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + lo[qq][ii] += o4_t(float4(v_elem)*float4(ss[qq*C + cc*NE + ty])); + } + } + } else { + device const vd4_t * pv4 = (device const vd4_t *) (v + i11*args.nb21); + + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + const short i = ii*NL + tx; + + v4_t mv; + + deq_v_t4(pv4 + i/nl_v, i%nl_v, mv); + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + lo[qq][ii] += o4_t(float4(mv)*float4(ss[qq*C + cc*NE + ty])); + } + } + } + } + } + } else if (is_same::value) { + device const v4_t * pv4 = (device const v4_t *) (v + ic*args.nb21); + + pv4 += ty*NS20/4 + tx; + + FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + const v4_t v_elem = pv4[cc*NE*NS20/4 + ii*NL]; + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + lo[qq][ii] += o4_t(float4(v_elem)*float4(ss[qq*C + cc*NE + ty])); + } + } + } + } else { + FOR_UNROLL (short cc = 0; cc < C/NE; ++cc) { + device const vd4_t * pv4 = (device const vd4_t *) (v + ((ic + NE*cc + ty)*args.nb21)); + + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + const short i = ii*NL + tx; + + v4_t mv; + deq_v_t4(pv4 + i/nl_v, i%nl_v, mv); + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + lo[qq][ii] += o4_t(float4(mv)*float4(ss[qq*C + NE*cc + ty])); + } + } + } + } + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + if (NE > 1) { + lo[qq][ii][0] += simd_shuffle_down(lo[qq][ii][0], 16); + lo[qq][ii][1] += simd_shuffle_down(lo[qq][ii][1], 16); + lo[qq][ii][2] += simd_shuffle_down(lo[qq][ii][2], 16); + lo[qq][ii][3] += simd_shuffle_down(lo[qq][ii][3], 16); + } + + if (NE > 2) { + lo[qq][ii][0] += simd_shuffle_down(lo[qq][ii][0], 8); + lo[qq][ii][1] += simd_shuffle_down(lo[qq][ii][1], 8); + lo[qq][ii][2] += simd_shuffle_down(lo[qq][ii][2], 8); + lo[qq][ii][3] += simd_shuffle_down(lo[qq][ii][3], 8); + } + + if (NE > 4) { + lo[qq][ii][0] += simd_shuffle_down(lo[qq][ii][0], 4); + lo[qq][ii][1] += simd_shuffle_down(lo[qq][ii][1], 4); + lo[qq][ii][2] += simd_shuffle_down(lo[qq][ii][2], 4); + lo[qq][ii][3] += simd_shuffle_down(lo[qq][ii][3], 4); + } + + if (NE > 8) { + lo[qq][ii][0] += simd_shuffle_down(lo[qq][ii][0], 2); + lo[qq][ii][1] += simd_shuffle_down(lo[qq][ii][1], 2); + lo[qq][ii][2] += simd_shuffle_down(lo[qq][ii][2], 2); + lo[qq][ii][3] += simd_shuffle_down(lo[qq][ii][3], 2); + } + + if (NE > 16) { + lo[qq][ii][0] += simd_shuffle_down(lo[qq][ii][0], 1); + lo[qq][ii][1] += simd_shuffle_down(lo[qq][ii][1], 1); + lo[qq][ii][2] += simd_shuffle_down(lo[qq][ii][2], 1); + lo[qq][ii][3] += simd_shuffle_down(lo[qq][ii][3], 1); + } + } + } + + if ((DV4/NL % NW == 0) || ty == 0) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + so4[qq*DV4 + ii*NL] += lo[qq][ii]; + } + } + } + } + } + + if (FC_flash_attn_ext_vec_has_sinks && sgitg == 0 && iwg == 0) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + const float m = M[qq]; + const float s = tiisg == 0 ? ((device const float *) sinks)[iq2] : -FLT_MAX/2; + + M[qq] = simd_max(max(M[qq], s)); + + const float ms = exp(m - M[qq]); + const float vs = exp(s - M[qq]); + + S[qq] = S[qq]*ms + simd_sum(vs); + + if ((DV4/NL % NW == 0) || ty == 0) { + FOR_UNROLL (short ii = 0; ii < DV4/NL; ++ii) { + so4[qq*DV4 + ii*NL] *= ms; + } + } + } + } + + // these are needed for reducing the results from the simdgroups (reuse the ss buffer) + if (tiisg == 0) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + ss[2*qq + 0] = (s_t) S[qq]; + ss[2*qq + 1] = (s_t) M[qq]; + } + } + } + + so4 -= tiisg; + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // parallel reduce + for (short r = NSG/2; r > 0; r >>= 1) { + if (sgitg < r) { + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + const float S0 = ss[ 2*qq + 0]; + const float S1 = ss[r*(SH/2) + 2*qq + 0]; + + const float M0 = ss[ 2*qq + 1]; + const float M1 = ss[r*(SH/2) + 2*qq + 1]; + + const float Mx = max(M0, M1); + + const float ms0 = exp(M0 - Mx); + const float ms1 = exp(M1 - Mx); + + const float Sx = S0*ms0 + S1*ms1; + + if (tiisg == 0) { + ss[2*qq + 0] = Sx; + ss[2*qq + 1] = Mx; + } + + // O_0 = diag(ms0)*O_0 + diag(ms1)*O_1 + for (short i = tiisg; i < DV4; i += NW) { + so4[qq*DV4 + i] = so4[qq*DV4 + i]*ms0 + so4[qq*DV4 + i + r*Q*PV4]*ms1; + } + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + // final rescale with 1/S and store to global memory + if (sgitg == 0) { + const int64_t nrows = args.ne3*args.ne2*args.ne1; + + device float4 * dst4 = (device float4 *) dst; + device float * dst1 = (device float *) dst + nrows*DV*NWG; // the S and M are stored after the results + + FOR_UNROLL (short qq = 0; qq < Q; ++qq) { + const int iq1_q = iq1*Q + qq; + if (iq1_q >= args.ne01) { + continue; + } + + const int64_t rid = iq3*args.ne2*args.ne1 + iq2 + iq1_q*args.ne1; + + const float Sval = NWG == 1 ? (ss[2*qq + 0] == 0.0f ? 0.0f : 1.0f/ss[2*qq + 0]) : 1.0f; + + // interleave the workgroup data + for (short i = tiisg; i < DV4; i += NW) { + dst4[rid*DV4*NWG + NWG*i + iwg] = (float4) so4[qq*DV4 + i]*Sval; + } + + // store S and M + if (NWG > 1) { + if (tiisg == 0) { + dst1[rid*(2*NWG) + 2*iwg + 0] = ss[2*qq + 0]; + dst1[rid*(2*NWG) + 2*iwg + 1] = ss[2*qq + 1]; + } + } + } + } + +#undef NWG +#undef NSG +#undef NS10 +#undef NS20 +} + +// note: I think the s_t can be half instead of float, because the Q*K scaling is done before storing to shared mem +// in the other (non-vec) kernel, we need s_t to also be float because we scale during the soft_max +// +#define FA_TYPES \ + half4, \ + half4, \ + half4, \ + float, \ + float, float4, \ + float4 + +#define FA_TYPES_F32 \ + half4, \ + float4, \ + float4, \ + float, \ + float, float4, \ + float4 + +typedef decltype(kernel_flash_attn_ext_vec) flash_attn_ext_vec_t; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk32_dv32")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk32_dv32_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk32_dv32_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk64_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv96")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv96_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv96_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv64")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv64_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk96_dv64_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk128_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv192_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk192_dv128_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk256_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk320_dv256_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q1_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q2_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q4_ne1")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk512_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + +template [[host_name("kernel_flash_attn_ext_vec_f32_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_f16_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_flash_attn_ext_vec_bf16_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +#endif +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_0_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q4_1_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_0_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q5_1_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512_q1_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512_q2_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512_q2_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512_q4_ne2")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; +template [[host_name("kernel_flash_attn_ext_vec_q8_0_dk576_dv512_q4_ne4")]] kernel flash_attn_ext_vec_t kernel_flash_attn_ext_vec; + + +#undef FA_TYPES +#undef FA_TYPES_F32 + +constant int32_t FC_flash_attn_ext_vec_reduce_DV [[function_constant(FC_FLASH_ATTN_EXT_VEC_REDUCE + 0)]]; +constant int32_t FC_flash_attn_ext_vec_reduce_NWG [[function_constant(FC_FLASH_ATTN_EXT_VEC_REDUCE + 1)]]; + +kernel void kernel_flash_attn_ext_vec_reduce( + constant ggml_metal_kargs_flash_attn_ext_vec_reduce & args, + device const char * htmp, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { +#define NWG (FC_flash_attn_ext_vec_reduce_NWG) +#define DV (FC_flash_attn_ext_vec_reduce_DV) + + const uint64_t rid = tgpig; + + const short iwg = tiisg; + + device const float * ss = (device const float *) htmp + (uint64_t)args.nrows*DV*NWG; + + float S = ss[rid*(2*NWG) + 2*iwg + 0]; + float M = ss[rid*(2*NWG) + 2*iwg + 1]; + + const float m = simd_max(M); + const float ms = exp(M - m); + + S = simd_sum(S*ms); + S = S == 0.0f ? 0.0f : 1.0f/S; + + const short DV4 = DV/4; + + device const float4 * htmp4 = (device const float4 *) htmp + rid*DV4*NWG; + device float4 * dst4 = (device float4 *) dst + rid*DV4; + + for (short i = sgitg; i < DV4; i += NWG) { + const float4 v = simd_sum(htmp4[i*NWG + iwg]*ms); + + if (iwg == 0) { + dst4[i] = v*S; + } + } + +#undef NWG +#undef DV +} + +template< + typename kd4x4_t, + short nl_k, + void (*deq_k)(device const kd4x4_t *, short, thread half4x4 &)> +kernel void kernel_lightning_indexer( + constant ggml_metal_kargs_lightning_indexer & args, + device const char * q, + device const char * k, + device const char * w, + device const char * m, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + constexpr short DK = OP_LIGHTNING_INDEXER_DK; + constexpr short NH = OP_LIGHTNING_INDEXER_NH; + constexpr short NHPTG = OP_LIGHTNING_INDEXER_NHPTG; + constexpr short NKPSG = OP_LIGHTNING_INDEXER_NKPSG; + constexpr short NSG = OP_LIGHTNING_INDEXER_NSG; + constexpr short NBPTG = OP_LIGHTNING_INDEXER_NBPTG; + + constexpr short DK4 = DK/4; + constexpr short DK8 = DK/8; + constexpr short DK16 = DK/16; + + constexpr short NK = NKPSG*NSG; // keys per threadgroup + constexpr short NTG = 32*NSG; // threads per threadgroup + + const int i_stream = tgpig.z; + const int i_kv_0 = tgpig.x*NK; // first key of this threadgroup + const int i_kv = i_kv_0 + sgitg*NKPSG; // first key of this simdgroup + + threadgroup half sk[NK * DK16 * 16]; + threadgroup half4x4 * sk4x4 = (threadgroup half4x4 *) sk; + + for (short i = tiitg; i < NK*DK16; i += NTG) { + const short ik = i/DK16; + const short i16 = i%DK16; + + half4x4 tmp; + + if (i_kv_0 + ik < args.n_kv) { + device const kd4x4_t * kr = (device const kd4x4_t *) (k + (i_kv_0 + ik)*args.nbk2 + i_stream*args.nbk3); + + deq_k(kr + i16/nl_k, i16%nl_k, tmp); + } else { + FOR_UNROLL (short j = 0; j < 4; ++j) { + tmp[j] = half4(0.0h); + } + } + + sk4x4[i] = tmp; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // K tile of this simdgroup, transposed to [DK, NKPSG] + simdgroup_half8x8 mk[DK8]; + + FOR_UNROLL (short i = 0; i < DK8; ++i) { + simdgroup_load(mk[i], sk + sgitg*NKPSG*DK + 8*i, DK, 0, true); + } + + threadgroup half4 sq4[NHPTG*DK4]; + threadgroup half * sq = (threadgroup half *) sq4; + + threadgroup float sw [NHPTG]; + threadgroup float sqk[NSG*NHPTG*NKPSG]; + + const int i_batch_0 = tgpig.y*NBPTG; + const int n_batch = min((int) NBPTG, args.n_batch - i_batch_0); + + for (short ib = 0; ib < n_batch; ++ib) { + const int i_batch = i_batch_0 + ib; + + device const char * pq = q + i_batch*args.nbq2 + i_stream*args.nbq3; + device const char * pw = w + i_batch*args.nbw1 + i_stream*args.nbw3; + + float score = 0.0f; + + FOR_UNROLL (short i_head = 0; i_head < NH; i_head += NHPTG) { + // stage the Q tile [DK, NHPTG] and the (prescaled) head weights + for (short i = tiitg; i < NHPTG*DK4; i += NTG) { + const short ih = i/DK4; + const short i4 = i%DK4; + + device const float4 * q4 = (device const float4 *) (pq + (i_head + ih)*args.nbq1); + + sq4[ih*DK4 + i4] = half4(q4[i4]); + } + + if (tiitg < NHPTG) { + sw[tiitg] = ((device const float *) pw)[i_head + tiitg]; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + simdgroup_float8x8 mqk = make_filled_simdgroup_matrix(0.0f); + + FOR_UNROLL (short i = 0; i < DK8; ++i) { + simdgroup_half8x8 mq; + + simdgroup_load(mq, sq + 8*i, DK, 0, false); + simdgroup_multiply_accumulate(mqk, mq, mk[i], mqk); + } + + threadgroup float * pqk = sqk + sgitg*NHPTG*NKPSG; + + simdgroup_store(mqk, pqk, NKPSG, 0, false); + simdgroup_barrier(mem_flags::mem_threadgroup); + + // one lane per key: ReLU, apply the head weight and accumulate over the head tile + if (tiisg < NKPSG) { + FOR_UNROLL (short ih = 0; ih < NHPTG; ++ih) { + score += max(pqk[ih*NKPSG + tiisg], 0.0f)*sw[ih]; + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + if (tiisg < NKPSG) { + const int ik = i_kv + tiisg; + if (ik < args.n_kv) { + device const half * pm = (device const half *) (m + i_batch*args.nbm1 + (i_stream % args.mask_ne3)*args.nbm3); + device float * pd = (device float *) (dst + i_batch*args.nb1 + i_stream*args.nb3); + + pd[ik] = score + (float) pm[ik]; + } + } + } +} + +typedef decltype(kernel_lightning_indexer) kernel_lightning_indexer_t; + +template [[host_name("kernel_lightning_indexer_f32")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +template [[host_name("kernel_lightning_indexer_f16")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; + +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_lightning_indexer_bf16")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +#endif + +template [[host_name("kernel_lightning_indexer_q4_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +template [[host_name("kernel_lightning_indexer_q4_1")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +template [[host_name("kernel_lightning_indexer_q5_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +template [[host_name("kernel_lightning_indexer_q5_1")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; +template [[host_name("kernel_lightning_indexer_q8_0")]] kernel kernel_lightning_indexer_t kernel_lightning_indexer; diff --git a/ggml/src/ggml-metal/kernels/gated_delta_net.metal b/ggml/src/ggml-metal/kernels/gated_delta_net.metal new file mode 100644 index 00000000..5e4861ec --- /dev/null +++ b/ggml/src/ggml-metal/kernels/gated_delta_net.metal @@ -0,0 +1,264 @@ +#include "common.h" + +constant short FC_gated_delta_net_ne20 [[function_constant(FC_GATED_DELTA_NET + 0)]]; +constant short FC_gated_delta_net_ne30 [[function_constant(FC_GATED_DELTA_NET + 1)]]; +constant short FC_gated_delta_net_K [[function_constant(FC_GATED_DELTA_NET + 2)]]; + +#if 1 +template +kernel void kernel_gated_delta_net_impl( + constant ggml_metal_kargs_gated_delta_net & args, + device const char * q, + device const char * k, + device const char * v, + device const char * g, + device const char * b, + device const char * s, + device char * dst, + device char * dst_fuse, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { +#define S_v FC_gated_delta_net_ne20 +#define G FC_gated_delta_net_ne30 +#define K FC_gated_delta_net_K + + const uint tx = tpitg.x; + const uint ty = tpitg.y; + + const uint i23 = tgpig.z; // B (n_seqs) + const uint i21 = tgpig.y; // H (head) + const uint i20 = tgpig.x*NSG + ty; // row within S_v + + const uint i01 = i21 % args.ne01; + const uint i11 = i21 % args.ne11; + + const float scale = 1.0f / sqrt((float)S_v); + + // input state layout [S_v, S_v, H, n_seqs] (s0 only): per-seq stride is H*D. + // state is stored transposed: M[i20][is] = S[is][i20], so row i20 is contiguous + const uint state_in_base = (i23*args.ne21 + i21)*S_v*S_v + i20*S_v; + device const float * s_ptr = (device const float *) (s) + state_in_base; + + float ls[NSG]; + + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + ls[j] = s_ptr[is]; + } + + device float * dst_attn = (device float *) (dst) + (i23*args.ne22*args.ne21 + i21)*S_v + i20; + + device const float * q_ptr = (device const float *) (q + i23*args.nb03 + i01*args.nb01); + device const float * k_ptr = (device const float *) (k + i23*args.nb13 + i11*args.nb11); + device const float * v_ptr = (device const float *) (v + i23*args.nb23 + i21*args.nb21); + + device const float * b_ptr = (device const float *) (b) + (i23*args.ne22*args.ne21 + i21); + device const float * g_ptr = (device const float *) (g) + (i23*args.ne22*args.ne21 + i21)*G; + + // snapshot slot mapping: slot 0 = most recent state, slot s = s tokens back. + // When n_tokens < K, only slots 0..n_tokens-1 are written; older slots are caller-owned. + + // output state base offset: after attention scores + const uint attn_size = args.ne22 * args.ne21 * S_v * args.ne23; + // output state per-slot size: S_v * S_v * H * n_seqs + const uint state_size_per_snap = S_v * S_v * args.ne21 * args.ne23; + // per-(seq,head) offset within a slot + const uint state_out_base = (i23*args.ne21 + i21)*S_v*S_v + i20*S_v; + + // when fused with the cache cpy, write the snapshots straight into the cache buffer using + // the slot stride; otherwise append them after the attn scores (nb_out == 0) + const bool fused = args.nb_out > 0; + const device float * state_out = fused ? (device float *)dst_fuse : (device float *)dst + attn_size; + const uint slot_stride = fused ? (uint)args.nb_out : state_size_per_snap; + + for (short t = 0; t < args.ne22; t++) { + float s_k = 0.0f; + + if (G == 1) { + const float g_exp = exp(g_ptr[0]); + + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + ls[j] *= g_exp; + + s_k += ls[j]*k_ptr[is]; + } + } else { + // KDA + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + ls[j] *= exp(g_ptr[is]); + + s_k += ls[j]*k_ptr[is]; + } + } + + s_k = simd_sum(s_k); + + const float d = (v_ptr[i20] - s_k)*b_ptr[0]; + + float y = 0.0f; + + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + ls[j] += k_ptr[is]*d; + + y += ls[j]*q_ptr[is]; + } + + y = simd_sum(y); + + if (tx == 0) { + dst_attn[t*args.ne21*S_v] = y*scale; + } + + q_ptr += args.ns02; + k_ptr += args.ns12; + v_ptr += args.ns22; + + b_ptr += args.ne21; + g_ptr += args.ne21*G; + + if (K > 1) { + const int target_slot = (int)args.ne22 - 1 - (int)t; + if (target_slot >= 0 && target_slot < (int)K) { + device float * dst_state = (device float *)state_out + (uint)target_slot * slot_stride + state_out_base; + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + dst_state[is] = ls[j]; + } + } + } + } + + if (K == 1) { + device float * dst_state = (device float *)state_out + state_out_base; + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + dst_state[is] = ls[j]; + } + } + +#undef S_v +#undef G +#undef K +} + +typedef decltype(kernel_gated_delta_net_impl<4>) kernel_gated_delta_net_t; + +template [[host_name("kernel_gated_delta_net_f32_1")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<1>; +template [[host_name("kernel_gated_delta_net_f32_2")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<2>; +template [[host_name("kernel_gated_delta_net_f32_4")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl<4>; + +#else +// a simplified version of the above +// no performance improvement, so keep the above version for now + +template +kernel void kernel_gated_delta_net_impl( + constant ggml_metal_kargs_gated_delta_net & args, + device const char * q, + device const char * k, + device const char * v, + device const char * g, + device const char * b, + device const char * s, + device char * dst, + device char * dst_fuse, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { +#define S_v FC_gated_delta_net_ne20 +#define G FC_gated_delta_net_ne30 + + const uint tx = tpitg.x; + const uint ty = tpitg.y; + + const uint i23 = tgpig.z; // B + const uint i21 = tgpig.y; // H + const uint i20 = tgpig.x*NSG + ty; + + const uint i01 = i21 % args.ne01; + const uint i11 = i21 % args.ne11; + + const float scale = 1.0f / sqrt((float)S_v); + + device const float * s_ptr = (device const float *) (s) + (i23*args.ne21 + i21)*S_v*S_v + i20; + + float lsf[NSG]; + + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + lsf[j] = s_ptr[is*S_v]; + } + + thread T * ls = (thread T *) (lsf); + + device float * dst_attn = (device float *) (dst) + (i23*args.ne22*args.ne21 + i21)*S_v + i20; + + device const float * q_ptr = (device const float *) (q + i23*args.nb03 + i01*args.nb01); + device const float * k_ptr = (device const float *) (k + i23*args.nb13 + i11*args.nb11); + device const float * v_ptr = (device const float *) (v + i23*args.nb23 + i21*args.nb21); + + device const float * b_ptr = (device const float *) (b) + (i23*args.ne22*args.ne21 + i21); + device const float * g_ptr = (device const float *) (g) + (i23*args.ne22*args.ne21 + i21)*G; + + for (short t = 0; t < args.ne22; t++) { + device const T * qt_ptr = (device const T *) (q_ptr); + device const T * kt_ptr = (device const T *) (k_ptr); + device const T * gt_ptr = (device const T *) (g_ptr); + + if (G == 1) { + *ls *= exp(g_ptr[0]); + } else { + // KDA + *ls *= exp(gt_ptr[tx]); + } + + const float s_k = simd_sum(dot(*ls, kt_ptr[tx])); + + const float d = (v_ptr[i20] - s_k)*b_ptr[0]; + + *ls += kt_ptr[tx]*d; + + const float y = simd_sum(dot(*ls, qt_ptr[tx])); + + if (tx == 0) { + *dst_attn = y*scale; + } + + q_ptr += args.ns02; + k_ptr += args.ns12; + v_ptr += args.ns22; + + b_ptr += args.ne21; + g_ptr += args.ne21*G; + + dst_attn += args.ne21*S_v; + } + + // when fused with the cache cpy, write the snapshots straight into the cache buffer using + // the slot stride; otherwise append them after the attn scores (nb_out == 0) + const bool fused = args.nb_out > 0; + const device float * state_out = fused ? (device float *)dst_fuse : (device float *)dst + args.ne23*args.ne22*args.ne21*S_v; + const uint slot_stride = fused ? (uint)args.nb_out : S_v*S_v; + + device float * dst_state = (device float *)state_out + (i23*args.ne21 + i21)*slot_stride + i20; + device T * dstt_state = (device T *) (dst_state); + + FOR_UNROLL (short j = 0; j < NSG; j++) { + const short is = tx*NSG + j; + dst_state[is*S_v] = lsf[j]; + } + +#undef S_v +#undef G +} + +typedef decltype(kernel_gated_delta_net_impl) kernel_gated_delta_net_t; + +template [[host_name("kernel_gated_delta_net_f32_1")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; +template [[host_name("kernel_gated_delta_net_f32_2")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; +template [[host_name("kernel_gated_delta_net_f32_4")]] kernel kernel_gated_delta_net_t kernel_gated_delta_net_impl; +#endif diff --git a/ggml/src/ggml-metal/kernels/misc.metal b/ggml/src/ggml-metal/kernels/misc.metal new file mode 100644 index 00000000..279d69f8 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/misc.metal @@ -0,0 +1,658 @@ +#include "common.h" + +kernel void kernel_argmax_f32( + constant ggml_metal_kargs_argmax & args, + device const char * src0, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint sgitg[[simdgroup_index_in_threadgroup]], + uint tiisg[[thread_index_in_simdgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const float * x_row = (device const float *) ((device const char *) src0 + tgpig * args.nb01); + + float lmax = -INFINITY; + int32_t larg = -1; + + for (int i00 = tpitg; i00 < args.ne00; i00 += ntg) { + if (x_row[i00] > lmax) { + lmax = x_row[i00]; + larg = i00; + } + } + + // find the argmax value in the block + float max_val = simd_max(lmax); + int32_t arg_val = simd_max(select(-1, larg, lmax == max_val)); + + device int32_t * dst_i32 = (device int32_t *) dst; + + threadgroup float * shared_maxval = (threadgroup float *) shmem; + threadgroup int32_t * shared_argmax = (threadgroup int32_t *) shmem + N_SIMDWIDTH; + + if (ntg > N_SIMDWIDTH) { + if (sgitg == 0) { + shared_maxval[tiisg] = -INFINITY; + shared_argmax[tiisg] = -1; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shared_maxval[sgitg] = max_val; + shared_argmax[sgitg] = arg_val; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + max_val = shared_maxval[tiisg]; + arg_val = shared_argmax[tiisg]; + + float max_val_reduced = simd_max(max_val); + int32_t arg_val_reduced = simd_max(select(-1, arg_val, max_val == max_val_reduced)); + + dst_i32[tgpig] = arg_val_reduced; + + return; + } + + dst_i32[tgpig] = arg_val; +} + +kernel void kernel_diag_f32( + constant ggml_metal_kargs_diag & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]]) { + constexpr short NW = N_SIMDWIDTH; + + const int32_t i3 = tgpig.z; + const int32_t i2 = tgpig.y; + const int32_t i1 = tgpig.x; + + device const float * src0_ptr = (device const float *)(src0 + i2*args.nb02 + i3*args.nb03); + device float * dst_ptr = (device float *)(dst + i1*args.nb01 + i2*args.nb2 + i3*args.nb3); + + for (int i0 = tiitg; i0 < args.ne0; i0 += NW) { + dst_ptr[i0] = i0 == i1 ? src0_ptr[i0] : 0.0f; + } +} + +kernel void kernel_roll_f32( + constant ggml_metal_kargs_roll & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t i3 = tgpig.z; + const int64_t i2 = tgpig.y; + const int64_t i1 = tgpig.x; + + device const float * src0_ptr = (device const float *) src0; + device float * dst_ptr = (device float *) dst; + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + // apply shifts and wrap around + int64_t i00 = i0 - args.s0; + int64_t i01 = i1 - args.s1; + int64_t i02 = i2 - args.s2; + int64_t i03 = i3 - args.s3; + + if (i00 < 0) { i00 += args.ne00; } else if (i00 >= args.ne00) { i00 -= args.ne00; } + if (i01 < 0) { i01 += args.ne01; } else if (i01 >= args.ne01) { i01 -= args.ne01; } + if (i02 < 0) { i02 += args.ne02; } else if (i02 >= args.ne02) { i02 -= args.ne02; } + if (i03 < 0) { i03 += args.ne03; } else if (i03 >= args.ne03) { i03 -= args.ne03; } + + int64_t src_idx = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00 + i00; + int64_t dst_idx = i3 *args.ne2 *args.ne1 *args.ne0 + i2 *args.ne1 *args.ne0 + i1 *args.ne0 + i0; + + dst_ptr[dst_idx] = src0_ptr[src_idx]; + } +} + +template +kernel void kernel_pad_impl( + constant ggml_metal_kargs_pad & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + const int32_t i3 = tgpig.z; + const int32_t i2 = tgpig.y; + const int32_t k0 = tgpig.x/args.ne1; + const int32_t i1 = tgpig.x - k0*args.ne1; + + const int32_t i03 = i3; + const int32_t i02 = i2; + const int32_t i01 = i1; + + device const T * src0_ptr = (device const T *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); + device T * dst_ptr = (device T *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); + + for (int32_t l0 = 0; l0 < 1024; l0 += ntg.x) { + const int32_t i0 = k0*1024 + tpitg.x + l0; + if (i0 >= args.ne0) { + break; + } + + if (i0 < args.ne00 && i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { + dst_ptr[i0] = src0_ptr[i0]; + } else { + dst_ptr[i0] = 0.0f; + } + } +} + +typedef decltype(kernel_pad_impl) kernel_pad_t; + +template [[host_name("kernel_pad_f32")]] kernel kernel_pad_t kernel_pad_impl; +template [[host_name("kernel_pad_f32_4")]] kernel kernel_pad_t kernel_pad_impl; + +// TODO: this is slow - optimize +kernel void kernel_pad_reflect_1d_f32( + constant ggml_metal_kargs_pad_reflect_1d & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tgpg[[threadgroups_per_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t i3 = tgpig.z; + const int64_t i2 = tgpig.y; + const int64_t i1 = tgpig.x; + + const int64_t i03 = i3; + const int64_t i02 = i2; + const int64_t i01 = i1; + + device const float * src0_ptr = (device const float *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); + device float * dst_ptr = (device float *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); + + if (i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + if (i0 < args.p0) { + dst_ptr[i0] = src0_ptr[args.p0 - i0]; + } else if (i0 < args.ne0 - args.p1) { + dst_ptr[i0] = src0_ptr[i0 - args.p0]; + } else { + dst_ptr[i0] = src0_ptr[(args.ne0 - args.p1 - args.p0) - (args.p1 + 1 - (args.ne0 - i0)) - 1]; + } + } + } +} + +kernel void kernel_arange_f32( + constant ggml_metal_kargs_arange & args, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + device float * dst_ptr = (device float *) dst; + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + dst_ptr[i0] = args.start + args.step * i0; + } +} + +kernel void kernel_timestep_embedding_f32( + constant ggml_metal_kargs_timestep_embedding & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + int i = tgpig.x; + device float * embed_data = (device float *)(dst + i*args.nb1); + + int half_ = args.dim / 2; + for (int j = tpitg.x; j < half_; j += ntg.x) { + float timestep = ((device float *)src0)[i]; + float freq = (float)exp(-log((float)args.max_period) * j / half_); + float arg = timestep * freq; + embed_data[j ] = cos(arg); + embed_data[j + half_] = sin(arg); + } + + if (args.dim % 2 != 0 && tpitg.x == 0) { + embed_data[2 * half_] = 0.f; + } +} + +kernel void kernel_opt_step_adamw_f32( + constant ggml_metal_kargs_opt_step_adamw & args, + device float * x, + device const float * g, + device float * g_m, + device float * g_v, + device const float * pars, + uint gid[[thread_position_in_grid]]) { + + if (gid >= args.np) { + return; + } + + const float alpha = pars[0]; + const float beta1 = pars[1]; + const float beta2 = pars[2]; + const float eps = pars[3]; + const float wd = pars[4]; + const float beta1h = pars[5]; + const float beta2h = pars[6]; + + const float gi = g[gid]; + const float gmi = g_m[gid] * beta1 + gi * (1.0f - beta1); + const float gvi = g_v[gid] * beta2 + gi * gi * (1.0f - beta2); + + g_m[gid] = gmi; + g_v[gid] = gvi; + + const float mh = gmi * beta1h; + const float vh = sqrt(gvi * beta2h) + eps; + + x[gid] = x[gid] * (1.0f - alpha * wd) - alpha * mh / vh; +} + +kernel void kernel_opt_step_sgd_f32( + constant ggml_metal_kargs_opt_step_sgd & args, + device float * x, + device const float * g, + device const float * pars, + uint gid[[thread_position_in_grid]]) { + + if (gid >= args.np) { + return; + } + + x[gid] = x[gid] * (1.0f - pars[0] * pars[1]) - pars[0] * g[gid]; +} + +template +kernel void kernel_memset( + constant ggml_metal_kargs_memset & args, + device T * dst, + uint tpig[[thread_position_in_grid]]) { + dst[tpig] = args.val; +} + +typedef decltype(kernel_memset) kernel_memset_t; + +template [[host_name("kernel_memset_i64")]] kernel kernel_memset_t kernel_memset; + +constant short FC_count_equal_nsg [[function_constant(FC_COUNT_EQUAL + 0)]]; + +template +kernel void kernel_count_equal( + constant ggml_metal_kargs_count_equal & args, + device const char * src0, + device const char * src1, + device atomic_int * dst, + threadgroup int32_t * shmem_i32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const short NSG = FC_count_equal_nsg; + + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = tgpig.x; + + if (i3 >= args.ne03 || i2 >= args.ne02 || i1 >= args.ne01) { + return; + } + + int sum = 0; + + device const char * base0 = src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03; + device const char * base1 = src1 + i1*args.nb11 + i2*args.nb12 + i3*args.nb13; + + for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { + const T v0 = *(device const T *)(base0 + i0*args.nb00); + const T v1 = *(device const T *)(base1 + i0*args.nb10); + sum += (v0 == v1); + } + + sum = simd_sum(sum); + + if (tiisg == 0) { + shmem_i32[sgitg] = sum; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (sgitg == 0) { + float v = 0.0f; + if (tpitg.x < NSG) { + v = shmem_i32[tpitg.x]; + } + + float total = simd_sum(v); + if (tpitg.x == 0) { + atomic_fetch_add_explicit(dst, (int32_t) total, memory_order_relaxed); + } + } +} + +typedef decltype(kernel_count_equal) kernel_count_equal_t; + +template [[host_name("kernel_count_equal_i32")]] kernel kernel_count_equal_t kernel_count_equal; + +template +kernel void kernel_snake( + constant ggml_metal_kargs_snake & args, + device const T * x, + device const float * a, + device const float * inv_b, + device T * dst, + uint tgpig [[threadgroup_position_in_grid]], + uint tpitg [[thread_position_in_threadgroup]], + uint ntg [[threads_per_threadgroup]]) { + + const int idx = tgpig * ntg + tpitg; + if (idx >= args.T * args.C) { + return; + } + + const int c = idx / args.T; // x is [T, C], a / inv_b collapse to [1, C] + const float xi = float(x[idx]); + const float si = sin(a[c] * xi); + dst[idx] = T(xi + si * si * inv_b[c]); +} + +template [[host_name("kernel_snake_f32")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const float *, device const float *, device const float *, device float *, uint, uint, uint); +template [[host_name("kernel_snake_f16")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const half *, device const float *, device const float *, device half *, uint, uint, uint); +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_snake_bf16")]] kernel void kernel_snake(constant ggml_metal_kargs_snake &, device const bfloat *, device const float *, device const float *, device bfloat *, uint, uint, uint); +#endif + +template +kernel void kernel_fwht( + constant ggml_metal_kargs_fwht & args, + device const src_t * src, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + constexpr int NW = N_SIMDWIDTH; + constexpr int NE = N / NW; + + const float scale = 1.0f / sqrt((float) N); + + const int sg_per_tg = ntg.x / NW; + const int64_t r = tgpig.x * sg_per_tg + sgitg; + if (r >= args.nrows) { + return; + } + + src += r * N; + dst += r * N; + + const int lane = tiisg; + + float reg[NE]; + for (int i = 0; i < NE; i++) { + reg[i] = float(src[i*NW + lane])*scale; + } + for (int i = 1; i < NW; i *= 2) { + for (int j = 0; j < NE; j++) { + const float val = reg[j]; + const float val2 = simd_shuffle_xor(val, i); + reg[j] = val2 - val + 2*((lane & i) == 0)*val; + } + } + + for (int i = NW; i < N; i *= 2) { + const int step = i / NW; + for (int j = 0; j < NE; j += (2 * step)) { + for (int k = 0; k < step; k++) { + const float x = reg[j + k ]; + const float y = reg[j + k + step]; + reg[j + k] = x + y; + reg[j + k + step] = x - y; + } + } + } + + for (int i = 0; i < NE; i++) { + dst[i*NW + lane] = reg[i]; + } +} + +typedef decltype(kernel_fwht<64, float>) kernel_fwht_f32_t; +typedef decltype(kernel_fwht<64, half>) kernel_fwht_f16_t; + +template [[host_name("kernel_fwht_f32_64")]] kernel kernel_fwht_f32_t kernel_fwht<64, float>; +template [[host_name("kernel_fwht_f32_128")]] kernel kernel_fwht_f32_t kernel_fwht<128, float>; +template [[host_name("kernel_fwht_f32_256")]] kernel kernel_fwht_f32_t kernel_fwht<256, float>; +template [[host_name("kernel_fwht_f32_512")]] kernel kernel_fwht_f32_t kernel_fwht<512, float>; + +template [[host_name("kernel_fwht_f16_64")]] kernel kernel_fwht_f16_t kernel_fwht<64, half>; +template [[host_name("kernel_fwht_f16_128")]] kernel kernel_fwht_f16_t kernel_fwht<128, half>; +template [[host_name("kernel_fwht_f16_256")]] kernel kernel_fwht_f16_t kernel_fwht<256, half>; +template [[host_name("kernel_fwht_f16_512")]] kernel kernel_fwht_f16_t kernel_fwht<512, half>; + +constant int FC_dsv4_hc_n_hc [[function_constant(FC_DSV4_HC + 0)]]; + +kernel void kernel_dsv4_hc_comb_f32( + constant ggml_metal_kargs_dsv4_hc_comb & args, + device const char * mixes, + device const char * scale, + device const char * base, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + constexpr ushort hc = 4; + constexpr ushort comb_offset = 2*hc; + + const int it = tgpig.x*ntg.y + sgitg; + if (it >= args.n_tokens) { + return; + } + + float scale_lane = 0.0f; + if (tiisg == 0) { + scale_lane = *(device const float *) (scale + 2*args.nb_s0); + } + const float scale_comb = simd_shuffle(scale_lane, 0); + + float v = 0.0f; + if (tiisg < hc*hc) { + v = *(device const float *) (mixes + (comb_offset + tiisg)*args.nb_m0 + it*args.nb_m1)*scale_comb + + *(device const float *) (base + (comb_offset + tiisg)*args.nb_b0); + } + + // Softmax across destinations (the four contiguous lanes for each source). + float vmax = max(v, simd_shuffle_xor(v, 1)); + vmax = max(vmax, simd_shuffle_xor(vmax, 2)); + v = exp(v - vmax); + + float sum = v + simd_shuffle_xor(v, 1); + sum += simd_shuffle_xor(sum, 2); + v = v/sum + args.eps; + + // Normalize columns: equal destination indices are four lanes apart. + sum = v + simd_shuffle_xor(v, 4); + sum += simd_shuffle_xor(sum, 8); + v /= sum + args.eps; + + for (int i = 1; i < args.n_iter; ++i) { + sum = v + simd_shuffle_xor(v, 1); + sum += simd_shuffle_xor(sum, 2); + v /= sum + args.eps; + + sum = v + simd_shuffle_xor(v, 4); + sum += simd_shuffle_xor(sum, 8); + v /= sum + args.eps; + } + + if (tiisg < hc*hc) { + const ushort idst = tiisg & 3; + const ushort isrc = tiisg >> 2; + *(device float *) (dst + idst*args.nb_d0 + isrc*args.nb_d1 + it*args.nb_d2) = v; + } +} + +kernel void kernel_dsv4_hc_pre_f32( + constant ggml_metal_kargs_dsv4_hc_pre & args, + device const char * x, + device const char * weights, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int it = tgpig.y; + const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; + + if (i0 >= args.n_embd) { + return; + } + + device const char * xb = x + i0*args.nb_x0 + it*args.nb_x2; + float result = 0.0f; + FOR_UNROLL (int ih = 0; ih < FC_dsv4_hc_n_hc; ++ih) { + const float xv = *(device const float *) (xb + ih*args.nb_x1); + const float wv = *(device const float *) (weights + ih*args.nb_w0 + it*args.nb_w1); + result = fma(xv, wv, result); + } + + *(device float *) (dst + i0*args.nb_d0 + it*args.nb_d1) = args.scale*result; +} + +kernel void kernel_dsv4_hc_pre_gated_f32( + constant ggml_metal_kargs_dsv4_hc_pre & args, + device const char * x, + device const char * gate, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int it = tgpig.y; + const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; + + if (i0 >= args.n_embd) { + return; + } + + device const char * xb = x + i0*args.nb_x0 + it*args.nb_x2; + device const char * gb = gate + i0*args.nb_w0 + it*args.nb_w2; + float result = 0.0f; + FOR_UNROLL (int ih = 0; ih < FC_dsv4_hc_n_hc; ++ih) { + const float g = 1.0f/(1.0f + exp(-*(device const float *) (gb + ih*args.nb_w1))); + const float xv = *(device const float *) (xb + ih*args.nb_x1); + result = fma(xv, g, result); + } + + *(device float *) (dst + i0*args.nb_d0 + it*args.nb_d1) = args.scale*result; +} + +kernel void kernel_dsv4_hc_post_nocomb_f32( + constant ggml_metal_kargs_dsv4_hc_post & args, + device const char * x, + device const char * residual, + device const char * post, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + constexpr ushort hc = 4; + + const int it = tgpig.y; + const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; + + float post_lane = 0.0f; + if (tiisg < hc) { + post_lane = *(device const float *) (post + tiisg*args.nb_p0 + it*args.nb_p1); + } + + float post_reg[hc]; + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + post_reg[idst] = simd_shuffle(post_lane, idst); + } + + if (i0 >= args.n_embd) { + return; + } + + const float xv = *(device const float *) (x + i0*args.nb_x0 + it*args.nb_x1); + device const char * rb = residual + i0*args.nb_r0 + it*args.nb_r2; + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + const float rv = *(device const float *) (rb + idst*args.nb_r1); + *(device float *) (dst + i0*args.nb_d0 + idst*args.nb_d1 + it*args.nb_d2) = xv*post_reg[idst] + rv; + } +} + +kernel void kernel_dsv4_hc_post_f32( + constant ggml_metal_kargs_dsv4_hc_post & args, + device const char * x, + device const char * residual, + device const char * post, + device const char * comb, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + constexpr ushort hc = 4; + + const int it = tgpig.y; + const int i0 = ((int) tgpig.x*ntg.y + sgitg)*32 + tiisg; + + float coeff_lane = 0.0f; + if (tiisg < hc) { + coeff_lane = *(device const float *) (post + tiisg*args.nb_p0 + it*args.nb_p1); + } else if (tiisg < hc + hc*hc) { + const ushort idx = tiisg - hc; + const ushort idst = idx & 3; + const ushort isrc = idx >> 2; + coeff_lane = *(device const float *) (comb + idst*args.nb_c0 + isrc*args.nb_c1 + it*args.nb_c2); + } + + float post_reg[hc]; + float comb_reg[hc][hc]; + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + post_reg[idst] = simd_shuffle(coeff_lane, idst); + } + FOR_UNROLL (ushort isrc = 0; isrc < hc; ++isrc) { + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + comb_reg[isrc][idst] = simd_shuffle(coeff_lane, hc + idst + hc*isrc); + } + } + + if (i0 >= args.n_embd) { + return; + } + + const float xv = *(device const float *) (x + i0*args.nb_x0 + it*args.nb_x1); + float result[hc]; + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + result[idst] = xv*post_reg[idst]; + } + + device const char * rb = residual + i0*args.nb_r0 + it*args.nb_r2; + FOR_UNROLL (ushort isrc = 0; isrc < hc; ++isrc) { + const float rv = *(device const float *) (rb + isrc*args.nb_r1); + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + result[idst] = fma(rv, comb_reg[isrc][idst], result[idst]); + } + } + + FOR_UNROLL (ushort idst = 0; idst < hc; ++idst) { + *(device float *) (dst + i0*args.nb_d0 + idst*args.nb_d1 + it*args.nb_d2) = result[idst]; + } +} diff --git a/ggml/src/ggml-metal/kernels/mul_mm.metal b/ggml/src/ggml-metal/kernels/mul_mm.metal new file mode 100644 index 00000000..a25838f9 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/mul_mm.metal @@ -0,0 +1,967 @@ +#include "common.h" +#include "dequantize.h" + +constant bool FC_mul_mm_bc_inp [[function_constant(FC_MUL_MM + 0)]]; +constant bool FC_mul_mm_bc_out [[function_constant(FC_MUL_MM + 1)]]; +constant short FC_mul_mm_ne12 [[function_constant(FC_MUL_MM + 2)]]; +constant short FC_mul_mm_ne13 [[function_constant(FC_MUL_MM + 3)]]; +constant short FC_mul_mm_r2 [[function_constant(FC_MUL_MM + 4)]]; +constant short FC_mul_mm_r3 [[function_constant(FC_MUL_MM + 5)]]; +constant bool FC_mul_mm_id_amax [[function_constant(FC_MUL_MM + 6)]]; + +// each block_q contains 16*nl weights +#ifdef GGML_METAL_HAS_TENSOR +template< + typename SA, typename SA_4x4, typename SA_8x8, + typename SB, typename SB_2x4, typename SB_8x8, + typename block_q, short nl, void (*dequantize_func)(device const block_q *, short, thread SA_4x4 &), + typename T0, typename T0_4x4, typename T1, typename T1_2x4> +kernel void kernel_mul_mm( + constant ggml_metal_kargs_mul_mm & args, + device const char * srcA, + device const char * srcB, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig [[threadgroup_position_in_grid]], + ushort tiitg [[thread_index_in_threadgroup]], + ushort sgitg [[simdgroup_index_in_threadgroup]]) { + (void) sgitg; + + // Matrix dimensions: A(M,K) x B(K,N) -> C(M,N) + const int K = args.ne00; + const int M = args.ne0; + const int N = args.ne1; + + // Batch dimension handling + const int im = tgpig.z; + const int i12 = im % FC_mul_mm_ne12; + const int i13 = im / FC_mul_mm_ne12; + + // Batch offsets for srcA and srcB + const uint64_t offset0 = (i12/FC_mul_mm_r2)*args.nb02 + (i13/FC_mul_mm_r3)*args.nb03; + + // Tile dimensions + constexpr int NRB = SZ_SIMDGROUP * N_MM_BLOCK_X * N_MM_SIMD_GROUP_X; + constexpr int NRA = SZ_SIMDGROUP * N_MM_BLOCK_Y * N_MM_SIMD_GROUP_Y; + + // Tile offsets in output matrix + const int ra = tgpig.y * NRA; + const int rb = tgpig.x * NRB; + + // Threadgroup memory for dequantized A tile only + threadgroup SA * sa = (threadgroup SA *)(shmem); + + // Work-item count for A loading + constexpr int A_WORK_ITEMS = NRA * N_MM_NK; + constexpr int NUM_THREADS = N_SIMDWIDTH * N_MM_SIMD_GROUP_X * N_MM_SIMD_GROUP_Y; + + // tA wraps threadgroup memory + auto tA = tensor(sa, dextents(N_MM_NK_TOTAL, NRA)); + + // tB wraps device memory directly + device T1 * ptrB = (device T1 *)(srcB + args.nb12*i12 + args.nb13*i13); + const int strideB = args.nb11 / sizeof(T1); + auto tB = tensor(ptrB, dextents(K, N), array({1, strideB})); + + // Configure matmul operation + // note: K is dynamic_extent (clamped to the valid range in PHASE 2), since a static + // N_MM_NK_TOTAL K tile would read src1 out of bounds when K % N_MM_NK_TOTAL != 0 + // ref: https://github.com/ggml-org/llama.cpp/pull/27064 + mpp::tensor_ops::matmul2d< + mpp::tensor_ops::matmul2d_descriptor( + NRB, NRA, static_cast(dynamic_extent), false, true, true, + mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate), + execution_simdgroups> mm; + + auto cT = mm.get_destination_cooperative_tensor(); + + // Accumulate partial results over K dimension + for (int loop_k = 0; loop_k < K; loop_k += N_MM_NK_TOTAL) { + // === PHASE 1: Dequantization of A into threadgroup memory === + for (int work = tiitg; work < A_WORK_ITEMS; work += NUM_THREADS) { + const int row = work / N_MM_NK; + const int k_chunk = work % N_MM_NK; + const int k_pos = loop_k + k_chunk * 16; + const short k_base = k_chunk * 16; + + // Bounds check: skip device read if row is out of matrix bounds + if (ra + row < M) { + if (is_same::value && FC_mul_mm_bc_inp) { + // Element-wise reads when K is not aligned (nb01 not aligned for half4x4/float4x4). + // MSL spec Table 2.5: half4x4 requires 8-byte alignment. When K is odd, + // nb01 = K*2 is not 8-byte aligned, so odd-row pointers are misaligned. + // Mirrors the legacy kernel's existing guard. + device const T0 * row_ptr = (device const T0 *)(srcA + args.nb01 * (ra + row) + offset0); + + FOR_UNROLL (short i = 0; i < 16; i++) { + sa[row * N_MM_NK_TOTAL + (k_base + i)] = (k_pos + i < K) ? (SA) row_ptr[k_pos + i] : (SA)0; + } + } else { + const int block_idx = k_pos / (16 * nl); + const short il = (k_pos / 16) % nl; + + device const block_q * row_ptr = (device const block_q *)(srcA + args.nb01 * (ra + row) + offset0); + + SA_4x4 temp_a; + dequantize_func(row_ptr + block_idx, il, temp_a); + + FOR_UNROLL (short i = 0; i < 16; i++) { + // Zero-pad A for K positions beyond valid range (handles partial K iterations) + sa[row * N_MM_NK_TOTAL + (k_base + i)] = (k_pos + i < K) ? temp_a[i/4][i%4] : (SA)0; + } + } + } else { + // Zero-pad rows beyond matrix bounds + FOR_UNROLL (short i = 0; i < 16; i++) { + sa[row * N_MM_NK_TOTAL + (k_base + i)] = (SA)0; + } + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // === PHASE 2: Tensor matmul === + // Clamp the K extent of both operand tensors to the remaining valid K range so + // the dynamic-K op never reads past the K extent of src1 (or the staged A tile). + const int kExt = min(N_MM_NK_TOTAL, K - loop_k); + + auto tAv = tensor(sa, dextents(kExt, NRA), array({1, N_MM_NK_TOTAL})); + auto tBv = tensor(ptrB + loop_k + rb * strideB, dextents(kExt, N - rb), array({1, strideB})); + + mm.run(tBv, tAv, cT); + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + // Store result tile to output matrix (with batch offset) + // cT.store handles bounds checking via tD's extents (M, N) + device float * dstBatch = (device float *)dst + im * N * M; + + auto tD = tensor(dstBatch, dextents(M, N), array({1, M})); + cT.store(tD.slice(ra, rb)); +} + +#else + +template< + typename S0, typename S0_4x4, typename S0_8x8, + typename S1, typename S1_2x4, typename S1_8x8, + typename block_q, short nl, void (*dequantize_func)(device const block_q *, short, thread S0_4x4 &), + typename T0, typename T0_4x4, typename T1, typename T1_2x4> +kernel void kernel_mul_mm( + constant ggml_metal_kargs_mul_mm & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + threadgroup S0 * sa = (threadgroup S0 *)(shmem); + threadgroup S1 * sb = (threadgroup S1 *)(shmem + 4096); + + constexpr int NR0 = 64; + constexpr int NR1 = 32; + + constexpr int NK = 32; + constexpr int NL0 = NK/16; + constexpr int NL1 = NK/8; + + const int im = tgpig.z; + const int r0 = tgpig.y*NR0; + const int r1 = tgpig.x*NR1; + + // if this block is of 64x32 shape or smaller + const short nr0 = (args.ne0 - r0 < NR0) ? (args.ne0 - r0) : NR0; + const short nr1 = (args.ne1 - r1 < NR1) ? (args.ne1 - r1) : NR1; + + // a thread shouldn't load data outside of the matrix + const short lr0 = ((short)tiitg/NL0) < nr0 ? ((short)tiitg/NL0) : nr0 - 1; // 0 .. 63 + const short lr1 = ((short)tiitg/NL1) < nr1 ? ((short)tiitg/NL1) : nr1 - 1; // 0 .. 31 + + const short il0 = (tiitg % NL0); + + short il = il0; + + const int i12 = im % FC_mul_mm_ne12; + const int i13 = im / FC_mul_mm_ne12; + + const uint64_t offset0 = (i12/FC_mul_mm_r2)*args.nb02 + (i13/FC_mul_mm_r3)*args.nb03; + const short offset1 = il0/nl; + + device const block_q * x = (device const block_q *)(src0 + args.nb01*(r0 + lr0) + offset0) + offset1; + + const short iy = 8*(tiitg % NL1); + + device const T1 * y = (device const T1 *)(src1 + + args.nb13*i13 + + args.nb12*i12 + + args.nb11*(r1 + lr1) + + args.nb10*iy); + + S0_8x8 ma[4]; + S1_8x8 mb[2]; + + simdgroup_float8x8 mc[8]; + + for (short i = 0; i < 8; i++){ + mc[i] = make_filled_simdgroup_matrix(0.f); + } + + for (int loop_k = 0; loop_k < args.ne00; loop_k += NK) { + // load data and store to threadgroup memory + if (is_same::value && FC_mul_mm_bc_inp) { + threadgroup_barrier(mem_flags::mem_threadgroup); + + // no need for dequantization + for (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + //const short lx = i%8; + //const short ly = (tiitg/NL0)%8; + const short lx = (tiitg/NL0)%8; + const short ly = i%8; + + const short ib = 8*sx + sy; + + *(sa + 64*ib + 8*ly + lx) = loop_k + 16*il + i < args.ne00 ? *((device T0 *) x + i) : 0; + } + } else { + S0_4x4 temp_a; + dequantize_func(x, il, temp_a); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + FOR_UNROLL (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + //const short lx = i%8; + //const short ly = (tiitg/NL0)%8; + const short lx = (tiitg/NL0)%8; + const short ly = i%8; + + const short ib = 8*sx + sy; + + // NOTE: this is massively slower.. WTF? + //sa[64*ib + 8*ly + lx] = temp_a[i/4][i%4]; + + *(sa + 64*ib + 8*ly + lx) = temp_a[i/4][i%4]; + } + } + + if (FC_mul_mm_bc_inp) { + for (short i = 0; i < 8; ++i) { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + const short lx = i; + const short ly = (tiitg/NL1)%8; + //const short lx = (tiitg/NL1)%8; + //const short ly = i; + + const short ib = 4*sx + sy; + + *(sb + 64*ib + 8*ly + lx) = loop_k + iy + i < args.ne00 ? (S1) *((device T1 *) y + i) : 0; + } + } else { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + //const short dx = sx; + //const short dy = sy; + + const short ly = (tiitg/NL1)%8; + + const short ib = 4*sx + sy; + + *(threadgroup S1_2x4 *)(sb + 64*ib + 8*ly) = (S1_2x4)(*((device T1_2x4 *) y)); + } + + il = (il + 2 < nl) ? il + 2 : il % 2; + x = (il < 2) ? x + (2 + nl - 1)/nl : x; + + y += NK; + + threadgroup_barrier(mem_flags::mem_threadgroup); + + // load matrices from threadgroup memory and conduct outer products + threadgroup const S0 * lsma = (sa + 4*64*(sgitg%2)); + threadgroup const S1 * lsmb = (sb + 2*64*(sgitg/2)); + + FOR_UNROLL (short ik = 0; ik < NK/8; ik++) { + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 4; i++) { + simdgroup_load(ma[i], lsma + 64*i, 8, 0, false); + } + + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 2; i++) { + simdgroup_load(mb[i], lsmb + 64*i, 8, 0, false); + } + + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 8; i++){ + simdgroup_multiply_accumulate(mc[i], mb[i/4], ma[i%4], mc[i]); + } + + lsma += 8*64; + lsmb += 4*64; + } + } + + if (!FC_mul_mm_bc_out || (r0 + NR0 <= args.ne0 && r1 + NR1 <= args.ne1)) { + // if no bounds checks on the output are needed, we can directly write to device memory + device float * C = (device float *) dst + + (r0 + 32*(sgitg & 1)) + \ + (r1 + 16*(sgitg >> 1)) * args.ne0 + im*args.ne1*args.ne0; + + for (short i = 0; i < 8; i++) { + simdgroup_store(mc[i], C + 8*(i%4) + 8*args.ne0*(i/4), args.ne0, 0, false); + } + } else { + // block is smaller than 64x32, we should avoid writing data outside of the matrix + threadgroup_barrier(mem_flags::mem_threadgroup); + + threadgroup float * temp_str = ((threadgroup float *) shmem) + 32*(sgitg&1) + (16*(sgitg >> 1))*NR0; + + for (short i = 0; i < 8; i++) { + simdgroup_store(mc[i], temp_str + 8*(i%4) + 8*NR0*(i/4), NR0, 0, false); + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (sgitg == 0) { + for (int j = tiitg; j < nr1; j += NR1) { + device float * D = (device float *) dst + r0 + (r1 + j)*args.ne0 + im*args.ne1*args.ne0; + device float4 * D4 = (device float4 *) D; + + threadgroup float * C = temp_str + (j*NR0); + threadgroup float4 * C4 = (threadgroup float4 *) C; + + int i = 0; + for (; i < nr0/4; i++) { + *(D4 + i) = *(C4 + i); + } + + i *= 4; + for (; i < nr0; i++) { + *(D + i) = *(C + i); + } + } + } + } +} + +#endif // GGML_METAL_HAS_TENSOR + +template // n_expert_used +kernel void kernel_mul_mm_id_map0( + constant ggml_metal_kargs_mul_mm_id_map0 & args, + device const char * src2, + device char * htpe, + device char * hids, + threadgroup char * shmem [[threadgroup(0)]], + ushort tpitg[[thread_position_in_threadgroup]], + ushort ntg[[threads_per_threadgroup]]) { + const short ide = tpitg; // expert id + + uint32_t n_all = 0; + + device int32_t * ids_i32 = (device int32_t *) hids + ide*args.ne21; + + for (int i21 = 0; i21 < args.ne21; i21 += ntg) { // n_tokens + if (i21 + tpitg < args.ne21) { + device const int32_t * src2_i32 = (device const int32_t *) (src2 + (i21 + tpitg)*args.nb21); + + threadgroup uint16_t * sids = (threadgroup uint16_t *) shmem + tpitg*ne20; + + #pragma unroll(ne20) + for (short i20 = 0; i20 < ne20; i20++) { + sids[i20] = src2_i32[i20]; + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (short t = 0; t < ntg; t++) { + if (i21 + t >= args.ne21) { + break; + } + + threadgroup const uint16_t * sids = (threadgroup const uint16_t *) shmem + t*ne20; + + short sel = 0; + #pragma unroll(ne20) + for (short i20 = 0; i20 < ne20; i20++) { + sel += (sids[i20] == ide)*(i20 + 1); + } + + ids_i32[n_all] = (i21 + t)*ne20 + sel - 1; + + n_all += sel > 0; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + device uint32_t * tpe_u32 = (device uint32_t *) (htpe); + tpe_u32[ide] = n_all; +} + +kernel void kernel_mul_mm_id_amax_part_f32( + constant ggml_metal_kargs_mul_mm_id_amax & args, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort ntg[[threads_per_threadgroup]]) { + const int nrow = args.ne01*args.ne02; + + float lmax = 0.0f; + + for (int ir = tgpig; ir < nrow; ir += N_MM_NPART_AMAX) { + const int i01 = ir % args.ne01; + const int i02 = ir / args.ne01; + + device const float * row = (device const float *) (src1 + i02*args.nb02 + i01*args.nb01); + + for (int i00 = tiitg; i00 < args.ne00; i00 += ntg) { + lmax = max(lmax, fabs(row[i00])); + } + } + + float amax = simd_max(lmax); + + threadgroup float * shared_amax = (threadgroup float *) shmem; + + if (ntg > N_SIMDWIDTH) { + if (sgitg == 0) { + shared_amax[tiisg] = 0.0f; + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shared_amax[sgitg] = amax; + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + amax = shared_amax[tiisg]; + amax = simd_max(amax); + } + + if (tiitg == 0) { + ((device float *) (dst + 8))[tgpig] = amax; + } +} + +kernel void kernel_mul_mm_id_amax_f32( + device char * dst, + ushort tiitg[[thread_index_in_threadgroup]]) { + device const float * part = (device const float *) (dst + 8); + + float amax = 0.0f; + + for (int i = tiitg; i < N_MM_NPART_AMAX; i += N_SIMDWIDTH) { + amax = max(amax, part[i]); + } + + amax = simd_max(amax); + + if (tiitg == 0) { + // leave a comfortable margin below the f16 max of 65504 + float scale = 1.0f; + + // isfinite: src1 already inf/nan is not ours to fix - keep the + // scale at 1.0 instead of turning it into a different failure + if (isfinite(amax) && amax > 32768.0f) { + scale = exp2(ceil(log2(amax)) - 15.0f); + } + + device float * d = (device float *) dst; + + d[0] = 1.0f/scale; // exact: scale is a power of two + d[1] = scale; + } +} + +typedef decltype(kernel_mul_mm_id_map0<1>) kernel_mul_mm_id_map0_t; + +template [[host_name("kernel_mul_mm_id_map0_ne20_1" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<1>; +template [[host_name("kernel_mul_mm_id_map0_ne20_2" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<2>; +template [[host_name("kernel_mul_mm_id_map0_ne20_4" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<4>; +template [[host_name("kernel_mul_mm_id_map0_ne20_5" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<5>; +template [[host_name("kernel_mul_mm_id_map0_ne20_6" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<6>; +template [[host_name("kernel_mul_mm_id_map0_ne20_8" )]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<8>; +template [[host_name("kernel_mul_mm_id_map0_ne20_10")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<10>; +template [[host_name("kernel_mul_mm_id_map0_ne20_16")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<16>; +template [[host_name("kernel_mul_mm_id_map0_ne20_22")]] kernel kernel_mul_mm_id_map0_t kernel_mul_mm_id_map0<22>; + +template +kernel void kernel_mul_mm_id( + constant ggml_metal_kargs_mul_mm_id & args, + device const char * src0, + device const char * src1, + device const char * htpe, + device const char * hids, + device char * dst, + device const char * amax, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + threadgroup S0 * sa = (threadgroup S0 *)(shmem); + threadgroup S1 * sb = (threadgroup S1 *)(shmem + 4096); + +#ifdef GGML_METAL_HAS_TENSOR + threadgroup float * sc = (threadgroup float *)(shmem); +#endif + + constexpr int NR0 = 64; + constexpr int NR1 = 32; + + constexpr int NK = 32; + constexpr int NL0 = NK/16; + constexpr int NL1 = NK/8; + + const int im = tgpig.z; // expert + const int r0 = tgpig.y*NR0; + const int r1 = tgpig.x*NR1; + + device const uint32_t * tpe_u32 = (device const uint32_t *) (htpe); + device const int32_t * ids_i32 = (device const int32_t *) (hids); + + const int32_t neh1 = tpe_u32[im]; + + if (r1 >= neh1) { + return; + } + + // if this block is of 64x32 shape or smaller + const short nr0 = (args.ne0 - r0 < NR0) ? (args.ne0 - r0) : NR0; + const short nr1 = ( neh1 - r1 < NR1) ? ( neh1 - r1) : NR1; + + // a thread shouldn't load data outside of the matrix + const short lr0 = ((short)tiitg/NL0) < nr0 ? ((short)tiitg/NL0) : nr0 - 1; // 0 .. 63 + const short lr1 = ((short)tiitg/NL1) < nr1 ? ((short)tiitg/NL1) : nr1 - 1; // 0 .. 31 + + const short il0 = (tiitg % NL0); + + short il = il0; + + const int id = ids_i32[im*args.ne21 + r1 + lr1]; + + const short i11 = (id % args.ne20) % args.ne11; + const short i12 = (id / args.ne20); + const short i13 = 0; + + const uint64_t offset0 = im*args.nb02 + i13*args.nb03; + const short offset1 = il0/nl; + + device const block_q * x = (device const block_q *)(src0 + args.nb01*(r0 + lr0) + offset0) + offset1; + + const short iy = 8*(tiitg % NL1); + + device const T1 * y = (device const T1 *)(src1 + + args.nb13*i13 + + args.nb12*i12 + + args.nb11*i11 + + args.nb10*iy); + + // skip the upper half of the token tile when the expert did not fill it + constexpr short NR1H = NR1/2; + + const bool has_hi = nr1 > NR1H; + + const short lb1 = (short) tiitg/NL1; // 0 .. NR1-1, this thread's row of the B tile + + // power-of-two rescaling + const float s1_inv = FC_mul_mm_id_amax ? ((device const float *) amax)[0] : 1.0f; + const float s1_scale = FC_mul_mm_id_amax ? ((device const float *) amax)[1] : 1.0f; + +#ifndef GGML_METAL_HAS_TENSOR + S0_8x8 ma[4]; + S1_8x8 mb[2]; + + simdgroup_float8x8 mc[8]; + + for (short i = 0; i < 8; i++){ + mc[i] = make_filled_simdgroup_matrix(0.f); + } + + // simdgroups 2,3 own rows NR1H..NR1-1 + const bool sg_active = has_hi || sgitg < 2; +#else + auto tA = tensor, tensor_inline>(sa, dextents(NK, NR0)); + + // sb is [NR1][NK] row-major + auto tB0 = tensor, tensor_inline>(sb, dextents(NK, NR1H)); + auto tB1 = tensor, tensor_inline>(sb + NR1H*NK, dextents(NK, NR1H)); + + mpp::tensor_ops::matmul2d< + mpp::tensor_ops::matmul2d_descriptor(NR1H, NR0, NK, false, true, false, mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate), + execution_simdgroups<4>> mm; + + auto cT0 = mm.get_destination_cooperative_tensor(); + auto cT1 = mm.get_destination_cooperative_tensor(); +#endif + + for (int loop_k = 0; loop_k < args.ne00; loop_k += NK) { +#ifndef GGML_METAL_HAS_TENSOR + // load data and store to threadgroup memory + if (is_same::value && FC_mul_mm_bc_inp) { + threadgroup_barrier(mem_flags::mem_threadgroup); + + // no need for dequantization + for (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + //const short lx = i%8; + //const short ly = (tiitg/NL0)%8; + const short lx = (tiitg/NL0)%8; + const short ly = i%8; + + const short ib = 8*sx + sy; + + *(sa + 64*ib + 8*ly + lx) = loop_k + 16*il + i < args.ne00 ? (S0) *((device T0 *) x + i) : (S0) 0; + } + } else { + S0_4x4 temp_a; + dequantize_func(x, il, temp_a); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + FOR_UNROLL (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + //const short lx = i%8; + //const short ly = (tiitg/NL0)%8; + const short lx = (tiitg/NL0)%8; + const short ly = i%8; + + const short ib = 8*sx + sy; + + // NOTE: this is massively slower.. WTF? + //sa[64*ib + 8*ly + lx] = temp_a[i/4][i%4]; + + *(sa + 64*ib + 8*ly + lx) = temp_a[i/4][i%4]; + } + } + + if (FC_mul_mm_bc_inp) { + for (short i = 0; i < 8; ++i) { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + const short lx = i; + const short ly = (tiitg/NL1)%8; + //const short lx = (tiitg/NL1)%8; + //const short ly = i; + + const short ib = 4*sx + sy; + + *(sb + 64*ib + 8*ly + lx) = loop_k + iy + i < args.ne00 ? (S1) (*((device T1 *) y + i) * (T1) s1_inv) : 0; + } + } else { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + //const short dx = sx; + //const short dy = sy; + + const short ly = (tiitg/NL1)%8; + + const short ib = 4*sx + sy; + + *(threadgroup S1_2x4 *)(sb + 64*ib + 8*ly) = (S1_2x4)((*((device T1_2x4 *) y)) * (T1) s1_inv); + } +#else + // load data and store to threadgroup memory + if (is_same::value && FC_mul_mm_bc_inp) { + threadgroup_barrier(mem_flags::mem_threadgroup); + + // no need for dequantization + for (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + const short lx = i%8; + const short ly = (tiitg/NL0)%8; + //const short lx = (tiitg/NL0)%8; + //const short ly = i%8; + + *(sa + NK*(8*sy + ly) + 8*sx + lx) = loop_k + 16*il + i < args.ne00 ? *((device T0 *) x + i) : 0; + } + } else { + S0_4x4 temp_a; + dequantize_func(x, il, temp_a); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + FOR_UNROLL (short i = 0; i < 16; i++) { + const short sx = 2*il0 + i/8; + const short sy = (tiitg/NL0)/8; + + const short lx = i%8; + const short ly = (tiitg/NL0)%8; + //const short lx = (tiitg/NL0)%8; + //const short ly = i%8; + + *(sa + NK*(8*sy + ly) + 8*sx + lx) = temp_a[i/4][i%4]; + } + } + + if (FC_mul_mm_bc_inp) { + for (short i = 0; i < 8; ++i) { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + const short lx = i; + const short ly = (tiitg/NL1)%8; + //const short lx = (tiitg/NL1)%8; + //const short ly = i; + + *(sb + NK*(8*sy + ly) + 8*sx + lx) = loop_k + iy + i < args.ne00 ? (S1) (*((device T1 *) y + i) * (T1) s1_inv) : 0; + } + } else { + const short sx = (tiitg%NL1); + const short sy = (tiitg/NL1)/8; + + //const short lx = i; + const short ly = (tiitg/NL1)%8; + //const short lx = (tiitg/NL1)%8; + //const short ly = i; + + *(threadgroup S1_2x4 *)(sb + NK*(8*sy + ly) + 8*sx) = (S1_2x4)((*((device T1_2x4 *) y)) * (T1) s1_inv); + } +#endif + + il = (il + 2 < nl) ? il + 2 : il % 2; + x = (il < 2) ? x + (2 + nl - 1)/nl : x; + + y += NK; + + threadgroup_barrier(mem_flags::mem_threadgroup); + +#ifndef GGML_METAL_HAS_TENSOR + if (sg_active) { + // load matrices from threadgroup memory and conduct outer products + threadgroup const S0 * lsma = (sa + 4*64*(sgitg%2)); + threadgroup const S1 * lsmb = (sb + 2*64*(sgitg/2)); + + FOR_UNROLL (short ik = 0; ik < NK/8; ik++) { + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 4; i++) { + simdgroup_load(ma[i], lsma + 64*i, 8, 0, false); + } + + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 2; i++) { + simdgroup_load(mb[i], lsmb + 64*i, 8, 0, false); + } + + simdgroup_barrier(mem_flags::mem_none); + + FOR_UNROLL (short i = 0; i < 8; i++){ + simdgroup_multiply_accumulate(mc[i], mb[i/4], ma[i%4], mc[i]); + } + + lsma += 8*64; + lsmb += 4*64; + } + } +#else + auto sA = tA.slice(0, 0); + auto sB0 = tB0.slice(0, 0); + + mm.run(sB0, sA, cT0); + + if (has_hi) { + auto sB1 = tB1.slice(0, 0); + + mm.run(sB1, sA, cT1); + } +#endif + } + + // block is smaller than 64x32, we should avoid writing data outside of the matrix + threadgroup_barrier(mem_flags::mem_threadgroup); + +#ifdef GGML_METAL_HAS_TENSOR + auto tC0 = tensor, tensor_inline>(sc, dextents(NR0, NR1H)); + cT0.store(tC0); + + if (has_hi) { + auto tC1 = tensor, tensor_inline>(sc + NR1H*NR0, dextents(NR0, NR1H)); + cT1.store(tC1); + } +#else + if (sg_active) { + threadgroup float * temp_str = ((threadgroup float *) shmem) + 32*(sgitg&1) + (16*(sgitg >> 1))*NR0; + + for (short i = 0; i < 8; i++) { + simdgroup_store(mc[i], temp_str + 8*(i%4) + 8*NR0*(i/4), NR0, 0, false); + } + } +#endif + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (short j = sgitg; j < nr1; j += 4) { + const int id = ids_i32[im*args.ne21 + r1 + j]; + + const short ide = id % args.ne20; + const short idt = id / args.ne20; + + device float * D = (device float *) dst + r0 + ide*args.ne0 + idt*args.ne1*args.ne0; + device float4 * D4 = (device float4 *) D; + + threadgroup float * C = (threadgroup float *) shmem + j*NR0; + threadgroup float4 * C4 = (threadgroup float4 *) C; + + int i = tiisg; + for (; i < nr0/4; i += 32) { + *(D4 + i) = *(C4 + i) * s1_scale; + } + + i = (4*(nr0/4)) + tiisg; + for (; i < nr0; i += 32) { + *(D + i) = *(C + i) * s1_scale; + } + } +} + +// +// matrix-matrix multiplication +// + +typedef decltype(kernel_mul_mm) mul_mm_t; + +template [[host_name("kernel_mul_mm_f32_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_f16_f32")]] kernel mul_mm_t kernel_mul_mm; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mm_bf16_f32")]] kernel mul_mm_t kernel_mul_mm; +#endif +template [[host_name("kernel_mul_mm_q1_0_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q2_0_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_0_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_1_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_0_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_1_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q8_0_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_mxfp4_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q2_K_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q3_K_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_K_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_K_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q6_K_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_xxs_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_xs_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq3_xxs_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq3_s_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_s_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq1_s_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq1_m_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq4_nl_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq4_xs_f32")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_tq2_0_f32")]] kernel mul_mm_t kernel_mul_mm; + +template [[host_name("kernel_mul_mm_f32_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_f16_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q1_0_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q2_0_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_0_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_1_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_0_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_1_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q8_0_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_mxfp4_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q2_K_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q3_K_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q4_K_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q5_K_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_q6_K_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_xxs_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_xs_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq3_xxs_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq3_s_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq2_s_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq1_s_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq1_m_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq4_nl_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_iq4_xs_f16")]] kernel mul_mm_t kernel_mul_mm; +template [[host_name("kernel_mul_mm_tq2_0_f16")]] kernel mul_mm_t kernel_mul_mm; + +// +// indirect matrix-matrix multiplication +// + +typedef decltype(kernel_mul_mm_id) mul_mm_id; + +template [[host_name("kernel_mul_mm_id_f32_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_f16_f32")]] kernel mul_mm_id kernel_mul_mm_id; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mm_id_bf16_f32")]] kernel mul_mm_id kernel_mul_mm_id; +#endif +template [[host_name("kernel_mul_mm_id_q1_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q2_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_1_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_1_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q8_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_mxfp4_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q2_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q3_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q6_K_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_xxs_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_xs_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq3_xxs_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq3_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq1_s_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq1_m_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq4_nl_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq4_xs_f32")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_tq2_0_f32")]] kernel mul_mm_id kernel_mul_mm_id; + +template [[host_name("kernel_mul_mm_id_f32_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_f16_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q1_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q2_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_1_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_1_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q8_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_mxfp4_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q2_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q3_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q4_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q5_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_q6_K_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_xxs_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_xs_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq3_xxs_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq3_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq2_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq1_s_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq1_m_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq4_nl_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_iq4_xs_f16")]] kernel mul_mm_id kernel_mul_mm_id; +template [[host_name("kernel_mul_mm_id_tq2_0_f16")]] kernel mul_mm_id kernel_mul_mm_id; diff --git a/ggml/src/ggml-metal/kernels/mul_mv.metal b/ggml/src/ggml-metal/kernels/mul_mv.metal new file mode 100644 index 00000000..0d1069f2 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/mul_mv.metal @@ -0,0 +1,3398 @@ +#include "common.h" +#include "dequantize.h" +// Q1_0 dot product: dot = d * (2 * Σ(yl[i] where bit=1) - sumy) +inline float block_q_n_dot_y(device const block_q1_0 * qb_curr, float sumy, thread float * yl, int il) { + device const uint8_t * qs = qb_curr->qs + il / 8; + const uint8_t b0 = qs[0]; + const uint8_t b1 = qs[1]; + + float acc = 0.0f; + + acc += select(0.0f, yl[ 0], bool(b0 & 0x01)); + acc += select(0.0f, yl[ 1], bool(b0 & 0x02)); + acc += select(0.0f, yl[ 2], bool(b0 & 0x04)); + acc += select(0.0f, yl[ 3], bool(b0 & 0x08)); + acc += select(0.0f, yl[ 4], bool(b0 & 0x10)); + acc += select(0.0f, yl[ 5], bool(b0 & 0x20)); + acc += select(0.0f, yl[ 6], bool(b0 & 0x40)); + acc += select(0.0f, yl[ 7], bool(b0 & 0x80)); + + acc += select(0.0f, yl[ 8], bool(b1 & 0x01)); + acc += select(0.0f, yl[ 9], bool(b1 & 0x02)); + acc += select(0.0f, yl[10], bool(b1 & 0x04)); + acc += select(0.0f, yl[11], bool(b1 & 0x08)); + acc += select(0.0f, yl[12], bool(b1 & 0x10)); + acc += select(0.0f, yl[13], bool(b1 & 0x20)); + acc += select(0.0f, yl[14], bool(b1 & 0x40)); + acc += select(0.0f, yl[15], bool(b1 & 0x80)); + + return qb_curr->d * (2.0f * acc - sumy); +} + +// Q2_0 dot: d * (sum_lo(y) + 2*sum_hi(y) - sumy) via per-bit conditional adds +inline float block_q_n_dot_y(device const block_q2_0 * qb_curr, float sumy, thread float * yl, int il) { + device const uint8_t * qs = qb_curr->qs + (il / 4); + const uint8_t b0 = qs[0]; + const uint8_t b1 = qs[1]; + const uint8_t b2 = qs[2]; + const uint8_t b3 = qs[3]; + + // Accumulate where low bit is set (bits 0,2,4,6 of each byte) + float acc_lo = 0.0f; + acc_lo += select(0.0f, yl[ 0], bool(b0 & 0x01)); + acc_lo += select(0.0f, yl[ 1], bool(b0 & 0x04)); + acc_lo += select(0.0f, yl[ 2], bool(b0 & 0x10)); + acc_lo += select(0.0f, yl[ 3], bool(b0 & 0x40)); + acc_lo += select(0.0f, yl[ 4], bool(b1 & 0x01)); + acc_lo += select(0.0f, yl[ 5], bool(b1 & 0x04)); + acc_lo += select(0.0f, yl[ 6], bool(b1 & 0x10)); + acc_lo += select(0.0f, yl[ 7], bool(b1 & 0x40)); + acc_lo += select(0.0f, yl[ 8], bool(b2 & 0x01)); + acc_lo += select(0.0f, yl[ 9], bool(b2 & 0x04)); + acc_lo += select(0.0f, yl[10], bool(b2 & 0x10)); + acc_lo += select(0.0f, yl[11], bool(b2 & 0x40)); + acc_lo += select(0.0f, yl[12], bool(b3 & 0x01)); + acc_lo += select(0.0f, yl[13], bool(b3 & 0x04)); + acc_lo += select(0.0f, yl[14], bool(b3 & 0x10)); + acc_lo += select(0.0f, yl[15], bool(b3 & 0x40)); + + // Accumulate where high bit is set (bits 1,3,5,7 of each byte) + float acc_hi = 0.0f; + acc_hi += select(0.0f, yl[ 0], bool(b0 & 0x02)); + acc_hi += select(0.0f, yl[ 1], bool(b0 & 0x08)); + acc_hi += select(0.0f, yl[ 2], bool(b0 & 0x20)); + acc_hi += select(0.0f, yl[ 3], bool(b0 & 0x80)); + acc_hi += select(0.0f, yl[ 4], bool(b1 & 0x02)); + acc_hi += select(0.0f, yl[ 5], bool(b1 & 0x08)); + acc_hi += select(0.0f, yl[ 6], bool(b1 & 0x20)); + acc_hi += select(0.0f, yl[ 7], bool(b1 & 0x80)); + acc_hi += select(0.0f, yl[ 8], bool(b2 & 0x02)); + acc_hi += select(0.0f, yl[ 9], bool(b2 & 0x08)); + acc_hi += select(0.0f, yl[10], bool(b2 & 0x20)); + acc_hi += select(0.0f, yl[11], bool(b2 & 0x80)); + acc_hi += select(0.0f, yl[12], bool(b3 & 0x02)); + acc_hi += select(0.0f, yl[13], bool(b3 & 0x08)); + acc_hi += select(0.0f, yl[14], bool(b3 & 0x20)); + acc_hi += select(0.0f, yl[15], bool(b3 & 0x80)); + + return qb_curr->d * (acc_lo + 2.0f * acc_hi - sumy); +} + +// function for calculate inner product between half a q4_0 block and 16 floats (yl), sumy is SUM(yl[i]) +// il indicates where the q4 quants begin (0 or QK4_0/4) +// we assume that the yl's have been multiplied with the appropriate scale factor +// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) +inline float block_q_n_dot_y(device const block_q4_0 * qb_curr, float sumy, thread float * yl, int il) { + float d = qb_curr->d; + + float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + + device const uint16_t * qs = ((device const uint16_t *) qb_curr + 1 + il/2); + + for (int i = 0; i < 8; i += 2) { + acc[0] += yl[i + 0] * (qs[i / 2] & 0x000F); + acc[1] += yl[i + 1] * (qs[i / 2] & 0x0F00); + acc[2] += yl[i + 8] * (qs[i / 2] & 0x00F0); + acc[3] += yl[i + 9] * (qs[i / 2] & 0xF000); + } + + return d * (sumy * -8.f + acc[0] + acc[1] + acc[2] + acc[3]); +} + +// function for calculate inner product between half a q4_1 block and 16 floats (yl), sumy is SUM(yl[i]) +// il indicates where the q4 quants begin (0 or QK4_0/4) +// we assume that the yl's have been multiplied with the appropriate scale factor +// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) +inline float block_q_n_dot_y(device const block_q4_1 * qb_curr, float sumy, thread float * yl, int il) { + float d = qb_curr->d; + float m = qb_curr->m; + + float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + + device const uint16_t * qs = ((device const uint16_t *) qb_curr + 2 + il/2); + + for (int i = 0; i < 8; i+=2) { + acc[0] += yl[i + 0] * (qs[i / 2] & 0x000F); + acc[1] += yl[i + 1] * (qs[i / 2] & 0x0F00); + acc[2] += yl[i + 8] * (qs[i / 2] & 0x00F0); + acc[3] += yl[i + 9] * (qs[i / 2] & 0xF000); + } + + return d * (acc[0] + acc[1] + acc[2] + acc[3]) + sumy * m; +} + +// function for calculate inner product between half a q5_0 block and 16 floats (yl), sumy is SUM(yl[i]) +// il indicates where the q5 quants begin (0 or QK5_0/4) +// we assume that the yl's have been multiplied with the appropriate scale factor +// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) +inline float block_q_n_dot_y(device const block_q5_0 * qb_curr, float sumy, thread float * yl, int il) { + float d = qb_curr->d; + + float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + + device const uint16_t * qs = ((device const uint16_t *)qb_curr + 3 + il/2); + const uint32_t qh = *((device const uint32_t *)qb_curr->qh); + + for (int i = 0; i < 8; i+=2) { + acc[0] += yl[i + 0] * ((qs[i / 2] & 0x000F) | ((qh >> (i+0+il ) << 4 ) & 0x00010)); + acc[1] += yl[i + 1] * ((qs[i / 2] & 0x0F00) | ((qh >> (i+1+il ) << 12) & 0x01000)); + acc[2] += yl[i + 8] * ((qs[i / 2] & 0x00F0) | ((qh >> (i+0+il+QK5_0/2) << 8 ) & 0x00100)); + acc[3] += yl[i + 9] * ((qs[i / 2] & 0xF000) | ((qh >> (i+1+il+QK5_0/2) << 16) & 0x10000)); + } + + return d * (sumy * -16.f + acc[0] + acc[1] + acc[2] + acc[3]); +} + +// function for calculate inner product between half a q5_1 block and 16 floats (yl), sumy is SUM(yl[i]) +// il indicates where the q5 quants begin (0 or QK5_1/4) +// we assume that the yl's have been multiplied with the appropriate scale factor +// that corresponds to the missing bit shifts (1, 1/16, 1/256, 1/4096) +inline float block_q_n_dot_y(device const block_q5_1 * qb_curr, float sumy, thread float * yl, int il) { + float d = qb_curr->d; + float m = qb_curr->m; + + float acc[4] = { 0.0f, 0.0f, 0.0f, 0.0f }; + + device const uint16_t * qs = ((device const uint16_t *)qb_curr + 4 + il/2); + const uint32_t qh = *((device const uint32_t *)qb_curr->qh); + + for (int i = 0; i < 8; i+=2) { + acc[0] += yl[i + 0] * ((qs[i / 2] & 0x000F) | ((qh >> (i+0+il ) << 4 ) & 0x00010)); + acc[1] += yl[i + 1] * ((qs[i / 2] & 0x0F00) | ((qh >> (i+1+il ) << 12) & 0x01000)); + acc[2] += yl[i + 8] * ((qs[i / 2] & 0x00F0) | ((qh >> (i+0+il+QK5_0/2) << 8 ) & 0x00100)); + acc[3] += yl[i + 9] * ((qs[i / 2] & 0xF000) | ((qh >> (i+1+il+QK5_0/2) << 16) & 0x10000)); + } + + return d * (acc[0] + acc[1] + acc[2] + acc[3]) + sumy * m; +} + +template +static inline void helper_mv_reduce_and_write( + device float * dst_f32, + float sumf[NR0], + const int r0, + const int ne01, + ushort tiisg, + ushort sgitg, + threadgroup char * shmem) { + constexpr short NW = N_SIMDWIDTH; + + threadgroup float * shmem_f32[NR0]; + + for (short row = 0; row < NR0; ++row) { + shmem_f32[row] = (threadgroup float *) shmem + NW*row; + + if (sgitg == 0) { + shmem_f32[row][tiisg] = 0.0f; + } + + sumf[row] = simd_sum(sumf[row]); + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (short row = 0; row < NR0; ++row) { + if (tiisg == 0) { + shmem_f32[row][sgitg] = sumf[row]; + } + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (short row = 0; row < NR0 && r0 + row < ne01; ++row) { + float tot = simd_sum(shmem_f32[row][tiisg]); + + if (tiisg == 0 && sgitg == 0) { + dst_f32[r0 + row] = tot; + } + } +} + +constant short FC_mul_mv_nsg [[function_constant(FC_MUL_MV + 0)]]; +constant short FC_mul_mv_nxpsg [[function_constant(FC_MUL_MV + 1)]]; +constant short FC_mul_mv_ne12 [[function_constant(FC_MUL_MV + 2)]]; +constant short FC_mul_mv_r2 [[function_constant(FC_MUL_MV + 3)]]; +constant short FC_mul_mv_r3 [[function_constant(FC_MUL_MV + 4)]]; +constant bool FC_mul_mv_split [[function_constant(FC_MUL_MV + 5)]]; + +template +void mul_vec_q_n_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NQ = 16; + + const int nb = args.ne00/QK4_0; + + const int r0 = (tgpig.x*NSG + sgitg)*NR0; + //const int r0 = tgpig.x*NR0; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + //device const block_q_type * x = (device const block_q_type *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + // pointers to src0 rows + device const block_q_type * ax[NR0]; + FOR_UNROLL (int row = 0; row < NR0; ++row) { + const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + + ax[row] = (device const block_q_type *) ((device char *) src0 + offset0); + } + + float sumf[NR0] = {0.f}; + + const short ix = (tiisg/(NW/NQ)); + const short il = (tiisg%(NW/NQ))*8; + + //const int ib0 = sgitg*NQ + ix; + const int ib0 = ix; + + float yl[16]; // src1 vector cache + + //device const float * yb = y + ix*QK4_0 + il; + device const float * yb = y + ib0*QK4_0 + il; + + // each thread in a SIMD group deals with half a block. + //for (int ib = ib0; ib < nb; ib += NSG*NQ) { + for (int ib = ib0; ib < nb; ib += NQ) { + float sumy[2] = { 0.f, 0.f }; + + FOR_UNROLL (short i = 0; i < 8; i += 2) { + sumy[0] += yb[i + 0] + yb[i + 1]; + yl[i + 0] = yb[i + 0]; + yl[i + 1] = yb[i + 1]/256.f; + + sumy[1] += yb[i + 16] + yb[i + 17]; + yl[i + 8] = yb[i + 16]/16.f; + yl[i + 9] = yb[i + 17]/4096.f; + } + + FOR_UNROLL (short row = 0; row < NR0; row++) { + sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy[0] + sumy[1], yl, il); + } + + yb += QK4_0 * 16; + //yb += NSG*NQ*QK4_0; + } + + device float * dst_f32 = (device float *) dst + im*args.ne0*args.ne1 + r1*args.ne0; + + //helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); + + for (int row = 0; row < NR0; ++row) { + const float tot = simd_sum(sumf[row]); + + if (tiisg == 0 && r0 + row < args.ne01) { + dst_f32[r0 + row] = tot; + } + } +} + +template +void kernel_mul_mv_q1_0_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK1_0; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset1 = r1*args.nb11 + (i12)*args.nb12 + (i13)*args.nb13; + + device const float * y = (device const float *) (src1 + offset1); + + device const block_q1_0 * ax[nr0]; + for (int row = 0; row < nr0; ++row) { + const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + ax[row] = (device const block_q1_0 *) ((device char *) src0 + offset0); + } + + float yl[16]; + float sumf[nr0] = {0.f}; + + const short ix = (tiisg/8); + const short il = (tiisg%8)*16; + + device const float * yb = y + ix*QK1_0 + il; + + for (int ib = ix; ib < nb; ib += N_SIMDWIDTH/8) { + float sumy = 0.f; + + FOR_UNROLL (short i = 0; i < 16; i++) { + yl[i] = yb[i]; + sumy += yb[i]; + } + + FOR_UNROLL (short row = 0; row < nr0; row++) { + sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy, yl, il); + } + + yb += QK1_0 * (N_SIMDWIDTH/8); + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0; ++row) { + const float tot = simd_sum(sumf[row]); + + if (tiisg == 0 && first_row + row < args.ne01) { + dst_f32[first_row + row] = tot; + } + } +} + +[[host_name("kernel_mul_mv_q1_0_f32")]] +kernel void kernel_mul_mv_q1_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_q1_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q2_0_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK2_0; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset1 = r1*args.nb11 + (i12)*args.nb12 + (i13)*args.nb13; + + device const float * y = (device const float *) (src1 + offset1); + + device const block_q2_0 * ax[nr0]; + for (int row = 0; row < nr0; ++row) { + const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + ax[row] = (device const block_q2_0 *) ((device char *) src0 + offset0); + } + + float yl[16]; + float sumf[nr0] = {0.f}; + + // group 64: 4 sub-blocks of 16 weights per Q2_0 block + const short ix = (tiisg/4); + const short il = (tiisg%4)*16; + + device const float * yb = y + ix*QK2_0 + il; + + for (int ib = ix; ib < nb; ib += N_SIMDWIDTH/4) { + float sumy = 0.f; + + FOR_UNROLL (short i = 0; i < 16; i++) { + yl[i] = yb[i]; + sumy += yb[i]; + } + + FOR_UNROLL (short row = 0; row < nr0; row++) { + sumf[row] += block_q_n_dot_y(ax[row] + ib, sumy, yl, il); + } + + yb += QK2_0 * (N_SIMDWIDTH/4); + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0; ++row) { + const float tot = simd_sum(sumf[row]); + + if (tiisg == 0 && first_row + row < args.ne01) { + dst_f32[first_row + row] = tot; + } + } +} + +[[host_name("kernel_mul_mv_q2_0_f32")]] +kernel void kernel_mul_mv_q2_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_q2_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +kernel void kernel_mul_mv_q4_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +kernel void kernel_mul_mv_q4_1_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +kernel void kernel_mul_mv_q5_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +kernel void kernel_mul_mv_q5_1_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + mul_vec_q_n_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q8_0_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NQ = 8; + + const int nb = args.ne00/QK8_0; + + const int r0 = tgpig.x*NR0; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + //device const block_q8_0 * x = (device const block_q8_0 *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + // pointers to src0 rows + device const block_q8_0 * ax[NR0]; + FOR_UNROLL (short row = 0; row < NR0; ++row) { + const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + + ax[row] = (device const block_q8_0 *) ((device char *) src0 + offset0); + } + + float sumf[NR0] = { 0.f }; + + const short ix = tiisg/(NW/NQ); + const short il = tiisg%(NW/NQ); + + const int ib0 = sgitg*NQ + ix; + + float yl[NQ]; + + device const float * yb = y + ib0*QK8_0 + il*NQ; + + // each thread in a SIMD group deals with NQ quants at a time + for (int ib = ib0; ib < nb; ib += NSG*NQ) { + for (short i = 0; i < NQ; ++i) { + yl[i] = yb[i]; + } + + for (short row = 0; row < NR0; row++) { + device const int8_t * qs = ax[row][ib].qs + il*NQ; + + float sumq = 0.f; + FOR_UNROLL (short i = 0; i < NQ; ++i) { + sumq += qs[i] * yl[i]; + } + + sumf[row] += sumq*ax[row][ib].d; + } + + yb += NSG*NQ*QK8_0; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); +} + +[[host_name("kernel_mul_mv_q8_0_f32")]] +kernel void kernel_mul_mv_q8_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_q8_0_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +// mat-vec kernel processing in chunks of float4 +// chpb - chunks per quantization block +template +void kernel_mul_mv_ext_q4_f32_impl( + constant ggml_metal_kargs_mul_mv_ext & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + const short NSG = FC_mul_mv_nsg; + const short nxpsg = FC_mul_mv_nxpsg; + + const short chpt = 4; // chunks per thread + + //const short nxpsg = (32); + const short nypsg = (32/nxpsg); + + const short tx = tiisg%nxpsg; + const short ty = tiisg/nxpsg; + + const int i01 = tgpig.x*(nypsg*NSG) + nypsg*sgitg + ty; + const int i11 = tgpig.y*r1ptg; + const int i1m = tgpig.z; + + const int i12 = i1m%FC_mul_mv_ne12; + const int i13 = i1m/FC_mul_mv_ne12; + + const uint64_t offset0 = i01*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = i11*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const q_t * xq = (i01 < args.ne01) ? (device const q_t *) (src0 + offset0) + tx/chpb : (device const q_t *) src0; + + device const float4 * y4[r1ptg]; + + for (int ir1 = 0; ir1 < r1ptg; ++ir1) { + y4[ir1] = (i11 + ir1 < args.ne11) ? (device const float4 *) (src1 + offset1 + ir1*args.nb11) + tx : (device const float4 *) src1; + } + + float sumf[r1ptg] = { [ 0 ... r1ptg - 1 ] = 0.0f }; + + short cch = tx%chpb; // current chunk index + + for (int ich = tx; 4*ich < args.ne00; ich += chpt*nxpsg) { + float4 lx[chpt]; + +#pragma unroll(chpt) + for (short ch = 0; ch < chpt; ++ch) { + deq_t4(xq, cch, lx[ch]); + + cch += nxpsg; + if (cch >= chpb) { + xq += cch/chpb; + cch %= chpb; + } + } + +#pragma unroll(chpt) + for (short ch = 0; ch < chpt; ++ch) { +#pragma unroll(r1ptg) + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + sumf[ir1] += dot(lx[ch], y4[ir1][ch*nxpsg]); + } + } + +#pragma unroll(r1ptg) + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + y4[ir1] += chpt*nxpsg; + } + } + + // reduce only the threads in each row + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + if (nxpsg >= 32) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 16); + } + if (nxpsg >= 16) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 8); + } + if (nxpsg >= 8) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 4); + } + if (nxpsg >= 4) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 2); + } + if (nxpsg >= 2) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 1); + } + + //sumf[ir1] = simd_sum(sumf[ir1]); + } + + if (tx == 0) { + for (short ir1 = 0; ir1 < r1ptg && i11 + ir1 < args.ne11; ++ir1) { + device float * dst_f32 = (device float *) dst + (uint64_t)i1m*args.ne0*args.ne1 + (uint64_t)(i11 + ir1)*args.ne0; + + if (i01 < args.ne01) { + dst_f32[i01] = sumf[ir1]; + } + } + } +} + +// mat-vec kernel processing in chunks of float4x4 +template +void kernel_mul_mv_ext_q4x4_f32_impl( + constant ggml_metal_kargs_mul_mv_ext & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + const short NSG = FC_mul_mv_nsg; + const short nxpsg = FC_mul_mv_nxpsg; + + const short chpt = 1; + + //const short nxpsg = (32); + const short nypsg = (32/nxpsg); + + const short tx = tiisg%nxpsg; + const short ty = tiisg/nxpsg; + + const int i01 = tgpig.x*(nypsg*NSG) + nypsg*sgitg + ty; + const int i11 = tgpig.y*r1ptg; + const int i1m = tgpig.z; + + const int i12 = i1m%FC_mul_mv_ne12; + const int i13 = i1m/FC_mul_mv_ne12; + + const uint64_t offset0 = i01*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = i11*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const q_t * xq = (i01 < args.ne01) ? (device const q_t *) (src0 + offset0) + tx/chpb : (device const q_t *) src0; + + device const float4x4 * y4x4[r1ptg]; + + for (int ir1 = 0; ir1 < r1ptg; ++ir1) { + y4x4[ir1] = (i11 + ir1 < args.ne11) ? (device const float4x4 *) (src1 + offset1 + ir1*args.nb11) + tx : (device const float4x4 *) src1; + } + + float sumf[r1ptg] = { [ 0 ... r1ptg - 1 ] = 0.0f }; + + short cch = tx%chpb; + + for (int ich = tx; 16*ich < args.ne00; ich += chpt*nxpsg) { + float4x4 lx[chpt]; + +#pragma unroll(chpt) + for (short ch = 0; ch < chpt; ++ch) { + deq_t4x4(xq, cch, lx[ch]); + + cch += nxpsg; + if (cch >= chpb) { + xq += cch/chpb; + cch %= chpb; + } + } + +#pragma unroll(chpt) + for (short ch = 0; ch < chpt; ++ch) { +#pragma unroll(r1ptg) + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + sumf[ir1] += + dot(lx[ch][0], y4x4[ir1][ch*nxpsg][0]) + + dot(lx[ch][1], y4x4[ir1][ch*nxpsg][1]) + + dot(lx[ch][2], y4x4[ir1][ch*nxpsg][2]) + + dot(lx[ch][3], y4x4[ir1][ch*nxpsg][3]); + + } + } + +#pragma unroll(r1ptg) + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + y4x4[ir1] += chpt*nxpsg; + } + } + + for (short ir1 = 0; ir1 < r1ptg; ++ir1) { + if (nxpsg >= 32) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 16); + } + if (nxpsg >= 16) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 8); + } + if (nxpsg >= 8) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 4); + } + if (nxpsg >= 4) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 2); + } + if (nxpsg >= 2) { + sumf[ir1] += simd_shuffle_down(sumf[ir1], 1); + } + + //sumf[ir1] = simd_sum(sumf[ir1]); + } + + if (tx == 0) { + for (short ir1 = 0; ir1 < r1ptg && i11 + ir1 < args.ne11; ++ir1) { + device float * dst_f32 = (device float *) dst + (uint64_t)i1m*args.ne0*args.ne1 + (uint64_t)(i11 + ir1)*args.ne0; + + if (i01 < args.ne01) { + dst_f32[i01] = sumf[ir1]; + } + } + } +} + +// dispatchers needed for compile-time nxpsg +// epb - elements per quantization block +template +kernel void kernel_mul_mv_ext_q4_f32_disp( + constant ggml_metal_kargs_mul_mv_ext & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_ext_q4_f32_impl(args, src0, src1, dst, tgpig, tiisg, sgitg); +} + +template +kernel void kernel_mul_mv_ext_q4x4_f32_disp( + constant ggml_metal_kargs_mul_mv_ext & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_ext_q4x4_f32_impl(args, src0, src1, dst, tgpig, tiisg, sgitg); +} + +typedef decltype(kernel_mul_mv_ext_q4_f32_disp <2, block_q8_0, 32, dequantize_q8_0_t4>) mul_mv_ext_q4_f32_t; +typedef decltype(kernel_mul_mv_ext_q4x4_f32_disp<2, block_q4_K, 256, dequantize_q4_K>) mul_mv_ext_q4x4_f32_t; + +template [[host_name("kernel_mul_mv_ext_f32_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, float4, 4, dequantize_f32_t4>; +template [[host_name("kernel_mul_mv_ext_f32_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, float4, 4, dequantize_f32_t4>; +template [[host_name("kernel_mul_mv_ext_f32_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, float4, 4, dequantize_f32_t4>; +template [[host_name("kernel_mul_mv_ext_f32_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, float4, 4, dequantize_f32_t4>; + +template [[host_name("kernel_mul_mv_ext_f16_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, half4, 4, dequantize_f16_t4>; +template [[host_name("kernel_mul_mv_ext_f16_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, half4, 4, dequantize_f16_t4>; +template [[host_name("kernel_mul_mv_ext_f16_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, half4, 4, dequantize_f16_t4>; +template [[host_name("kernel_mul_mv_ext_f16_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, half4, 4, dequantize_f16_t4>; + +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, bfloat4, 4, dequantize_bf16_t4>; +template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, bfloat4, 4, dequantize_bf16_t4>; +template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, bfloat4, 4, dequantize_bf16_t4>; +template [[host_name("kernel_mul_mv_ext_bf16_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, bfloat4, 4, dequantize_bf16_t4>; +#endif + +template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q1_0, 128, dequantize_q1_0_t4>; +template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q1_0, 128, dequantize_q1_0_t4>; +template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q1_0, 128, dequantize_q1_0_t4>; +template [[host_name("kernel_mul_mv_ext_q1_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q1_0, 128, dequantize_q1_0_t4>; + +template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q2_0, 64, dequantize_q2_0_t4>; +template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q2_0, 64, dequantize_q2_0_t4>; +template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q2_0, 64, dequantize_q2_0_t4>; +template [[host_name("kernel_mul_mv_ext_q2_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q2_0, 64, dequantize_q2_0_t4>; + +template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q4_0, 32, dequantize_q4_0_t4>; +template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q4_0, 32, dequantize_q4_0_t4>; +template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q4_0, 32, dequantize_q4_0_t4>; +template [[host_name("kernel_mul_mv_ext_q4_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q4_0, 32, dequantize_q4_0_t4>; + +template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q4_1, 32, dequantize_q4_1_t4>; +template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q4_1, 32, dequantize_q4_1_t4>; +template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q4_1, 32, dequantize_q4_1_t4>; +template [[host_name("kernel_mul_mv_ext_q4_1_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q4_1, 32, dequantize_q4_1_t4>; + +template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q5_0, 32, dequantize_q5_0_t4>; +template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q5_0, 32, dequantize_q5_0_t4>; +template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q5_0, 32, dequantize_q5_0_t4>; +template [[host_name("kernel_mul_mv_ext_q5_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q5_0, 32, dequantize_q5_0_t4>; + +template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q5_1, 32, dequantize_q5_1_t4>; +template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q5_1, 32, dequantize_q5_1_t4>; +template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q5_1, 32, dequantize_q5_1_t4>; +template [[host_name("kernel_mul_mv_ext_q5_1_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q5_1, 32, dequantize_q5_1_t4>; + +template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_q8_0, 32, dequantize_q8_0_t4>; +template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_q8_0, 32, dequantize_q8_0_t4>; +template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_q8_0, 32, dequantize_q8_0_t4>; +template [[host_name("kernel_mul_mv_ext_q8_0_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_q8_0, 32, dequantize_q8_0_t4>; + +template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_mxfp4, 32, dequantize_mxfp4_t4>; +template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_mxfp4, 32, dequantize_mxfp4_t4>; +template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_mxfp4, 32, dequantize_mxfp4_t4>; +template [[host_name("kernel_mul_mv_ext_mxfp4_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_mxfp4, 32, dequantize_mxfp4_t4>; + +template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_2")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<2, block_iq4_nl, 32, dequantize_iq4_nl_t4>; +template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_3")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<3, block_iq4_nl, 32, dequantize_iq4_nl_t4>; +template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_4")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<4, block_iq4_nl, 32, dequantize_iq4_nl_t4>; +template [[host_name("kernel_mul_mv_ext_iq4_nl_f32_r1_5")]] kernel mul_mv_ext_q4_f32_t kernel_mul_mv_ext_q4_f32_disp<5, block_iq4_nl, 32, dequantize_iq4_nl_t4>; + +template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q4_K, 256, dequantize_q4_K>; +template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q4_K, 256, dequantize_q4_K>; +template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q4_K, 256, dequantize_q4_K>; +template [[host_name("kernel_mul_mv_ext_q4_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q4_K, 256, dequantize_q4_K>; + +template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q5_K, 256, dequantize_q5_K>; +template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q5_K, 256, dequantize_q5_K>; +template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q5_K, 256, dequantize_q5_K>; +template [[host_name("kernel_mul_mv_ext_q5_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q5_K, 256, dequantize_q5_K>; + +template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q6_K, 256, dequantize_q6_K>; +template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q6_K, 256, dequantize_q6_K>; +template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q6_K, 256, dequantize_q6_K>; +template [[host_name("kernel_mul_mv_ext_q6_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q6_K, 256, dequantize_q6_K>; + +template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q2_K, 256, dequantize_q2_K>; +template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q2_K, 256, dequantize_q2_K>; +template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q2_K, 256, dequantize_q2_K>; +template [[host_name("kernel_mul_mv_ext_q2_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q2_K, 256, dequantize_q2_K>; + +template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_2")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<2, block_q3_K, 256, dequantize_q3_K>; +template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_3")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<3, block_q3_K, 256, dequantize_q3_K>; +template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_4")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<4, block_q3_K, 256, dequantize_q3_K>; +template [[host_name("kernel_mul_mv_ext_q3_K_f32_r1_5")]] kernel mul_mv_ext_q4x4_f32_t kernel_mul_mv_ext_q4x4_f32_disp<5, block_q3_K, 256, dequantize_q3_K>; + +template +void kernel_mul_mv_t_t_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NB = 32; + constexpr short NF = 8; + + const int nb = args.ne00/NB; + + const int r0 = tgpig.x*NR0; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + //device const T0 * x = (device const T0 *) (src0 + offset0); + device const T1 * y = (device const T1 *) (src1 + offset1); + + // pointers to src0 rows + device const T0 * ax [NR0]; + FOR_UNROLL (short row = 0; row < NR0; ++row) { + const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + + ax[row] = (device const T0 *) ((device char *) src0 + offset0); + } + + float sumf[NR0] = { 0.f }; + + const short ix = tiisg/(NW/NF); + const short il = tiisg%(NW/NF); + + const int ib0 = sgitg*NF + ix; + + T1 yl[NF]; + + device const T1 * yb = y + (ib0*NB + il*NF); + + for (int ib = ib0; ib < nb; ib += NSG*NF) { + for (short i = 0; i < NF; ++i) { + yl[i] = yb[i]; + } + + for (short row = 0; row < NR0; row++) { + device const T0 * xb = ax[row] + (ib*NB + il*NF); + + float sumq = 0.f; + FOR_UNROLL (short i = 0; i < NF; ++i) { + sumq += xb[i] * yl[i]; + } + + sumf[row] += sumq; + } + + yb += NSG*NF*NW; + } + + for (int i = nb*NB + sgitg*NW + tiisg; i < args.ne00; i += NW*NSG) { + for (short row = 0; row < NR0; row++) { + sumf[row] += ax[row][i] * y[i]; + } + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); +} + +template +void kernel_mul_mv_t_t_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + switch (args.nr0) { + //case 1: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + case 2: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + //case 3: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + //case 4: kernel_mul_mv_t_t_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + } +} + +template +kernel void kernel_mul_mv_t_t( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_t_t_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +typedef decltype(kernel_mul_mv_t_t) mul_mv_t_t; + +template [[host_name("kernel_mul_mv_f32_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +template [[host_name("kernel_mul_mv_f16_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +template [[host_name("kernel_mul_mv_f16_f16")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_bf16_f32")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +template [[host_name("kernel_mul_mv_bf16_bf16")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +template [[host_name("kernel_mul_mv_f32_bf16")]] kernel mul_mv_t_t kernel_mul_mv_t_t; +#endif + +template +void kernel_mul_mv_t_t_4_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr short NW = N_SIMDWIDTH; + constexpr short NB = 32; + constexpr short NF = 16; + constexpr short NF4 = NF/4; + + const int nb = args.ne00/NB; + + const int r0 = tgpig.x*NR0; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + //const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const T1 * y = (device const T1 *) (src1 + offset1); + device const T14 * y4 = (device const T14 *) (src1 + offset1); + + // pointers to src0 rows + device const T0 * ax [NR0]; + device const T04 * ax4[NR0]; + FOR_UNROLL (short row = 0; row < NR0; ++row) { + const uint64_t offset0 = (r0 + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + + ax [row] = (device const T0 *) ((device char *) src0 + offset0); + ax4[row] = (device const T04 *) ((device char *) src0 + offset0); + } + + float sumf[NR0] = { 0.f }; + + const short ix = tiisg/(NW/NF); + const short il = tiisg%(NW/NF); + + const int ib0 = sgitg*NF + ix; + + T14 yl4[NF4]; + + device const T14 * yb4 = y4 + (ib0*NB + il*NF)/4; + + for (int ib = ib0; ib < nb; ib += NSG*NF) { + for (short i = 0; i < NF4; ++i) { + yl4[i] = yb4[i]; + } + + for (short row = 0; row < NR0; row++) { + device const T04 * xb4 = ax4[row] + (ib*NB + il*NF)/4; + + float sumq = 0.f; + FOR_UNROLL (short i = 0; i < NF4; ++i) { + sumq += dot(float4(xb4[i]), float4(yl4[i])); + } + + sumf[row] += sumq; + } + + yb4 += NSG*NF*NW/4; + } + + for (int i = nb*NB + sgitg*NW + tiisg; i < args.ne00; i += NW*NSG) { + for (short row = 0; row < NR0; row++) { + sumf[row] += ax[row][i] * y[i]; + } + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + helper_mv_reduce_and_write(dst_f32, sumf, r0, args.ne01, tiisg, sgitg, shmem); +} + +template +void kernel_mul_mv_t_t_4_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + switch (args.nr0) { + //case 1: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + case 2: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + //case 3: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + //case 4: kernel_mul_mv_t_t_4_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); break; + }; +} + +template +kernel void kernel_mul_mv_t_t_4( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_t_t_4_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +typedef decltype(kernel_mul_mv_t_t_4) mul_mv_t_t_4; + +template [[host_name("kernel_mul_mv_f32_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +template [[host_name("kernel_mul_mv_f16_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +template [[host_name("kernel_mul_mv_f16_f16_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_bf16_f32_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +template [[host_name("kernel_mul_mv_bf16_bf16_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +template [[host_name("kernel_mul_mv_f32_bf16_4")]] kernel mul_mv_t_t_4 kernel_mul_mv_t_t_4; +#endif + +template +void kernel_mul_mv_t_t_short_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig, + ushort tiisg) { + const int r0 = tgpig.x*32 + tiisg; + const int r1 = tgpig.y; + const int im = tgpig.z; + + if (r0 >= args.ne01) { + return; + } + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = r0*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + + device const T0 * x = (device const T0 *) (src0 + offset0); + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1; + + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const T1 * y = (device const T1 *) (src1 + offset1); + + float res = 0.0f; + + for (int i = 0; i < args.ne00; ++i) { + res += (float) x[i] * (float) y[i]; + } + + dst_f32[(uint64_t)r1*args.ne0 + r0] = res; +} + +template +kernel void kernel_mul_mv_t_t_short( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]]) { + kernel_mul_mv_t_t_short_impl( + args, + src0, + src1, + dst, + tgpig, + tiisg); +} + +typedef decltype(kernel_mul_mv_t_t_short) mul_mv_t_t_short_t; + +template [[host_name("kernel_mul_mv_f32_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +template [[host_name("kernel_mul_mv_f16_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +template [[host_name("kernel_mul_mv_f16_f16_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_bf16_f32_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +template [[host_name("kernel_mul_mv_bf16_bf16_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +template [[host_name("kernel_mul_mv_f32_bf16_short")]] kernel mul_mv_t_t_short_t kernel_mul_mv_t_t_short; +#endif + +template +void kernel_mul_mv_q2_K_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_q2_K * x = (device const block_q2_K *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + const short ix = tiisg/8; // 0...3 + const short it = tiisg%8; // 0...7 + const short iq = it/4; // 0 or 1 + const short ir = it%4; // 0...3 + const short is = (8*ir)/16;// 0 or 1 + + device const float * y4 = y + ix * QK_K + 128 * iq + 8 * ir; + + for (int ib = ix; ib < nb; ib += 4) { + float4 sumy = {0.f, 0.f, 0.f, 0.f}; + for (short i = 0; i < 8; ++i) { + yl[i+ 0] = y4[i+ 0]; sumy[0] += yl[i+ 0]; + yl[i+ 8] = y4[i+32]; sumy[1] += yl[i+ 8]; + yl[i+16] = y4[i+64]; sumy[2] += yl[i+16]; + yl[i+24] = y4[i+96]; sumy[3] += yl[i+24]; + } + + device const uint8_t * sc = (device const uint8_t *)x[ib].scales + 8*iq + is; + device const uint16_t * qs = (device const uint16_t *)x[ib].qs + 16 * iq + 4 * ir; + device const half * dh = &x[ib].d; + + for (short row = 0; row < nr0; row++) { + float4 acc1 = {0.f, 0.f, 0.f, 0.f}; + float4 acc2 = {0.f, 0.f, 0.f, 0.f}; + for (int i = 0; i < 8; i += 2) { + acc1[0] += yl[i+ 0] * (qs[i/2] & 0x0003); + acc2[0] += yl[i+ 1] * (qs[i/2] & 0x0300); + acc1[1] += yl[i+ 8] * (qs[i/2] & 0x000c); + acc2[1] += yl[i+ 9] * (qs[i/2] & 0x0c00); + acc1[2] += yl[i+16] * (qs[i/2] & 0x0030); + acc2[2] += yl[i+17] * (qs[i/2] & 0x3000); + acc1[3] += yl[i+24] * (qs[i/2] & 0x00c0); + acc2[3] += yl[i+25] * (qs[i/2] & 0xc000); + } + float dall = dh[0]; + float dmin = dh[1] * 1.f/16.f; + sumf[row] += dall * ((acc1[0] + 1.f/256.f * acc2[0]) * (sc[0] & 0xF) * 1.f/ 1.f + + (acc1[1] + 1.f/256.f * acc2[1]) * (sc[2] & 0xF) * 1.f/ 4.f + + (acc1[2] + 1.f/256.f * acc2[2]) * (sc[4] & 0xF) * 1.f/16.f + + (acc1[3] + 1.f/256.f * acc2[3]) * (sc[6] & 0xF) * 1.f/64.f) - + dmin * (sumy[0] * (sc[0] & 0xF0) + sumy[1] * (sc[2] & 0xF0) + sumy[2] * (sc[4] & 0xF0) + sumy[3] * (sc[6] & 0xF0)); + + qs += args.nb01/2; + sc += args.nb01; + dh += args.nb01/2; + } + + y4 += 4 * QK_K; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_q2_K_f32")]] +kernel void kernel_mul_mv_q2_K_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_q2_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q3_K_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_q3_K * x = (device const block_q3_K *) (src0 + offset0); + device const float * yy = (device const float *) (src1 + offset1); + + float yl[32]; + + //const uint16_t kmask1 = 0x3030; + //const uint16_t kmask2 = 0x0f0f; + + const short tid = tiisg/4; + const short ix = tiisg%4; + const short ip = tid/4; // 0 or 1 + const short il = 2*((tid%4)/2); // 0 or 2 + const short ir = tid%2; + const short l0 = 8*ir; + + // One would think that the Metal compiler would figure out that ip and il can only have + // 4 possible states, and optimize accordingly. Well, no. It needs help, and we do it + // with these two tales. + // + // Possible masks for the high bit + const ushort4 mm[4] = {{0x0001, 0x0100, 0x0002, 0x0200}, // ip = 0, il = 0 + {0x0004, 0x0400, 0x0008, 0x0800}, // ip = 0, il = 2 + {0x0010, 0x1000, 0x0020, 0x2000}, // ip = 1, il = 0 + {0x0040, 0x4000, 0x0080, 0x8000}}; // ip = 1, il = 2 + + // Possible masks for the low 2 bits + const int4 qm[2] = {{0x0003, 0x0300, 0x000c, 0x0c00}, {0x0030, 0x3000, 0x00c0, 0xc000}}; + + const ushort4 hm = mm[2*ip + il/2]; + + const short shift = 2*il; + + const float v1 = il == 0 ? 4.f : 64.f; + const float v2 = 4.f * v1; + + const uint16_t s_shift1 = 4*ip; + const uint16_t s_shift2 = s_shift1 + il; + + const short q_offset = 32*ip + l0; + const short y_offset = 128*ip + 32*il + l0; + + device const float * y1 = yy + ix*QK_K + y_offset; + + uint32_t scales32, aux32; + thread uint16_t * scales16 = (thread uint16_t *)&scales32; + thread const int8_t * scales = (thread const int8_t *)&scales32; + + float sumf1[nr0] = {0.f}; + float sumf2[nr0] = {0.f}; + + for (int i = ix; i < nb; i += 4) { + for (short l = 0; l < 8; ++l) { + yl[l+ 0] = y1[l+ 0]; + yl[l+ 8] = y1[l+16]; + yl[l+16] = y1[l+32]; + yl[l+24] = y1[l+48]; + } + + device const uint16_t * q = (device const uint16_t *)(x[i].qs + q_offset); + device const uint16_t * h = (device const uint16_t *)(x[i].hmask + l0); + device const uint16_t * a = (device const uint16_t *)(x[i].scales); + device const half * dh = &x[i].d; + + for (short row = 0; row < nr0; ++row) { + const float d_all = (float)dh[0]; + + scales16[0] = a[4]; + scales16[1] = a[5]; + aux32 = ((scales32 >> s_shift2) << 4) & 0x30303030; + scales16[0] = a[il+0]; + scales16[1] = a[il+1]; + scales32 = ((scales32 >> s_shift1) & 0x0f0f0f0f) | aux32; + + float s1 = 0, s2 = 0, s3 = 0, s4 = 0, s5 = 0, s6 = 0; + for (short l = 0; l < 8; l += 2) { + const int32_t qs = q[l/2]; + s1 += yl[l+0] * (qs & qm[il/2][0]); + s2 += yl[l+1] * (qs & qm[il/2][1]); + s3 += ((h[l/2] & hm[0]) ? 0.f : yl[l+0]) + ((h[l/2] & hm[1]) ? 0.f : yl[l+1]); + s4 += yl[l+16] * (qs & qm[il/2][2]); + s5 += yl[l+17] * (qs & qm[il/2][3]); + s6 += ((h[l/2] & hm[2]) ? 0.f : yl[l+16]) + ((h[l/2] & hm[3]) ? 0.f : yl[l+17]); + } + float d1 = d_all * (s1 + 1.f/256.f * s2 - s3*v1); + float d2 = d_all * (s4 + 1.f/256.f * s5 - s6*v2); + sumf1[row] += d1 * (scales[0] - 32); + sumf2[row] += d2 * (scales[2] - 32); + + s1 = s2 = s3 = s4 = s5 = s6 = 0; + for (short l = 0; l < 8; l += 2) { + const int32_t qs = q[l/2+8]; + s1 += yl[l+8] * (qs & qm[il/2][0]); + s2 += yl[l+9] * (qs & qm[il/2][1]); + s3 += ((h[l/2+8] & hm[0]) ? 0.f : yl[l+8]) + ((h[l/2+8] & hm[1]) ? 0.f : yl[l+9]); + s4 += yl[l+24] * (qs & qm[il/2][2]); + s5 += yl[l+25] * (qs & qm[il/2][3]); + s6 += ((h[l/2+8] & hm[2]) ? 0.f : yl[l+24]) + ((h[l/2+8] & hm[3]) ? 0.f : yl[l+25]); + } + d1 = d_all * (s1 + 1.f/256.f * s2 - s3*v1); + d2 = d_all * (s4 + 1.f/256.f * s5 - s6*v2); + sumf1[row] += d1 * (scales[1] - 32); + sumf2[row] += d2 * (scales[3] - 32); + + q += args.nb01/2; + h += args.nb01/2; + a += args.nb01/2; + dh += args.nb01/2; + } + + y1 += 4 * QK_K; + } + + for (int row = 0; row < nr0; ++row) { + const float sumf = (sumf1[row] + 0.25f * sumf2[row]) / (1 << shift); + sumf1[row] = simd_sum(sumf); + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + if (tiisg == 0) { + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + dst_f32[first_row + row] = sumf1[row]; + } + } +} + +[[host_name("kernel_mul_mv_q3_K_f32")]] +kernel void kernel_mul_mv_q3_K_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_q3_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q4_K_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr uint16_t kmask1 = 0x3f3f; + constexpr uint16_t kmask2 = 0x0f0f; + constexpr uint16_t kmask3 = 0xc0c0; + + const short ix = tiisg/8; // 0...3 + const short it = tiisg%8; // 0...7 + const short iq = it/4; // 0 or 1 + const short ir = it%4; // 0...3 + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_q4_K * x = (device const block_q4_K *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[16]; + float yh[16]; + + float sumf[nr0]={0.f}; + + device const float * y4 = y + ix * QK_K + 64 * iq + 8 * ir; + + uint16_t sc16[4]; + thread const uint8_t * sc8 = (thread const uint8_t *)sc16; + + for (int ib = ix; ib < nb; ib += 4) { + float4 sumy = {0.f, 0.f, 0.f, 0.f}; + + for (short i = 0; i < 8; ++i) { + yl[i+0] = y4[i+ 0]; sumy[0] += yl[i+0]; + yl[i+8] = y4[i+ 32]; sumy[1] += yl[i+8]; + yh[i+0] = y4[i+128]; sumy[2] += yh[i+0]; + yh[i+8] = y4[i+160]; sumy[3] += yh[i+8]; + } + + device const uint16_t * sc = (device const uint16_t *)x[ib].scales + iq; + device const uint16_t * q1 = (device const uint16_t *)x[ib].qs + 16 * iq + 4 * ir; + device const half * dh = &x[ib].d; + + for (short row = 0; row < nr0; row++) { + sc16[0] = sc[0] & kmask1; + sc16[1] = sc[2] & kmask1; + sc16[2] = ((sc[4] >> 0) & kmask2) | ((sc[0] & kmask3) >> 2); + sc16[3] = ((sc[4] >> 4) & kmask2) | ((sc[2] & kmask3) >> 2); + + device const uint16_t * q2 = q1 + 32; + + float4 acc1 = {0.f, 0.f, 0.f, 0.f}; + float4 acc2 = {0.f, 0.f, 0.f, 0.f}; + + FOR_UNROLL (short i = 0; i < 4; ++i) { + acc1[0] += yl[2*i + 0] * (q1[i] & 0x000F); + acc1[1] += yl[2*i + 1] * (q1[i] & 0x0F00); + acc1[2] += yl[2*i + 8] * (q1[i] & 0x00F0); + acc1[3] += yl[2*i + 9] * (q1[i] & 0xF000); + acc2[0] += yh[2*i + 0] * (q2[i] & 0x000F); + acc2[1] += yh[2*i + 1] * (q2[i] & 0x0F00); + acc2[2] += yh[2*i + 8] * (q2[i] & 0x00F0); + acc2[3] += yh[2*i + 9] * (q2[i] & 0xF000); + } + + sumf[row] += dh[0] * ((acc1[0] + 1.f/256.f * acc1[1]) * sc8[0] + + (acc1[2] + 1.f/256.f * acc1[3]) * sc8[1] * 1.f/16.f + + (acc2[0] + 1.f/256.f * acc2[1]) * sc8[4] + + (acc2[2] + 1.f/256.f * acc2[3]) * sc8[5] * 1.f/16.f) - + dh[1] * (sumy[0] * sc8[2] + sumy[1] * sc8[3] + sumy[2] * sc8[6] + sumy[3] * sc8[7]); + + q1 += args.nb01/2; + sc += args.nb01/2; + dh += args.nb01/2; + } + + y4 += 4 * QK_K; + } + + device float * dst_f32 = (device float *) dst + (int64_t)im*args.ne0*args.ne1 + (int64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_q4_K_f32")]] +kernel void kernel_mul_mv_q4_K_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_q4_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q5_K_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_q5_K * x = (device const block_q5_K *) (src0 + offset0); + device const float * yy = (device const float *) (src1 + offset1); + + float sumf[nr0]={0.f}; + + float yl[16], yh[16]; + + constexpr uint16_t kmask1 = 0x3f3f; + constexpr uint16_t kmask2 = 0x0f0f; + constexpr uint16_t kmask3 = 0xc0c0; + + const short tid = tiisg/4; + const short ix = tiisg%4; + const short iq = tid/4; + const short ir = tid%4; + + const short l0 = 8*ir; + const short q_offset = 32*iq + l0; + const short y_offset = 64*iq + l0; + + const uint8_t hm1 = 1u << (2*iq); + const uint8_t hm2 = hm1 << 1; + const uint8_t hm3 = hm1 << 4; + const uint8_t hm4 = hm2 << 4; + + uint16_t sc16[4]; + thread const uint8_t * sc8 = (thread const uint8_t *)sc16; + + device const float * y1 = yy + ix*QK_K + y_offset; + + for (int i = ix; i < nb; i += 4) { + device const uint8_t * q1 = x[i].qs + q_offset; + device const uint8_t * qh = x[i].qh + l0; + device const half * dh = &x[i].d; + device const uint16_t * a = (device const uint16_t *)x[i].scales + iq; + + device const float * y2 = y1 + 128; + float4 sumy = {0.f, 0.f, 0.f, 0.f}; + for (short l = 0; l < 8; ++l) { + yl[l+0] = y1[l+ 0]; sumy[0] += yl[l+0]; + yl[l+8] = y1[l+32]; sumy[1] += yl[l+8]; + yh[l+0] = y2[l+ 0]; sumy[2] += yh[l+0]; + yh[l+8] = y2[l+32]; sumy[3] += yh[l+8]; + } + + for (short row = 0; row < nr0; ++row) { + device const uint8_t * q2 = q1 + 64; + + sc16[0] = a[0] & kmask1; + sc16[1] = a[2] & kmask1; + sc16[2] = ((a[4] >> 0) & kmask2) | ((a[0] & kmask3) >> 2); + sc16[3] = ((a[4] >> 4) & kmask2) | ((a[2] & kmask3) >> 2); + + float4 acc1 = {0.f}; + float4 acc2 = {0.f}; + FOR_UNROLL (short l = 0; l < 8; ++l) { + uint8_t h = qh[l]; + acc1[0] += yl[l+0] * (q1[l] & 0x0F); + acc1[1] += yl[l+8] * (q1[l] & 0xF0); + acc1[2] += yh[l+0] * (q2[l] & 0x0F); + acc1[3] += yh[l+8] * (q2[l] & 0xF0); + acc2[0] += h & hm1 ? yl[l+0] : 0.f; + acc2[1] += h & hm2 ? yl[l+8] : 0.f; + acc2[2] += h & hm3 ? yh[l+0] : 0.f; + acc2[3] += h & hm4 ? yh[l+8] : 0.f; + } + + sumf[row] += dh[0] * (sc8[0] * (acc1[0] + 16.f*acc2[0]) + + sc8[1] * (acc1[1]/16.f + 16.f*acc2[1]) + + sc8[4] * (acc1[2] + 16.f*acc2[2]) + + sc8[5] * (acc1[3]/16.f + 16.f*acc2[3])) - + dh[1] * (sumy[0] * sc8[2] + sumy[1] * sc8[3] + sumy[2] * sc8[6] + sumy[3] * sc8[7]); + + q1 += args.nb01; + qh += args.nb01; + dh += args.nb01/2; + a += args.nb01/2; + } + + y1 += 4 * QK_K; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + const float tot = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = tot; + } + } +} + +[[host_name("kernel_mul_mv_q5_K_f32")]] +kernel void kernel_mul_mv_q5_K_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_q5_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_q6_K_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + constexpr uint8_t kmask1 = 0x03; + constexpr uint8_t kmask2 = 0x0C; + constexpr uint8_t kmask3 = 0x30; + constexpr uint8_t kmask4 = 0xC0; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_q6_K * x = (device const block_q6_K *) (src0 + offset0); + device const float * yy = (device const float *) (src1 + offset1); + + float sumf[nr0] = { 0.f }; + + float yl[16]; + + const short tid = tiisg/2; + const short ix = tiisg%2; + const short ip = tid/8; // 0 or 1 + const short il = tid%8; + const short l0 = 4*il; + const short is = 8*ip + l0/16; + + const short y_offset = 128*ip + l0; + const short q_offset_l = 64*ip + l0; + const short q_offset_h = 32*ip + l0; + + for (int i = ix; i < nb; i += 2) { + device const uint8_t * q1 = x[i].ql + q_offset_l; + device const uint8_t * q2 = q1 + 32; + device const uint8_t * qh = x[i].qh + q_offset_h; + device const int8_t * sc = x[i].scales + is; + device const half * dh = &x[i].d; + + device const float * y = yy + i * QK_K + y_offset; + + for (short l = 0; l < 4; ++l) { + yl[4*l + 0] = y[l + 0]; + yl[4*l + 1] = y[l + 32]; + yl[4*l + 2] = y[l + 64]; + yl[4*l + 3] = y[l + 96]; + } + + for (short row = 0; row < nr0; ++row) { + float4 sums = {0.f, 0.f, 0.f, 0.f}; + + FOR_UNROLL (short l = 0; l < 4; ++l) { + sums[0] += yl[4*l + 0] * ((int8_t)((q1[l] & 0xF) | ((qh[l] & kmask1) << 4)) - 32); + sums[1] += yl[4*l + 1] * ((int8_t)((q2[l] & 0xF) | ((qh[l] & kmask2) << 2)) - 32); + sums[2] += yl[4*l + 2] * ((int8_t)((q1[l] >> 4) | ((qh[l] & kmask3) << 0)) - 32); + sums[3] += yl[4*l + 3] * ((int8_t)((q2[l] >> 4) | ((qh[l] & kmask4) >> 2)) - 32); + } + + sumf[row] += dh[0] * (sums[0] * sc[0] + sums[1] * sc[2] + sums[2] * sc[4] + sums[3] * sc[6]); + + q1 += args.nb01; + q2 += args.nb01; + qh += args.nb01; + sc += args.nb01; + dh += args.nb01/2; + } + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_q6_K_f32")]] +kernel void kernel_mul_mv_q6_K_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_q6_K_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +// ======================= "True" 2-bit + +template +void kernel_mul_mv_iq2_xxs_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq2_xxs * x = (device const block_iq2_xxs *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + threadgroup uint64_t * svalues = (threadgroup uint64_t *)(shmem); + threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 256); + { + int nval = 4; + int pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2xxs_grid[pos + i]; + nval = 2; + pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq2_xxs * xr = x + ibl; + device const uint16_t * q2 = xr->qs + 4 * ib; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + const float db = dh[0]; + device const uint8_t * aux8 = (device const uint8_t *)q2; + const uint32_t aux32 = q2[2] | (q2[3] << 16); + const float d = db * (0.5f + (aux32 >> 28)); + + float sum = 0; + for (short l = 0; l < 4; ++l) { + const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + aux8[l]); + const uint8_t signs = ssigns[(aux32 >> 7*l) & 127]; + for (short j = 0; j < 8; ++j) { + sum += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); + } + } + sumf[row] += d * sum; + + dh += args.nb01/2; + q2 += args.nb01/2; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all * 0.25f; + } + } +} + +template +void kernel_mul_mv_iq2_xxs_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq2_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq2_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq2_xxs_f32")]] +kernel void kernel_mul_mv_iq2_xxs_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + kernel_mul_mv_iq2_xxs_f32_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq2_xs_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq2_xs * x = (device const block_iq2_xs *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + threadgroup uint64_t * svalues = (threadgroup uint64_t *)(shmem); + threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 512); + { + int nval = 8; + int pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2xs_grid[pos + i]; + nval = 2; + pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq2_xs * xr = x + ibl; + device const uint16_t * q2 = xr->qs + 4 * ib; + device const uint8_t * sc = xr->scales + ib; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + const float db = dh[0]; + const uint8_t ls1 = sc[0] & 0xf; + const uint8_t ls2 = sc[0] >> 4; + const float d1 = db * (0.5f + ls1); + const float d2 = db * (0.5f + ls2); + + float sum1 = 0, sum2 = 0; + for (short l = 0; l < 2; ++l) { + const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + (q2[l] & 511)); + const uint8_t signs = ssigns[(q2[l] >> 9)]; + for (short j = 0; j < 8; ++j) { + sum1 += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); + } + } + for (short l = 2; l < 4; ++l) { + const threadgroup uint8_t * grid = (const threadgroup uint8_t *)(svalues + (q2[l] & 511)); + const uint8_t signs = ssigns[(q2[l] >> 9)]; + for (short j = 0; j < 8; ++j) { + sum2 += yl[8*l + j] * grid[j] * (signs & kmask_iq2xs[j] ? -1.f : 1.f); + } + } + sumf[row] += d1 * sum1 + d2 * sum2; + + dh += args.nb01/2; + q2 += args.nb01/2; + sc += args.nb01; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all * 0.25f; + } + } +} + +template +void kernel_mul_mv_iq2_xs_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq2_xs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq2_xs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq2_xs_f32")]] +kernel void kernel_mul_mv_iq2_xs_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq2_xs_f32_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +// FC_mul_mv_split: for nb32 < 32 (nb32 divides 32), 32/nb32 threads share each chunk and each takes a slice of the rows +template +void kernel_mul_mv_iq3_xxs_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq3_xxs * x = (device const block_iq3_xxs *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + threadgroup uint32_t * svalues = (threadgroup uint32_t *)(shmem); + threadgroup uint8_t * ssigns = (threadgroup uint8_t *)(svalues + 256); + { + int nval = 4; + int pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) svalues[pos + i] = iq3xxs_grid[pos + i]; + nval = 2; + pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) ssigns[pos+i] = ksigns_iq2xs[pos+i]; + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq3_xxs * xr = x + ibl; + device const uint8_t * q3 = xr->qs + 8 * ib; + device const uint16_t * gas = (device const uint16_t *)(xr->qs + QK_K/4) + 2 * ib; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + const float db = dh[0]; + const uint32_t aux32 = gas[0] | (gas[1] << 16); + const float d = db * (0.5f + (aux32 >> 28)); + + float2 sum = {0}; + for (short l = 0; l < 4; ++l) { + const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(svalues + q3[2*l+0]); + const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(svalues + q3[2*l+1]); + const uint8_t signs = ssigns[(aux32 >> 7*l) & 127]; + for (short j = 0; j < 4; ++j) { + sum[0] += yl[8*l + j + 0] * grid1[j] * (signs & kmask_iq2xs[j+0] ? -1.f : 1.f); + sum[1] += yl[8*l + j + 4] * grid2[j] * (signs & kmask_iq2xs[j+4] ? -1.f : 1.f); + } + } + sumf[row] += d * (sum[0] + sum[1]); + + dh += args.nb01/2; + q3 += args.nb01; + gas += args.nb01/2; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all * 0.5f; + } + } +} + +template +void kernel_mul_mv_iq3_xxs_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq3_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq3_xxs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq3_xxs_f32")]] +kernel void kernel_mul_mv_iq3_xxs_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq3_xxs_f32_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq3_s_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq3_s * x = (device const block_iq3_s *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + threadgroup uint32_t * svalues = (threadgroup uint32_t *) shmem; + { + int nval = 8; + int pos = (32*sgitg + tiisg)*nval; + for (int i = 0; i < nval; ++i) svalues[pos + i] = iq3s_grid[pos + i]; + threadgroup_barrier(mem_flags::mem_threadgroup); + } + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq3_s * xr = x + ibl; + device const uint8_t * qs = xr->qs + 8 * ib; + device const uint8_t * qh = xr->qh + ib; + device const uint8_t * sc = xr->scales + (ib/2); + device const uint8_t * signs = xr->signs + 4 * ib; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + const float db = dh[0]; + const float d = db * (1 + 2*((sc[0] >> 4*(ib%2)) & 0xf)); + + float2 sum = {0}; + for (short l = 0; l < 4; ++l) { + const threadgroup uint32_t * table1 = qh[0] & kmask_iq2xs[2*l+0] ? svalues + 256 : svalues; + const threadgroup uint32_t * table2 = qh[0] & kmask_iq2xs[2*l+1] ? svalues + 256 : svalues; + const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(table1 + qs[2*l+0]); + const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(table2 + qs[2*l+1]); + for (short j = 0; j < 4; ++j) { + sum[0] += yl[8*l + j + 0] * grid1[j] * select(1, -1, signs[l] & kmask_iq2xs[j+0]); + sum[1] += yl[8*l + j + 4] * grid2[j] * select(1, -1, signs[l] & kmask_iq2xs[j+4]); + } + } + sumf[row] += d * (sum[0] + sum[1]); + + dh += args.nb01/2; + qs += args.nb01; + qh += args.nb01; + sc += args.nb01; + signs += args.nb01; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +template +void kernel_mul_mv_iq3_s_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq3_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq3_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq3_s_f32")]] +kernel void kernel_mul_mv_iq3_s_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq3_s_f32_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq2_s_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq2_s * x = (device const block_iq2_s *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + //threadgroup uint64_t * svalues = (threadgroup uint64_t *) shmem; + //{ + // int nval = 32; + // int pos = (32*sgitg + tiisg)*nval; + // for (int i = 0; i < nval; ++i) svalues[pos + i] = iq2s_grid[pos + i]; + // threadgroup_barrier(mem_flags::mem_threadgroup); + //} + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq2_s * xr = x + ibl; + device const uint8_t * qs = xr->qs + 4 * ib; + device const uint8_t * qh = xr->qh + ib; + device const uint8_t * sc = xr->scales + ib; + device const uint8_t * signs = qs + QK_K/8; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + const float db = dh[0]; + const float d1 = db * (0.5f + (sc[0] & 0xf)); + const float d2 = db * (0.5f + (sc[0] >> 4)); + + float2 sum = {0}; + for (short l = 0; l < 2; ++l) { + //const threadgroup uint8_t * grid1 = (const threadgroup uint8_t *)(svalues + (qs[l+0] | ((qh[0] << (8-2*l)) & 0x300))); + //const threadgroup uint8_t * grid2 = (const threadgroup uint8_t *)(svalues + (qs[l+2] | ((qh[0] << (4-2*l)) & 0x300))); + constant uint8_t * grid1 = (constant uint8_t *)(iq2s_grid + (qs[l+0] | ((qh[0] << (8-2*l)) & 0x300))); + constant uint8_t * grid2 = (constant uint8_t *)(iq2s_grid + (qs[l+2] | ((qh[0] << (4-2*l)) & 0x300))); + for (short j = 0; j < 8; ++j) { + sum[0] += yl[8*l + j + 0] * grid1[j] * select(1, -1, signs[l+0] & kmask_iq2xs[j]); + sum[1] += yl[8*l + j + 16] * grid2[j] * select(1, -1, signs[l+2] & kmask_iq2xs[j]); + } + } + sumf[row] += d1 * sum[0] + d2 * sum[1]; + + dh += args.nb01/2; + qs += args.nb01; + qh += args.nb01; + sc += args.nb01; + signs += args.nb01; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all * 0.25f; + } + } +} + +template +void kernel_mul_mv_iq2_s_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq2_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq2_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq2_s_f32")]] +kernel void kernel_mul_mv_iq2_s_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq2_s_f32_disp(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq1_s_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq1_s * x = (device const block_iq1_s *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + device const float * y4 = y + 32 * ix; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + float sumy = 0; + for (short i = 0; i < 32; ++i) { + yl[i] = y4[i]; + sumy += yl[i]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq1_s * xr = x + ibl; + device const uint8_t * qs = xr->qs + 4 * ib; + device const uint16_t * qh = xr->qh + ib; + device const half * dh = &xr->d; + + for (short row = row0; row < row1; row++) { + constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); + constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 5) & 0x700))); + constant uint8_t * grid3 = (constant uint8_t *)(iq1s_grid_gpu + (qs[2] | ((qh[0] << 2) & 0x700))); + constant uint8_t * grid4 = (constant uint8_t *)(iq1s_grid_gpu + (qs[3] | ((qh[0] >> 1) & 0x700))); + + float sum = 0; + for (short j = 0; j < 4; ++j) { + sum += yl[j+ 0] * (grid1[j] & 0xf) + yl[j+ 4] * (grid1[j] >> 4) + + yl[j+ 8] * (grid2[j] & 0xf) + yl[j+12] * (grid2[j] >> 4) + + yl[j+16] * (grid3[j] & 0xf) + yl[j+20] * (grid3[j] >> 4) + + yl[j+24] * (grid4[j] & 0xf) + yl[j+28] * (grid4[j] >> 4); + } + sumf[row] += (float)dh[0] * (sum + sumy * (qh[0] & 0x8000 ? -1 - IQ1S_DELTA : -1 + IQ1S_DELTA)) * (2*((qh[0] >> 12) & 7) + 1); + + dh += args.nb01/2; + qs += args.nb01; + qh += args.nb01/2; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +template +void kernel_mul_mv_iq1_s_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq1_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq1_s_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq1_s_f32")]] +kernel void kernel_mul_mv_iq1_s_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq1_s_f32_disp(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq1_m_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const int nb32 = nb * (QK_K / 32); + + const short ntx = FC_mul_mv_split ? nb32 : 32; + const short nrep = 32 / ntx; + + const short ix = tiisg % ntx; + const short irep = tiisg / ntx; + + const short row0 = (nr0 * irep ) / nrep; + const short row1 = (nr0 * (irep + 1)) / nrep; + + const uint64_t offset0 = (first_row + row0)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq1_m * x = (device const block_iq1_m *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + float yl[32]; + float sumf[nr0]={0.f}; + + device const float * y4 = y + 32 * ix; + + iq1m_scale_t scale; + + for (int ib32 = ix; ib32 < nb32; ib32 += ntx) { + float4 sumy = {0.f}; + for (short i = 0; i < 8; ++i) { + yl[i+ 0] = y4[i+ 0]; sumy[0] += yl[i+ 0]; + yl[i+ 8] = y4[i+ 8]; sumy[1] += yl[i+ 8]; + yl[i+16] = y4[i+16]; sumy[2] += yl[i+16]; + yl[i+24] = y4[i+24]; sumy[3] += yl[i+24]; + } + + const int ibl = ib32 / (QK_K / 32); + const int ib = ib32 % (QK_K / 32); + + device const block_iq1_m * xr = x + ibl; + device const uint8_t * qs = xr->qs + 4 * ib; + device const uint8_t * qh = xr->qh + 2 * ib; + device const uint16_t * sc = (device const uint16_t *)xr->scales; + + for (short row = row0; row < row1; row++) { + scale.u16 = (sc[0] >> 12) | ((sc[1] >> 8) & 0x00f0) | ((sc[2] >> 4) & 0x0f00) | (sc[3] & 0xf000); + + constant uint8_t * grid1 = (constant uint8_t *)(iq1s_grid_gpu + (qs[0] | ((qh[0] << 8) & 0x700))); + constant uint8_t * grid2 = (constant uint8_t *)(iq1s_grid_gpu + (qs[1] | ((qh[0] << 4) & 0x700))); + constant uint8_t * grid3 = (constant uint8_t *)(iq1s_grid_gpu + (qs[2] | ((qh[1] << 8) & 0x700))); + constant uint8_t * grid4 = (constant uint8_t *)(iq1s_grid_gpu + (qs[3] | ((qh[1] << 4) & 0x700))); + + float2 sum = {0.f}; + for (short j = 0; j < 4; ++j) { + sum[0] += yl[j+ 0] * (grid1[j] & 0xf) + yl[j+ 4] * (grid1[j] >> 4) + + yl[j+ 8] * (grid2[j] & 0xf) + yl[j+12] * (grid2[j] >> 4); + sum[1] += yl[j+16] * (grid3[j] & 0xf) + yl[j+20] * (grid3[j] >> 4) + + yl[j+24] * (grid4[j] & 0xf) + yl[j+28] * (grid4[j] >> 4); + } + const float delta1 = sumy[0] * (qh[0] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA) + sumy[1] * (qh[0] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); + const float delta2 = sumy[2] * (qh[1] & 0x08 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA) + sumy[3] * (qh[1] & 0x80 ? -1 - IQ1M_DELTA : -1 + IQ1M_DELTA); + + sumf[row] += (float)scale.f16 * ((sum[0] + delta1) * (2*((sc[ib/2] >> (6*(ib%2)+0)) & 7) + 1) + + (sum[1] + delta2) * (2*((sc[ib/2] >> (6*(ib%2)+3)) & 7) + 1)); + + sc += args.nb01/2; + qs += args.nb01; + qh += args.nb01; + } + + y4 += 32 * ntx; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +template +void kernel_mul_mv_iq1_m_f32_disp( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + if (FC_mul_mv_split) { + kernel_mul_mv_iq1_m_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } else { + kernel_mul_mv_iq1_m_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); + } +} + +[[host_name("kernel_mul_mv_iq1_m_f32")]] +kernel void kernel_mul_mv_iq1_m_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq1_m_f32_disp(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq4_nl_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + threadgroup float * shmem_f32 = (threadgroup float *) shmem; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * NR0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq4_nl * x = (device const block_iq4_nl *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + const int nb = args.ne00/QK4_NL; + const int ns01 = args.nb01/args.nb00; + + const short ix = tiisg/2; // 0...15 + const short it = tiisg%2; // 0 or 1 + + shmem_f32[tiisg] = kvalues_iq4nl_f[tiisg%16]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + float4 yl[4]; + float sumf[NR0]={0.f}; + + device const float * yb = y + ix*QK4_NL + it*8; + + uint32_t aux32[2]; + thread const uint8_t * q8 = (thread const uint8_t *)aux32; + + float4 qf1, qf2; + + // [TAG_MUL_MV_WEIRD] + for (int ib = ix; ib < nb && ib < ns01; ib += 16) { + device const float4 * y4 = (device const float4 *)yb; + yl[0] = y4[0]; + yl[1] = y4[4]; + yl[2] = y4[1]; + yl[3] = y4[5]; + + for (short row = 0; row < NR0; row++) { + device const block_iq4_nl & xb = x[row*ns01 + ib]; + device const uint16_t * q4 = (device const uint16_t *)(xb.qs + 8*it); + + float4 acc1 = {0.f}, acc2 = {0.f}; + + aux32[0] = q4[0] | (q4[1] << 16); + aux32[1] = (aux32[0] >> 4) & 0x0f0f0f0f; + aux32[0] &= 0x0f0f0f0f; + qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; + qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; + acc1 += yl[0] * qf1; + acc2 += yl[1] * qf2; + + aux32[0] = q4[2] | (q4[3] << 16); + aux32[1] = (aux32[0] >> 4) & 0x0f0f0f0f; + aux32[0] &= 0x0f0f0f0f; + qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; + qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; + acc1 += yl[2] * qf1; + acc2 += yl[3] * qf2; + + acc1 += acc2; + + sumf[row] += (float)xb.d * (acc1[0] + acc1[1] + acc1[2] + acc1[3]); + } + + yb += 16 * QK4_NL; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_iq4_nl_f32")]] +kernel void kernel_mul_mv_iq4_nl_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq4_nl_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_iq4_xs_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + threadgroup float * shmem_f32 = (threadgroup float *) shmem; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + const int first_row = (r0 * NSG + sgitg) * NR0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_iq4_xs * x = (device const block_iq4_xs *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + const int nb = args.ne00/QK_K; + const int ns01 = args.nb01/args.nb00; + + const short ix = tiisg/16; // 0 or 1 + const short it = tiisg%16; // 0...15 + const short ib = it/2; + const short il = it%2; + + shmem_f32[tiisg] = kvalues_iq4nl_f[tiisg%16]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + float4 yl[4]; + float sumf[NR0]={0.f}; + + device const float * yb = y + ix * QK_K + ib * 32 + il * 8; + + uint32_t aux32[2]; + thread const uint8_t * q8 = (thread const uint8_t *)aux32; + + float4 qf1, qf2; + + // [TAG_MUL_MV_WEIRD] + for (int ibl = ix; ibl < nb && ibl < ns01; ibl += 2) { + device const float4 * y4 = (device const float4 *)yb; + yl[0] = y4[0]; + yl[1] = y4[4]; + yl[2] = y4[1]; + yl[3] = y4[5]; + + for (short row = 0; row < NR0; ++row) { + device const block_iq4_xs & xb = x[row*ns01 + ibl]; + device const uint32_t * q4 = (device const uint32_t *)(xb.qs + 16*ib + 8*il); + + float4 acc1 = {0.f}, acc2 = {0.f}; + + aux32[0] = (q4[0] ) & 0x0f0f0f0f; + aux32[1] = (q4[0] >> 4) & 0x0f0f0f0f; + qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; + qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; + acc1 += yl[0] * qf1; + acc2 += yl[1] * qf2; + + aux32[0] = (q4[1] ) & 0x0f0f0f0f; + aux32[1] = (q4[1] >> 4) & 0x0f0f0f0f; + qf1 = {shmem_f32[q8[0]], shmem_f32[q8[1]], shmem_f32[q8[2]], shmem_f32[q8[3]]}; + qf2 = {shmem_f32[q8[4]], shmem_f32[q8[5]], shmem_f32[q8[6]], shmem_f32[q8[7]]}; + acc1 += yl[2] * qf1; + acc2 += yl[3] * qf2; + + acc1 += acc2; + + const int ls = (((xb.scales_l[ib/2] >> 4*(ib%2)) & 0xf) | (((xb.scales_h >> 2*ib) & 3) << 4)) - 32; + sumf[row] += (float)xb.d * ls * (acc1[0] + acc1[1] + acc1[2] + acc1[3]); + } + + yb += 2 * QK_K; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_iq4_xs_f32")]] +kernel void kernel_mul_mv_iq4_xs_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_iq4_xs_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_mxfp4_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + threadgroup float * shmem_f32 = (threadgroup float *) shmem; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * NR0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset0 = first_row*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const block_mxfp4 * x = (device const block_mxfp4 *) (src0 + offset0); + device const float * y = (device const float *) (src1 + offset1); + + const int nb = args.ne00/QK_MXFP4; + const int ns01 = args.nb01/args.nb00; // this can be larger than nb for permuted src0 tensors + + const short ix = tiisg/2; // 0...15 + const short it = tiisg%2; // 0 or 1 + + shmem_f32[tiisg] = kvalues_mxfp4_f[tiisg%16]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + float4 yl[4]; + float sumf[NR0]={0.f}; + + device const float * yb = y + ix*QK_MXFP4 + it*8; + + // note: just the check `ib < nb` is enough, but adding the redundant `&& ib < ns01` check makes the kernel a bit faster + // no idea why that is - needs some deeper investigation [TAG_MUL_MV_WEIRD] + for (int ib = ix; ib < nb && ib < ns01; ib += 16) { + device const float4 * y4 = (device const float4 *) yb; + + yl[0] = y4[0]; + yl[1] = y4[4]; + yl[2] = y4[1]; + yl[3] = y4[5]; + + FOR_UNROLL (short row = 0; row < NR0; row++) { + device const block_mxfp4 & xb = x[row*ns01 + ib]; + device const uint8_t * q2 = (device const uint8_t *)(xb.qs + 8*it); + + float4 acc1 = yl[0]*float4(shmem_f32[q2[0] & 0x0F], shmem_f32[q2[1] & 0x0F], shmem_f32[q2[2] & 0x0F], shmem_f32[q2[3] & 0x0F]); + float4 acc2 = yl[1]*float4(shmem_f32[q2[0] >> 4 ], shmem_f32[q2[1] >> 4 ], shmem_f32[q2[2] >> 4 ], shmem_f32[q2[3] >> 4 ]); + float4 acc3 = yl[2]*float4(shmem_f32[q2[4] & 0x0F], shmem_f32[q2[5] & 0x0F], shmem_f32[q2[6] & 0x0F], shmem_f32[q2[7] & 0x0F]); + float4 acc4 = yl[3]*float4(shmem_f32[q2[4] >> 4 ], shmem_f32[q2[5] >> 4 ], shmem_f32[q2[6] >> 4 ], shmem_f32[q2[7] >> 4 ]); + + acc1 = (acc1 + acc3) + (acc2 + acc4); + + sumf[row] += e8m0_to_fp32(xb.e) * ((acc1[0] + acc1[1]) + (acc1[2] + acc1[3])); + } + + yb += 16 * QK_MXFP4; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < NR0 && first_row + row < args.ne0; ++row) { + float sum_all = simd_sum(sumf[row]); + if (tiisg == 0) { + dst_f32[first_row + row] = sum_all; + } + } +} + +[[host_name("kernel_mul_mv_mxfp4_f32")]] +kernel void kernel_mul_mv_mxfp4_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_mxfp4_f32_impl(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +template +void kernel_mul_mv_tq2_0_f32_impl( + args_t args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg) { + const short NSG = FC_mul_mv_nsg; + + const int nb = args.ne00/QK_K; + + const int r0 = tgpig.x; + const int r1 = tgpig.y; + const int im = tgpig.z; + + const int first_row = (r0 * NSG + sgitg) * nr0; + + const uint i12 = im%FC_mul_mv_ne12; + const uint i13 = im/FC_mul_mv_ne12; + + const uint64_t offset1 = r1*args.nb11 + (i12 )*args.nb12 + (i13 )*args.nb13; + + device const float * y = (device const float *) (src1 + offset1); + + device const block_tq2_0 * ax[nr0]; + for (int row = 0; row < nr0; ++row) { + const uint64_t offset0 = (first_row + row)*args.nb01 + (i12/FC_mul_mv_r2)*args.nb02 + (i13/FC_mul_mv_r3)*args.nb03; + ax[row] = (device const block_tq2_0 *) ((device char *) src0 + offset0); + } + + float sumf[nr0] = {0.f}; + + // 8 threads per block, NBLOCK blocks per pass, 2 halves per block per pass + constexpr short NBLOCK = 4; + + constexpr short NB = N_SIMDWIDTH/NBLOCK; // threads per block + + const short blk = tiisg / NB; // 0..NBLOCK-1, block handled by this thread + const short htg = tiisg % NB; // 0..NB-1, thread within block (0..7) + + // byte and y base offsets within the block (32 elements per thread, 4 per byte) + device const float4 * yb4 = (device const float4 *)(y + 4*htg + blk*QK_K); + + // hoisted per-byte coefficients (from y) and total y-sum, shared across rows + // ref: https://github.com/ggml-org/llama.cpp/pull/26980 + float4 coef[4]; + + for (int ib = blk; ib < nb; ib += NBLOCK) { + FOR_UNROLL (short h0 = 0; h0 < 2; ++h0) { + const float4 y0 = yb4[ 0 + 32*h0]; + const float4 y1 = yb4[ 8 + 32*h0]; + const float4 y2 = yb4[16 + 32*h0]; + const float4 y3 = yb4[24 + 32*h0]; + + float sumy = 0.f; + FOR_UNROLL (short j = 0; j < 4; ++j) { + coef[j] = float4( + y0[j], + y1[j] - 4.0f*y0[j], + y2[j] - 4.0f*y1[j], + y3[j] - 4.0f*y2[j]); + + sumy += (y0[j] + y1[j]) + (y2[j] + y3[j]); + } + + FOR_UNROLL (short row = 0; row < nr0; ++row) { + device const block_tq2_0 & xb = ax[row][ib]; + device const uchar * qs = xb.qs + 4*htg + 32*h0; + + float sum = -sumy; + FOR_UNROLL (short j = 0; j < 4; ++j) { + // express the 2-bit field shifts (v>>2, v>>4, v>>6) as float floor ops + const float v = (float)qs[j]; + + const float f0 = v; + const float f1 = floor(v*0.25f); // v>>2 + const float f2 = floor(v*0.0625); // v>>4 + const float f3 = floor(v*0.015625); // v>>6 + + sum += coef[j][0]*f0 + coef[j][1]*f1 + coef[j][2]*f2 + coef[j][3]*f3; + } + + sumf[row] += xb.d * sum; + } + } + + yb4 += QK_K * NBLOCK / 4; + } + + device float * dst_f32 = (device float *) dst + (uint64_t)im*args.ne0*args.ne1 + (uint64_t)r1*args.ne0; + + for (int row = 0; row < nr0; ++row) { + const float tot = simd_sum(sumf[row]); + if (tiisg == 0 && first_row + row < args.ne01) { + dst_f32[first_row + row] = tot; + } + } +} + +[[host_name("kernel_mul_mv_tq2_0_f32")]] +kernel void kernel_mul_mv_tq2_0_f32( + constant ggml_metal_kargs_mul_mv & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + + kernel_mul_mv_tq2_0_f32_impl(args, src0, src1, dst, nullptr, tgpig, tiisg, sgitg); +} + +// +// matrix-vector multiplication +// + +typedef void (kernel_mul_mv_disp_t)( + ggml_metal_kargs_mul_mv args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig, + ushort tiisg); + +typedef void (kernel_mul_mv2_disp_t)( + ggml_metal_kargs_mul_mv args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiisg, + ushort sgitg); + +template +void mmv_fn( + ggml_metal_kargs_mul_mv args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiitg, + ushort tiisg, + ushort sgitg) { + disp_fn(args, src0, src1, dst, tgpig, tiisg); +} + +template +void mmv_fn( + ggml_metal_kargs_mul_mv args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem, + uint3 tgpig, + ushort tiitg, + ushort tiisg, + ushort sgitg) { + disp_fn(args, src0, src1, dst, shmem, tgpig, tiisg, sgitg); +} + +typedef decltype(mmv_fn>) mul_mv_disp_fn_t; + +template +kernel void kernel_mul_mv_id( + constant ggml_metal_kargs_mul_mv_id & args, + device const char * src0s, + device const char * src1, + device char * dst, + device const char * ids, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]]) { + const int iid1 = tgpig.z/args.nei0; + const int idx = tgpig.z%args.nei0; + + tgpig.z = 0; + + const int32_t i02 = ((device const int32_t *) (ids + iid1*args.nbi1))[idx]; + + const int64_t i11 = idx % args.ne11; + const int64_t i12 = iid1; + + const int64_t i1 = idx; + const int64_t i2 = i12; + + device const char * src0_cur = src0s + i02*args.nb02; + device const char * src1_cur = src1 + i11*args.nb11 + i12*args.nb12; + + device char * dst_cur = dst + (i1*args.ne0 + i2*args.ne1*args.ne0)*sizeof(float); + + ggml_metal_kargs_mul_mv args0 = { + /*.ne00 =*/ args.ne00, + /*.ne01 =*/ args.ne01, + /*.ne02 =*/ 1, // args.ne02, + /*.nb00 =*/ args.nb00, + /*.nb01 =*/ args.nb01, + /*.nb02 =*/ args.nb02, + /*.nb03 =*/ args.nb02, // args.ne02 == 1 + /*.ne10 =*/ args.ne10, + /*.ne11 =*/ 1, // args.ne11, + /*.ne12 =*/ 1, // args.ne12, + /*.nb10 =*/ args.nb10, + /*.nb11 =*/ args.nb11, + /*.nb12 =*/ args.nb12, + /*.nb13 =*/ args.nb12, // ne12 == 1 + /*.ne0 =*/ args.ne0, + /*.ne1 =*/ 1, // args.ne1, + /*.nr0 =*/ args.nr0, + /*.r2 =*/ 1, + /*.r3 =*/ 1, + }; + + disp_fn( + args0, + /* src0 */ src0_cur, + /* src1 */ src1_cur, + /* dst */ dst_cur, + shmem, + tgpig, + tiitg, + tiisg, + sgitg); +} + +typedef decltype(kernel_mul_mv_id>>) kernel_mul_mv_id_t; + +typedef decltype(kernel_mul_mv_id>>) kernel_mul_mv_id_4_t; + +template [[host_name("kernel_mul_mv_id_f32_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_f16_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_id_bf16_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +#endif +template [[host_name("kernel_mul_mv_id_f32_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_f16_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_mul_mv_id_bf16_f32_4")]] kernel kernel_mul_mv_id_4_t kernel_mul_mv_id>>; +#endif + +template [[host_name("kernel_mul_mv_id_q8_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; + +template [[host_name("kernel_mul_mv_id_q1_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q2_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q4_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q4_1_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q5_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q5_1_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; + +template [[host_name("kernel_mul_mv_id_mxfp4_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; + +template [[host_name("kernel_mul_mv_id_q2_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q3_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q4_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q5_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_q6_K_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq1_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq1_m_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq2_xxs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq2_xs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq3_xxs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq3_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq2_s_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq4_nl_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_iq4_xs_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; +template [[host_name("kernel_mul_mv_id_tq2_0_f32")]] kernel kernel_mul_mv_id_t kernel_mul_mv_id>>; diff --git a/ggml/src/ggml-metal/kernels/norm.metal b/ggml/src/ggml-metal/kernels/norm.metal new file mode 100644 index 00000000..c76d0c7f --- /dev/null +++ b/ggml/src/ggml-metal/kernels/norm.metal @@ -0,0 +1,318 @@ +#include "common.h" + +constant bool FC_norm_use_scale [[function_constant(FC_NORM + 0)]]; + +// F == 1 : norm (no fuse) +// F == 2 : norm + mul +// F == 3 : norm + mul + add +template +kernel void kernel_norm_fuse_impl( + constant ggml_metal_kargs_norm & args, + device const char * src0, + device const char * src1_0, + device const char * src1_1, + device char * dst, + threadgroup float * shmem_f32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + if (sgitg == 0) { + shmem_f32[tiisg] = 0.0f; + } + + const int i01 = tgpig.x; + const int i02 = tgpig.y; + const int i03 = tgpig.z; + + device const T * x = (device const T *) (src0 + i03*args.nbf3[0] + i02*args.nbf2[0] + i01*args.nbf1[0]); + + device const T * f0 = (device const T *) (src1_0 + (i03%args.nef3[1])*args.nbf3[1] + (i02%args.nef2[1])*args.nbf2[1] + (i01%args.nef1[1])*args.nbf1[1]); + device const T * f1 = (device const T *) (src1_1 + (i03%args.nef3[2])*args.nbf3[2] + (i02%args.nef2[2])*args.nbf2[2] + (i01%args.nef1[2])*args.nbf1[2]); + + T sumft(0.0f); + + float sumf = 0.0f; + + for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { + sumft += x[i00]; + } + sumf = dot(sumft, T(1.0f)); + sumf = simd_sum(sumf); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shmem_f32[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sumf = shmem_f32[tiisg]; + sumf = simd_sum(sumf); + + const float mean = sumf/args.ne00; + + device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); + + sumf = 0.0f; + for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { + y[i00] = x[i00] - mean; + sumf += dot(y[i00], y[i00]); + } + sumf = simd_sum(sumf); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shmem_f32[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sumf = shmem_f32[tiisg]; + sumf = simd_sum(sumf); + + const float variance = sumf/args.ne00; + + const float scale = 1.0f/sqrt(variance + args.eps); + for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { + if (F == 1) { + y[i00] = (y[i00]*scale); + } + if (F == 2) { + if (FC_norm_use_scale) { + y[i00] = (y[i00]*scale) * args.scale; + } else { + y[i00] = (y[i00]*scale)*f0[i00]; + } + } + if (F == 3) { + y[i00] = (y[i00]*scale)*f0[i00] + f1[i00]; + } + } +} + +typedef decltype(kernel_norm_fuse_impl) kernel_norm_fuse_t; + +template [[host_name("kernel_norm_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; +template [[host_name("kernel_norm_mul_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; +template [[host_name("kernel_norm_mul_add_f32")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; + +template [[host_name("kernel_norm_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; +template [[host_name("kernel_norm_mul_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; +template [[host_name("kernel_norm_mul_add_f32_4")]] kernel kernel_norm_fuse_t kernel_norm_fuse_impl; + +// F == 1 : rms_norm (no fuse) +// F == 2 : rms_norm + mul +// F == 3 : rms_norm + mul + add +template +kernel void kernel_rms_norm_fuse_impl( + constant ggml_metal_kargs_norm & args, + device const char * src0, + device const char * src1_0, + device const char * src1_1, + device char * dst, + threadgroup float * shmem_f32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + if (sgitg == 0) { + shmem_f32[tiisg] = 0.0f; + } + + const int i01 = tgpig.x; + const int i02 = tgpig.y; + const int i03 = tgpig.z; + + device const T * x = (device const T *) (src0 + i03*args.nbf3[0] + i02*args.nbf2[0] + i01*args.nbf1[0]); + + device const T * f0 = (device const T *) (src1_0 + (i03%args.nef3[1])*args.nbf3[1] + (i02%args.nef2[1])*args.nbf2[1] + (i01%args.nef1[1])*args.nbf1[1]); + device const T * f1 = (device const T *) (src1_1 + (i03%args.nef3[2])*args.nbf3[2] + (i02%args.nef2[2])*args.nbf2[2] + (i01%args.nef1[2])*args.nbf1[2]); + + float sumf = 0.0f; + + // parallel sum + for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { + sumf += dot(x[i00], x[i00]); + } + sumf = simd_sum(sumf); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shmem_f32[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sumf = shmem_f32[tiisg]; + sumf = simd_sum(sumf); + + const float mean = sumf/args.ne00; + const float scale = 1.0f/sqrt(mean + args.eps); + + device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); + for (int i00 = tpitg.x; i00 < args.ne00_t; i00 += ntg.x) { + if (F == 1) { + y[i00] = (x[i00]*scale); + } + if (F == 2) { + if (FC_norm_use_scale) { + y[i00] = (x[i00]*scale) * args.scale; + } else { + y[i00] = (x[i00]*scale)*f0[i00]; + } + } + if (F == 3) { + y[i00] = (x[i00]*scale)*f0[i00] + f1[i00]; + } + } +} + +typedef decltype(kernel_rms_norm_fuse_impl) kernel_rms_norm_fuse_t; + +template [[host_name("kernel_rms_norm_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; +template [[host_name("kernel_rms_norm_mul_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; +template [[host_name("kernel_rms_norm_mul_add_f32")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; + +template [[host_name("kernel_rms_norm_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; +template [[host_name("kernel_rms_norm_mul_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; +template [[host_name("kernel_rms_norm_mul_add_f32_4")]] kernel kernel_rms_norm_fuse_t kernel_rms_norm_fuse_impl; + +template +kernel void kernel_l2_norm_impl( + constant ggml_metal_kargs_l2_norm & args, + device const char * src0, + device char * dst, + threadgroup float * shmem_f32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int i03 = tgpig.z; + const int i02 = tgpig.y; + const int i01 = tgpig.x; + + if (sgitg == 0) { + shmem_f32[tiisg] = 0.0f; + } + + device const T0 * x = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); + device T * y = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1); + + float sumf = 0.0f; + + // parallel sum + for (int i00 = tpitg.x; i00 < args.ne00; i00 += ntg.x) { + sumf += dot(x[i00], x[i00]); + } + sumf = simd_sum(sumf); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shmem_f32[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sumf = shmem_f32[tiisg]; + sumf = simd_sum(sumf); + + const float scale = 1.0f/max(sqrt(sumf), args.eps); + + for (int i00 = tpitg.x; i00 < args.ne00; i00 += ntg.x) { + y[i00] = x[i00] * scale; + } +} + +typedef decltype(kernel_l2_norm_impl) kernel_l2_norm_t; + +template [[host_name("kernel_l2_norm_f32_f32")]] kernel kernel_l2_norm_t kernel_l2_norm_impl; +template [[host_name("kernel_l2_norm_f32_f32_4")]] kernel kernel_l2_norm_t kernel_l2_norm_impl; + +kernel void kernel_group_norm_f32( + constant ggml_metal_kargs_group_norm & args, + device const float * src0, + device float * dst, + threadgroup float * buf [[threadgroup(0)]], + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint sgitg[[simdgroup_index_in_threadgroup]], + uint tiisg[[thread_index_in_simdgroup]], + uint ntg[[threads_per_threadgroup]]) { + const int64_t ne = args.ne00*args.ne01*args.ne02; + const int64_t gs = args.ne00*args.ne01*((args.ne02 + args.ngrp - 1) / args.ngrp); + + int start = tgpig * gs; + int end = start + gs; + + start += tpitg; + + if (end >= ne) { + end = ne; + } + + float tmp = 0.0f; // partial sum for thread in warp + + for (int j = start; j < end; j += ntg) { + tmp += src0[j]; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + tmp = simd_sum(tmp); + if (ntg > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = 0.0f; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = tmp; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + tmp = buf[tiisg]; + tmp = simd_sum(tmp); + } + + const float mean = tmp / gs; + tmp = 0.0f; + + for (int j = start; j < end; j += ntg) { + float xi = src0[j] - mean; + dst[j] = xi; + tmp += xi * xi; + } + + tmp = simd_sum(tmp); + if (ntg > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = 0.0f; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = tmp; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + tmp = buf[tiisg]; + tmp = simd_sum(tmp); + } + + const float variance = tmp / gs; + const float scale = 1.0f/sqrt(variance + args.eps); + for (int j = start; j < end; j += ntg) { + dst[j] *= scale; + } +} diff --git a/ggml/src/ggml-metal/kernels/pool.metal b/ggml/src/ggml-metal/kernels/pool.metal new file mode 100644 index 00000000..13d355b9 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/pool.metal @@ -0,0 +1,148 @@ +#include "common.h" + +kernel void kernel_pool_2d_max_f32( + constant ggml_metal_kargs_pool_2d & args, + device const float * src0, + device float * dst, + uint gid[[thread_position_in_grid]]) { + + if (gid >= args.np) { + return; + } + + const int idx = gid; + const int I_HW = args.IH * args.IW; + const int O_HW = args.OH * args.OW; + const int nc = idx / O_HW; + const int cur_oh = idx % O_HW / args.OW; + const int cur_ow = idx % O_HW % args.OW; + + device const float * i_ptr = src0 + nc * I_HW; + device float * o_ptr = dst + nc * O_HW; + + const int start_h = cur_oh * args.s1 - args.p1; + const int bh = MAX(0, start_h); + const int eh = MIN(args.IH, start_h + args.k1); + const int start_w = cur_ow * args.s0 - args.p0; + const int bw = MAX(0, start_w); + const int ew = MIN(args.IW, start_w + args.k0); + + float res = -INFINITY; + + for (int i = bh; i < eh; i += 1) { + for (int j = bw; j < ew; j += 1) { + res = MAX(res, i_ptr[i * args.IW + j]); + } + } + + o_ptr[cur_oh * args.OW + cur_ow] = res; +} + +kernel void kernel_pool_2d_avg_f32( + constant ggml_metal_kargs_pool_2d & args, + device const float * src0, + device float * dst, + uint gid[[thread_position_in_grid]]) { + + if (gid >= args.np) { + return; + } + + const int idx = gid; + const int I_HW = args.IH * args.IW; + const int O_HW = args.OH * args.OW; + const int nc = idx / O_HW; + const int cur_oh = idx % O_HW / args.OW; + const int cur_ow = idx % O_HW % args.OW; + + device const float * i_ptr = src0 + nc * I_HW; + device float * o_ptr = dst + nc * O_HW; + + const int start_h = cur_oh * args.s1 - args.p1; + const int bh = MAX(0, start_h); + const int eh = MIN(args.IH, start_h + args.k1); + const int start_w = cur_ow * args.s0 - args.p0; + const int bw = MAX(0, start_w); + const int ew = MIN(args.IW, start_w + args.k0); + // const float scale = 1. / ((eh - bh) * (ew - bw)); + const float scale = 1. / (args.k0 * args.k1); + + float res = 0; + + for (int i = bh; i < eh; i += 1) { + for (int j = bw; j < ew; j += 1) { + float cur = i_ptr[i * args.IW + j]; + res += cur * scale; + } + } + + o_ptr[cur_oh * args.OW + cur_ow] = res; +} + + +kernel void kernel_pool_1d_max_f32( + constant ggml_metal_kargs_pool_1d & args, + device const float * src, + device float * dst, + uint gid [[thread_position_in_grid]] +) { + + if (gid >= args.np) { + return; + } + + const int ow = (int)gid % args.OW; + const int row = (int)gid / args.OW; + + const int base = ow * args.s0 - args.p0; + + float acc = -INFINITY; + + const int src_off = row * args.IW; + const int dst_off = row * args.OW; + + for (int ki = 0; ki < args.k0; ++ki) { + int j = base + ki; + if (j < 0 || j >= args.IW){ + continue; + } + float v = src[src_off + j]; + acc = max(acc, v); + } + + dst[dst_off + ow] = acc; +} + +kernel void kernel_pool_1d_avg_f32( + constant ggml_metal_kargs_pool_1d & args, + device const float * src, + device float * dst, + uint gid [[thread_position_in_grid]] +) { + + if (gid >= args.np) { + return; + } + + const int ow = (int)gid % args.OW; + const int row = (int)gid / args.OW; + + const int base = ow * args.s0 - args.p0; + + float acc = 0.0f; + int cnt = 0; + + const int src_off = row * args.IW; + const int dst_off = row * args.OW; + + for (int ki = 0; ki < args.k0; ++ki) { + const int j = base + ki; + if (j < 0 || j >= args.IW) { + continue; + } + acc += src[src_off + j]; + cnt += 1; + } + + dst[dst_off + ow] = (cnt > 0) ? (acc / (float)cnt) : 0.0f; +} diff --git a/ggml/src/ggml-metal/kernels/quantize.h b/ggml/src/ggml-metal/kernels/quantize.h new file mode 100644 index 00000000..0741b222 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/quantize.h @@ -0,0 +1,262 @@ +#pragma once + +#include "common.h" + +void quantize_q1_0(device const float * src, device block_q1_0 & dst) { + float sum_abs = 0.0f; + for (int j = 0; j < QK1_0; j++) { + sum_abs += fabs(src[j]); + } + dst.d = sum_abs / QK1_0; + + for (int j = 0; j < QK1_0 / 8; j++) { + dst.qs[j] = 0; + } + for (int j = 0; j < QK1_0; j++) { + if (src[j] >= 0.0f) { + dst.qs[j / 8] |= (1 << (j % 8)); + } + } +} + +void quantize_q2_0(device const float * src, device block_q2_0 & dst) { + float amax = 0.0f; + for (int j = 0; j < QK2_0; j++) { + float a = fabs(src[j]); + if (a > amax) amax = a; + } + const float d = amax; + dst.d = d; + + const float id = d > 0.0f ? 1.0f / d : 0.0f; + + for (int j = 0; j < QK2_0 / 4; j++) { + dst.qs[j] = 0; + } + for (int j = 0; j < QK2_0; j++) { + int q = (int)round(src[j] * id) + 1; + q = max(0, min(3, q)); + dst.qs[j / 4] |= (q << (2 * (j % 4))); + } +} + +void quantize_q4_0(device const float * src, device block_q4_0 & dst) { +#pragma METAL fp math_mode(safe) + float amax = 0.0f; // absolute max + float max = 0.0f; + + for (int j = 0; j < QK4_0; j++) { + const float v = src[j]; + if (amax < fabs(v)) { + amax = fabs(v); + max = v; + } + } + + const float d = max / -8; + const float id = d ? 1.0f/d : 0.0f; + + dst.d = d; + + for (int j = 0; j < QK4_0/2; ++j) { + const float x0 = src[0 + j]*id; + const float x1 = src[QK4_0/2 + j]*id; + + const uint8_t xi0 = MIN(15, (int8_t)(x0 + 8.5f)); + const uint8_t xi1 = MIN(15, (int8_t)(x1 + 8.5f)); + + dst.qs[j] = xi0; + dst.qs[j] |= xi1 << 4; + } +} + +void quantize_q4_1(device const float * src, device block_q4_1 & dst) { +#pragma METAL fp math_mode(safe) + float min = FLT_MAX; + float max = -FLT_MAX; + + for (int j = 0; j < QK4_1; j++) { + const float v = src[j]; + if (min > v) min = v; + if (max < v) max = v; + } + + const float d = (max - min) / ((1 << 4) - 1); + const float id = d ? 1.0f/d : 0.0f; + + dst.d = d; + dst.m = min; + + for (int j = 0; j < QK4_1/2; ++j) { + const float x0 = (src[0 + j] - min)*id; + const float x1 = (src[QK4_1/2 + j] - min)*id; + + const uint8_t xi0 = MIN(15, (int8_t)(x0 + 0.5f)); + const uint8_t xi1 = MIN(15, (int8_t)(x1 + 0.5f)); + + dst.qs[j] = xi0; + dst.qs[j] |= xi1 << 4; + } +} + +void quantize_q5_0(device const float * src, device block_q5_0 & dst) { +#pragma METAL fp math_mode(safe) + float amax = 0.0f; // absolute max + float max = 0.0f; + + for (int j = 0; j < QK5_0; j++) { + const float v = src[j]; + if (amax < fabs(v)) { + amax = fabs(v); + max = v; + } + } + + const float d = max / -16; + const float id = d ? 1.0f/d : 0.0f; + + dst.d = d; + + uint32_t qh = 0; + for (int j = 0; j < QK5_0/2; ++j) { + const float x0 = src[0 + j]*id; + const float x1 = src[QK5_0/2 + j]*id; + + const uint8_t xi0 = MIN(31, (int8_t)(x0 + 16.5f)); + const uint8_t xi1 = MIN(31, (int8_t)(x1 + 16.5f)); + + dst.qs[j] = (xi0 & 0xf) | ((xi1 & 0xf) << 4); + qh |= ((xi0 & 0x10u) >> 4) << (j + 0); + qh |= ((xi1 & 0x10u) >> 4) << (j + QK5_0/2); + } + + thread const uint8_t * qh8 = (thread const uint8_t *)&qh; + + for (int j = 0; j < 4; ++j) { + dst.qh[j] = qh8[j]; + } +} + +void quantize_q5_1(device const float * src, device block_q5_1 & dst) { +#pragma METAL fp math_mode(safe) + float max = src[0]; + float min = src[0]; + + for (int j = 1; j < QK5_1; j++) { + const float v = src[j]; + min = v < min ? v : min; + max = v > max ? v : max; + } + + const float d = (max - min) / 31; + const float id = d ? 1.0f/d : 0.0f; + + dst.d = d; + dst.m = min; + + uint32_t qh = 0; + for (int j = 0; j < QK5_1/2; ++j) { + const float x0 = (src[0 + j] - min)*id; + const float x1 = (src[QK5_1/2 + j] - min)*id; + + const uint8_t xi0 = (uint8_t)(x0 + 0.5f); + const uint8_t xi1 = (uint8_t)(x1 + 0.5f); + + dst.qs[j] = (xi0 & 0xf) | ((xi1 & 0xf) << 4); + qh |= ((xi0 & 0x10u) >> 4) << (j + 0); + qh |= ((xi1 & 0x10u) >> 4) << (j + QK5_1/2); + } + + thread const uint8_t * qh8 = (thread const uint8_t *)&qh; + + for (int j = 0; j < 4; ++j) { + dst.qh[j] = qh8[j]; + } +} + +void quantize_q8_0(device const float * src, device block_q8_0 & dst) { +#pragma METAL fp math_mode(safe) + float amax = 0.0f; // absolute max + + for (int j = 0; j < QK8_0; j++) { + const float v = src[j]; + amax = MAX(amax, fabs(v)); + } + + const float d = amax / ((1 << 7) - 1); + const float id = d ? 1.0f/d : 0.0f; + + dst.d = d; + + for (int j = 0; j < QK8_0; ++j) { + const float x0 = src[j]*id; + + dst.qs[j] = round(x0); + } +} + +void quantize_iq4_nl(device const float * src, device block_iq4_nl & dst) { +#pragma METAL fp math_mode(safe) + float amax = 0.0f; // absolute max + float max = 0.0f; + + for (int j = 0; j < QK4_NL; j++) { + const float v = src[j]; + if (amax < fabs(v)) { + amax = fabs(v); + max = v; + } + } + + const float d = max / kvalues_iq4nl_f[0]; + const float id = d ? 1.0f/d : 0.0f; + + float sumqx = 0, sumq2 = 0; + for (int j = 0; j < QK4_NL/2; ++j) { + const float x0 = src[0 + j]*id; + const float x1 = src[QK4_NL/2 + j]*id; + + const uint8_t xi0 = best_index_int8(16, kvalues_iq4nl_f, x0); + const uint8_t xi1 = best_index_int8(16, kvalues_iq4nl_f, x1); + + dst.qs[j] = xi0 | (xi1 << 4); + + const float v0 = kvalues_iq4nl_f[xi0]; + const float v1 = kvalues_iq4nl_f[xi1]; + const float w0 = src[0 + j]*src[0 + j]; + const float w1 = src[QK4_NL/2 + j]*src[QK4_NL/2 + j]; + sumqx += w0*v0*src[j] + w1*v1*src[QK4_NL/2 + j]; + sumq2 += w0*v0*v0 + w1*v1*v1; + + } + + dst.d = sumq2 > 0 ? sumqx/sumq2 : d; +} + +void quantize_tq2_0(device const float * src, device block_tq2_0 & dst) { +#pragma METAL fp math_mode(safe) + float amax = 0.0f; // absolute max + + for (int j = 0; j < QK_K; j++) { + const float v = src[j]; + amax = MAX(amax, fabs(v)); + } + + const float d = amax; + const float id = d ? 1.0f/d : 0.0f; + + dst.d = (half) d; + + for (int j = 0; j < QK_K/4; j += 32) { + for (int m = 0; m < 32; ++m) { + uint8_t q = 0; + for (int n = 0; n < 4; ++n) { + // -1, 0, 1 -> 0, 1, 2 + int xi = (int)round(src[m + n*32] * id) + 1; + q += (uint8_t)((xi & 3) << (2*n)); + } + dst.qs[j + m] = q; + } + src += 4*32; + } +} diff --git a/ggml/src/ggml-metal/kernels/quantize.metal b/ggml/src/ggml-metal/kernels/quantize.metal new file mode 100644 index 00000000..42ca6d74 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/quantize.metal @@ -0,0 +1,480 @@ +#include "common.h" +#include "dequantize.h" +#include "quantize.h" + +template +kernel void kernel_cpy_t_t( + constant ggml_metal_kargs_cpy & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int32_t i03 = tgpig[2]; + const int32_t i02 = tgpig[1]; + const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; + const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; + + if (i01 >= args.ne01) { + return; + } + + const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; + + const int32_t i3 = n/(args.ne2*args.ne1*args.ne0); + const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0)/(args.ne1*args.ne0); + const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0)/args.ne0; + const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0); + + device T1 * dst_data = (device T1 *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.ne00;) { + device const T0 * src = (device T0 *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + i00*args.nb00); + dst_data[i00] = (T1) src[0]; + break; + } +} + +typedef decltype(kernel_cpy_t_t) kernel_cpy_t; + +template [[host_name("kernel_cpy_f32_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_f32_f16")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_f32_i32")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_i32_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_i32_i32")]] kernel kernel_cpy_t kernel_cpy_t_t; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_cpy_f32_bf16")]] kernel kernel_cpy_t kernel_cpy_t_t; +#endif +template [[host_name("kernel_cpy_f16_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_f16_f16")]] kernel kernel_cpy_t kernel_cpy_t_t; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_cpy_bf16_f32")]] kernel kernel_cpy_t kernel_cpy_t_t; +template [[host_name("kernel_cpy_bf16_bf16")]] kernel kernel_cpy_t kernel_cpy_t_t; +#endif + +template +kernel void kernel_cpy_f32_q( + constant ggml_metal_kargs_cpy & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int32_t i03 = tgpig[2]; + const int32_t i02 = tgpig[1]; + const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; + const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; + + if (i01 >= args.ne01) { + return; + } + + const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; + + const int32_t i3 = n / (args.ne2*args.ne1*args.ne0); + const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0) / (args.ne1*args.ne0); + const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0) / args.ne0; + const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0)/QK; + + device block_q * dst_data = (device block_q *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.nk0;) { + device const float * src = (device const float *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + (i00*QK)*args.nb00); + + quantize_func(src, dst_data[i00]); + + break; + } +} + +typedef decltype(kernel_cpy_f32_q) cpy_f_q_t; + +template [[host_name("kernel_cpy_f32_q8_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q1_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q2_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q4_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q4_1")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q5_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_q5_1")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_iq4_nl")]] kernel cpy_f_q_t kernel_cpy_f32_q; +template [[host_name("kernel_cpy_f32_tq2_0")]] kernel cpy_f_q_t kernel_cpy_f32_q; + +template +kernel void kernel_cpy_q_f32( + constant ggml_metal_kargs_cpy & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int32_t i03 = tgpig[2]; + const int32_t i02 = tgpig[1]; + const int32_t i01 = ntg[1] == 1 ? tgpig[0]%args.ne01 : tgpig[0]*ntg[1] + tpitg.y; + const int32_t iw0 = ntg[1] == 1 ? tgpig[0]/args.ne01 : 0; + + if (i01 >= args.ne01) { + return; + } + + const int64_t n = i03*args.ne02*args.ne01*args.ne00 + i02*args.ne01*args.ne00 + i01*args.ne00; + + const int32_t i3 = n/(args.ne2*args.ne1*args.ne0); + const int32_t i2 = (n - i3*args.ne2*args.ne1*args.ne0)/(args.ne1*args.ne0); + const int32_t i1 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0)/args.ne0; + const int32_t i0 = (n - i3*args.ne2*args.ne1*args.ne0 - i2*args.ne1*args.ne0 - i1*args.ne0); + + device const block_q * src_data = (device const block_q *)(src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); + device T4x4 * dst_data = (device T4x4 *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + for (int32_t i00 = iw0*ntg[0] + tpitg.x; i00 < args.nk0;) { + T4x4 temp; + dequantize_func(src_data + i00/nl, i00%nl, temp); + dst_data[i00] = temp; + + break; + } +} + +typedef decltype(kernel_cpy_q_f32) cpy_q_f_t; + +template [[host_name("kernel_cpy_q1_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q2_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q4_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q4_1_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q5_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q5_1_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q8_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; + +template [[host_name("kernel_cpy_tq2_0_f32")]] kernel cpy_q_f_t kernel_cpy_q_f32; + +template [[host_name("kernel_cpy_q1_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q2_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q4_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q4_1_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q5_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q5_1_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; +template [[host_name("kernel_cpy_q8_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; + +template [[host_name("kernel_cpy_tq2_0_f16")]] kernel cpy_q_f_t kernel_cpy_q_f32; + +template +kernel void kernel_concat( + constant ggml_metal_kargs_concat & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = ntg.y == 1 ? tgpig.x : tgpig.x*ntg.y + tpitg.y; + + if (i1 >= args.ne1) { + return; + } + + int o[4] = {0, 0, 0, 0}; + o[args.dim] = args.dim == 0 ? args.ne00 : (args.dim == 1 ? args.ne01 : (args.dim == 2 ? args.ne02 : args.ne03)); + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + device const T * x; + + if (i0 < args.ne00 && i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { + x = (device const T *)(src0 + (i3 )*args.nb03 + (i2 )*args.nb02 + (i1 )*args.nb01 + (i0 )*args.nb00); + } else { + x = (device const T *)(src1 + (i3 - o[3])*args.nb13 + (i2 - o[2])*args.nb12 + (i1 - o[1])*args.nb11 + (i0 - o[0])*args.nb10); + } + + device T * y = (device T *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + *y = *x; + } +} + +typedef decltype(kernel_concat) kernel_concat_t; + +template [[host_name("kernel_concat_f32")]] kernel kernel_concat_t kernel_concat; +template [[host_name("kernel_concat_f16")]] kernel kernel_concat_t kernel_concat; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_concat_bf16")]] kernel kernel_concat_t kernel_concat; +#endif +template [[host_name("kernel_concat_i8")]] kernel kernel_concat_t kernel_concat; +template [[host_name("kernel_concat_i16")]] kernel kernel_concat_t kernel_concat; +template [[host_name("kernel_concat_i32")]] kernel kernel_concat_t kernel_concat; +template [[host_name("kernel_concat_i64")]] kernel kernel_concat_t kernel_concat; + +template +kernel void kernel_concat_q( + constant ggml_metal_kargs_concat & args, + device const char * src0, + device const char * src1, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + // note: for quantized types, the args are in units of blocks (nb0 == type_size) + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = ntg.y == 1 ? tgpig.x : tgpig.x*ntg.y + tpitg.y; + + if (i1 >= args.ne1) { + return; + } + + int o[4] = {0, 0, 0, 0}; + o[args.dim] = args.dim == 0 ? args.ne00 : (args.dim == 1 ? args.ne01 : (args.dim == 2 ? args.ne02 : args.ne03)); + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + device const block_q * x; + + if (i0 < args.ne00 && i1 < args.ne01 && i2 < args.ne02 && i3 < args.ne03) { + x = (device const block_q *)(src0 + (i3 )*args.nb03 + (i2 )*args.nb02 + (i1 )*args.nb01 + (i0 )*args.nb00); + } else { + x = (device const block_q *)(src1 + (i3 - o[3])*args.nb13 + (i2 - o[2])*args.nb12 + (i1 - o[1])*args.nb11 + (i0 - o[0])*args.nb10); + } + + device block_q * y = (device block_q *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + *y = *x; + } +} + +typedef decltype(kernel_concat_q) kernel_concat_q_t; + +template [[host_name("kernel_concat_q4_0")]] kernel kernel_concat_q_t kernel_concat_q; +template [[host_name("kernel_concat_q4_1")]] kernel kernel_concat_q_t kernel_concat_q; +template [[host_name("kernel_concat_q5_0")]] kernel kernel_concat_q_t kernel_concat_q; +template [[host_name("kernel_concat_q5_1")]] kernel kernel_concat_q_t kernel_concat_q; +template [[host_name("kernel_concat_q8_0")]] kernel kernel_concat_q_t kernel_concat_q; + +template +kernel void kernel_get_rows_q( + constant ggml_metal_kargs_get_rows & args, + device const void * src0, + device const void * src1, + device void * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 ntg [[threads_per_threadgroup]]) { + const int32_t iw0 = tgpig.x/args.ne10; + const int32_t i10 = tgpig.x%args.ne10; + const int32_t i11 = tgpig.y; + const int32_t i12 = tgpig.z; + + const int32_t r = ((const device int32_t *) ((const device char *) src1 + i12*args.nb12 + i11*args.nb11 + i10*args.nb10))[0]; + + const int32_t i02 = i11; + const int32_t i03 = i12; + + auto psrc = (device const block_q *) ((const device char *) src0 + i03*args.nb03 + i02*args.nb02 + r*args.nb01); + auto pdst = (device float4x4 *) (( device char *) dst + i12*args.nb3 + i11*args.nb2 + i10*args.nb1); + + for (int ind = iw0*ntg.x + tiitg; ind < args.ne00t;) { + float4x4 temp; + dequantize_func(psrc + ind/nl, ind%nl, temp); + pdst[ind] = temp; + + break; + } +} + +template +kernel void kernel_get_rows_f( + constant ggml_metal_kargs_get_rows & args, + device const void * src0, + device const void * src1, + device void * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 ntg [[threads_per_threadgroup]]) { + const int32_t iw0 = tgpig.x/args.ne10; + const int32_t i10 = tgpig.x%args.ne10; + const int32_t i11 = tgpig.y; + const int32_t i12 = tgpig.z; + + const int32_t r = ((const device int32_t *) ((const device char *) src1 + i12*args.nb12 + i11*args.nb11 + i10*args.nb10))[0]; + + const int32_t i02 = i11; + const int32_t i03 = i12; + + auto psrc = (const device T0 *) ((const device char *) src0 + i03*args.nb03 + i02*args.nb02 + r*args.nb01); + auto pdst = ( device T *) (( device char *) dst + i12*args.nb3 + i11*args.nb2 + i10*args.nb1); + + for (int ind = iw0*ntg.x + tiitg; ind < args.ne00t;) { + pdst[ind] = psrc[ind]; + + break; + } +} + +typedef decltype(kernel_get_rows_f) get_rows_f_t; + +template [[host_name("kernel_get_rows_f32")]] kernel get_rows_f_t kernel_get_rows_f; +template [[host_name("kernel_get_rows_f16")]] kernel get_rows_f_t kernel_get_rows_f; +template [[host_name("kernel_get_rows_i32")]] kernel get_rows_f_t kernel_get_rows_f; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_get_rows_bf16")]] kernel get_rows_f_t kernel_get_rows_f; +#endif + +typedef decltype(kernel_get_rows_q) get_rows_q_t; + +template [[host_name("kernel_get_rows_q1_0")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q2_0")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q4_0")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q4_1")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q5_0")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q5_1")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q8_0")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_mxfp4")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q2_K")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q3_K")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q4_K")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q5_K")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_q6_K")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq2_xxs")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq2_xs")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq3_xxs")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq3_s")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq2_s")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq1_s")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq1_m")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq4_nl")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_iq4_xs")]] kernel get_rows_q_t kernel_get_rows_q; +template [[host_name("kernel_get_rows_tq2_0")]] kernel get_rows_q_t kernel_get_rows_q; + +template +kernel void kernel_set_rows_q( + constant ggml_metal_kargs_set_rows & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint tiitg[[thread_index_in_threadgroup]], + uint3 tptg [[threads_per_threadgroup]]) { + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + + const int32_t i12 = i03%args.ne12; + const int32_t i11 = i02%args.ne11; + + const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; + if (i01 >= args.ne01) { + return; + } + + const int32_t i10 = i01; + const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; + + device block_q * dst_row = ( device block_q *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); + const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); + + for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { + quantize_func(src_row + QK*ind, dst_row[ind]); + } +} + +template +kernel void kernel_set_rows_q32( + constant ggml_metal_kargs_set_rows & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint tiitg[[thread_index_in_threadgroup]], + uint3 tptg [[threads_per_threadgroup]]) { + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + + const int32_t i12 = i03%args.ne12; + const int32_t i11 = i02%args.ne11; + + const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; + if (i01 >= args.ne01) { + return; + } + + const int32_t i10 = i01; + const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; + + device block_q * dst_row = ( device block_q *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); + const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); + + for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { + quantize_func(src_row + 32*ind, dst_row[ind]); + } +} + +template +kernel void kernel_set_rows_f( + constant ggml_metal_kargs_set_rows & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint tiitg[[thread_index_in_threadgroup]], + uint3 tptg [[threads_per_threadgroup]]) { + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + + const int32_t i12 = i03%args.ne12; + const int32_t i11 = i02%args.ne11; + + const int32_t i01 = tgpig.x*tptg.y + tiitg/tptg.x; + if (i01 >= args.ne01) { + return; + } + + const int32_t i10 = i01; + const TI i1 = ((const device TI *) ((const device char *) src1 + i10*args.nb10 + i11*args.nb11 + i12*args.nb12))[0]; + + device TD * dst_row = ( device TD *) (( device char *) dst + i1*args.nb1 + i02*args.nb2 + i03*args.nb3); + const device TS * src_row = (const device TS *) ((const device char *) src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); + + for (int ind = tiitg%tptg.x; ind < args.nk0; ind += tptg.x) { + dst_row[ind] = (TD) src_row[ind]; + } +} + +typedef decltype(kernel_set_rows_f) set_rows_f_t; + +template [[host_name("kernel_set_rows_f32_i64_f32")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_f32_i32_f32")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_f32_i64_f16")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_f32_i32_f16")]] kernel set_rows_f_t kernel_set_rows_f; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_set_rows_f32_i64_bf16")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_f32_i32_bf16")]] kernel set_rows_f_t kernel_set_rows_f; +#endif + +template [[host_name("kernel_set_rows_f16_i64_f16")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_f16_i32_f16")]] kernel set_rows_f_t kernel_set_rows_f; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_set_rows_bf16_i64_bf16")]] kernel set_rows_f_t kernel_set_rows_f; +template [[host_name("kernel_set_rows_bf16_i32_bf16")]] kernel set_rows_f_t kernel_set_rows_f; +#endif + +typedef decltype(kernel_set_rows_q32) set_rows_q32_t; + +template [[host_name("kernel_set_rows_f32_i64_q8_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_q8_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i64_q4_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_q4_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i64_q4_1")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_q4_1")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i64_q5_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_q5_0")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i64_q5_1")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_q5_1")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i64_iq4_nl")]] kernel set_rows_q32_t kernel_set_rows_q32; +template [[host_name("kernel_set_rows_f32_i32_iq4_nl")]] kernel set_rows_q32_t kernel_set_rows_q32; + +typedef decltype(kernel_set_rows_q) set_rows_qK_t; + +template [[host_name("kernel_set_rows_f32_i64_tq2_0")]] kernel set_rows_qK_t kernel_set_rows_q; +template [[host_name("kernel_set_rows_f32_i32_tq2_0")]] kernel set_rows_qK_t kernel_set_rows_q; + diff --git a/ggml/src/ggml-metal/kernels/reduce.metal b/ggml/src/ggml-metal/kernels/reduce.metal new file mode 100644 index 00000000..0af9e4f6 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/reduce.metal @@ -0,0 +1,228 @@ +#include "common.h" + +kernel void kernel_op_sum_f32( + constant ggml_metal_kargs_sum & args, + device const float * src0, + device float * dst, + threadgroup float * shmem_f32 [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + + if (args.np == 0) { + return; + } + + // TODO: become function constant + const uint nsg = (ntg.x + 31) / 32; + + float sumf = 0; + + for (uint64_t i0 = tpitg.x; i0 < args.np; i0 += ntg.x) { + sumf += src0[i0]; + } + + sumf = simd_sum(sumf); + + if (tiisg == 0) { + shmem_f32[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + float total = 0; + + if (sgitg == 0) { + float v = 0; + + if (tpitg.x < nsg) { + v = shmem_f32[tpitg.x]; + } + + total = simd_sum(v); + + if (tpitg.x == 0) { + dst[0] = total; + } + } +} + +constant short FC_sum_rows_op [[function_constant(FC_SUM_ROWS + 0)]]; + +template +kernel void kernel_sum_rows_impl( + constant ggml_metal_kargs_sum_rows & args, + device const char * src0, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { +#define FC_OP FC_sum_rows_op + + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = tgpig.x; + + threadgroup T0 * shmem_t = (threadgroup T0 *) shmem; + + if (sgitg == 0) { + shmem_t[tiisg] = 0.0f; + } + + device const T0 * src_row = (device const T0 *) (src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03); + device T * dst_row = (device T *) (dst + i1*args.nb1 + i2*args.nb2 + i3*args.nb3); + + T0 sumf = T0(0.0f); + + for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { + sumf += src_row[i0]; + } + + sumf = simd_sum(sumf); + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + shmem_t[sgitg] = sumf; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sumf = shmem_t[tiisg]; + sumf = simd_sum(sumf); + + if (tpitg.x == 0) { + if (FC_OP == OP_SUM_ROWS_NUM_MEAN) { + if (is_same::value) { + dst_row[0] = sum(sumf) / (4*args.ne00); + } else { + dst_row[0] = sum(sumf) / args.ne00; + } + } else { + dst_row[0] = sum(sumf); + } + } + +#undef FC_OP +} + +typedef decltype(kernel_sum_rows_impl) kernel_sum_rows_t; + +template [[host_name("kernel_sum_rows_f32_f32")]] kernel kernel_sum_rows_t kernel_sum_rows_impl; +template [[host_name("kernel_sum_rows_f32_f32_4")]] kernel kernel_sum_rows_t kernel_sum_rows_impl; + +template +kernel void kernel_cumsum_blk( + constant ggml_metal_kargs_cumsum_blk & args, + device const char * src0, + device char * tmp, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int ib = tgpig[0]/args.ne01; + + const int i00 = ib*ntg.x; + const int i01 = tgpig[0]%args.ne01; + const int i02 = tgpig[1]; + const int i03 = tgpig[2]; + + device const float * src0_row = (device const float *) (src0 + + args.nb01*i01 + + args.nb02*i02 + + args.nb03*i03); + + threadgroup float * shmem_f32 = (threadgroup float *) shmem; + + float v = 0.0f; + + if (i00 + tpitg.x < args.ne00) { + v = src0_row[i00 + tpitg.x]; + } + + float s = simd_prefix_inclusive_sum(v); + + if (tiisg == N_SIMDWIDTH - 1) { + shmem_f32[sgitg] = s; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (sgitg == 0) { + shmem_f32[tiisg] = simd_prefix_exclusive_sum(shmem_f32[tiisg]); + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + s += shmem_f32[sgitg]; + + device float * dst_row = (device float *) dst + + args.ne00*i01 + + args.ne00*args.ne01*i02 + + args.ne00*args.ne01*args.ne02*i03; + + if (i00 + tpitg.x < args.ne00) { + dst_row[i00 + tpitg.x] = s; + } + + if (args.outb && tpitg.x == ntg.x - 1) { + device float * tmp_row = (device float *) tmp + + args.net0*i01 + + args.net0*args.net1*i02 + + args.net0*args.net1*args.net2*i03; + + tmp_row[ib] = s; + } +} + +typedef decltype(kernel_cumsum_blk) kernel_cumsum_blk_t; + +template [[host_name("kernel_cumsum_blk_f32")]] kernel kernel_cumsum_blk_t kernel_cumsum_blk; + +template +kernel void kernel_cumsum_add( + constant ggml_metal_kargs_cumsum_add & args, + device const char * tmp, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int ib = tgpig[0]/args.ne01; + + if (ib == 0) { + return; + } + + const int i00 = ib*ntg.x; + const int i01 = tgpig[0]%args.ne01; + const int i02 = tgpig[1]; + const int i03 = tgpig[2]; + + device const float * tmp_row = (device const float *) (tmp + + args.nbt1*i01 + + args.nbt2*i02 + + args.nbt3*i03); + + device float * dst_row = (device float *) dst + + args.ne00*i01 + + args.ne00*args.ne01*i02 + + args.ne00*args.ne01*args.ne02*i03; + + if (i00 + tpitg.x < args.ne00) { + dst_row[i00 + tpitg.x] += tmp_row[ib - 1]; + } +} + +typedef decltype(kernel_cumsum_add) kernel_cumsum_add_t; + +template [[host_name("kernel_cumsum_add_f32")]] kernel kernel_cumsum_add_t kernel_cumsum_add; diff --git a/ggml/src/ggml-metal/kernels/rope.metal b/ggml/src/ggml-metal/kernels/rope.metal new file mode 100644 index 00000000..401ceacb --- /dev/null +++ b/ggml/src/ggml-metal/kernels/rope.metal @@ -0,0 +1,333 @@ +#include "common.h" + +constant bool FC_rope_is_imrope [[function_constant(FC_ROPE + 0)]]; +constant bool FC_rope_is_back [[function_constant(FC_ROPE + 1)]]; + +static float rope_yarn_ramp(const float low, const float high, const int i0) { + const float y = (i0 / 2 - low) / max(0.001f, high - low); + return 1.0f - min(1.0f, max(0.0f, y)); +} + +// YaRN algorithm based on LlamaYaRNScaledRotaryEmbedding.py from https://github.com/jquesnelle/yarn +// MIT licensed. Copyright (c) 2023 Jeffrey Quesnelle and Bowen Peng. +static void rope_yarn( + float theta_extrap, float freq_scale, float corr_dims[2], int i0, float ext_factor, float mscale, + thread float * cos_theta, thread float * sin_theta) { + // Get n-d rotational scaling corrected for extrapolation + float theta_interp = freq_scale * theta_extrap; + float theta = theta_interp; + if (ext_factor != 0.0f) { + float ramp_mix = rope_yarn_ramp(corr_dims[0], corr_dims[1], i0) * ext_factor; + theta = theta_interp * (1 - ramp_mix) + theta_extrap * ramp_mix; + + // Get n-d magnitude scaling corrected for interpolation + mscale *= 1.0f + 0.1f * log(1.0f / freq_scale); + } + *cos_theta = cos(theta) * mscale; + *sin_theta = sin(theta) * mscale; + if (FC_rope_is_back) { + *sin_theta *= -1.0f; + } +} + +// Apparently solving `n_rot = 2pi * x * base^((2 * max_pos_emb) / n_dims)` for x, we get +// `corr_fac(n_rot) = n_dims * log(max_pos_emb / (n_rot * 2pi)) / (2 * log(base))` +static float rope_yarn_corr_factor(int n_dims, int n_ctx_orig, float n_rot, float base) { + return n_dims * log(n_ctx_orig / (n_rot * 2 * M_PI_F)) / (2 * log(base)); +} + +static void rope_yarn_corr_dims( + int n_dims, int n_ctx_orig, float freq_base, float beta_fast, float beta_slow, float dims[2] +) { + // start and end correction dims + dims[0] = max(0.0f, floor(rope_yarn_corr_factor(n_dims, n_ctx_orig, beta_fast, freq_base))); + dims[1] = min(n_dims - 1.0f, ceil(rope_yarn_corr_factor(n_dims, n_ctx_orig, beta_slow, freq_base))); +} + +template +kernel void kernel_rope_norm( + constant ggml_metal_kargs_rope & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 tptg [[threads_per_threadgroup]], + uint3 tgpig[[threadgroup_position_in_grid]]) { + const int i3 = tgpig[2]; + const int i2 = tgpig[1]; + const int i1 = tgpig[0]; + + float corr_dims[2]; + rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); + + device const int32_t * pos = (device const int32_t *) src1; + + const float theta_base = (float) pos[i2]; + const float inv_ndims = -1.f/args.n_dims; + + float cos_theta; + float sin_theta; + + for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { + if (i0 >= args.n_offs && i0 < args.n_offs + args.n_dims) { + const int iw = i0 - args.n_offs; // relative idx + const int ic = iw/2; + + const float theta = theta_base * pow(args.freq_base, inv_ndims*iw); + + const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; + + rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, iw, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + const float x0 = src[0]; + const float x1 = src[1]; + + dst_data[0] = x0*cos_theta - x1*sin_theta; + dst_data[1] = x0*sin_theta + x1*cos_theta; + } else { + if (args.inplace) { + continue; + } + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + dst_data[0] = src[0]; + dst_data[1] = src[1]; + } + } +} + +template +kernel void kernel_rope_neox( + constant ggml_metal_kargs_rope & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 tptg [[threads_per_threadgroup]], + uint3 tgpig[[threadgroup_position_in_grid]]) { + const int i3 = tgpig[2]; + const int i2 = tgpig[1]; + const int i1 = tgpig[0]; + + float corr_dims[2]; + rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); + + device const int32_t * pos = (device const int32_t *) src1; + + const float theta_base = (float) pos[i2]; + const float inv_ndims = -1.f/args.n_dims; + + float cos_theta; + float sin_theta; + + for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { + if (i0 >= args.n_offs && i0 < args.n_offs + args.n_dims) { + const int iw = i0 - args.n_offs; // relative idx + const int ic = iw/2; + + const float theta = theta_base * pow(args.freq_base, inv_ndims*iw); + + const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; + + rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, iw, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + (args.n_offs + ic)*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + (args.n_offs + ic)*args.nb0); + + const float x0 = src[0]; + const float x1 = src[args.n_dims/2]; + + dst_data[0] = x0*cos_theta - x1*sin_theta; + dst_data[args.n_dims/2] = x0*sin_theta + x1*cos_theta; + } else { + if (args.inplace) { + continue; + } + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + dst_data[0] = src[0]; + dst_data[1] = src[1]; + } + } +} + +template +kernel void kernel_rope_multi( + constant ggml_metal_kargs_rope & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 tptg [[threads_per_threadgroup]], + uint3 tgpig[[threadgroup_position_in_grid]]) { + const int i3 = tgpig[2]; + const int i2 = tgpig[1]; + const int i1 = tgpig[0]; + + float corr_dims[2]; + rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); + + device const int32_t * pos = (device const int32_t *) src1; + + const float inv_ndims = -1.f/args.n_dims; + + float cos_theta; + float sin_theta; + + for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { + if (i0 >= args.n_offs && i0 < args.n_offs + args.n_dims) { + const int iw = i0 - args.n_offs; // relative idx + const int ic = iw/2; + + // mrope theta calculations + // note: the rest is the same as kernel_rope_neox + const int sect_dims = args.sect_0 + args.sect_1 + args.sect_2 + args.sect_3; + const int sec_w01 = args.sect_0 + args.sect_1; // end of section 1 + const int sec_w012 = args.sect_0 + args.sect_1 + args.sect_2; // end of section 2 + const int sector = ic % sect_dims; + + float theta_base; + if (FC_rope_is_imrope) { + if (sector % 3 == 1 && sector < 3 * args.sect_1) { // h + theta_base = (float) pos[i2 + args.ne02 * 1]; + } else if (sector % 3 == 2 && sector < 3 * args.sect_2) { // w + theta_base = (float) pos[i2 + args.ne02 * 2]; + } else if (sector % 3 == 0 && sector < 3 * args.sect_0) { // t + theta_base = (float) pos[i2 + args.ne02 * 0]; + } else { // e + theta_base = (float) pos[i2 + args.ne02 * 3]; + } + } else { + if (sector < args.sect_0) { + theta_base = (float) pos[i2]; + } else if (sector < sec_w01) { + theta_base = (float) pos[i2 + args.ne02 * 1]; + } else if (sector < sec_w012) { + theta_base = (float) pos[i2 + args.ne02 * 2]; + } else { + theta_base = (float) pos[i2 + args.ne02 * 3]; + } + } + // end of mrope + + const float theta = theta_base * pow(args.freq_base, inv_ndims*iw); + + const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; + + rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, iw, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + (args.n_offs + ic)*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + (args.n_offs + ic)*args.nb0); + + const float x0 = src[0]; + const float x1 = src[args.n_dims/2]; + + dst_data[0] = x0*cos_theta - x1*sin_theta; + dst_data[args.n_dims/2] = x0*sin_theta + x1*cos_theta; + } else { + if (args.inplace) { + continue; + } + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + dst_data[0] = src[0]; + dst_data[1] = src[1]; + } + } +} + +template +kernel void kernel_rope_vision( + constant ggml_metal_kargs_rope & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + ushort tiitg[[thread_index_in_threadgroup]], + ushort3 tptg [[threads_per_threadgroup]], + uint3 tgpig[[threadgroup_position_in_grid]]) { + const int i3 = tgpig[2]; + const int i2 = tgpig[1]; + const int i1 = tgpig[0]; + + float corr_dims[2]; + rope_yarn_corr_dims(args.n_dims, args.n_ctx_orig, args.freq_base, args.beta_fast, args.beta_slow, corr_dims); + + device const int32_t * pos = (device const int32_t *) src1; + + const float inv_ndims = -1.f/args.n_dims; + + float cos_theta; + float sin_theta; + + for (int i0 = 2*tiitg; i0 < args.ne0; i0 += 2*tptg.x) { + if (i0 < 2*args.n_dims) { // different from kernel_rope_multi + const int ic = i0/2; + + // mrope theta calculations (only support 2 dimensions) + const int sect_dims = args.sect_0 + args.sect_1; + const int sector = ic % sect_dims; + + float p; + float theta_base; + if (sector < args.sect_1) { + p = (float) sector; + theta_base = (float) pos[i2]; + } else { + p = (float) sector - args.sect_0; + theta_base = (float) pos[i2 + args.ne02]; + } + + const float theta = theta_base * pow(args.freq_base, 2.0f * inv_ndims * p); + // end of mrope + + const float freq_factor = args.src2 ? ((device const float *) src2)[ic] : 1.0f; + + rope_yarn(theta/freq_factor, args.freq_scale, corr_dims, i0, args.ext_factor, args.attn_factor, &cos_theta, &sin_theta); + + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + ic*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + ic*args.nb0); + + const float x0 = src[0]; + const float x1 = src[args.n_dims]; // different from kernel_rope_multi + + dst_data[0] = x0*cos_theta - x1*sin_theta; + dst_data[args.n_dims] = x0*sin_theta + x1*cos_theta; // different from kernel_rope_multi + } else { + device const T * const src = (device T *)(src0 + i3*args.nb03 + i2*args.nb02 + i1*args.nb01 + i0*args.nb00); + device T * dst_data = (device T *)( dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + dst_data[0] = src[0]; + dst_data[1] = src[1]; + } + } +} + +typedef decltype(kernel_rope_norm) kernel_rope_norm_t; +typedef decltype(kernel_rope_neox) kernel_rope_neox_t; +typedef decltype(kernel_rope_multi) kernel_rope_multi_t; +typedef decltype(kernel_rope_vision) kernel_rope_vision_t; + +template [[host_name("kernel_rope_norm_f32")]] kernel kernel_rope_norm_t kernel_rope_norm; +template [[host_name("kernel_rope_norm_f16")]] kernel kernel_rope_norm_t kernel_rope_norm; + +template [[host_name("kernel_rope_neox_f32")]] kernel kernel_rope_neox_t kernel_rope_neox; +template [[host_name("kernel_rope_neox_f16")]] kernel kernel_rope_neox_t kernel_rope_neox; + +template [[host_name("kernel_rope_multi_f32")]] kernel kernel_rope_multi_t kernel_rope_multi; +template [[host_name("kernel_rope_multi_f16")]] kernel kernel_rope_multi_t kernel_rope_multi; + +template [[host_name("kernel_rope_vision_f32")]] kernel kernel_rope_vision_t kernel_rope_vision; +template [[host_name("kernel_rope_vision_f16")]] kernel kernel_rope_vision_t kernel_rope_vision; diff --git a/ggml/src/ggml-metal/kernels/softmax.metal b/ggml/src/ggml-metal/kernels/softmax.metal new file mode 100644 index 00000000..f32fe293 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/softmax.metal @@ -0,0 +1,223 @@ +#include "common.h" + +template +kernel void kernel_soft_max( + constant ggml_metal_kargs_soft_max & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + threadgroup float * buf [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint sgitg[[simdgroup_index_in_threadgroup]], + uint tiisg[[thread_index_in_simdgroup]], + uint3 tptg[[threads_per_threadgroup]]) { + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + const int32_t i01 = tgpig.x; + + const int32_t i13 = i03%args.ne13; + const int32_t i12 = i02%args.ne12; + const int32_t i11 = i01; + + device const float * psrc0 = (device const float *) (src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); + device const T * pmask = src1 != src0 ? (device const T * ) (src1 + i11*args.nb11 + i12*args.nb12 + i13*args.nb13) : nullptr; + device const float * psrc2 = src2 != src0 ? (device const float *) (src2) : nullptr; + device float * pdst = (device float *) (dst + i01*args.nb1 + i02*args.nb2 + i03*args.nb3); + + float slope = 1.0f; + + // ALiBi + if (args.max_bias > 0.0f) { + const int32_t h = i02; + + const float base = h < args.n_head_log2 ? args.m0 : args.m1; + const int exp = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; + + slope = pow(base, exp); + } + + // parallel max + float lmax = psrc2 ? psrc2[i02] : -INFINITY; + + for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { + lmax = MAX(lmax, psrc0[i00]*args.scale + (pmask ? slope*pmask[i00] : 0.0f)); + } + + // find the max value in the block + float max_val = simd_max(lmax); + if (tptg.x > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = -INFINITY; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = max_val; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + max_val = buf[tiisg]; + max_val = simd_max(max_val); + } + + // parallel sum + float lsum = 0.0f; + for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { + const float exp_psrc0 = exp((psrc0[i00]*args.scale + (pmask ? slope*pmask[i00] : 0.0f)) - max_val); + lsum += exp_psrc0; + pdst[i00] = exp_psrc0; + } + + // This barrier fixes a failing test + // ref: https://github.com/ggml-org/ggml/pull/621#discussion_r1425156335 + threadgroup_barrier(mem_flags::mem_none); + + float sum = simd_sum(lsum); + + if (tptg.x > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = 0.0f; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = sum; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sum = buf[tiisg]; + sum = simd_sum(sum); + } + + if (psrc2) { + sum += exp(psrc2[i02] - max_val); + } + + const float inv_sum = 1.0f/sum; + + for (int i00 = tpitg.x; i00 < args.ne00; i00 += tptg.x) { + pdst[i00] *= inv_sum; + } +} + +template +kernel void kernel_soft_max_4( + constant ggml_metal_kargs_soft_max & args, + device const char * src0, + device const char * src1, + device const char * src2, + device char * dst, + threadgroup float * buf [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint sgitg[[simdgroup_index_in_threadgroup]], + uint tiisg[[thread_index_in_simdgroup]], + uint3 tptg[[threads_per_threadgroup]]) { + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + const int32_t i01 = tgpig.x; + + const int32_t i13 = i03%args.ne13; + const int32_t i12 = i02%args.ne12; + const int32_t i11 = i01; + + device const float4 * psrc4 = (device const float4 *) (src0 + i01*args.nb01 + i02*args.nb02 + i03*args.nb03); + device const T * pmask = src1 != src0 ? (device const T * ) (src1 + i11*args.nb11 + i12*args.nb12 + i13*args.nb13) : nullptr; + device const float * psrc2 = src2 != src0 ? (device const float * ) (src2) : nullptr; + device float4 * pdst4 = (device float4 *) (dst + i01*args.nb1 + i02*args.nb2 + i03*args.nb3); + + float slope = 1.0f; + + if (args.max_bias > 0.0f) { + const int32_t h = i02; + + const float base = h < args.n_head_log2 ? args.m0 : args.m1; + const int exp = h < args.n_head_log2 ? h + 1 : 2*(h - args.n_head_log2) + 1; + + slope = pow(base, exp); + } + + // parallel max + float4 lmax4 = psrc2 ? psrc2[i02] : -INFINITY; + + for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { + lmax4 = fmax(lmax4, psrc4[i00]*args.scale + (float4)((pmask ? slope*pmask[i00] : 0.0f))); + } + + const float lmax = MAX(MAX(lmax4[0], lmax4[1]), MAX(lmax4[2], lmax4[3])); + + float max_val = simd_max(lmax); + if (tptg.x > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = -INFINITY; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = max_val; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + max_val = buf[tiisg]; + max_val = simd_max(max_val); + } + + // parallel sum + float4 lsum4 = 0.0f; + for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { + const float4 exp_psrc4 = exp((psrc4[i00]*args.scale + (float4)((pmask ? slope*pmask[i00] : 0.0f))) - max_val); + lsum4 += exp_psrc4; + pdst4[i00] = exp_psrc4; + } + + const float lsum = lsum4[0] + lsum4[1] + lsum4[2] + lsum4[3]; + + // This barrier fixes a failing test + // ref: https://github.com/ggml-org/ggml/pull/621#discussion_r1425156335 + threadgroup_barrier(mem_flags::mem_none); + + float sum = simd_sum(lsum); + + if (tptg.x > N_SIMDWIDTH) { + if (sgitg == 0) { + buf[tiisg] = 0.0f; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiisg == 0) { + buf[sgitg] = sum; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + sum = buf[tiisg]; + sum = simd_sum(sum); + } + + if (psrc2) { + sum += exp(psrc2[i02] - max_val); + } + + const float inv_sum = 1.0f/sum; + + for (int i00 = tpitg.x; i00 < args.ne00/4; i00 += tptg.x) { + pdst4[i00] *= inv_sum; + } +} + +typedef decltype(kernel_soft_max) kernel_soft_max_t; +typedef decltype(kernel_soft_max_4) kernel_soft_max_4_t; + +template [[host_name("kernel_soft_max_f16")]] kernel kernel_soft_max_t kernel_soft_max; +template [[host_name("kernel_soft_max_f32")]] kernel kernel_soft_max_t kernel_soft_max; +template [[host_name("kernel_soft_max_f16_4")]] kernel kernel_soft_max_4_t kernel_soft_max_4; +template [[host_name("kernel_soft_max_f32_4")]] kernel kernel_soft_max_4_t kernel_soft_max_4; diff --git a/ggml/src/ggml-metal/kernels/solve_tri.metal b/ggml/src/ggml-metal/kernels/solve_tri.metal new file mode 100644 index 00000000..50f16fac --- /dev/null +++ b/ggml/src/ggml-metal/kernels/solve_tri.metal @@ -0,0 +1,75 @@ +#include "common.h" + +constant short FC_solve_tri_nsg [[function_constant(FC_SOLVE_TRI + 0)]]; +constant short FC_solve_tri_n [[function_constant(FC_SOLVE_TRI + 1)]]; +constant short FC_solve_tri_k [[function_constant(FC_SOLVE_TRI + 2)]]; + +kernel void kernel_solve_tri_f32( + constant ggml_metal_kargs_solve_tri & args, + device const char * src0, + device const char * src1, + device char * dst, + threadgroup char * shmem [[threadgroup(0)]], + ushort3 tgpig[[threadgroup_position_in_grid]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + constexpr short NW = N_SIMDWIDTH; + + const short NSG = FC_solve_tri_nsg; + const short N = FC_solve_tri_n; + const short K = FC_solve_tri_k; + const short NP = PAD2(N, NW); + + const int32_t i03 = tgpig.z; + const int32_t i02 = tgpig.y; + const int32_t i01 = tgpig.x*NSG + sgitg; + + threadgroup float * sh0 = (threadgroup float *) shmem; + + device const float * src0_ptr = (device const float *)(src0 + i02 * args.nb02 + i03 * args.nb03) + sgitg*N; + device const float * src1_ptr = (device const float *)(src1 + i02 * args.nb12 + i03 * args.nb13) + i01; + device float * dst_ptr = (device float *)(dst + i02 * args.nb2 + i03 * args.nb3) + i01; + + for (short rr = 0; rr < N; rr += NSG) { + threadgroup_barrier(mem_flags::mem_threadgroup); + + { + threadgroup float * sh0_cur = sh0 + sgitg*NP; + + for (short t = 0; t*NW < N; ++t) { + const short idx = t*NW + tiisg; + sh0_cur[idx] = src0_ptr[idx]; + } + + src0_ptr += NSG*N; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (i01 >= args.ne10) { + continue; + } + + for (short ir = 0; ir < NSG && rr + ir < N; ++ir) { + const short r = rr + ir; + + threadgroup float * sh0_cur = sh0 + ir*NP; + + float sum = 0.0f; + + for (short t = 0; t*NW < r; ++t) { + const short idx = t*NW + tiisg; + sum += sh0_cur[idx] * dst_ptr[idx*K] * (idx < r); + } + + sum = simd_sum(sum); + + if (tiisg == 0) { + const float diag = sh0_cur[r]; + + dst_ptr[r*K] = (src1_ptr[r*K] - sum) / diag; + } + } + } +} diff --git a/ggml/src/ggml-metal/kernels/ssm.metal b/ggml/src/ggml-metal/kernels/ssm.metal new file mode 100644 index 00000000..b21c53b7 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/ssm.metal @@ -0,0 +1,476 @@ +#include "common.h" + +constant bool FC_ssm_conv_silu [[function_constant(FC_SSM_CONV + 1)]]; +constant int FC_ssm_conv_nc [[function_constant(FC_SSM_CONV + 2)]]; + +// ref: ggml.c:ggml_compute_forward_ssm_conv_f32 +kernel void kernel_ssm_conv_f32_f32( + constant ggml_metal_kargs_ssm_conv & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + const int64_t ir = tgpig.x; + const int64_t i2 = tgpig.y; + const int64_t i3 = tgpig.z; + + const int64_t nc = FC_ssm_conv_nc; + //const int64_t ncs = args.ne00; + //const int64_t nr = args.ne01; + //const int64_t n_t = args.ne1; + //const int64_t n_s = args.ne2; + + device const float * s = (device const float *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); + device const float * c = (device const float *) ((device const char *) src1 + ir*args.nb11); + device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); + + float sumf = 0.0f; + + FOR_UNROLL (int64_t i0 = 0; i0 < nc; ++i0) { + sumf += s[i0] * c[i0]; + } + + x[0] = FC_ssm_conv_silu ? sumf/(1.0f + exp(-sumf)) : sumf; +} + +kernel void kernel_ssm_conv_f32_f32_4( + constant ggml_metal_kargs_ssm_conv & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + const int64_t ir = tgpig.x; + const int64_t i2 = tgpig.y; + const int64_t i3 = tgpig.z; + + const int64_t nc = FC_ssm_conv_nc; + //const int64_t ncs = args.ne00; + //const int64_t nr = args.ne01; + //const int64_t n_t = args.ne1; + //const int64_t n_s = args.ne2; + + device const float4 * s = (device const float4 *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); + device const float4 * c = (device const float4 *) ((device const char *) src1 + ir*args.nb11); + device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); + + float sumf = 0.0f; + + FOR_UNROLL (int64_t i0 = 0; i0 < nc/4; ++i0) { + sumf += dot(s[i0], c[i0]); + } + + x[0] = FC_ssm_conv_silu ? sumf/(1.0f + exp(-sumf)) : sumf; +} + +constant short FC_ssm_conv_bs [[function_constant(FC_SSM_CONV + 0)]]; + +// Batched version: each threadgroup processes multiple tokens for better efficiency +// Thread layout: each thread handles one token, threadgroup covers BATCH_SIZE tokens +kernel void kernel_ssm_conv_f32_f32_batched( + constant ggml_metal_kargs_ssm_conv & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + // tgpig.x = row index (ir) + // tgpig.y = batch of tokens (i2_base / BATCH_SIZE) + // tgpig.z = sequence index (i3) + // tpitg.x = thread within batch (0..BATCH_SIZE-1) + const short BATCH_SIZE = FC_ssm_conv_bs; + + const int64_t ir = tgpig.x; + const int64_t i2_base = tgpig.y * BATCH_SIZE; + const int64_t i3 = tgpig.z; + const int64_t i2_off = tpitg.x; + const int64_t i2 = i2_base + i2_off; + + const int64_t nc = FC_ssm_conv_nc; // conv kernel size (typically 4) + const int64_t n_t = args.ne1; // number of tokens + + // Bounds check for partial batches at the end + if (i2 >= n_t) { + return; + } + + // Load conv weights (shared across all tokens for this row) + device const float * c = (device const float *) ((device const char *) src1 + ir*args.nb11); + + // Load source for this specific token + device const float * s = (device const float *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); + + // Output location for this token + device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); + + float sumf = 0.0f; + FOR_UNROLL (int64_t i0 = 0; i0 < nc; ++i0) { + sumf += s[i0] * c[i0]; + } + + x[0] = FC_ssm_conv_silu ? sumf/(1.0f + exp(-sumf)) : sumf; +} + +kernel void kernel_ssm_conv_f32_f32_batched_4( + constant ggml_metal_kargs_ssm_conv & args, + device const void * src0, + device const void * src1, + device float * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + // tgpig.x = row index (ir) + // tgpig.y = batch of tokens (i2_base / BATCH_SIZE) + // tgpig.z = sequence index (i3) + // tpitg.x = thread within batch (0..BATCH_SIZE-1) + const short BATCH_SIZE = FC_ssm_conv_bs; + + const int64_t ir = tgpig.x; + const int64_t i2_base = tgpig.y * BATCH_SIZE; + const int64_t i3 = tgpig.z; + const int64_t i2_off = tpitg.x; + const int64_t i2 = i2_base + i2_off; + + const int64_t nc = FC_ssm_conv_nc; // conv kernel size (typically 4) + const int64_t n_t = args.ne1; // number of tokens + + // Bounds check for partial batches at the end + if (i2 >= n_t) { + return; + } + + // Load conv weights (shared across all tokens for this row) + device const float4 * c = (device const float4 *) ((device const char *) src1 + ir*args.nb11); + + // Load source for this specific token + device const float4 * s = (device const float4 *) ((device const char *) src0 + ir*args.nb01 + i2*args.nb00 + i3*args.nb02); + + // Output location for this token + device float * x = (device float *) ((device char *) dst + ir*args.nb0 + i2*args.nb1 + i3*args.nb2); + + float sumf = 0.0f; + FOR_UNROLL (int64_t i0 = 0; i0 < nc/4; ++i0) { + sumf += dot(s[i0], c[i0]); + } + + x[0] = FC_ssm_conv_silu ? sumf/(1.0f + exp(-sumf)) : sumf; +} + +// ref: ggml.c:ggml_compute_forward_ssm_scan_f32, Mamba-2 part +// Optimized version: reduces redundant memory loads by having one thread load shared values +// TAIL == false is the whole-sequence / decode path: token_offset folds away at compile time. +template +kernel void kernel_ssm_scan_impl( + constant ggml_metal_kargs_ssm_scan & args, + device const void * src0, + device const void * src1, + device const void * src2, + device const void * src3, + device const void * src4, + device const void * src5, + device const void * src6, + device float * dst, + threadgroup float * shared [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]], + ushort sgptg[[simdgroups_per_threadgroup]], + uint3 tgpg[[threadgroups_per_grid]]) { + constexpr short NW = N_SIMDWIDTH; + + // Shared memory layout: + // [0..sgptg*NW-1]: partial sums for reduction (existing) + // [sgptg*NW..sgptg*NW+sgptg-1]: pre-computed x_dt values for each token in batch + // [sgptg*NW+sgptg..sgptg*NW+2*sgptg-1]: pre-computed dA values for each token in batch + threadgroup float * shared_sums = shared; + threadgroup float * shared_x_dt = shared + sgptg * NW; + threadgroup float * shared_dA = shared + sgptg * NW + sgptg; + + shared_sums[tpitg.x] = 0.0f; + + const int32_t i0 = tpitg.x; + const int32_t i1 = tgpig.x; + const int32_t ir = tgpig.y; // current head + const int32_t i3 = tgpig.z; // current seq + + const int32_t nc = args.d_state; + const int32_t nr = args.d_inner; + const int32_t nh = args.n_head; + const int32_t ng = args.n_group; + const int32_t n_t = args.n_seq_tokens; + const int32_t n_s = args.n_seqs; + const int32_t K = args.K; + const int32_t n_t_total = TAIL ? args.n_seq_tokens_total : n_t; + const int32_t t_off = TAIL ? args.token_offset : 0; + + const int32_t s_off = args.s_off; + + device const int32_t * ids = (device const int32_t *) src6; + + device float * s_buff = (device float *) ((device char *) dst + ir*args.nb02 + i3*args.nb03 + s_off); + device const float * s0_buff = t_off != 0 ? + s_buff : + (device const float *) ((device const char *) src0 + ir*args.nb02 + ids[i3]*args.nb03); + + const int32_t i = i0 + i1*nc; + const int32_t g = ir / (nh / ng); // repeat_interleave + + float s0 = s0_buff[i]; + float s = 0.0f; + + device const float * A = (device const float *) ((device const char *) src3 + ir*args.nb31); // {ne30, nh} + + const float A0 = A[i0%args.ne30]; + + device const float * x = (device const float *)((device const char *) src1 + i1*args.nb10 + ir*args.nb11 + t_off*args.nb12 + i3*args.nb13); // {dim, nh, nt, ns} + device const float * dt = (device const float *)((device const char *) src2 + ir*args.nb20 + t_off*args.nb21 + i3*args.nb22); // {nh, nt, ns} + device const float * B = (device const float *)((device const char *) src4 + g*args.nb41 + t_off*args.nb42 + i3*args.nb43); // {d_state, ng, nt, ns} + device const float * C = (device const float *)((device const char *) src5 + g*args.nb51 + t_off*args.nb52 + i3*args.nb53); // {d_state, ng, nt, ns} + + device float * y = dst + (i1 + ir*nr + t_off*nh*nr + i3*(n_t_total*nh*nr)); // {dim, nh, nt, ns} + + for (int i2 = 0; i2 < n_t; i2 += sgptg) { + threadgroup_barrier(mem_flags::mem_threadgroup); + + // Pre-compute x_dt and dA for this batch of tokens + // Only first sgptg threads do the loads and expensive math + if (i0 < sgptg && i2 + i0 < n_t) { + // ns12 and ns21 are element strides (nb12/nb10, nb21/nb20) + device const float * x_t = x + i0 * args.ns12; + device const float * dt_t = dt + i0 * args.ns21; + + const float dt0 = dt_t[0]; + const float dtsp = dt0 <= 20.0f ? log(1.0f + exp(dt0)) : dt0; + shared_x_dt[i0] = x_t[0] * dtsp; + shared_dA[i0] = dtsp; // Store dtsp, compute exp(dtsp * A0) per-thread since A0 varies + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + + for (int t = 0; t < sgptg && i2 + t < n_t; t++) { + const float x_dt = shared_x_dt[t]; + const float dA = exp(shared_dA[t] * A0); + + s = (s0 * dA) + (B[i0] * x_dt); + + const float sumf = simd_sum(s * C[i0]); + + if (tiisg == 0) { + shared_sums[t*NW + sgitg] = sumf; + } + + // recurse + s0 = s; + + const int32_t slot = n_t - 1 - (i2 + t); + if (slot > 0 && slot < K) { + device float * s_snapshot = (device float *) ((device char *) s_buff + (int64_t) slot*n_s*args.nb03); + s_snapshot[i] = s; + } + + B += args.ns42; + C += args.ns52; + } + + // Advance pointers for next batch + x += sgptg * args.ns12; + dt += sgptg * args.ns21; + + threadgroup_barrier(mem_flags::mem_threadgroup); + + const float sumf = simd_sum(shared_sums[sgitg*NW + tiisg]); + + if (tiisg == 0 && i2 + sgitg < n_t) { + y[sgitg*nh*nr] = sumf; + } + + y += sgptg*nh*nr; + } + + s_buff[i] = s; +} + +typedef decltype(kernel_ssm_scan_impl) kernel_ssm_scan_t; + +template [[host_name("kernel_ssm_scan_f32")]] kernel kernel_ssm_scan_t kernel_ssm_scan_impl; +template [[host_name("kernel_ssm_scan_f32_tail")]] kernel kernel_ssm_scan_t kernel_ssm_scan_impl; + +// Chunked SSD SSM scan via Metal simdgroup MMatrix Multiply-Accumulate (simdgroup_float8x8) fast path. +// One threadgroup per (head, sequence) and tokens are processed in chunks. +// C*B^T computed in each chunk one time and reused across the head_dim channel tiles. +kernel void kernel_ssm_scan_ssd_mma_f32( + constant ggml_metal_kargs_ssm_scan & args, + device const void * src0, + device const void * src1, + device const void * src2, + device const void * src3, + device const void * src4, + device const void * src5, + device const void * src6, + device float * dst, + threadgroup float * shared [[threadgroup(0)]], + uint3 tgpig[[threadgroup_position_in_grid]], + ushort tiitg[[thread_index_in_threadgroup]], + ushort sgitg[[simdgroup_index_in_threadgroup]], + ushort tiisg[[thread_index_in_simdgroup]]) { + constexpr short CS = OP_SSM_SCAN_SSD_CS; + constexpr short TC = 8; // Tile Count of each edge in a simdgroup 8x8 tile + constexpr short HD = OP_SSM_SCAN_SSD_HD; + constexpr short NSG = OP_SSM_SCAN_SSD_NSG; + + // acs/exp(acs)/state-decay vectors, dtX[CS][HD], four private SAM row tiles [8][CS], + // and two 8x8 scratch tiles per simdgroup. Total: 26.75 KiB. + threadgroup float * shared_acs = shared; + threadgroup float * shared_exp_acs = shared + CS; + threadgroup float * shared_state_decay = shared + 2*CS; + threadgroup float * shared_dtx = shared + 3*CS; + threadgroup float * shared_sam = shared + 3*CS + CS*HD; + threadgroup float * sam_rows = shared_sam + sgitg*TC*CS; + threadgroup float * shared_tile = shared_sam + NSG*TC*CS; + threadgroup float * tile0 = shared_tile + sgitg*2*TC*TC; + threadgroup float * tile1 = tile0 + TC*TC; + + const int32_t ir = tgpig.y; // current head + const int32_t i3 = tgpig.z; // current seq + + const int32_t nc = args.d_state; + const int32_t nr = args.d_inner; + const int32_t nh = args.n_head; + const int32_t ng = args.n_group; + const int32_t n_t = args.n_seq_tokens; + const int32_t n_t_total = args.n_seq_tokens_total; + const int32_t g = ir / (nh / ng); + + device const int32_t * ids = (device const int32_t *) src6; + + device const float * s0_buff = (device const float *) ((device const char *) src0 + ir*args.nb02 + ids[i3]*args.nb03); + device float * s_buff = (device float *) ((device char *) dst + ir*args.nb02 + i3*args.nb03 + args.s_off); + + device const float * A = (device const float *) ((device const char *) src3 + ir*args.nb31); + device const float * x = (device const float *) ((device const char *) src1 + ir*args.nb11 + i3*args.nb13); + device const float * dt = (device const float *) ((device const char *) src2 + ir*args.nb20 + i3*args.nb22); + device const float * B = (device const float *) ((device const char *) src4 + g*args.nb41 + i3*args.nb43); + device const float * C = (device const float *) ((device const char *) src5 + g*args.nb51 + i3*args.nb53); + + device float * y = dst + (ir*nr + i3*(n_t_total*nh*nr)); + + for (int32_t t0 = 0; t0 < n_t; t0 += CS) { + for (int32_t idx = tiitg; idx < CS*HD; idx += NSG*N_SIMDWIDTH) { + const int32_t t = idx / HD; + const int32_t c = idx % HD; + const float dt0 = dt[(t0 + t) * (int32_t) args.ns21]; + const float dtsp = dt0 <= 20.0f ? log(1.0f + exp(dt0)) : dt0; + shared_dtx[idx] = x[(t0 + t) * (int32_t) args.ns12 + c] * dtsp; + } + if (tiitg < CS) { + const float dt0 = dt[(t0 + tiitg) * (int32_t) args.ns21]; + const float dtsp = dt0 <= 20.0f ? log(1.0f + exp(dt0)) : dt0; + shared_acs[tiitg] = dtsp * A[0]; + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + if (tiitg == 0) { + float acc = 0.0f; + for (short t = 0; t < CS; ++t) { + acc += shared_acs[t]; + shared_acs[t] = acc; + } + } + threadgroup_barrier(mem_flags::mem_threadgroup); + if (tiitg < CS) { + shared_exp_acs[tiitg] = exp(shared_acs[tiitg]); + shared_state_decay[tiitg] = exp(shared_acs[CS - 1] - shared_acs[tiitg]); + } + threadgroup_barrier(mem_flags::mem_threadgroup); + + device const float * state = t0 == 0 ? s0_buff : s_buff; + + // Build one 8x64 row tile of SAM per simdgroup, then reuse it across every channel tile. + for (short ib = sgitg; ib < CS/TC; ib += NSG) { + for (short jb = 0; jb <= ib; ++jb) { + simdgroup_float8x8 cb = make_filled_simdgroup_matrix(0.0f); + + for (int32_t k0 = 0; k0 < nc; k0 += TC) { + simdgroup_float8x8 mc; + simdgroup_float8x8 mb; + simdgroup_load(mc, C + (t0 + ib*TC)*(int32_t) args.ns52 + k0, args.ns52); + simdgroup_load(mb, B + (t0 + jb*TC)*(int32_t) args.ns42 + k0, args.ns42, 0, true); + simdgroup_multiply_accumulate(cb, mc, mb, cb); + } + + threadgroup float * sam = sam_rows + jb*TC; + simdgroup_store(cb, sam, CS); + simdgroup_barrier(mem_flags::mem_threadgroup); + for (short e = tiisg; e < TC*TC; e += N_SIMDWIDTH) { + const short ri = e / TC; + const short rj = e % TC; + const short i = ib*TC + ri; + const short j = jb*TC + rj; + sam[ri*CS + rj] = j <= i ? + sam[ri*CS + rj] * exp(shared_acs[i] - shared_acs[j]) : 0.0f; + } + simdgroup_barrier(mem_flags::mem_threadgroup); + } + + for (short ch = 0; ch < HD/TC; ++ch) { + simdgroup_float8x8 y_diag = make_filled_simdgroup_matrix(0.0f); + simdgroup_float8x8 y_inter = make_filled_simdgroup_matrix(0.0f); + + for (short jb = 0; jb <= ib; ++jb) { + simdgroup_float8x8 sam; + simdgroup_float8x8 mdtx; + simdgroup_load(sam, sam_rows + jb*TC, CS); + simdgroup_load(mdtx, shared_dtx + jb*TC*HD + ch*TC, HD); + simdgroup_multiply_accumulate(y_diag, sam, mdtx, y_diag); + } + + for (int32_t k0 = 0; k0 < nc; k0 += TC) { + simdgroup_float8x8 mc; + simdgroup_float8x8 ms; + simdgroup_load(mc, C + (t0 + ib*TC)*(int32_t) args.ns52 + k0, args.ns52); + simdgroup_load(ms, state + ch*TC*nc + k0, nc, 0, true); + simdgroup_multiply_accumulate(y_inter, mc, ms, y_inter); + } + + simdgroup_store(y_diag, tile0, TC); + simdgroup_store(y_inter, tile1, TC); + simdgroup_barrier(mem_flags::mem_threadgroup); + for (short e = tiisg; e < TC*TC; e += N_SIMDWIDTH) { + const short ri = e / TC; + const short ci = e % TC; + const int32_t token = t0 + ib*TC + ri; + y[token*nh*nr + ch*TC + ci] = + tile0[e] + shared_exp_acs[ib*TC + ri] * tile1[e]; + } + simdgroup_barrier(mem_flags::mem_threadgroup); + } + } + + // All simdgroups must finish reading s_buff before any thread overwrites it. + threadgroup_barrier(mem_flags::mem_device | mem_flags::mem_threadgroup); + + // Keep the carried-state reduction in token order. Reassociating this particular product + // with MMA compounds rounding differences at every chunk boundary; CB, y_diag, and C*S + // remain on the matrix unit. + const float chunk_decay = exp(shared_acs[CS - 1]); + for (int32_t idx = tiitg; idx < nc*HD; idx += NSG*N_SIMDWIDTH) { + const int32_t ci = idx / nc; + const int32_t si = idx % nc; + float state_c = 0.0f; + for (short t = 0; t < CS; ++t) { + state_c += shared_state_decay[t] * + B[(t0 + t)*(int32_t) args.ns42 + si] * + shared_dtx[t*HD + ci]; + } + s_buff[idx] = chunk_decay * state[idx] + state_c; + } + + // All state tiles must be visible before the next chunk consumes s_buff as S_prev. + threadgroup_barrier(mem_flags::mem_device | mem_flags::mem_threadgroup); + } +} diff --git a/ggml/src/ggml-metal/kernels/tri.metal b/ggml/src/ggml-metal/kernels/tri.metal new file mode 100644 index 00000000..862f7867 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/tri.metal @@ -0,0 +1,69 @@ +#include "common.h" + +template +bool _ggml_vec_tri_cmp(const int i, const int r); + +template<> +bool _ggml_vec_tri_cmp(const int i, const int r) { + return i < r; +} + +template<> +bool _ggml_vec_tri_cmp(const int i, const int r) { + return i <= r; +} + +template<> +bool _ggml_vec_tri_cmp(const int i, const int r) { + return i > r; +} + +template<> +bool _ggml_vec_tri_cmp(const int i, const int r) { + return i >= r; +} + +template +kernel void kernel_tri( + constant ggml_metal_kargs_tri & args, + device const char * src0, + device const char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { + const int i3 = tgpig.z; + const int i2 = tgpig.y; + const int i1 = tgpig.x; + + if (i3 >= args.ne03 || i2 >= args.ne02 || i1 >= args.ne01) { + return; + } + + device const T * src_row = (device const T *) ((device const char *) src0 + i1*args.nb01 + i2*args.nb02 + i3*args.nb03); + device T * dst_row = (device T *) ((device char *) dst + i1*args.nb1 + i2*args.nb2 + i3*args.nb3); + + // Each thread is a single element of the row if ne00 < max threads per + // threadgroup, so this will loop once for each index that this thread is + // responsible for + for (int64_t i0 = tpitg.x; i0 < args.ne00; i0 += ntg.x) { + // Use the comparison as a mask for branchless + dst_row[i0] = static_cast(_ggml_vec_tri_cmp(i0, i1)) * src_row[i0]; + } +} + +typedef decltype(kernel_tri) kernel_tri_t; + +template [[host_name("kernel_tri_f32_0")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f32_1")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f32_2")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f32_3")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f16_0")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f16_1")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f16_2")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_f16_3")]] kernel kernel_tri_t kernel_tri; +#if defined(GGML_METAL_HAS_BF16) +template [[host_name("kernel_tri_bf16_0")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_bf16_1")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_bf16_2")]] kernel kernel_tri_t kernel_tri; +template [[host_name("kernel_tri_bf16_3")]] kernel kernel_tri_t kernel_tri; +#endif diff --git a/ggml/src/ggml-metal/kernels/unary.metal b/ggml/src/ggml-metal/kernels/unary.metal new file mode 100644 index 00000000..e50a6486 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/unary.metal @@ -0,0 +1,400 @@ +#include "common.h" + +constant short FC_unary_op [[function_constant(FC_UNARY + 0)]]; +constant bool FC_unary_cnt[[function_constant(FC_UNARY + 1)]]; + +template +kernel void kernel_unary_impl( + constant ggml_metal_kargs_unary & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + ushort3 tpitg[[thread_position_in_threadgroup]], + ushort3 ntg[[threads_per_threadgroup]]) { +#define FC_OP FC_unary_op +#define FC_CNT FC_unary_cnt + + device const T0 * src0_ptr; + device T * dst_ptr; + + int i0; + + if (FC_CNT) { + i0 = tgpig.x; + + src0_ptr = (device const T0 *) (src0); + dst_ptr = (device T *) (dst); + } else { + const int i03 = tgpig.z; + const int i02 = tgpig.y; + const int k0 = tgpig.x/args.ne01; + const int i01 = tgpig.x - k0*args.ne01; + + i0 = k0*ntg.x + tpitg.x; + + src0_ptr = (device const T0 *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01); + dst_ptr = (device T *) (dst + i03*args.nb3 + i02*args.nb2 + i01*args.nb1 ); + } + + { + //threadgroup_barrier(mem_flags::mem_none); + + if (!FC_CNT) { + if (i0 >= args.ne0) { + return; + } + } + + const TC x = (TC) src0_ptr[i0]; + + if (FC_OP == OP_UNARY_NUM_SCALE) { + dst_ptr[i0] = (T) (args.scale * x + args.bias); + } + + if (FC_OP == OP_UNARY_NUM_FILL) { + dst_ptr[i0] = (T) args.val; + } + + if (FC_OP == OP_UNARY_NUM_CLAMP) { + dst_ptr[i0] = (T) clamp(x, args.min, args.max); + } + + if (FC_OP == OP_UNARY_NUM_SQR) { + dst_ptr[i0] = (T) (x * x); + } + + if (FC_OP == OP_UNARY_NUM_SQRT) { + dst_ptr[i0] = (T) sqrt(x); + } + + if (FC_OP == OP_UNARY_NUM_SIN) { + dst_ptr[i0] = (T) sin(x); + } + + if (FC_OP == OP_UNARY_NUM_COS) { + dst_ptr[i0] = (T) cos(x); + } + + if (FC_OP == OP_UNARY_NUM_LOG) { + dst_ptr[i0] = (T) log(x); + } + + if (FC_OP == OP_UNARY_NUM_LEAKY_RELU) { + dst_ptr[i0] = (T) (TC(x > 0)*x + TC(x <= 0)*(x * args.slope)); + } + + if (FC_OP == OP_UNARY_NUM_TANH) { + dst_ptr[i0] = (T) precise::tanh(x); + } + + if (FC_OP == OP_UNARY_NUM_RELU) { + dst_ptr[i0] = (T) fmax(0, x); + } + + if (FC_OP == OP_UNARY_NUM_SIGMOID) { + dst_ptr[i0] = (T) (1 / (1 + exp(-x))); + } + + if (FC_OP == OP_UNARY_NUM_GELU) { + dst_ptr[i0] = (T) (0.5*x*(1 + precise::tanh(SQRT_2_OVER_PI*x*(1 + GELU_COEF_A*x*x)))); + } + + if (FC_OP == OP_UNARY_NUM_GELU_ERF) { + dst_ptr[i0] = (T) (0.5*x*(1 + erf_approx(SQRT_2_INV*x))); + } + + if (FC_OP == OP_UNARY_NUM_GELU_QUICK) { + dst_ptr[i0] = (T) (x * (1/(1 + exp(GELU_QUICK_COEF*x)))); + } + + if (FC_OP == OP_UNARY_NUM_SILU) { + dst_ptr[i0] = (T) (x / (1 + exp(-x))); + } + + if (FC_OP == OP_UNARY_NUM_ELU) { + dst_ptr[i0] = (T) elu_approx(x); + } + + if (FC_OP == OP_UNARY_NUM_NEG) { + dst_ptr[i0] = (T) -x; + } + + if (FC_OP == OP_UNARY_NUM_ABS) { + dst_ptr[i0] = (T) fabs(x); + } + + if (FC_OP == OP_UNARY_NUM_SGN) { + dst_ptr[i0] = T(x > 0) - T(x < 0); + } + + if (FC_OP == OP_UNARY_NUM_STEP) { + dst_ptr[i0] = T(x > 0); + } + + if (FC_OP == OP_UNARY_NUM_HARDSWISH) { + dst_ptr[i0] = (T) (x * fmax(0, fmin(1, x/6 + 0.5))); + } + + if (FC_OP == OP_UNARY_NUM_HARDSIGMOID) { + dst_ptr[i0] = (T) fmax(0, fmin(1, x/6 + 0.5)); + } + + if (FC_OP == OP_UNARY_NUM_EXP) { + dst_ptr[i0] = (T) exp(x); + } + + if (FC_OP == OP_UNARY_NUM_SOFTPLUS) { + dst_ptr[i0] = (T) select(log(1 + exp(x)), x, x > 20); + } + + if (FC_OP == OP_UNARY_NUM_EXPM1) { + // TODO: precise implementation + dst_ptr[i0] = (T) (exp(x) - 1); + } + + if (FC_OP == OP_UNARY_NUM_FLOOR) { + dst_ptr[i0] = (T) floor(x); + } + + if (FC_OP == OP_UNARY_NUM_CEIL) { + dst_ptr[i0] = (T) ceil(x); + } + + if (FC_OP == OP_UNARY_NUM_ROUND) { + dst_ptr[i0] = (T) round(x); + } + + if (FC_OP == OP_UNARY_NUM_TRUNC) { + dst_ptr[i0] = (T) trunc(x); + } + + if (FC_OP == OP_UNARY_NUM_XIELU) { + const TC xi = x; + const TC gate = TC(xi > TC(0.0f)); + const TC clamped = fmin(xi, TC(args.val)); + const TC y_pos = TC(args.scale) * xi * xi + TC(args.bias) * xi; + const TC y_neg = (exp(clamped) - TC(1.0f) - xi) * TC(args.slope) + TC(args.bias) * xi; + dst_ptr[i0] = (T) (gate * y_pos + (TC(1.0f) - gate) * y_neg); + } + } + +#undef FC_OP +#undef FC_CNT +} + +typedef decltype(kernel_unary_impl) kernel_unary_t; + +template [[host_name("kernel_unary_f32_f32")]] kernel kernel_unary_t kernel_unary_impl; +template [[host_name("kernel_unary_f32_f32_4")]] kernel kernel_unary_t kernel_unary_impl; +template [[host_name("kernel_unary_f16_f16")]] kernel kernel_unary_t kernel_unary_impl; +template [[host_name("kernel_unary_f16_f16_4")]] kernel kernel_unary_t kernel_unary_impl; + +kernel void kernel_silu_back_f32( + constant ggml_metal_kargs_silu_back & args, + device const float * dy, + device const float * x, + device float * dx, + uint gid [[thread_position_in_grid]]) { + if (gid >= args.ne) { + return; + } + + const float s = 1.0f / (1.0f + exp(-x[gid])); + dx[gid] = dy[gid] * s * (1.0f + x[gid] * (1.0f - s)); +} + +template +kernel void kernel_reglu( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float x0 = src0_row[i0]; + const float x1 = src1_row[i0]; + + dst_row[i0] = (T)(x0*x1*(x0 > 0.0f)); + } +} + +typedef decltype(kernel_reglu) kernel_reglu_t; + +template [[host_name("kernel_reglu_f32")]] kernel kernel_reglu_t kernel_reglu; +template [[host_name("kernel_reglu_f16")]] kernel kernel_reglu_t kernel_reglu; + +template +kernel void kernel_geglu( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float x0 = src0_row[i0]; + const float x1 = src1_row[i0]; + + const float gelu = 0.5f*x0*(1.0f + precise::tanh(SQRT_2_OVER_PI*x0*(1.0f + GELU_COEF_A*x0*x0))); + + dst_row[i0] = (T)(gelu*x1); + } +} + +typedef decltype(kernel_geglu) kernel_geglu_t; + +template [[host_name("kernel_geglu_f32")]] kernel kernel_geglu_t kernel_geglu; +template [[host_name("kernel_geglu_f16")]] kernel kernel_geglu_t kernel_geglu; + +template +kernel void kernel_swiglu( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float x0 = src0_row[i0]; + const float x1 = src1_row[i0]; + + const float silu = x0 / (1.0f + exp(-x0)); + + dst_row[i0] = (T)(silu*x1); + } +} + +typedef decltype(kernel_swiglu) kernel_swiglu_t; + +template [[host_name("kernel_swiglu_f32")]] kernel kernel_swiglu_t kernel_swiglu; +template [[host_name("kernel_swiglu_f16")]] kernel kernel_swiglu_t kernel_swiglu; + +template +kernel void kernel_swiglu_oai( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + float x0 = src0_row[i0]; + float x1 = src1_row[i0]; + + x0 = min(x0, args.limit); + x1 = max(min(x1, args.limit), -args.limit); + + float out_glu = x0 / (1.0f + exp(-x0 * args.alpha)); + out_glu = out_glu * (1.0f + x1); + + dst_row[i0] = (T)out_glu; + } +} + +typedef decltype(kernel_swiglu_oai) kernel_swiglu_oai_t; + +template [[host_name("kernel_swiglu_oai_f32")]] kernel kernel_swiglu_oai_t kernel_swiglu_oai; +template [[host_name("kernel_swiglu_oai_f16")]] kernel kernel_swiglu_oai_t kernel_swiglu_oai; + +template +kernel void kernel_swiglu_clamp( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float gate = min((float) src0_row[i0], args.limit); + const float up = clamp((float) src1_row[i0], -args.limit, args.limit); + + dst_row[i0] = (T)(gate / (1.0f + exp(-gate)) * up); + } +} + +typedef decltype(kernel_swiglu_clamp) kernel_swiglu_clamp_t; + +template [[host_name("kernel_swiglu_clamp_f32")]] kernel kernel_swiglu_clamp_t kernel_swiglu_clamp; +template [[host_name("kernel_swiglu_clamp_f16")]] kernel kernel_swiglu_clamp_t kernel_swiglu_clamp; + +template +kernel void kernel_geglu_erf( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float x0 = src0_row[i0]; + const float x1 = src1_row[i0]; + + const float gelu_erf = 0.5f*x0*(1.0f+erf_approx(x0*SQRT_2_INV)); + + dst_row[i0] = (T)(gelu_erf*x1); + } +} + +typedef decltype(kernel_geglu_erf) kernel_geglu_erf_t; + +template [[host_name("kernel_geglu_erf_f32")]] kernel kernel_geglu_erf_t kernel_geglu_erf; +template [[host_name("kernel_geglu_erf_f16")]] kernel kernel_geglu_erf_t kernel_geglu_erf; + +template +kernel void kernel_geglu_quick( + constant ggml_metal_kargs_glu & args, + device const char * src0, + device const char * src1, + device char * dst, + uint tgpig[[threadgroup_position_in_grid]], + uint tpitg[[thread_position_in_threadgroup]], + uint ntg[[threads_per_threadgroup]]) { + device const T * src0_row = (device const T *) ((device const char *) src0 + tgpig*args.nb01) + args.i00; + device const T * src1_row = (device const T *) ((device const char *) src1 + tgpig*args.nb11) + args.i10; + device T * dst_row = (device T *) ((device char *) dst + tgpig*args.nb1); + + for (int i0 = tpitg; i0 < args.ne0; i0 += ntg) { + const float x0 = src0_row[i0]; + const float x1 = src1_row[i0]; + + const float gelu_quick = x0*(1.0f/(1.0f+exp(GELU_QUICK_COEF*x0))); + + dst_row[i0] = (T)(gelu_quick*x1); + } +} + +typedef decltype(kernel_geglu_quick) kernel_geglu_quick_t; + +template [[host_name("kernel_geglu_quick_f32")]] kernel kernel_geglu_quick_t kernel_geglu_quick; +template [[host_name("kernel_geglu_quick_f16")]] kernel kernel_geglu_quick_t kernel_geglu_quick; diff --git a/ggml/src/ggml-metal/kernels/upscale.metal b/ggml/src/ggml-metal/kernels/upscale.metal new file mode 100644 index 00000000..8bac1308 --- /dev/null +++ b/ggml/src/ggml-metal/kernels/upscale.metal @@ -0,0 +1,179 @@ +#include "common.h" + +constant bool FC_upscale_aa [[function_constant(FC_UPSCALE + 0)]]; + +kernel void kernel_upscale_nearest_f32( + constant ggml_metal_kargs_upscale & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t i3 = tgpig.z; + const int64_t i2 = tgpig.y; + const int64_t i1 = tgpig.x; + + const int64_t i03 = i3/args.sf3; + const int64_t i02 = i2/args.sf2; + const int64_t i01 = i1/args.sf1; + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const int64_t i00 = i0/args.sf0; + + device const float * src0_ptr = (device const float *) (src0 + i03*args.nb03 + i02*args.nb02 + i01*args.nb01 + i00*args.nb00); + device float * dst_ptr = (device float *) (dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1 + i0*args.nb0); + + dst_ptr[0] = src0_ptr[0]; + } +} + +static inline float bilinear_tri(float x) { + return MAX(0.0f, 1.0f - fabs(x)); +} + +kernel void kernel_upscale_bilinear_f32( + constant ggml_metal_kargs_upscale & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t i3 = tgpig.z; + const int64_t i2 = tgpig.y; + const int64_t i1 = tgpig.x; + + const int64_t i03 = i3 / args.sf3; + const int64_t i02 = i2 / args.sf2; + + const float f01 = ((float)i1 + args.poffs) / args.sf1 - args.poffs; + const int64_t i01 = MAX(0, MIN(args.ne01 - 1, (int64_t)floor(f01))); + const int64_t i01p = MAX(0, MIN(args.ne01 - 1, i01 + 1)); + const float fd1 = MAX(0.0f, MIN(1.0f, f01 - (float)i01)); + + src0 += i03*args.nb03 + i02*args.nb02; + + device float * dst_ptr = (device float *)(dst + i3*args.nb3 + i2*args.nb2 + i1*args.nb1); + + if (FC_upscale_aa) { + const float support0 = MAX(1.0f, 1.0f / args.sf0); + const float invscale0 = 1.0f / support0; + const float support1 = MAX(1.0f, 1.0f / args.sf1); + const float invscale1 = 1.0f / support1; + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; + + int64_t x_min = MAX((int64_t)0, (int64_t)floor(f00 - support0 + args.poffs)); + int64_t x_max = MIN(args.ne00, (int64_t)ceil (f00 + support0 + args.poffs)); + + int64_t y_min = MAX((int64_t)0, (int64_t)floor(f01 - support1 + args.poffs)); + int64_t y_max = MIN(args.ne01, (int64_t)ceil (f01 + support1 + args.poffs)); + + float sum = 0.0f; + float wsum = 0.0f; + + for (int64_t sy = y_min; sy < y_max; ++sy) { + const float wy = MAX(0.0f, 1.0f - fabs((float)sy - f01) * invscale1); + for (int64_t sx = x_min; sx < x_max; ++sx) { + const float wx = MAX(0.0f, 1.0f - fabs((float)sx - f00) * invscale0); + const float w = wx * wy; + device const float * src_ptr = (device const float *)(src0 + sy*args.nb01 + sx*args.nb00); + sum += (*src_ptr) * w; + wsum += w; + } + } + + const float v = (wsum > 0.0f) ? (sum / wsum) : 0.0f; + dst_ptr[i0] = v; + } + } else { + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; + const int64_t i00 = MAX(0, MIN(args.ne00 - 1, (int64_t)floor(f00))); + const int64_t i00p = MAX(0, MIN(args.ne00 - 1, i00 + 1)); + const float fd0 = MAX(0.0f, MIN(1.0f, f00 - (float)i00)); + + device const float * src00 = (device const float *)(src0 + i01*args.nb01 + i00*args.nb00); + device const float * src10 = (device const float *)(src0 + i01*args.nb01 + i00p*args.nb00); + device const float * src01 = (device const float *)(src0 + i01p*args.nb01 + i00*args.nb00); + device const float * src11 = (device const float *)(src0 + i01p*args.nb01 + i00p*args.nb00); + + const float v = + (*src00) * (1.0f - fd0) * (1.0f - fd1) + + (*src10) * fd0 * (1.0f - fd1) + + (*src01) * (1.0f - fd0) * fd1 + + (*src11) * fd0 * fd1; + + dst_ptr[i0] = v; + } + } +} + +static inline float bicubic_weight1(float x) { + const float a = -0.75f; + return ((a + 2) * x - (a + 3)) * x * x + 1; +} + +static inline float bicubic_weight2(float x) { + const float a = -0.75f; + return ((a * x - 5 * a) * x + 8 * a) * x - 4 * a; +} + +kernel void kernel_upscale_bicubic_f32( + constant ggml_metal_kargs_upscale & args, + device const char * src0, + device char * dst, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const int64_t i3 = tgpig.z; + const int64_t i2 = tgpig.y; + const int64_t i1 = tgpig.x; + + const int64_t i03 = i3 / args.sf3; + const int64_t i02 = i2 / args.sf2; + + const float f01 = ((float)i1 + args.poffs) / args.sf1 - args.poffs; + const int64_t i01 = (int64_t)floor(f01); + const float fd1 = f01 - (float)i01; + + const float w_y0 = bicubic_weight2(fd1 + 1.0f); + const float w_y1 = bicubic_weight1(fd1); + const float w_y2 = bicubic_weight1(1.0f - fd1); + const float w_y3 = bicubic_weight2(2.0f - fd1); + + const device char * src_slice = src0 + i03 * args.nb03 + i02 * args.nb02; + + device float * dst_ptr = (device float *)(dst + i3 * args.nb3 + i2 * args.nb2 + i1 * args.nb1); + + for (int i0 = tpitg.x; i0 < args.ne0; i0 += ntg.x) { + const float f00 = ((float)i0 + args.poffs) / args.sf0 - args.poffs; + const int64_t i00 = (int64_t)floor(f00); + const float fd0 = f00 - (float)i00; + + const float w_x0 = bicubic_weight2(fd0 + 1.0f); + const float w_x1 = bicubic_weight1(fd0); + const float w_x2 = bicubic_weight1(1.0f - fd0); + const float w_x3 = bicubic_weight2(2.0f - fd0); + + float sum = 0.0f; + + for (int dy = -1; dy <= 2; ++dy) { + const int64_t iy = MAX(0, MIN(args.ne01 - 1, i01 + dy)); + const float wy = (dy == -1) ? w_y0 : (dy == 0) ? w_y1 : (dy == 1) ? w_y2 : w_y3; + + for (int dx = -1; dx <= 2; ++dx) { + const int64_t ix = MAX(0, MIN(args.ne00 - 1, i00 + dx)); + const float wx = (dx == -1) ? w_x0 : (dx == 0) ? w_x1 : (dx == 1) ? w_x2 : w_x3; + + device const float * src_ptr = (device const float *)(src_slice + iy * args.nb01 + ix * args.nb00); + sum += (*src_ptr) * wx * wy; + } + } + + dst_ptr[i0] = sum; + } +} diff --git a/ggml/src/ggml-metal/kernels/wkv.metal b/ggml/src/ggml-metal/kernels/wkv.metal new file mode 100644 index 00000000..8767581c --- /dev/null +++ b/ggml/src/ggml-metal/kernels/wkv.metal @@ -0,0 +1,179 @@ +#include "common.h" + +kernel void kernel_rwkv_wkv6_f32( + device const float * k, + device const float * v, + device const float * r, + device const float * tf, + device const float * td, + device const float * state_in, + device float * dst, + constant uint & B, + constant uint & T, + constant uint & C, + constant uint & H, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const uint head_size = 64; // TODO: support head_size = 128 + const uint batch_id = tgpig.x / H; + const uint head_id = tgpig.x % H; + const uint tid = tpitg.x; + + if (batch_id >= B || head_id >= H) { + return; + } + + const uint state_size = C * head_size; + const uint n_seq_tokens = T / B; + + threadgroup float _k[head_size]; + threadgroup float _r[head_size]; + threadgroup float _tf[head_size]; + threadgroup float _td[head_size]; + + float state[head_size]; + + for (uint i = 0; i < head_size; i++) { + state[i] = state_in[batch_id * state_size + head_id * head_size * head_size + + i * head_size + tid]; + } + + threadgroup_barrier(mem_flags::mem_threadgroup); + _tf[tid] = tf[head_id * head_size + tid]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + const uint start_t = batch_id * n_seq_tokens * C + head_id * head_size + tid; + const uint end_t = (batch_id + 1) * n_seq_tokens * C + head_id * head_size + tid; + + for (uint t = start_t; t < end_t; t += C) { + threadgroup_barrier(mem_flags::mem_threadgroup); + _k[tid] = k[t]; + _r[tid] = r[t]; + _td[tid] = td[t]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + const float v_val = v[t]; + float y = 0.0; + + for (uint j = 0; j < head_size; j += 4) { + float4 k_vec = float4(_k[j], _k[j+1], _k[j+2], _k[j+3]); + float4 r_vec = float4(_r[j], _r[j+1], _r[j+2], _r[j+3]); + float4 tf_vec = float4(_tf[j], _tf[j+1], _tf[j+2], _tf[j+3]); + float4 td_vec = float4(_td[j], _td[j+1], _td[j+2], _td[j+3]); + float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); + + float4 kv = k_vec * v_val; + + float4 temp = tf_vec * kv + s_vec; + y += dot(r_vec, temp); + + s_vec = s_vec * td_vec + kv; + state[j] = s_vec[0]; + state[j+1] = s_vec[1]; + state[j+2] = s_vec[2]; + state[j+3] = s_vec[3]; + } + + dst[t] = y; + } + + for (uint i = 0; i < head_size; i++) { + dst[T * C + batch_id * state_size + head_id * head_size * head_size + + i * head_size + tid] = state[i]; + } +} + +kernel void kernel_rwkv_wkv7_f32( + device const float * r, + device const float * w, + device const float * k, + device const float * v, + device const float * a, + device const float * b, + device const float * state_in, + device float * dst, + constant uint & B, + constant uint & T, + constant uint & C, + constant uint & H, + uint3 tgpig[[threadgroup_position_in_grid]], + uint3 tpitg[[thread_position_in_threadgroup]], + uint3 ntg[[threads_per_threadgroup]]) { + + const uint head_size = 64; // TODO: support head_size = 128 + const uint batch_id = tgpig.x / H; + const uint head_id = tgpig.x % H; + const uint tid = tpitg.x; + + if (batch_id >= B || head_id >= H) { + return; + } + + const uint state_size = C * head_size; + const uint n_seq_tokens = T / B; + + threadgroup float _r[head_size]; + threadgroup float _w[head_size]; + threadgroup float _k[head_size]; + threadgroup float _a[head_size]; + threadgroup float _b[head_size]; + + float state[head_size]; + + for (uint i = 0; i < head_size; i++) { + state[i] = state_in[batch_id * state_size + head_id * head_size * head_size + + tid * head_size + i]; + } + + const uint start_t = batch_id * n_seq_tokens * C + head_id * head_size + tid; + const uint end_t = (batch_id + 1) * n_seq_tokens * C + head_id * head_size + tid; + + for (uint t = start_t; t < end_t; t += C) { + threadgroup_barrier(mem_flags::mem_threadgroup); + _r[tid] = r[t]; + _w[tid] = w[t]; + _k[tid] = k[t]; + _a[tid] = a[t]; + _b[tid] = b[t]; + threadgroup_barrier(mem_flags::mem_threadgroup); + + const float v_val = v[t]; + float y = 0.0, sa = 0.0; + + float4 sa_vec(0.0); + + for (uint j = 0; j < head_size; j += 4) { + float4 a_vec = float4(_a[j], _a[j+1], _a[j+2], _a[j+3]); + float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); + sa_vec += a_vec * s_vec; + } + sa = sa_vec[0] + sa_vec[1] + sa_vec[2] + sa_vec[3]; + + for (uint j = 0; j < head_size; j += 4) { + float4 r_vec = float4(_r[j], _r[j+1], _r[j+2], _r[j+3]); + float4 w_vec = float4(_w[j], _w[j+1], _w[j+2], _w[j+3]); + float4 k_vec = float4(_k[j], _k[j+1], _k[j+2], _k[j+3]); + float4 b_vec = float4(_b[j], _b[j+1], _b[j+2], _b[j+3]); + float4 s_vec = float4(state[j], state[j+1], state[j+2], state[j+3]); + + float4 kv = k_vec * v_val; + + s_vec = s_vec * w_vec + kv + sa * b_vec; + y += dot(s_vec, r_vec); + + state[j] = s_vec[0]; + state[j+1] = s_vec[1]; + state[j+2] = s_vec[2]; + state[j+3] = s_vec[3]; + } + + dst[t] = y; + } + + for (uint i = 0; i < head_size; i++) { + dst[T * C + batch_id * state_size + head_id * head_size * head_size + + tid * head_size + i] = state[i]; + } +} diff --git a/ggml/src/ggml-musa/CMakeLists.txt b/ggml/src/ggml-musa/CMakeLists.txt index cc53c812..82b754f4 100644 --- a/ggml/src/ggml-musa/CMakeLists.txt +++ b/ggml/src/ggml-musa/CMakeLists.txt @@ -43,17 +43,8 @@ if (MUSAToolkit_FOUND) add_compile_definitions(GGML_MUSA_MUDNN_COPY) endif() - if (GGML_CUDA_FA_ALL_QUANTS) - file(GLOB SRCS "../ggml-cuda/template-instances/fattn-vec*.cu") - list(APPEND GGML_SOURCES_MUSA ${SRCS}) - add_compile_definitions(GGML_CUDA_FA_ALL_QUANTS) - else() - list(APPEND GGML_SOURCES_MUSA - ../ggml-cuda/template-instances/fattn-vec-instance-f16-f16.cu - ../ggml-cuda/template-instances/fattn-vec-instance-q4_0-q4_0.cu - ../ggml-cuda/template-instances/fattn-vec-instance-q8_0-q8_0.cu - ../ggml-cuda/template-instances/fattn-vec-instance-bf16-bf16.cu) - endif() + ggml_cuda_fattn_vec_instances(${CMAKE_CURRENT_SOURCE_DIR}/../ggml-cuda SRCS) + list(APPEND GGML_SOURCES_MUSA ${SRCS}) set_source_files_properties(${GGML_SOURCES_MUSA} PROPERTIES LANGUAGE CXX) foreach(SOURCE ${GGML_SOURCES_MUSA}) @@ -75,7 +66,6 @@ if (MUSAToolkit_FOUND) endif() add_compile_definitions(GGML_USE_MUSA) - add_compile_definitions(GGML_CUDA_PEER_MAX_BATCH_SIZE=${GGML_CUDA_PEER_MAX_BATCH_SIZE}) if (GGML_MUSA_GRAPHS) add_compile_definitions(GGML_MUSA_GRAPHS) diff --git a/ggml/src/ggml-opencl/CMakeLists.txt b/ggml/src/ggml-opencl/CMakeLists.txt index 1dc70717..ff5e8ef4 100644 --- a/ggml/src/ggml-opencl/CMakeLists.txt +++ b/ggml/src/ggml-opencl/CMakeLists.txt @@ -63,6 +63,7 @@ endfunction() set(GGML_OPENCL_KERNELS add add_id + moe_add_id_glu argsort tri fill @@ -84,6 +85,7 @@ set(GGML_OPENCL_KERNELS mul_mv_f16_f32_1row mul_mv_f16_f32_l4 mul_mv_f16_f32 + mul_mv_f16_f32_mrow mul_mv_f32_f32 mul_mv_q1_0_f32 mul_mv_q1_0_f32_flat @@ -168,6 +170,7 @@ set(GGML_OPENCL_KERNELS gemv_noshuffle_q4_0_f32 gemv_noshuffle_q4_0_f32_spec gemm_noshuffle_q4_0_f32 + gemv_noshuffle_q4_0_f32_32b_trans gemv_noshuffle_q4_1_f32 gemm_noshuffle_q4_1_f32 gemv_noshuffle_q5_0_f32 @@ -179,9 +182,16 @@ set(GGML_OPENCL_KERNELS gemv_noshuffle_q8_0_f32 gemm_noshuffle_q8_0_f32 gemv_noshuffle_q4_k_f32 + gemv_noshuffle_q4_k_f32_o4 + gemv_noshuffle_q4_k_f32_tiled gemm_noshuffle_q4_k_f32 + gemv_noshuffle_q4_k_f32_32b_trans gemv_noshuffle_q6_k_f32 + gemv_noshuffle_q6_k_f32_o4 + gemv_noshuffle_q6_k_f32_tiled gemm_noshuffle_q6_k_f32 + gemm_noshuffle_q6_k_f32_tiled + gemv_noshuffle_q6_k_f32_32b_trans gemv_noshuffle_q5_k_f32 gemm_noshuffle_q5_k_f32 mul @@ -202,6 +212,7 @@ set(GGML_OPENCL_KERNELS sqr sqrt ssm_conv + ssm_scan gated_delta_net sub sum_rows @@ -214,6 +225,7 @@ set(GGML_OPENCL_KERNELS exp expm1 abs + unary_ext softplus pad repeat @@ -221,6 +233,7 @@ set(GGML_OPENCL_KERNELS mul_mm_f16_f32_kq_kqv conv2d conv2d_f16_f32 + flash_attn_repack flash_attn_pre_f16 flash_attn_f32_f16 flash_attn_f32_q8_0 @@ -230,7 +243,7 @@ set(GGML_OPENCL_KERNELS ) if (GGML_OPENCL_USE_ADRENO_KERNELS) - list(APPEND GGML_OPENCL_KERNELS gemm_xmem_f16_f32_os8) + list(APPEND GGML_OPENCL_KERNELS gemm_xmem_f16_f32_os8 sdpa_xmem_f32_f16_os8) endif () foreach (K ${GGML_OPENCL_KERNELS}) diff --git a/ggml/src/ggml-opencl/ggml-opencl.cpp b/ggml/src/ggml-opencl/ggml-opencl.cpp index 25790860..91cd1a0e 100644 --- a/ggml/src/ggml-opencl/ggml-opencl.cpp +++ b/ggml/src/ggml-opencl/ggml-opencl.cpp @@ -203,39 +203,67 @@ static ggml_cl_version get_opencl_platform_version(cl_platform_id platform) { return parse_cl_version(param_value); } +// Returns the DEVICE's OpenCL version. On an error returns ggml_cl_version with all zeroes. +static ggml_cl_version get_opencl_device_version(cl_device_id device) { + size_t param_size; + if (clGetDeviceInfo(device, CL_DEVICE_VERSION, 0, nullptr, ¶m_size) != CL_SUCCESS || !param_size) { + return {}; + } + std::unique_ptr param_storage(new char[param_size]); + if (clGetDeviceInfo(device, CL_DEVICE_VERSION, param_size, param_storage.get(), nullptr) != CL_SUCCESS) { + return {}; + } + + auto param_value = std::string_view(param_storage.get(), param_size); + const std::string version_prefix = "OpenCL "; // "OpenCL . " + if (param_value.find(version_prefix) != 0) { + return {}; + } + param_value.remove_prefix(version_prefix.length()); + return parse_cl_version(param_value); +} + // Return a version to use in OpenCL C compilation. On an error returns ggml_cl_version with all zeroes. static ggml_cl_version get_opencl_c_version(ggml_cl_version platform_version, cl_device_id device) { size_t param_size; #if CL_TARGET_OPENCL_VERSION >= 300 - if (platform_version.major >= 3) { - CL_CHECK(clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_ALL_VERSIONS, 0, nullptr, ¶m_size)); - if (!param_size) { - return {}; - } + // CL_DEVICE_OPENCL_C_ALL_VERSIONS is an OpenCL 3.0 *device* query, so gating it on the + // *platform* version is not enough: a 3.0 platform can expose 2.0 devices, where the + // query returns CL_INVALID_VALUE and the old CL_CHECK aborted during backend init. + // Gate on the device version, and treat a failure as "fall back to the legacy query" + // rather than fatal -- a device may advertise 3.0 and still refuse the property. + const ggml_cl_version device_version = get_opencl_device_version(device); + if (platform_version.major >= 3 && device_version.major >= 3) { + cl_int err = clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_ALL_VERSIONS, 0, nullptr, ¶m_size); + if (err == CL_SUCCESS && param_size) { + std::unique_ptr versions(new cl_name_version[param_size]); + err = clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_ALL_VERSIONS, param_size, versions.get(), nullptr); + if (err == CL_SUCCESS) { + unsigned versions_count = param_size / sizeof(cl_name_version); - std::unique_ptr versions(new cl_name_version[param_size]); - CL_CHECK(clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_ALL_VERSIONS, param_size, versions.get(), nullptr)); - unsigned versions_count = param_size / sizeof(cl_name_version); + cl_version version_max = 0; + for (unsigned i = 0; i < versions_count; i++) { + version_max = std::max(versions[i].version, version_max); + } - cl_version version_max = 0; - for (unsigned i = 0; i < versions_count; i++) { - version_max = std::max(versions[i].version, version_max); + return { CL_VERSION_MAJOR(version_max), CL_VERSION_MINOR(version_max) }; + } } - - return { CL_VERSION_MAJOR(version_max), CL_VERSION_MINOR(version_max) }; + // fall through to CL_DEVICE_OPENCL_C_VERSION below } #else GGML_UNUSED(platform_version); #endif // CL_TARGET_OPENCL_VERSION >= 300 - CL_CHECK(clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_VERSION, 0, nullptr, ¶m_size)); - if (!param_size) { + if (clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_VERSION, 0, nullptr, ¶m_size) != CL_SUCCESS || !param_size) { return {}; } std::unique_ptr param_storage(new char[param_size]); - CL_CHECK(clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_VERSION, param_size, param_storage.get(), nullptr)); + if (clGetDeviceInfo(device, CL_DEVICE_OPENCL_C_VERSION, param_size, param_storage.get(), nullptr) != CL_SUCCESS) { + return {}; + } auto param_value = std::string_view(param_storage.get(), param_size); const std::string version_prefix = "OpenCL C "; // Suffix: "XX.YY " @@ -417,6 +445,10 @@ static void populateProfilingInfo( struct ggml_backend_opencl_context; +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static void ggml_cl_adreno_xmem_attn_release_scratch(ggml_backend_opencl_context * backend_ctx); +#endif + // backend device context struct ggml_backend_opencl_device_context { cl_platform_id platform; @@ -535,7 +567,65 @@ struct ggml_opencl_fa_kernels { // attempted (variant, (dk, dv)) // all attempted FA kernels appear here, but those not registered failed compilation std::set>> variant_attempted; + + // FA bin kernels +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + cl_kernel kernel_flash_attn_f32_f16_bin; + + cl_kernel kernel_repack_q_for_wmm; + cl_kernel kernel_repack_k_for_wmm; + cl_kernel kernel_repack_v_for_wmm; + cl_kernel kernel_repack_mask_for_wmm; +#endif +}; + +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +struct ggml_cl_adreno_xmem_attn_scratch { + cl_mem q_img = nullptr; + cl_mem k_img = nullptr; + cl_mem v_img = nullptr; + cl_mem out_img = nullptr; + cl_mem k_transpose_buf = nullptr; + cl_mem k_transpose_img1d = nullptr; + cl_mem k_packed_buf = nullptr; + cl_mem v_packed_buf = nullptr; + cl_mem score_buf = nullptr; + cl_mem prob_buf = nullptr; + cl_mem score_img1d = nullptr; + cl_mem prob_img1d = nullptr; + cl_mem softmax_stats_img2d = nullptr; + cl_mem xmem_qk = nullptr; + cl_mem xmem_pv = nullptr; + + int n_q = 0; + int n_kv = 0; + int n_kv_padded = 0; + int d_head_q = 0; + int d_head_v = 0; + int q_width = 0; + int kv_heads_total = 0; +}; + +struct ggml_cl_adreno_xmem_attn_state { + bool compiled = false; + bool logged = false; + + cl_kernel kernel_q_f32_to_img_scaled = nullptr; + cl_kernel kernel_kv_f32_to_img_gqa = nullptr; + cl_kernel kernel_kv_f16_to_img_gqa = nullptr; + cl_kernel kernel_img_to_f32 = nullptr; + cl_kernel kernel_k_gather = nullptr; + cl_kernel kernel_pack_k = nullptr; + cl_kernel kernel_qk_gemm = nullptr; + cl_kernel kernel_softmax_reduce_basic = nullptr; + cl_kernel kernel_softmax_apply_basic = nullptr; + cl_kernel kernel_mask_scores = nullptr; + cl_kernel kernel_pack_v = nullptr; + cl_kernel kernel_pv_gemm = nullptr; + + ggml_cl_adreno_xmem_attn_scratch scratch; }; +#endif // backend context struct ggml_backend_opencl_context { @@ -568,6 +658,10 @@ struct ggml_backend_opencl_context { bool has_integer_dot = false; // cl_khr_integer_dot_product or cl_qcom_dot_product8 bool has_qcom_subgroup_shuffle = false; // specifically cl_qcom_subgroup_shuffle bool disable_fusion; + bool fuse_mm_glu = true; // opt-out GGML_OPENCL_FUSE_MM_GLU=0 (byte-identical gate+up GEMV + GLU, q4_K FFN) + bool fuse_rms_add = true; // opt-out GGML_OPENCL_FUSE_RMS_ADD=0 (fused rms_norm*w + residual) + bool f16_mrow = true; // opt-out GGML_OPENCL_F16_MROW=0 (multi-row-per-WG f16 decode GEMV for attn proj + lm_head) + int f16_mrow_rpt = 1; // GGML_OPENCL_F16_MROW_RPT={1,2,4,8,16} rows-per-subgroup register blocking // ragged moe, use int to directly pass to kernel cl_uint adreno_use_moe_ragged; @@ -577,11 +671,19 @@ struct ggml_backend_opencl_context { // whether fuse moe combine cl_uint fuse_moe_combine; + // whether to fold the MoE bias adds into swiglu_oai + cl_uint fuse_moe_bias_glu; + + // whether to fold the MoE down-projection bias add into the combine + cl_uint fuse_moe_bias_combine; + bool adreno_has_large_buffer; bool adreno_use_large_buffer; bool adreno_use_bin_kernels; get_adreno_bin_kernel_func_t get_adreno_bin_kernel_func = nullptr; ggml_cl_compiler_version adreno_cl_compiler_version; + // The q6_K flat mul_mat codegen workarounds are needed by old E031 compilers only. + bool q6_k_flat_old_compiler; std::string kernel_compile_opts; // cached for lazy-compiled kernels. @@ -611,6 +713,7 @@ struct ggml_backend_opencl_context { ggml_cl_buffer prealloc_moe_sa; // per-block s [tok_slots * ne00/32] (half) // scratch copy of the router weights to avoid dst aliasing ggml_cl_buffer prealloc_moe_combine_w; + ggml_cl_buffer prealloc_splitk_partial; // [ksplit * M] partials for split-K GEMV // pool of persistent image1d_buffer views over kv-cache layers, keyed by // (parent buffer, offset within parent) @@ -656,6 +759,7 @@ struct ggml_backend_opencl_context { cl_program program_add; cl_program program_add_id; + cl_program program_moe_add_id_glu; cl_program program_clamp; cl_program program_cvt; cl_program program_diag_mask_inf; @@ -721,6 +825,7 @@ struct ggml_backend_opencl_context { cl_kernel kernel_div, kernel_div_row, kernel_div_f16, kernel_div_row_f16; cl_kernel kernel_sub, kernel_sub_row, kernel_sub_f16, kernel_sub_row_f16; cl_kernel kernel_add_id; + cl_kernel kernel_add_id_add_id_swiglu_oai; cl_kernel kernel_scale_f32, kernel_scale_f32_4; cl_kernel kernel_sqr_cont_f32, kernel_sqr_cont_f32_4, kernel_sqr_cont_f16, kernel_sqr_cont_f16_4; cl_kernel kernel_sqrt_cont_f32, kernel_sqrt_cont_f32_4, kernel_sqrt_cont_f16, kernel_sqrt_cont_f16_4; @@ -734,10 +839,12 @@ struct ggml_backend_opencl_context { cl_kernel kernel_tri; cl_kernel kernel_fill; cl_kernel kernel_clamp; - cl_kernel kernel_geglu, kernel_reglu, kernel_swiglu, kernel_swiglu_oai, kernel_geglu_erf, kernel_geglu_quick, - kernel_geglu_f16, kernel_reglu_f16, kernel_swiglu_f16, kernel_geglu_erf_f16, kernel_geglu_quick_f16; + cl_kernel kernel_geglu, kernel_reglu, kernel_swiglu, kernel_swiglu_oai, kernel_swiglu_clamp, kernel_geglu_erf, + kernel_geglu_quick, kernel_geglu_f16, kernel_reglu_f16, kernel_swiglu_f16, kernel_swiglu_clamp_f16, + kernel_geglu_erf_f16, kernel_geglu_quick_f16; cl_kernel kernel_norm, kernel_norm_mul_add; cl_kernel kernel_rms_norm, kernel_rms_norm_mul; + cl_kernel kernel_rms_norm_mul_add = nullptr; // fused rms_norm(x)*w + b (residual) cl_kernel kernel_l2_norm_f32; cl_kernel kernel_group_norm, kernel_group_norm_mul_add; cl_kernel kernel_diag_mask_inf, kernel_diag_mask_inf_8; @@ -745,6 +852,9 @@ struct ggml_backend_opencl_context { cl_kernel kernel_soft_max, kernel_soft_max_4; cl_kernel kernel_soft_max_f16, kernel_soft_max_4_f16; ggml_opencl_fa_kernels fa; +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + ggml_cl_adreno_xmem_attn_state adreno_xmem_attn; +#endif cl_kernel kernel_get_rows_f32, kernel_get_rows_f16, kernel_get_rows_q4_0; cl_kernel kernel_set_rows_f32_i64, kernel_set_rows_f32_i32, kernel_set_rows_f16_i64, kernel_set_rows_f16_i32; cl_kernel kernel_set_rows_q8_0_i64, kernel_set_rows_q8_0_i32; @@ -754,9 +864,16 @@ struct ggml_backend_opencl_context { cl_kernel kernel_rope_norm_f32, kernel_rope_norm_f16, kernel_rope_neox_f32, kernel_rope_neox_f16; cl_kernel kernel_rope_multi_f32, kernel_rope_multi_f16, kernel_rope_vision_f32, kernel_rope_vision_f16; cl_kernel kernel_cpy_f16_f16, kernel_cpy_f16_f32, kernel_cpy_f32_f16, kernel_cpy_f32_f32, kernel_cpy_f32_f32_pack, kernel_cpy_i32_i32; + cl_kernel kernel_cpy_f32_f32_flat = nullptr; cl_kernel kernel_mul_mat_f32_f32; cl_kernel kernel_mul_mat_f16_f16; cl_kernel kernel_mul_mat_f16_f32_1row; + cl_program program_mul_mv_f16_f32_mrow; + cl_kernel kernel_mul_mat_f16_f32_mrow = nullptr; // multi-row decode GEMV (attn proj + lm_head) + cl_kernel kernel_mul_mat_f16_f32_mrow_r2 = nullptr; + cl_kernel kernel_mul_mat_f16_f32_mrow_r4 = nullptr; + cl_kernel kernel_mul_mat_f16_f32_mrow_h8 = nullptr; + cl_kernel kernel_mul_mat_f16_f32_mrow_h8r2 = nullptr; cl_kernel kernel_mul_mat_f16_f32; cl_kernel kernel_mul_mat_f16_f32_l4; cl_kernel kernel_mul_mat_f16_f32_l4_dr; @@ -854,11 +971,20 @@ struct ggml_backend_opencl_context { cl_kernel kernel_expm1_f16, kernel_expm1_f16_4, kernel_expm1_f16_nc; cl_kernel kernel_abs_f32, kernel_abs_f32_4, kernel_abs_f32_nc; cl_kernel kernel_abs_f16, kernel_abs_f16_4, kernel_abs_f16_nc; + cl_kernel kernel_sgn_f32, kernel_sgn_f32_4, kernel_sgn_f32_nc, kernel_sgn_f16, kernel_sgn_f16_4, kernel_sgn_f16_nc; + cl_kernel kernel_step_f32, kernel_step_f32_4, kernel_step_f32_nc, kernel_step_f16, kernel_step_f16_4, kernel_step_f16_nc; + cl_kernel kernel_elu_f32, kernel_elu_f32_4, kernel_elu_f32_nc, kernel_elu_f16, kernel_elu_f16_4, kernel_elu_f16_nc; + cl_kernel kernel_hardswish_f32, kernel_hardswish_f32_4, kernel_hardswish_f32_nc, kernel_hardswish_f16, kernel_hardswish_f16_4, kernel_hardswish_f16_nc; + cl_kernel kernel_hardsigmoid_f32, kernel_hardsigmoid_f32_4, kernel_hardsigmoid_f32_nc, kernel_hardsigmoid_f16, kernel_hardsigmoid_f16_4, kernel_hardsigmoid_f16_nc; + cl_kernel kernel_floor_f32, kernel_floor_f32_4, kernel_floor_f32_nc, kernel_floor_f16, kernel_floor_f16_4, kernel_floor_f16_nc; + cl_kernel kernel_ceil_f32, kernel_ceil_f32_4, kernel_ceil_f32_nc, kernel_ceil_f16, kernel_ceil_f16_4, kernel_ceil_f16_nc; + cl_kernel kernel_round_f32, kernel_round_f32_4, kernel_round_f32_nc, kernel_round_f16, kernel_round_f16_4, kernel_round_f16_nc; + cl_kernel kernel_trunc_f32, kernel_trunc_f32_4, kernel_trunc_f32_nc, kernel_trunc_f16, kernel_trunc_f16_4, kernel_trunc_f16_nc; cl_kernel kernel_softplus_f32, kernel_softplus_f32_4, kernel_softplus_f32_nc; cl_kernel kernel_softplus_f16, kernel_softplus_f16_4, kernel_softplus_f16_nc; cl_kernel kernel_upscale; cl_kernel kernel_upscale_bilinear; - cl_kernel kernel_concat_f32, kernel_concat_f32_pack; + cl_kernel kernel_concat_b1, kernel_concat_b2, kernel_concat_b4, kernel_concat_b8, kernel_concat_b4_pack; cl_kernel kernel_conv_2d_f16; cl_kernel kernel_conv_2d_f32; cl_kernel kernel_conv_2d_f16_f32; @@ -866,6 +992,10 @@ struct ggml_backend_opencl_context { // [size_idx][kda][tgpp] where size_idx: 0=S_V=16, 1=32, 2=64, 3=128; kda: 0 or 1. // tgpp 0 = TG variant (COLS_PER_LANE_GROUP=1), tgpp 1 = prefill variant (COLS_PER_LANE_GROUP=4). cl_kernel kernel_gated_delta_net_f32[4][2][2] = {}; + cl_kernel kernel_ssm_scan_f32 = nullptr; + cl_kernel kernel_ssm_scan_f32_mamba2_d128 = nullptr; + cl_kernel kernel_ssm_scan_f32_mamba2_d256 = nullptr; + cl_kernel kernel_timestep_embedding; cl_kernel kernel_gemv_moe_q4_0_f32_ns, kernel_gemm_moe_q4_0_f32_ns, kernel_gemm_moe_q4_0_f32_ns_bin; cl_kernel kernel_gemm_moe_q8_0_f32_ns; @@ -890,14 +1020,19 @@ struct ggml_backend_opencl_context { cl_kernel kernel_gemv_moe_mxfp4_f32_ns_wimg = nullptr; // weight-as-texture MoE decode GEMV cl_kernel kernel_gemm_moe_mxfp4_q8_1_dp4a = nullptr; // dp4a (int8) mxfp4 MoE prefill GEMM cl_kernel kernel_gemm_moe_q4_0_q8_1_dp4a = nullptr; // dp4a (int8) q4_0 MoE prefill GEMM + cl_kernel kernel_gemm_moe_mxfp4_q8_1_dp4a_bin = nullptr; // binary dp4a (int8) mxfp4 MoE prefill GEMM + cl_kernel kernel_gemm_moe_q4_0_q8_1_dp4a_bin = nullptr; // binary dp4a (int8) q4_0 MoE prefill GEMM cl_kernel kernel_moe_reorder_b; cl_kernel kernel_moe_histogram, kernel_moe_scan, kernel_moe_fill, kernel_moe_scatter; + cl_kernel kernel_moe_scatter_stable = nullptr; // deterministic slot assignment cl_kernel kernel_moe_combine_f32 = nullptr; // fused router-weight mul + cross-expert sum + cl_kernel kernel_moe_combine_bias_f32 = nullptr; // same, with the down-projection bias add folded in cl_kernel kernel_mul_mv_id_q4_0_f32_8x_flat; cl_kernel kernel_mul_mv_id_q8_0_f32, kernel_mul_mv_id_q8_0_f32_flat; cl_kernel kernel_mul_mv_id_mxfp4_f32; cl_kernel kernel_mul_mv_id_mxfp4_f32_flat; cl_kernel kernel_mul_mm_f32_f32_l4_lm; + cl_kernel kernel_gemv_f32_f32_mc; // multi-column (small-N) f32 GEMV for spec/MTP verify cl_kernel kernel_mul_mm_f16_f32_l4_lm; cl_kernel kernel_mul_mm_q1_0_f32_l4_lm; cl_kernel kernel_mul_mm_q4_0_f32_l4_lm; @@ -1019,6 +1154,18 @@ struct ggml_backend_opencl_context { } void enqueue_ndrange_kernel(cl_kernel kernel, cl_uint work_dim, size_t *global_work_size, size_t *local_work_size, const ggml_tensor * tensor) { + // From the spec on clEnqueueNDRangeKernel: + // If the device associated with command_queue is an OpenCL 2.1 or newer device, + // and global_work_size is NULL or the value in any passed dimension is zero, + // then the kernel command will trivially succeed after its event dependencies + // are satisfied and subsequently update its completion event. + // So this ensures such cases always return trivially without causing errors in + // case of an older device. + for (cl_uint i = 0; i < work_dim; i++) { + if (global_work_size[i] == 0) { + return; + } + } #ifdef GGML_OPENCL_PROFILING cl_event evt; CL_CHECK(clEnqueueNDRangeKernel(queue, kernel, work_dim, NULL, global_work_size, local_work_size, 0, NULL, &evt)); @@ -1063,28 +1210,58 @@ struct ggml_backend_opencl_context { // Gemm and Gemv related programs, kernels, etc cl_kernel kernel_gemm_noshuffle_q4_0_f32; cl_kernel kernel_gemv_noshuffle_q4_0_f32; + cl_kernel kernel_gemv_noshuffle_q4_0_f32_mc3; // multi-column (N=3) verify GEMV (spec/MTP) + cl_kernel kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin; + cl_kernel kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8_bin; + cl_kernel kernel_gemv_noshuffle_q4_0_f32_32b_trans; cl_kernel kernel_gemv_noshuffle_q4_0_f32_4096_1_11008; cl_kernel kernel_gemv_noshuffle_q4_0_f32_4096_1_4096; cl_kernel kernel_gemv_noshuffle_q4_0_f32_11008_1_4096; cl_kernel kernel_gemv_noshuffle_q4_0_f32_32000_1_4096; cl_kernel kernel_gemv_noshuffle_q4_1_f32; + cl_kernel kernel_gemv_noshuffle_q4_1_f32_mc3; // multi-column (N=3) verify GEMV (spec/MTP) cl_kernel kernel_gemm_noshuffle_q4_1_f32; cl_kernel kernel_gemm_noshuffle_q8_0_f32, kernel_gemm_noshuffle_q8_0_f32_bin; cl_kernel kernel_gemm_noshuffle_q8_0_q8_1_dp4a = nullptr; // dp4a (int8) dense q8_0 prefill GEMM (opt-in) cl_kernel kernel_gemm_noshuffle_q8_0_q8_1_dp4a_wimg = nullptr; // q8_0 dense dp4a, weights via texture (opt-in) cl_kernel kernel_gemv_noshuffle_q8_0_f32; + cl_kernel kernel_gemv_noshuffle_q8_0_f32_splitk; // split-K across WGs (small-M decode) cl_kernel kernel_gemm_noshuffle_q1_0_f32; cl_kernel kernel_gemv_noshuffle_q1_0_f32; cl_kernel kernel_gemv_noshuffle_q4_k_f32; + cl_kernel kernel_gemv_noshuffle_q4_k_f32_o4; // 4-output-per-WI, long-vocab lm_head + cl_kernel kernel_gemv_noshuffle_q4_k_f32_tiled; // tiled-wide layout (opt-in) + cl_kernel kernel_gemv_noshuffle_q4_k_f32_splitk; // split-K across WGs (small-M decode) + cl_kernel kernel_gemv_splitk_reduce_f32; // sums split-K per-slice partials + cl_kernel kernel_gemv_noshuffle_q4_k_f32_glu; // fused gate+up GEMV + GLU (FFN) + cl_kernel kernel_convert_block_q4_k_tiled_ns; // tiled-wide convert (opt-in) + cl_kernel kernel_gemv_noshuffle_q4_k_f32_mc3; // multi-column (N=3) verify GEMV cl_kernel kernel_gemm_noshuffle_q4_k_f32; + cl_kernel kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin; + cl_kernel kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8_bin; + cl_kernel kernel_gemv_noshuffle_q4_k_f32_32b_trans; cl_kernel kernel_gemm_noshuffle_q4_k_q8_1_dp4a = nullptr; // dp4a (int8) dense prefill GEMM cl_kernel kernel_gemm_noshuffle_q4_k_q8_1_dp4a_wimg = nullptr; // dp4a dense prefill GEMM, weights via texture (X1 opt-in) cl_kernel kernel_gemm_noshuffle_q5_k_q8_1_dp4a = nullptr; // dp4a (int8) dense q5_K prefill GEMM cl_kernel kernel_gemm_noshuffle_q6_k_q8_1_dp4a = nullptr; // dp4a (int8) dense q6_K prefill GEMM cl_kernel kernel_quant_a_q8_1; // plain activation q8_1 pre-pass + cl_kernel kernel_gemm_noshuffle_q4_k_f32_r1; + cl_kernel kernel_gemm_noshuffle_q4_k_f32_kimg; + cl_kernel kernel_gemm_noshuffle_q4_k_f32_cok; cl_kernel kernel_gemv_noshuffle_q6_K_f32; + cl_kernel kernel_gemv_noshuffle_q6_K_f32_o4; + cl_kernel kernel_gemv_noshuffle_q6_K_f32_o4_global; // weights via __global (opt-in) + cl_kernel kernel_gemv_noshuffle_q6_K_f32_tiled; // tiled-wide layout (opt-in) + cl_kernel kernel_gemv_noshuffle_q6_K_f32_tiled_mc3; // tiled multi-column (N=3) verify lm_head + cl_kernel kernel_gemm_noshuffle_q6_K_f32_tiled; // batched (N>1) over the tiled layout + cl_kernel kernel_convert_block_q6_k_tiled_ns; // tiled-wide convert (opt-in) + cl_kernel kernel_gemv_noshuffle_q6_K_f32_mc3; // multi-column (N=3) verify GEMV cl_kernel kernel_gemm_noshuffle_q6_K_f32; + cl_kernel kernel_gemm_noshuffle_q6_K_f32_cok; + cl_kernel kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin; + cl_kernel kernel_gemv_noshuffle_q6_k_f32_32b_trans; cl_kernel kernel_gemv_noshuffle_q5_k_f32; + cl_kernel kernel_gemv_noshuffle_q5_k_f32_mc3; // multi-column (N=3) verify GEMV (spec/MTP) cl_kernel kernel_gemm_noshuffle_q5_k_f32; cl_kernel kernel_gemv_noshuffle_q5_0_f32; cl_kernel kernel_gemm_noshuffle_q5_0_f32; @@ -1123,6 +1300,9 @@ struct ggml_backend_opencl_context { if (kv.second.image) { CL_CHECK(clReleaseMemObject(kv.second.image)); } } dequant_f16_pool.clear(); +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + ggml_cl_adreno_xmem_attn_release_scratch(this); +#endif } } }; @@ -1274,6 +1454,7 @@ static void load_cl_kernels_argsort(ggml_backend_opencl_context *backend_ctx) { static bool use_adreno_bin_kernels(ggml_backend_opencl_context * backend_ctx) { #ifndef GGML_OPENCL_USE_ADRENO_BIN_KERNELS + GGML_UNUSED(backend_ctx); return false; #else if (backend_ctx->gpu_family != GPU_FAMILY::ADRENO) { @@ -1340,6 +1521,23 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // moe_add_id_glu + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "moe_add_id_glu.cl.h" + }; +#else + const std::string kernel_src = read_file("moe_add_id_glu.cl"); +#endif + backend_ctx->program_moe_add_id_glu = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + + CL_CHECK((backend_ctx->kernel_add_id_add_id_swiglu_oai = + clCreateKernel(backend_ctx->program_moe_add_id_glu, "kernel_add_id_add_id_swiglu_oai", &err), err)); + GGML_LOG_CONT("."); + } + // tri { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -1409,6 +1607,13 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK((backend_ctx->kernel_cpy_f32_f16 = clCreateKernel(prog, "kernel_cpy_f32_f16", &err), err)); CL_CHECK((backend_ctx->kernel_cpy_f32_f32 = clCreateKernel(prog, "kernel_cpy_f32_f32", &err), err)); CL_CHECK((backend_ctx->kernel_cpy_f32_f32_pack = clCreateKernel(prog, "kernel_cpy_f32_f32_pack", &err), err)); + { // optional: without it ggml_cl_cpy keeps the row-mapped kernel + cl_int err_flat = CL_SUCCESS; + cl_kernel k = clCreateKernel(prog, "kernel_cpy_f32_f32_flat", &err_flat); + if (err_flat == CL_SUCCESS) { + backend_ctx->kernel_cpy_f32_f32_flat = k; + } + } CL_CHECK((backend_ctx->kernel_cpy_i32_i32 = clCreateKernel(prog, "kernel_cpy_i32_i32", &err), err)); GGML_LOG_CONT("."); } @@ -1453,10 +1658,16 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK((backend_ctx->kernel_restore_block_q5_1_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_restore_block_q5_1_trans4_ns", &err), err)); CL_CHECK((backend_ctx->kernel_convert_block_q4_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_q4_k_trans4_ns", &err), err)); CL_CHECK((backend_ctx->kernel_restore_block_q4_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_restore_block_q4_k_trans4_ns", &err), err)); +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + CL_CHECK((backend_ctx->kernel_convert_block_q4_k_tiled_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_q4_k_tiled_ns", &err), err)); +#endif CL_CHECK((backend_ctx->kernel_convert_block_q5_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_q5_k_trans4_ns", &err), err)); CL_CHECK((backend_ctx->kernel_restore_block_q5_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_restore_block_q5_k_trans4_ns", &err), err)); CL_CHECK((backend_ctx->kernel_convert_block_q6_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_q6_k_trans4_ns", &err), err)); CL_CHECK((backend_ctx->kernel_restore_block_q6_k_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_restore_block_q6_k_trans4_ns", &err), err)); +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + CL_CHECK((backend_ctx->kernel_convert_block_q6_k_tiled_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_q6_k_tiled_ns", &err), err)); +#endif CL_CHECK((backend_ctx->kernel_convert_block_mxfp4 = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_mxfp4", &err), err)); CL_CHECK((backend_ctx->kernel_convert_block_mxfp4_trans = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_mxfp4_trans", &err), err)); CL_CHECK((backend_ctx->kernel_convert_block_mxfp4_trans4_ns = clCreateKernel(backend_ctx->program_cvt, "kernel_convert_block_mxfp4_trans4_ns", &err), err)); @@ -1567,11 +1778,13 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK((backend_ctx->kernel_reglu = clCreateKernel(backend_ctx->program_glu, "kernel_reglu", &err), err)); CL_CHECK((backend_ctx->kernel_swiglu = clCreateKernel(backend_ctx->program_glu, "kernel_swiglu", &err), err)); CL_CHECK((backend_ctx->kernel_swiglu_oai = clCreateKernel(backend_ctx->program_glu, "kernel_swiglu_oai", &err), err)); + CL_CHECK((backend_ctx->kernel_swiglu_clamp = clCreateKernel(backend_ctx->program_glu, "kernel_swiglu_clamp", &err), err)); CL_CHECK((backend_ctx->kernel_geglu_erf = clCreateKernel(backend_ctx->program_glu, "kernel_geglu_erf", &err), err)); CL_CHECK((backend_ctx->kernel_geglu_quick = clCreateKernel(backend_ctx->program_glu, "kernel_geglu_quick", &err), err)); CL_CHECK((backend_ctx->kernel_geglu_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_geglu_f16", &err), err)); CL_CHECK((backend_ctx->kernel_reglu_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_reglu_f16", &err), err)); CL_CHECK((backend_ctx->kernel_swiglu_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_swiglu_f16", &err), err)); + CL_CHECK((backend_ctx->kernel_swiglu_clamp_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_swiglu_clamp_f16", &err), err)); CL_CHECK((backend_ctx->kernel_geglu_erf_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_geglu_erf_f16", &err), err)); CL_CHECK((backend_ctx->kernel_geglu_quick_f16 = clCreateKernel(backend_ctx->program_glu, "kernel_geglu_quick_f16", &err), err)); GGML_LOG_CONT("."); @@ -1927,8 +2140,14 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { #else const std::string kernel_src = read_file("mul_mv_q6_k_f32_flat.cl"); #endif + // The codegen workarounds in this kernel are a measured 13-20% loss on + // compilers that do not need them, so only the affected ones build them; + // everyone else gets the original source. + const std::string q6k_opts = backend_ctx->q6_k_flat_old_compiler + ? compile_opts + " -DADRENO_OLD_COMPILER=1" + : compile_opts; cl_program prog = - build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + build_program_from_source(backend_ctx, kernel_src.c_str(), q6k_opts); CL_CHECK((backend_ctx->kernel_mul_mv_q6_K_f32_flat = clCreateKernel(prog, "kernel_mul_mv_q6_K_f32_flat", &err), err)); CL_CHECK(clReleaseProgram(prog)); @@ -2099,6 +2318,26 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // mul_mv_f16_f32_mrow (multi-row decode GEMV) + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "mul_mv_f16_f32_mrow.cl.h" + }; +#else + const std::string kernel_src = read_file("mul_mv_f16_f32_mrow.cl"); +#endif + backend_ctx->program_mul_mv_f16_f32_mrow = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + + CL_CHECK((backend_ctx->kernel_mul_mat_f16_f32_mrow = clCreateKernel(backend_ctx->program_mul_mv_f16_f32_mrow, "kernel_mul_mat_f16_f32_mrow", &err), err)); + CL_CHECK((backend_ctx->kernel_mul_mat_f16_f32_mrow_r2 = clCreateKernel(backend_ctx->program_mul_mv_f16_f32_mrow, "kernel_mul_mat_f16_f32_mrow_r2", &err), err)); + CL_CHECK((backend_ctx->kernel_mul_mat_f16_f32_mrow_r4 = clCreateKernel(backend_ctx->program_mul_mv_f16_f32_mrow, "kernel_mul_mat_f16_f32_mrow_r4", &err), err)); + CL_CHECK((backend_ctx->kernel_mul_mat_f16_f32_mrow_h8 = clCreateKernel(backend_ctx->program_mul_mv_f16_f32_mrow, "kernel_mul_mat_f16_f32_mrow_h8", &err), err)); + CL_CHECK((backend_ctx->kernel_mul_mat_f16_f32_mrow_h8r2 = clCreateKernel(backend_ctx->program_mul_mv_f16_f32_mrow, "kernel_mul_mat_f16_f32_mrow_h8r2", &err), err)); + GGML_LOG_CONT("."); + } + // mul_mv_f16_f32_l4 { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -2239,6 +2478,49 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { } #endif // GGML_OPENCL_USE_ADRENO_KERNELS +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // Adreno xmem SDPA + if (backend_ctx->gpu_family == GPU_FAMILY::ADRENO) { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "sdpa_xmem_f32_f16_os8.cl.h" + }; +#else + const std::string kernel_src = read_file("sdpa_xmem_f32_f16_os8.cl"); +#endif + cl_program program = build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + + auto & xmem_attn = backend_ctx->adreno_xmem_attn; + CL_CHECK((xmem_attn.kernel_q_f32_to_img_scaled = + clCreateKernel(program, "adreno_xmem_attn_q_f32_to_img_scaled", &err), err)); + CL_CHECK((xmem_attn.kernel_kv_f32_to_img_gqa = + clCreateKernel(program, "adreno_xmem_attn_kv_f32_to_img_gqa", &err), err)); + CL_CHECK((xmem_attn.kernel_kv_f16_to_img_gqa = + clCreateKernel(program, "adreno_xmem_attn_kv_f16_to_img_gqa", &err), err)); + CL_CHECK((xmem_attn.kernel_img_to_f32 = + clCreateKernel(program, "adreno_xmem_attn_img_to_f32", &err), err)); + CL_CHECK((xmem_attn.kernel_k_gather = + clCreateKernel(program, "adreno_xmem_attn_k_gather", &err), err)); + CL_CHECK((xmem_attn.kernel_pack_k = + clCreateKernel(program, "adreno_xmem_attn_pack_k", &err), err)); + CL_CHECK((xmem_attn.kernel_qk_gemm = + clCreateKernel(program, "adreno_xmem_attn_qk_gemm", &err), err)); + CL_CHECK((xmem_attn.kernel_softmax_reduce_basic = + clCreateKernel(program, "adreno_xmem_attn_softmax_reduce_basic", &err), err)); + CL_CHECK((xmem_attn.kernel_softmax_apply_basic = + clCreateKernel(program, "adreno_xmem_attn_softmax_apply_basic", &err), err)); + CL_CHECK((xmem_attn.kernel_mask_scores = + clCreateKernel(program, "adreno_xmem_attn_mask_scores", &err), err)); + CL_CHECK((xmem_attn.kernel_pack_v = + clCreateKernel(program, "adreno_xmem_attn_pack_v", &err), err)); + CL_CHECK((xmem_attn.kernel_pv_gemm = + clCreateKernel(program, "adreno_xmem_attn_pv_gemm", &err), err)); + CL_CHECK(clReleaseProgram(program)); + xmem_attn.compiled = true; + GGML_LOG_CONT("."); + } +#endif // GGML_OPENCL_USE_ADRENO_KERNELS + // mul_mm_f32_f32_l4_lm { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -2252,6 +2534,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); CL_CHECK((backend_ctx->kernel_mul_mm_f32_f32_l4_lm = clCreateKernel(backend_ctx->program_mul_mm_f32_f32_l4_lm, "kernel_mul_mm_f32_f32_l4_lm", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_f32_f32_mc = clCreateKernel(backend_ctx->program_mul_mm_f32_f32_l4_lm, "kernel_gemv_f32_f32_mc", &err), err)); GGML_LOG_CONT("."); } @@ -2521,6 +2804,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK((backend_ctx->kernel_rms_norm = clCreateKernel(backend_ctx->program_rms_norm, "kernel_rms_norm", &err), err)); CL_CHECK((backend_ctx->kernel_rms_norm_mul = clCreateKernel(backend_ctx->program_rms_norm, "kernel_rms_norm_mul", &err), err)); + CL_CHECK((backend_ctx->kernel_rms_norm_mul_add = clCreateKernel(backend_ctx->program_rms_norm, "kernel_rms_norm_mul_add", &err), err)); GGML_LOG_CONT("."); } @@ -2976,6 +3260,38 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // unary_ext (sgn, step, elu, hardswish, hardsigmoid, floor, ceil, round, trunc) + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "unary_ext.cl.h" + }; +#else + const std::string kernel_src = read_file("unary_ext.cl"); +#endif + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); +#define CL_UNARY_EXT_K(op) \ + CL_CHECK((backend_ctx->kernel_##op##_f32 = clCreateKernel(prog, "kernel_" #op "_f32", &err), err)); \ + CL_CHECK((backend_ctx->kernel_##op##_f32_4 = clCreateKernel(prog, "kernel_" #op "_f32_4", &err), err)); \ + CL_CHECK((backend_ctx->kernel_##op##_f32_nc = clCreateKernel(prog, "kernel_" #op "_f32_nc", &err), err)); \ + CL_CHECK((backend_ctx->kernel_##op##_f16 = clCreateKernel(prog, "kernel_" #op "_f16", &err), err)); \ + CL_CHECK((backend_ctx->kernel_##op##_f16_4 = clCreateKernel(prog, "kernel_" #op "_f16_4", &err), err)); \ + CL_CHECK((backend_ctx->kernel_##op##_f16_nc = clCreateKernel(prog, "kernel_" #op "_f16_nc", &err), err)); + CL_UNARY_EXT_K(sgn) + CL_UNARY_EXT_K(step) + CL_UNARY_EXT_K(elu) + CL_UNARY_EXT_K(hardswish) + CL_UNARY_EXT_K(hardsigmoid) + CL_UNARY_EXT_K(floor) + CL_UNARY_EXT_K(ceil) + CL_UNARY_EXT_K(round) + CL_UNARY_EXT_K(trunc) +#undef CL_UNARY_EXT_K + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + // softplus { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -3040,8 +3356,11 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { #endif cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); - CL_CHECK((backend_ctx->kernel_concat_f32 = clCreateKernel(prog, "kernel_concat_f32", &err), err)); - CL_CHECK((backend_ctx->kernel_concat_f32_pack = clCreateKernel(prog, "kernel_concat_f32_pack", &err), err)); + CL_CHECK((backend_ctx->kernel_concat_b1 = clCreateKernel(prog, "kernel_concat_b1", &err), err)); + CL_CHECK((backend_ctx->kernel_concat_b2 = clCreateKernel(prog, "kernel_concat_b2", &err), err)); + CL_CHECK((backend_ctx->kernel_concat_b4 = clCreateKernel(prog, "kernel_concat_b4", &err), err)); + CL_CHECK((backend_ctx->kernel_concat_b8 = clCreateKernel(prog, "kernel_concat_b8", &err), err)); + CL_CHECK((backend_ctx->kernel_concat_b4_pack = clCreateKernel(prog, "kernel_concat_b4_pack", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3154,6 +3473,50 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // ssm_scan + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "ssm_scan.cl.h" + }; +#else + const std::string kernel_src = read_file("ssm_scan.cl"); +#endif + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + + CL_CHECK((backend_ctx->kernel_ssm_scan_f32 = clCreateKernel(prog, "kernel_ssm_scan_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_ssm_scan_f32_mamba2_d128 = clCreateKernel(prog, "kernel_ssm_scan_f32_mamba2_d128", &err), err)); + CL_CHECK((backend_ctx->kernel_ssm_scan_f32_mamba2_d256 = clCreateKernel(prog, "kernel_ssm_scan_f32_mamba2_d256", &err), err)); + + cl_kernel * kernels[] = { + &backend_ctx->kernel_ssm_scan_f32_mamba2_d128, + &backend_ctx->kernel_ssm_scan_f32_mamba2_d256 + }; + + // specialized kernels use subgroups and assume subgroup size is 64, + // if device does not support subgroups or subgroup size is not 64, + // release these kernels + for (int i = 0; i < 2; ++i) { + size_t subgroup_size = 0; +#if CL_TARGET_OPENCL_VERSION >= 210 + const size_t local_work_size[] = { 64, 1 }; + const cl_int subgroup_err = clGetKernelSubGroupInfo(*kernels[i], backend_ctx->device, CL_KERNEL_MAX_SUB_GROUP_SIZE_FOR_NDRANGE, + sizeof(local_work_size), local_work_size, sizeof(subgroup_size), &subgroup_size, nullptr); + if (subgroup_err != CL_SUCCESS) { + subgroup_size = 0; + } +#endif + // The specialized kernels reduce over one 64-lane subgroup. + if (subgroup_size != 64) { + CL_CHECK(clReleaseKernel(*kernels[i])); + *kernels[i] = nullptr; + } + } + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + // gated_delta_net: one kernel per (S_V, KDA, tgpp) triple. { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -3246,6 +3609,8 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { backend_ctx, kernel_src.c_str(), compile_opts); CL_CHECK((backend_ctx->kernel_moe_combine_f32 = clCreateKernel(prog, "kernel_moe_combine_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_moe_combine_bias_f32 = + clCreateKernel(prog, "kernel_moe_combine_bias_f32", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3412,6 +3777,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { cl_program prog = build_program_from_source(backend_ctx, kernel_src_CL_gemv_general.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_0_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_0_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_0_f32_mc3 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_0_f32_mc3", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3507,6 +3873,55 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32b_trans = nullptr; + backend_ctx->kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin = nullptr; + backend_ctx->kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8_bin = nullptr; + if (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E) { + { + std::string opts = std::string("-cl-std=") + opencl_c_std + + " -cl-mad-enable " + " -DSIMDGROUP_WIDTH=" + + std::to_string(backend_ctx->adreno_wave_size); +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q4_0_f32_32b_trans.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q4_0_f32_32b_trans.cl"); +#endif + cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32b_trans = + clCreateKernel(prog, "kernel_gemv_noshuffle_q4_0_f32_32b_trans", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + if (use_adreno_bin_kernels(backend_ctx)) { + size_t bin_size = 0; + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_noshuffle_q4_0_f32_32b_trans_ila_a8", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program bin_prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, "", bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin = + clCreateKernel(bin_prog, "kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8", &err), err)); + CL_CHECK(clReleaseProgram(bin_prog)); + GGML_LOG_CONT("."); + } + + kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program bin_prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, "", bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8_bin = + clCreateKernel(bin_prog, "kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8", &err), err)); + CL_CHECK(clReleaseProgram(bin_prog)); + GGML_LOG_CONT("."); + } + } + } + // gemm_noshuffle_q4_1_f32 { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -3541,6 +3956,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_1_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_1_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_1_f32_mc3 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_1_f32_mc3", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3755,6 +4171,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { cl_program prog = build_program_from_source(backend_ctx, kernel_src_CL_gemv_general.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q8_0_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q8_0_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q8_0_f32_splitk = clCreateKernel(prog, "kernel_gemv_noshuffle_q8_0_f32_splitk", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3770,6 +4187,9 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { #endif cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_f32 = clCreateKernel(prog, "kernel_gemm_noshuffle_q4_k_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_f32_r1 = clCreateKernel(prog, "kernel_gemm_noshuffle_q4_k_f32_r1", &err), err)); + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_f32_kimg = clCreateKernel(prog, "kernel_gemm_noshuffle_q4_k_f32_kimg", &err), err)); + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_f32_cok = clCreateKernel(prog, "kernel_gemm_noshuffle_q4_k_f32_cok", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -3864,6 +4284,18 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { if (backend_ctx->has_vector_subgroup_broadcast) { CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST "; } + // Opt-in: dequant-once-per-block mc3 verify GEMV (factors q4_K dequant + // out of the 3-column loop; byte-identical, lower spill). A/B vs the + // shipped inline mc3 in the same binary. + if (getenv("GGML_OPENCL_Q4K_MC3_DQ")) { + CL_gemv_compile_opts += " -DQ4K_MC3_DEQUANT_ONCE "; + } + // Opt-in: LDS-staged dequant mc3 verify GEMV (stages the dequantized + // q4_K weights in __local instead of private regs that spill to slow + // global on Adreno; byte-identical). A/B vs inline + dequant-once. + if (getenv("GGML_OPENCL_Q4K_MC3_LDS")) { + CL_gemv_compile_opts += " -DQ4K_MC3_DEQUANT_LDS "; + } #ifdef GGML_OPENCL_EMBED_KERNELS const std::string kernel_src { @@ -3876,10 +4308,140 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_mc3 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32_mc3", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_splitk = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32_splitk", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_splitk_reduce_f32 = clCreateKernel(prog, "kernel_gemv_splitk_reduce_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_glu = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32_glu", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + // gemv_noshuffle_q4_k_f32_o4 — 4-output-per-WI variant for the long-vocab + // q4_K lm_head/embed GEMV (shares one activation read across 4 output rows). + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q4_k_f32_o4.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q4_k_f32_o4.cl"); +#endif + std::string CL_gemv_compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable "; + if (backend_ctx->has_vector_subgroup_broadcast) { + CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST "; + } + cl_program prog = build_program_from_source( + backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_o4 = clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32_o4", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + // gemv_noshuffle_q4_k_f32_tiled — tiled-wide canonical layout, default ON + // (opt out: GGML_OPENCL_Q4K_GEMV_TILED=0; separate convert + GEMV; weights via __global). + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q4_k_f32_tiled.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q4_k_f32_tiled.cl"); +#endif + std::string compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable "; + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_tiled = + clCreateKernel(prog, "kernel_gemv_noshuffle_q4_k_f32_tiled", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } + backend_ctx->kernel_gemv_noshuffle_q4_k_f32_32b_trans = nullptr; + backend_ctx->kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin = nullptr; + backend_ctx->kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8_bin = nullptr; + if (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E) { + { + std::string opts = std::string("-cl-std=") + opencl_c_std + + " -cl-mad-enable " + " -DSIMDGROUP_WIDTH=" + + std::to_string(backend_ctx->adreno_wave_size); +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q4_k_f32_32b_trans.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q4_k_f32_32b_trans.cl"); +#endif + cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q4_k_f32_32b_trans = + clCreateKernel(prog, "gemv_noshuffle_q4_k_f32_32b_trans", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + if (use_adreno_bin_kernels(backend_ctx)) { + size_t bin_size = 0; + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_noshuffle_q4_k_f32_32b_trans_ila_a8", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program bin_prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, "", bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin = + clCreateKernel(bin_prog, "kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8", &err), err)); + CL_CHECK(clReleaseProgram(bin_prog)); + GGML_LOG_CONT("."); + } + + kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program bin_prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, "", bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8_bin = + clCreateKernel(bin_prog, "kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8", &err), err)); + CL_CHECK(clReleaseProgram(bin_prog)); + GGML_LOG_CONT("."); + } + } + } + + backend_ctx->kernel_gemv_noshuffle_q6_k_f32_32b_trans = nullptr; + backend_ctx->kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin = nullptr; + if (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E) { + { + std::string opts = std::string("-cl-std=") + opencl_c_std + + " -cl-mad-enable " + " -DSIMDGROUP_WIDTH=" + + std::to_string(backend_ctx->adreno_wave_size); +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q6_k_f32_32b_trans.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q6_k_f32_32b_trans.cl"); +#endif + cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_k_f32_32b_trans = + clCreateKernel(prog, "kernel_gemv_noshuffle_q6_k_f32_32b_trans", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + if (use_adreno_bin_kernels(backend_ctx)) { + size_t bin_size = 0; + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_noshuffle_q6_k_f32_32b_trans_ila_a8", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program bin_prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, "", bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin = + clCreateKernel(bin_prog, "kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8", &err), err)); + CL_CHECK(clReleaseProgram(bin_prog)); + GGML_LOG_CONT("."); + } + } + } + std::string CL_moe_compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable " " -cl-fast-relaxed-math"; @@ -4190,6 +4752,24 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // gemm_moe_mxfp4_q8_1_dp4a_bin (dp4a prefill GEMM) + if (backend_ctx->has_integer_dot) { + size_t bin_size = 0; + backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a_bin = nullptr; + + if (use_adreno_bin_kernels(backend_ctx)) { + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_moe_mxfp4_q8_1_dp4a_ila", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, CL_moe_compile_opts, bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a_bin = clCreateKernel(prog, "kernel_gemm_moe_mxfp4_q8_1_dp4a_ila", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + } + } + // gemm_moe_q4_0_q8_1_dp4a (dp4a prefill GEMM) if (backend_ctx->has_integer_dot) { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -4207,6 +4787,24 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { GGML_LOG_CONT("."); } + // gemm_moe_q4_0_q8_1_dp4a_bin (dp4a prefill GEMM) + if (backend_ctx->has_integer_dot) { + size_t bin_size = 0; + backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a_bin = nullptr; + + if (use_adreno_bin_kernels(backend_ctx)) { + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("gemm_moe_q4_0_q8_1_dp4a_ila", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, CL_moe_compile_opts, bin_size); + + CL_CHECK((backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a_bin = clCreateKernel(prog, "kernel_gemm_moe_q4_0_q8_1_dp4a_ila", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + } + } + // gemm_moe_q8_1_dp4a (generic dp4a MoE GEMM; MOE_QT=80 -> q8_0 expert variant) if (backend_ctx->has_integer_dot) { #ifdef GGML_OPENCL_EMBED_KERNELS @@ -4442,6 +5040,7 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK((backend_ctx->kernel_moe_scan = clCreateKernel(prog, "kernel_moe_scan", &err), err)); CL_CHECK((backend_ctx->kernel_moe_fill = clCreateKernel(prog, "kernel_moe_fill", &err), err)); CL_CHECK((backend_ctx->kernel_moe_scatter = clCreateKernel(prog, "kernel_moe_scatter", &err), err)); + CL_CHECK((backend_ctx->kernel_moe_scatter_stable = clCreateKernel(prog, "kernel_moe_scatter_stable", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -4466,44 +5065,131 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q6_K_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32_mc3 = clCreateKernel(prog, "kernel_gemv_noshuffle_q6_K_f32_mc3", &err), err)); + if (getenv("GGML_OPENCL_MC3_PROBE")) { + cl_ulong pm6 = 0, pm4 = 0; size_t wg6 = 0, wg4 = 0, mult = 0; + clGetKernelWorkGroupInfo(backend_ctx->kernel_gemv_noshuffle_q6_K_f32_mc3, backend_ctx->device, CL_KERNEL_PRIVATE_MEM_SIZE, sizeof(pm6), &pm6, NULL); + clGetKernelWorkGroupInfo(backend_ctx->kernel_gemv_noshuffle_q6_K_f32_mc3, backend_ctx->device, CL_KERNEL_WORK_GROUP_SIZE, sizeof(wg6), &wg6, NULL); + clGetKernelWorkGroupInfo(backend_ctx->kernel_gemv_noshuffle_q4_k_f32_mc3, backend_ctx->device, CL_KERNEL_PRIVATE_MEM_SIZE, sizeof(pm4), &pm4, NULL); + clGetKernelWorkGroupInfo(backend_ctx->kernel_gemv_noshuffle_q4_k_f32_mc3, backend_ctx->device, CL_KERNEL_WORK_GROUP_SIZE, sizeof(wg4), &wg4, NULL); + clGetKernelWorkGroupInfo(backend_ctx->kernel_gemv_noshuffle_q6_K_f32_mc3, backend_ctx->device, CL_KERNEL_PREFERRED_WORK_GROUP_SIZE_MULTIPLE, sizeof(mult), &mult, NULL); + fprintf(stderr, "[MC3-PROBE] q4K_mc3 private=%llu wg_cap=%zu | q6K_mc3 private=%llu wg_cap=%zu | pref_mult=%zu\n", + (unsigned long long)pm4, wg4, (unsigned long long)pm6, wg6, mult); + fflush(stderr); + } GGML_LOG_CONT("."); } - // gemm_noshuffle_q6_k_f32 + // gemv_noshuffle_q6_k_f32_o4 — 4-output-per-WI variant, opt-in via + // GGML_OPENCL_Q6K_GEMV_O4=1 (~3x fewer dispatches on long-vocab lm_head). { #ifdef GGML_OPENCL_EMBED_KERNELS const std::string kernel_src { - #include "gemm_noshuffle_q6_k_f32.cl.h" + #include "gemv_noshuffle_q6_k_f32_o4.cl.h" }; #else - const std::string kernel_src = read_file("gemm_noshuffle_q6_k_f32.cl"); + const std::string kernel_src = read_file("gemv_noshuffle_q6_k_f32_o4.cl"); #endif - cl_program prog = - build_program_from_source(backend_ctx, kernel_src.c_str(), CL_moe_compile_opts); - - CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q6_K_f32 = clCreateKernel(prog, "kernel_gemm_noshuffle_q6_K_f32", &err), err)); - GGML_LOG_CONT("."); - } - // gemv_noshuffle_q5_k_f32 - { std::string CL_gemv_compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable "; if (backend_ctx->has_vector_subgroup_broadcast) { - CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST "; + CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAT "; } -#ifdef GGML_OPENCL_EMBED_KERNELS - const std::string kernel_src { - #include "gemv_noshuffle_q5_k_f32.cl.h" - }; -#else - const std::string kernel_src = read_file("gemv_noshuffle_q5_k_f32.cl"); + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); + + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32_o4 = clCreateKernel(prog, "kernel_gemv_noshuffle_q6_K_f32_o4", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + + // Global-read variant: weights read from __global coalesced instead of + // image1d_buffer (the texture cache caps the streaming lm_head read + // bandwidth). Opt-in via GGML_OPENCL_Q6K_GEMV_O4_GLOBAL. + cl_program prog_g = build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts + " -DQ6K_O4_GLOBAL"); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32_o4_global = + clCreateKernel(prog_g, "kernel_gemv_noshuffle_q6_K_f32_o4_global", &err), err)); + CL_CHECK(clReleaseProgram(prog_g)); + GGML_LOG_CONT("."); + } + + // gemv_noshuffle_q6_k_f32_tiled — tiled-wide canonical layout, default ON + // (opt out: GGML_OPENCL_Q6K_GEMV_TILED=0; separate convert + GEMV; weights via __global). + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q6_k_f32_tiled.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q6_k_f32_tiled.cl"); +#endif + std::string compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable "; + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32_tiled = + clCreateKernel(prog, "kernel_gemv_noshuffle_q6_K_f32_tiled", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q6_K_f32_tiled_mc3 = + clCreateKernel(prog, "kernel_gemv_noshuffle_q6_K_f32_tiled_mc3", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + // gemm_noshuffle_q6_k_f32_tiled — batched (N>1) GEMM over the same tiled-wide + // canonical layout, so batched lm_head/embed stays correct + on GPU. + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemm_noshuffle_q6_k_f32_tiled.cl.h" + }; +#else + const std::string kernel_src = read_file("gemm_noshuffle_q6_k_f32_tiled.cl"); +#endif + std::string compile_opts = std::string("-cl-std=") + opencl_c_std + " -cl-mad-enable "; + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q6_K_f32_tiled = + clCreateKernel(prog, "kernel_gemm_noshuffle_q6_K_f32_tiled", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + + // gemm_noshuffle_q6_k_f32 + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemm_noshuffle_q6_k_f32.cl.h" + }; +#else + const std::string kernel_src = read_file("gemm_noshuffle_q6_k_f32.cl"); +#endif + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), CL_moe_compile_opts); + + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q6_K_f32 = clCreateKernel(prog, "kernel_gemm_noshuffle_q6_K_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemm_noshuffle_q6_K_f32_cok = clCreateKernel(prog, "kernel_gemm_noshuffle_q6_K_f32_cok", &err), err)); + GGML_LOG_CONT("."); + } + + // gemv_noshuffle_q5_k_f32 + { + std::string CL_gemv_compile_opts = std::string("-cl-std=") + opencl_c_std + + " -cl-mad-enable "; + if (backend_ctx->has_vector_subgroup_broadcast) { + CL_gemv_compile_opts += " -DVECTOR_SUB_GROUP_BROADCAST "; + } + +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "gemv_noshuffle_q5_k_f32.cl.h" + }; +#else + const std::string kernel_src = read_file("gemv_noshuffle_q5_k_f32.cl"); #endif cl_program prog = build_program_from_source(backend_ctx, kernel_src.c_str(), CL_gemv_compile_opts); CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q5_k_f32 = clCreateKernel(prog, "kernel_gemv_noshuffle_q5_k_f32", &err), err)); + CL_CHECK((backend_ctx->kernel_gemv_noshuffle_q5_k_f32_mc3 = clCreateKernel(prog, "kernel_gemv_noshuffle_q5_k_f32_mc3", &err), err)); CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } @@ -4522,6 +5208,43 @@ static void load_cl_kernels(ggml_backend_opencl_context *backend_ctx) { CL_CHECK(clReleaseProgram(prog)); GGML_LOG_CONT("."); } + + // repack + { +#ifdef GGML_OPENCL_EMBED_KERNELS + const std::string kernel_src { + #include "flash_attn_repack.cl.h" + }; +#else + const std::string kernel_src = read_file("flash_attn_repack.cl"); +#endif + cl_program prog = + build_program_from_source(backend_ctx, kernel_src.c_str(), compile_opts); + + CL_CHECK((backend_ctx->fa.kernel_repack_q_for_wmm = clCreateKernel(prog, "kernel_repack_q_for_wmm", &err), err)); + CL_CHECK((backend_ctx->fa.kernel_repack_k_for_wmm = clCreateKernel(prog, "kernel_repack_k_for_wmm", &err), err)); + CL_CHECK((backend_ctx->fa.kernel_repack_v_for_wmm = clCreateKernel(prog, "kernel_repack_v_for_wmm", &err), err)); + CL_CHECK((backend_ctx->fa.kernel_repack_mask_for_wmm = clCreateKernel(prog, "kernel_repack_mask_for_wmm", &err), err)); + GGML_LOG_CONT("."); + } + + // kernel_flash_attn_f32_f16_bin + { + size_t bin_size = 0; + backend_ctx->fa.kernel_flash_attn_f32_f16_bin = nullptr; + + if (use_adreno_bin_kernels(backend_ctx)) { + const char * kernel_bin = (const char *)backend_ctx->get_adreno_bin_kernel("flash_attn_f32_f16_wmm", &bin_size); + if (kernel_bin && bin_size > 0) { + cl_program prog = + build_program_from_binary(backend_ctx->context, backend_ctx->device, kernel_bin, CL_moe_compile_opts, bin_size); + + CL_CHECK((backend_ctx->fa.kernel_flash_attn_f32_f16_bin = clCreateKernel(prog, "flash_attn_f32_f16", &err), err)); + CL_CHECK(clReleaseProgram(prog)); + GGML_LOG_CONT("."); + } + } + } #endif // GGML_OPENCL_USE_ADRENO_KERNELS GGML_LOG_CONT("\n"); backend_ctx->kernels_loaded = true; @@ -5699,6 +6422,8 @@ static void ggml_opencl_print_backend_info(ggml_backend_opencl_device_context * auto * backend_ctx = dev_ctx->backend_ctx; + GGML_LOG_INFO("ggml_opencl: OpenCL device: %s\n", + backend_ctx->device_name.c_str()); GGML_LOG_INFO("ggml_opencl: OpenCL driver: %s\n", backend_ctx->driver_version.c_str()); GGML_LOG_INFO("ggml_opencl: vector subgroup broadcast support: %s\n", @@ -5715,11 +6440,11 @@ static void ggml_opencl_print_backend_info(ggml_backend_opencl_device_context * backend_ctx->global_mem_size/1024/1024); GGML_LOG_INFO("ggml_opencl: max mem alloc size: %zu MB\n", backend_ctx->max_alloc_size/1024/1024); - GGML_LOG_INFO("ggml_opencl: device max image buffer size (pixels): %lu\n", + GGML_LOG_INFO("ggml_opencl: device max image buffer size (pixels): %zu\n", backend_ctx->image_max_buffer_size); - GGML_LOG_INFO("ggml_opencl: device max image2d size: %lu x %lu\n", + GGML_LOG_INFO("ggml_opencl: device max image2d size: %zu x %zu\n", backend_ctx->image2d_max_width, backend_ctx->image2d_max_height); - GGML_LOG_INFO("ggml_opencl: device max workgroup size: %lu\n", + GGML_LOG_INFO("ggml_opencl: device max workgroup size: %zu\n", backend_ctx->max_workgroup_size); GGML_LOG_INFO("ggml_opencl: SVM coarse grain buffer support: %s\n", backend_ctx->svm_caps & CL_DEVICE_SVM_COARSE_GRAIN_BUFFER ? "true" : "false"); @@ -5894,6 +6619,16 @@ static ggml_backend_opencl_context * ggml_cl_init(ggml_backend_dev_t dev) { (backend_ctx->adreno_cl_compiler_version.type == E031 && backend_ctx->adreno_cl_compiler_version.major >= 47) || (backend_ctx->adreno_cl_compiler_version.type == DX && backend_ctx->adreno_cl_compiler_version.major >= 17); + // The q6_K flat mul_mat miscompile is a defect of the older E031 compilers, not a + // property of any GPU generation: it reproduces on E031.38 (Adreno 642L) and E031.41 + // (Adreno 740) and is fixed by E031.45 (Adreno 619). Gate on the compiler so parts + // that do not need the workarounds do not pay for them. The explicit type check is + // required: newer_than_or_same() is false for every non-E031 compiler, so negating it + // alone would enable the workarounds on E17/DX. + backend_ctx->q6_k_flat_old_compiler = + backend_ctx->adreno_cl_compiler_version.type == E031 && + !backend_ctx->adreno_cl_compiler_version.newer_than_or_same(E031, 45, 0, 0); + size_t ext_str_size; clGetDeviceInfo(device, CL_DEVICE_EXTENSIONS, 0, NULL, &ext_str_size); char *ext_buffer = (char *)alloca(ext_str_size + 1); @@ -5952,9 +6687,13 @@ static ggml_backend_opencl_context * ggml_cl_init(ggml_backend_dev_t dev) { } #ifdef GGML_OPENCL_USE_ADRENO_KERNELS - // determine whether to use Adreno xmem GEMM - backend_ctx->adreno_xmem_gemm_enabled = getenv("GGML_OPENCL_ADRENO_XMEM_GEMM") != nullptr && - backend_ctx->gpu_family == GPU_FAMILY::ADRENO; + // Adreno xmem F16xF32 GEMM, default on adreno, opt out with GGML_OPENCL_ADRENO_XMEM_GEMM=0. + // This helps models with f16 attention weights, e.g., gpt-oss-20b-f16 + { + const char * xmem_env = getenv("GGML_OPENCL_ADRENO_XMEM_GEMM"); + backend_ctx->adreno_xmem_gemm_enabled = backend_ctx->gpu_family == GPU_FAMILY::ADRENO && + (xmem_env ? atoi(xmem_env) != 0 : true); + } #endif // determine whether to use large buffer for Adreno @@ -5971,6 +6710,12 @@ static ggml_backend_opencl_context * ggml_cl_init(ggml_backend_dev_t dev) { backend_ctx->adreno_moe_ragged_skip_gran = (ragged_gran_env != NULL) ? atoi(ragged_gran_env) : 8; // whether fuse moe combine + static const char * fuse_moe_bias_glu_env = getenv("GGML_OPENCL_FUSE_MOE_BIAS_GLU"); + backend_ctx->fuse_moe_bias_glu = fuse_moe_bias_glu_env == NULL ? 1 : (atoi(fuse_moe_bias_glu_env) != 0); + + static const char * fuse_moe_bias_combine_env = getenv("GGML_OPENCL_FUSE_MOE_BIAS_COMBINE"); + backend_ctx->fuse_moe_bias_combine = fuse_moe_bias_combine_env == NULL ? 1 : (atoi(fuse_moe_bias_combine_env) != 0); + static const char * fuse_moe_combine_env = getenv("GGML_OPENCL_FUSE_MOE_COMBINE"); backend_ctx->fuse_moe_combine = fuse_moe_combine_env == NULL ? 1 : (atoi(fuse_moe_combine_env) != 0); @@ -6046,6 +6791,19 @@ static ggml_backend_opencl_context * ggml_cl_init(ggml_backend_dev_t dev) { #endif // GGML_OPENCL_USE_ADRENO_KERNELS backend_ctx->disable_fusion = getenv("GGML_OPENCL_DISABLE_FUSION") != nullptr; + if (const char * env = getenv("GGML_OPENCL_FUSE_MM_GLU")) { + backend_ctx->fuse_mm_glu = atoi(env) != 0; + } + if (const char * env = getenv("GGML_OPENCL_FUSE_RMS_ADD")) { + backend_ctx->fuse_rms_add = atoi(env) != 0; + } + if (const char * env = getenv("GGML_OPENCL_F16_MROW")) { + backend_ctx->f16_mrow = atoi(env) != 0; + } + if (const char * env = getenv("GGML_OPENCL_F16_MROW_RPT")) { + const int v = atoi(env); + backend_ctx->f16_mrow_rpt = (v == 2 || v == 4 || v == 8 || v == 16) ? v : 1; + } dev_ctx->backend_ctx = backend_ctx.release(); return dev_ctx->backend_ctx; @@ -6227,11 +6985,10 @@ struct ggml_tensor_extra_cl_q4_0 { CL_CHECK(clReleaseMemObject(q_img)); q_img = nullptr; } - // Currently, q_img and d_img are only initialized when SMALL_ALLOC is - // enabled. They point to the images in ggml_backend_opencl_buffer_context. - // So, there is no need to release them here. - // TODO: initialize them for non SMALL_PATH path, or remove them. - d_img = nullptr; + if (d_img != nullptr) { + CL_CHECK(clReleaseMemObject(d_img)); + d_img = nullptr; + } size_q = 0; size_d = 0; } @@ -6649,6 +7406,8 @@ struct ggml_tensor_extra_cl_q6_K { cl_mem ql_img = nullptr; // Upper 2 bits of quantized weights. cl_mem qh = nullptr; + // Upper 2 bits as image1d_buffer_t + cl_mem qh_img = nullptr; // Scales for each block. cl_mem s = nullptr; // Scales for each super block. @@ -6684,6 +7443,10 @@ struct ggml_tensor_extra_cl_q6_K { CL_CHECK(clReleaseMemObject(ql_img)); ql_img = nullptr; } + if (qh_img != nullptr) { + CL_CHECK(clReleaseMemObject(qh_img)); + qh_img = nullptr; + } size_ql = 0; size_qh = 0; @@ -6839,6 +7602,300 @@ static bool ggml_opencl_can_fuse_moe_combine(const struct ggml_cgraph * cgraph, return true; } +// Detect the gpt-oss MoE bias+activation epilogue on the PREFILL path: +// {MUL_MAT_ID(gate), ADD_ID(gate_bias), MUL_MAT_ID(up), ADD_ID(up_bias), GLU(swiglu_oai)}. +// The two matmuls still run as their own dispatches (the prefill GEMM is the vendor's); +// what collapses is the epilogue — both add_id passes are in-place read-modify-writes of a +// tensor the GLU immediately reads again, so they are three full passes over the same +// [n_ff, n_expert_used, n_tokens] f32 tensor where one suffices. +// +// The decode counterpart is handled by the mxfp4 fused GEMV arm in ggml_opencl_can_fuse, +// which folds the matmul too; this one deliberately fires only when that cannot (ne[2] > 1). +static bool ggml_opencl_can_fuse_moe_bias_glu(const struct ggml_cgraph * cgraph, int node_idx) { + if (node_idx + 4 >= cgraph->n_nodes) { + return false; + } + + const enum ggml_op mg_ops[] = { GGML_OP_MUL_MAT_ID, GGML_OP_ADD_ID, GGML_OP_MUL_MAT_ID, GGML_OP_ADD_ID, GGML_OP_GLU }; + const int mg_out[] = { node_idx + 4 }; + if (!ggml_can_fuse_subgraph(cgraph, node_idx, 5, mg_ops, mg_out, 1)) { + return false; + } + + const ggml_tensor * gmm = cgraph->nodes[node_idx]; + const ggml_tensor * gad = cgraph->nodes[node_idx+1]; + const ggml_tensor * umm = cgraph->nodes[node_idx+2]; + const ggml_tensor * uad = cgraph->nodes[node_idx+3]; + const ggml_tensor * glu = cgraph->nodes[node_idx+4]; + + if (ggml_get_glu_op(glu) != GGML_GLU_OP_SWIGLU_OAI) { + return false; + } + // Prefill only — at one token the mxfp4 arm above folds the matmul as well. + if (gmm->src[1]->ne[2] == 1) { + return false; + } + // Wiring: both matmuls share the activation and the expert selection, each add_id + // biases its own matmul, and the GLU consumes the two biased results as separate + // operands (so the same-buffer ne00_off/ne10_off split path is not in play). + if (gad->src[0] != gmm || uad->src[0] != umm || + glu->src[0] != gad || glu->src[1] != uad || + umm->src[1] != gmm->src[1] || umm->src[2] != gmm->src[2]) { + return false; + } + // A swapped GLU would exchange the gate/up roles the fused kernel hard-codes. + if (ggml_get_op_params_i32(glu, 1)) { + return false; + } + if (gad->type != GGML_TYPE_F32 || uad->type != GGML_TYPE_F32 || glu->type != GGML_TYPE_F32) { + return false; + } + if (!gad->src[1] || gad->src[1]->type != GGML_TYPE_F32 || + !uad->src[1] || uad->src[1]->type != GGML_TYPE_F32) { + return false; + } + if (!gad->src[2] || gad->src[2]->type != GGML_TYPE_I32 || uad->src[2] != gad->src[2]) { + return false; + } + // Full width on both operands: the kernel writes one output element per input pair. + if (!ggml_are_same_shape(gad, uad) || glu->ne[0] != gad->ne[0] || + glu->ne[1] != gad->ne[1] || glu->ne[2] != gad->ne[2] || glu->ne[3] != gad->ne[3]) { + return false; + } + if (gad->ne[3] != 1) { + return false; + } + // The destination is addressed by (expert slot, token) rather than the GLU's flat row + // walk; those agree only for a contiguous destination. + if (!ggml_is_contiguous(glu) || !ggml_is_contiguous(gmm) || !ggml_is_contiguous(umm)) { + return false; + } + return true; +} + +static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); + +// Runs the gate and up matmuls unchanged, then one kernel in place of +// add_id(gate) + add_id(up) + swiglu_oai. See ggml_opencl_can_fuse_moe_bias_glu. +static void ggml_cl_moe_bias_glu_fused(ggml_backend_t backend, ggml_tensor * gate_mm, const ggml_tensor * gate_add, + ggml_tensor * up_mm, const ggml_tensor * up_add, const ggml_tensor * glu) { + ggml_backend_opencl_context * backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_cl_mul_mat_id(backend, gate_mm->src[0], gate_mm->src[1], gate_mm); + ggml_cl_mul_mat_id(backend, up_mm->src[0], up_mm->src[1], up_mm); + + const ggml_tensor * gbias = gate_add->src[1]; + const ggml_tensor * ubias = up_add->src[1]; + const ggml_tensor * ids = gate_add->src[2]; + + ggml_tensor_extra_cl * eg = (ggml_tensor_extra_cl *)gate_mm->extra; + ggml_tensor_extra_cl * egb = (ggml_tensor_extra_cl *)gbias->extra; + ggml_tensor_extra_cl * eu = (ggml_tensor_extra_cl *)up_mm->extra; + ggml_tensor_extra_cl * eub = (ggml_tensor_extra_cl *)ubias->extra; + ggml_tensor_extra_cl * ei = (ggml_tensor_extra_cl *)ids->extra; + ggml_tensor_extra_cl * ed = (ggml_tensor_extra_cl *)glu->extra; + + cl_ulong off_g = eg->offset + gate_mm->view_offs; + cl_ulong off_gb = egb->offset + gbias->view_offs; + cl_ulong off_u = eu->offset + up_mm->view_offs; + cl_ulong off_ub = eub->offset + ubias->view_offs; + cl_ulong off_i = ei->offset + ids->view_offs; + cl_ulong off_d = ed->offset + glu->view_offs; + + const cl_ulong nb01_g = gate_mm->nb[1]; + const cl_ulong nb02_g = gate_mm->nb[2]; + const cl_ulong nb01_u = up_mm->nb[1]; + const cl_ulong nb02_u = up_mm->nb[2]; + const cl_ulong nb11_g = gbias->nb[1]; + const cl_ulong nb11_u = ubias->nb[1]; + const cl_ulong nb21 = ids->nb[1]; + const cl_ulong nbd1 = glu->nb[1]; + const cl_ulong nbd2 = glu->nb[2]; + + const int ne0 = (int)glu->ne[0]; + const float alpha = ggml_get_op_params_f32(glu, 2); + const float limit = ggml_get_op_params_f32(glu, 3); + + cl_kernel kernel = backend_ctx->kernel_add_id_add_id_swiglu_oai; + + int i = 0; + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &eg->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_g)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &egb->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_gb)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &eu->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_u)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &eub->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_ub)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &ei->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_i)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_mem), &ed->data_device)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &off_d)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb01_g)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb02_g)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb01_u)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb02_u)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb11_g)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb11_u)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nb21)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nbd1)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(cl_ulong), &nbd2)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(int), &ne0)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(float), &limit)); + CL_CHECK(clSetKernelArg(kernel, i++, sizeof(float), &alpha)); + + const int nth = MIN(ne0, (int) backend_ctx->get_kernel_workgroup_size(kernel)); + size_t global_work_size[] = { (size_t)glu->ne[1]*nth, (size_t)glu->ne[2], 1 }; + size_t local_work_size[] = { (size_t)nth, 1, 1 }; + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, glu); +} + +// Fusion B: the MoE down-projection bias add feeding the combine. +// +// The graph runs ADD_ID(down_bias) and then immediately the combine subgraph +// {MUL(router weights), k VIEWs, k-1 ADDs}, and the ADD_ID's only consumer is that +// MUL. Since the ADD_ID is an in-place read-modify-write of a tensor the combine +// reads once more, the bias can be added inside the combine instead, dropping a +// full pass over [n_embd, k, n_tokens]. +// +// Shape checks for the combine tail are delegated to ggml_opencl_can_fuse_moe_combine +// (which also owns the n_nodes >= 32 bail and the experts/dst aliasing bail); what is +// added here is the ADD_ID wiring plus a subgraph check over the WHOLE run, so that +// the intermediate bias result is confirmed not to escape. +static bool ggml_opencl_can_fuse_moe_bias_combine(const struct ggml_cgraph * cgraph, int node_idx, + const ggml_tensor ** out_final_add) { + if (node_idx + 1 >= cgraph->n_nodes) { + return false; + } + const ggml_tensor * add = cgraph->nodes[node_idx]; + if (add->op != GGML_OP_ADD_ID) { + return false; + } + const ggml_tensor * mul = cgraph->nodes[node_idx+1]; + if (mul->op != GGML_OP_MUL || mul->src[0] != add) { + return false; + } + + const ggml_tensor * final_add = NULL; + if (!ggml_opencl_can_fuse_moe_combine(cgraph, node_idx+1, &final_add)) { + return false; + } + + const ggml_tensor * raw = add->src[0]; + const ggml_tensor * bias = add->src[1]; + const ggml_tensor * ids = add->src[2]; + if (!raw || !bias || !ids) { + return false; + } + if (raw->type != GGML_TYPE_F32 || bias->type != GGML_TYPE_F32 || + ids->type != GGML_TYPE_I32 || add->type != GGML_TYPE_F32) { + return false; + } + // The combine reads the raw matmul output with the strides it computed from the + // add_id result, so the two must have the same layout. + if (!ggml_are_same_shape(raw, add) || !ggml_is_contiguous(raw)) { + return false; + } + if (raw->nb[1] != add->nb[1] || raw->nb[2] != add->nb[2]) { + return false; + } + // ids is indexed as [expert slot, token]; the combine walks the same two axes. + if (ids->ne[0] < add->ne[1] || ids->ne[1] < add->ne[2]) { + return false; + } + + // Whole-run escape check: ADD_ID + MUL + k VIEWs + (k-1) ADDs, only the last node escapes. + const int k = (int)add->ne[1]; + const int n_nodes = 2 + k + (k - 1); + if (n_nodes >= 32 || node_idx + n_nodes > cgraph->n_nodes) { + return false; + } + enum ggml_op ops[32]; + int n = 0; + ops[n++] = GGML_OP_ADD_ID; + ops[n++] = GGML_OP_MUL; + for (int j = 0; j < k; ++j) ops[n++] = GGML_OP_VIEW; + for (int j = 0; j < k - 1; ++j) ops[n++] = GGML_OP_ADD; + const int outs[] = { node_idx + n_nodes - 1 }; + if (!ggml_can_fuse_subgraph(cgraph, node_idx, n_nodes, ops, outs, 1)) { + return false; + } + + *out_final_add = final_add; + return true; +} + + +// Fusion B dispatch: the combine, reading the RAW matmul output and adding the +// per-expert bias row inline. See ggml_opencl_can_fuse_moe_bias_combine. +static void ggml_cl_moe_bias_combine_fused(ggml_backend_t backend, const ggml_tensor * add, + const ggml_tensor * mul, const ggml_tensor * dst) { + ggml_backend_opencl_context * backend_ctx = (ggml_backend_opencl_context *)backend->context; + + const ggml_tensor * experts = add->src[0]; // raw matmul output, bias not yet applied + const ggml_tensor * bias = add->src[1]; + const ggml_tensor * ids = add->src[2]; + const ggml_tensor * weights = mul->src[1]; + + ggml_tensor_extra_cl * ee = (ggml_tensor_extra_cl *)experts->extra; + ggml_tensor_extra_cl * eb = (ggml_tensor_extra_cl *)bias->extra; + ggml_tensor_extra_cl * ei = (ggml_tensor_extra_cl *)ids->extra; + ggml_tensor_extra_cl * ew = (ggml_tensor_extra_cl *)weights->extra; + ggml_tensor_extra_cl * ed = (ggml_tensor_extra_cl *)dst->extra; + cl_ulong off_e = ee->offset + experts->view_offs; + cl_ulong off_b = eb->offset + bias->view_offs; + cl_ulong off_i = ei->offset + ids->view_offs; + cl_ulong off_w = ew->offset + weights->view_offs; + cl_ulong off_d = ed->offset + dst->view_offs; + + const int n_embd4 = (int)(experts->ne[0] / 4); + const int k = (int)experts->ne[1]; + const int nt = (int)experts->ne[2]; + const cl_uint e1 = (cl_uint)(experts->nb[1] / sizeof(float)); + const cl_uint e2 = (cl_uint)(experts->nb[2] / sizeof(float)); + const cl_uint w1 = (cl_uint)(weights->nb[1] / sizeof(float)); + const cl_uint w2 = (cl_uint)(weights->nb[2] / sizeof(float)); + const cl_uint d1 = (cl_uint)(dst->nb[1] / sizeof(float)); + const cl_ulong nb_b1 = bias->nb[1]; + const cl_ulong nb_i1 = ids->nb[1]; + + const size_t w_bytes = ggml_nbytes(weights); + backend_ctx->prealloc_moe_combine_w.allocate(backend_ctx->context, w_bytes); + CL_CHECK(clEnqueueCopyBuffer(backend_ctx->queue, ew->data_device, backend_ctx->prealloc_moe_combine_w.buffer, + off_w, 0, w_bytes, 0, NULL, NULL)); + cl_mem w_dev = backend_ctx->prealloc_moe_combine_w.buffer; + cl_ulong w_off = 0; + + cl_kernel kernel = backend_ctx->kernel_moe_combine_bias_f32; + int a = 0; + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_mem), &ee->data_device)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &off_e)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_mem), &w_dev)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &w_off)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_mem), &eb->data_device)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &off_b)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_mem), &ei->data_device)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &off_i)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_mem), &ed->data_device)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &off_d)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(int), &n_embd4)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(int), &k)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(int), &nt)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_uint), &e1)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_uint), &e2)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_uint), &w1)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_uint), &w2)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_uint), &d1)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &nb_b1)); + CL_CHECK(clSetKernelArg(kernel, a++, sizeof(cl_ulong), &nb_i1)); + + size_t lws[2] = { 64, 1 }; + size_t gws[2] = { (size_t)(((n_embd4 + 63) / 64) * 64), (size_t)nt }; + backend_ctx->enqueue_ndrange_kernel(kernel, 2, gws, lws, dst); +} + + static void ggml_cl_moe_combine_fused(ggml_backend_t backend, const ggml_tensor * mul, const ggml_tensor * dst) { ggml_backend_opencl_context * backend_ctx = (ggml_backend_opencl_context *)backend->context; const ggml_tensor * experts = mul->src[0]; @@ -6891,7 +7948,78 @@ static void ggml_cl_moe_combine_fused(ggml_backend_t backend, const ggml_tensor backend_ctx->enqueue_ndrange_kernel(kernel, 2, gws, lws, dst); } -static bool ggml_opencl_can_fuse(const struct ggml_cgraph * cgraph, int node_idx, std::initializer_list ops) { +inline bool use_q4k_tiled(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor); // defined below (used by the GLU-subgraph fuse check) +inline bool use_q4_k_bin_kernels(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor); +inline bool use_adreno_kernels(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor); // defined below + +static bool ggml_opencl_can_fuse(const ggml_backend_opencl_context * backend_ctx, const struct ggml_cgraph * cgraph, int node_idx, std::initializer_list ops) { + + // glu(mul_mat(Wg,x), mul_mat(Wu,x)) — the FFN gate/up GEMVs + GLU. This is a + // non-linear subgraph (up does NOT consume gate), so the contiguous + // ggml_can_fuse below rejects it; use ggml_can_fuse_subgraph with the glu as + // the sole output and validate the edges explicitly. q4_K decode only; + // byte-identical to the per-op path. + if (ops.size() == 3 && ops.begin()[0] == GGML_OP_MUL_MAT && + ops.begin()[1] == GGML_OP_MUL_MAT && ops.begin()[2] == GGML_OP_GLU) { + const enum ggml_op glu_ops[] = { GGML_OP_MUL_MAT, GGML_OP_MUL_MAT, GGML_OP_GLU }; + const int glu_out[] = { node_idx + 2 }; + if (!ggml_can_fuse_subgraph(cgraph, node_idx, 3, glu_ops, glu_out, 1)) { + return false; + } + + const ggml_tensor *gate = cgraph->nodes[node_idx]; + const ggml_tensor *up = cgraph->nodes[node_idx+1]; + const ggml_tensor *glu = cgraph->nodes[node_idx+2]; + + // decode GEMV path only (single token); prefill GEMM is separate + if (gate->ne[1] != 1 || up->ne[1] != 1) { + return false; + } + // both projections must be q4_K weights, f32 activation/output + if (gate->src[0]->type != GGML_TYPE_Q4_K || up->src[0]->type != GGML_TYPE_Q4_K || + gate->src[1]->type != GGML_TYPE_F32 || up->src[1]->type != GGML_TYPE_F32 || + gate->type != GGML_TYPE_F32 || up->type != GGML_TYPE_F32 || glu->type != GGML_TYPE_F32) { + return false; + } + // gate and up must share the same activation and have matching shape/stride + if (gate->src[1] != up->src[1] || + !ggml_are_same_shape(gate->src[0], up->src[0]) || + !ggml_are_same_stride(gate->src[0], up->src[0])) { + return false; + } + // GLU must read gate as src[0] and up as src[1], no swap (the fused + // epilogue applies the activation to gate, multiplies by up) + if (glu->src[0] != gate || glu->src[1] != up) { + return false; + } + if (ggml_get_op_params_i32(glu, 1) /* swapped */) { + return false; + } + // SWIGLU_OAI carries extra alpha/limit params -> not handled by the fused kernel + if (ggml_get_glu_op(glu) == GGML_GLU_OP_SWIGLU_OAI) { + return false; + } + // the fused kernel reads the standard noshuffle image layout; the tiled + // layout packs weights differently -> defer those to the per-op path + if (use_q4k_tiled(backend_ctx, gate->src[0]) || use_q4k_tiled(backend_ctx, up->src[0])) { + return false; + } + // q4_K bin kernel requires 32b transposed layout, not compatible with the fused gemv + if (use_q4_k_bin_kernels(backend_ctx, gate->src[0]) || use_q4_k_bin_kernels(backend_ctx, up->src[0])) { + return false; + } + // that noshuffle layout is only produced at set_tensor time when + // use_adreno_kernels() accepts the weight (ne0 >= 512 && ne1 >= 512). + // Smaller weights stay in the plain q4_K layout, which this kernel would + // misread -> defer them to the per-op path. Real FFN gate/up weights are + // far above the threshold, so production dispatch is unchanged. + if (!use_adreno_kernels(backend_ctx, gate->src[0]) || + !use_adreno_kernels(backend_ctx, up->src[0])) { + return false; + } + return true; + } + if (!ggml_can_fuse(cgraph, node_idx, ops)) { return false; } @@ -6939,10 +8067,42 @@ static bool ggml_opencl_can_fuse(const struct ggml_cgraph * cgraph, int node_idx if (!ggml_is_contiguous(norm->src[0]) || !ggml_is_contiguous(w) || !ggml_is_contiguous(b)) { return false; } - } else if (ops.size() == 3 && ops.begin()[0] == GGML_OP_GROUP_NORM && ops.begin()[1] == GGML_OP_MUL && ops.begin()[2] == GGML_OP_ADD) { - const ggml_tensor *gn = cgraph->nodes[node_idx]; - const ggml_tensor *mul = cgraph->nodes[node_idx+1]; - const ggml_tensor *add = cgraph->nodes[node_idx+2]; + } else if (ops.size() == 3 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL && ops.begin()[2] == GGML_OP_ADD) { + // rms_norm(x) * w + b, fused (residual). Mirrors the RMS_NORM+MUL gate + // plus the residual-add operand's constraints. + const ggml_tensor *rms_norm = cgraph->nodes[node_idx]; + const ggml_tensor *mul = cgraph->nodes[node_idx+1]; + const ggml_tensor *add = cgraph->nodes[node_idx+2]; + const ggml_tensor *w = mul->src[0] == rms_norm ? mul->src[1] : mul->src[0]; + const ggml_tensor *b = add->src[0] == mul ? add->src[1] : add->src[0]; + + GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); + GGML_ASSERT(rms_norm->type == GGML_TYPE_F32); + + if (w->type != GGML_TYPE_F32 || mul->type != GGML_TYPE_F32 || + b->type != GGML_TYPE_F32 || add->type != GGML_TYPE_F32) { + return false; + } + if (rms_norm->src[0]->ne[0] % 4 != 0) { + return false; + } + // if rms_norm is the B operand of mul, broadcast is not handled + if (rms_norm == mul->src[1] && !ggml_are_same_shape(mul->src[0], rms_norm)) { + return false; + } + // the residual must match the normed output shape (no add broadcast) + if (!ggml_are_same_shape(b, add)) { + return false; + } + // rms_norm assumes contiguous rows + if (!ggml_is_contiguous_rows(mul->src[0]) || !ggml_is_contiguous_rows(mul->src[1]) || + !ggml_is_contiguous_rows(b)) { + return false; + } + } else if (ops.size() == 3 && ops.begin()[0] == GGML_OP_GROUP_NORM && ops.begin()[1] == GGML_OP_MUL && ops.begin()[2] == GGML_OP_ADD) { + const ggml_tensor *gn = cgraph->nodes[node_idx]; + const ggml_tensor *mul = cgraph->nodes[node_idx+1]; + const ggml_tensor *add = cgraph->nodes[node_idx+2]; const ggml_tensor *w = mul->src[0] == gn ? mul->src[1] : mul->src[0]; const ggml_tensor *b = add->src[0] == mul ? add->src[1] : add->src[0]; @@ -6962,6 +8122,215 @@ static void ggml_opencl_op_rms_norm_fused(ggml_backend_t backend, ggml_tensor * static void ggml_opencl_op_norm_fused(ggml_backend_t backend, ggml_tensor * norm_tensor, ggml_tensor * mul_tensor, ggml_tensor * add_tensor); static void ggml_opencl_op_group_norm_fused(ggml_backend_t backend, ggml_tensor * gn_tensor, ggml_tensor * mul_tensor, ggml_tensor * add_tensor); +static void ggml_cl_mul_mat_q4_k_glu_fused(ggml_backend_t backend, ggml_tensor * gate_tensor, ggml_tensor * up_tensor, ggml_tensor * glu_tensor) { +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + GGML_ASSERT(gate_tensor && up_tensor && glu_tensor); + + const ggml_tensor * Wg = gate_tensor->src[0]; + const ggml_tensor * Wu = up_tensor->src[0]; + const ggml_tensor * src1 = gate_tensor->src[1]; // == up_tensor->src[1] + const ggml_tensor * dst = glu_tensor; + + GGML_ASSERT(Wg && Wg->extra); + GGML_ASSERT(Wu && Wu->extra); + GGML_ASSERT(src1 && src1->extra); + GGML_ASSERT(dst && dst->extra); + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *)src1->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + ggml_tensor_extra_cl_q4_K * extra_g = (ggml_tensor_extra_cl_q4_K *)Wg->extra; + ggml_tensor_extra_cl_q4_K * extra_u = (ggml_tensor_extra_cl_q4_K *)Wu->extra; + + cl_ulong offset1 = extra1->offset + src1->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + const int K = Wg->ne[0]; // ne00 + const int M = Wg->ne[1]; // ne01 (= ffn intermediate width) + const int N = 1; // decode GEMV + + const cl_uchar mask_d6 = 0x3F, mask_d4 = 0x0F, mask_hi2 = 0xC0; + const int glu_op = (int)ggml_get_glu_op(dst); + + cl_context context = backend_ctx->context; + cl_int err; + cl_image_format img_fmt; + cl_image_desc img_desc; + cl_buffer_region region; + + // q images for the two weight matrices (standard noshuffle layout) + img_fmt = { CL_R, CL_UNSIGNED_INT32 }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * K / 2 / 4; + img_desc.buffer = extra_g->q; + cl_mem qg_img = nullptr, qu_img = nullptr; + CL_CHECK((qg_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + img_desc.buffer = extra_u->q; + CL_CHECK((qu_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + // shared activation image (one column at decode) + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + cl_mem b_sub_buf = nullptr, b_img = nullptr; + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + img_fmt = { CL_RGBA, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N / 4; + img_desc.buffer = b_sub_buf; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + cl_kernel kernel = backend_ctx->kernel_gemv_noshuffle_q4_k_f32_glu; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &qg_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra_g->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra_g->dm)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra_g->s)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &qu_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extra_u->d)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_mem), &extra_u->dm)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_mem), &extra_u->s)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_int), &K)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_int), &M)); + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_int), &glu_op)); + CL_CHECK(clSetKernelArg(kernel, 14, sizeof(cl_uchar), &mask_d6)); + CL_CHECK(clSetKernelArg(kernel, 15, sizeof(cl_uchar), &mask_d4)); + CL_CHECK(clSetKernelArg(kernel, 16, sizeof(cl_uchar), &mask_hi2)); + + // K-split = nsg_y subgroups. HARD-CAP at 8 (512 work-items): the fused + // kernel's cross-subgroup reduce uses a float4 reduceLM (gate+up packed) = + // 2x the LDS of the base GEMV's float2 reduce, so 16 co-resident subgroups + // exceed the per-CU LDS budget on X2 and the WG barrier DEADLOCKS -> GPU TDR + // (reproduced on upstream gemma-4 E4B decode, K=2560 M=10240). This used to + // be masked: get_kernel_workgroup_size reported 896 for this kernel (so the + // cap loop fell to 8), but it now returns 1024 and the Adreno per-kernel WG + // query is unreliable (over-reports), so cap explicitly instead of trusting + // it. nsg_y < 16 also means the cross-subgroup accumulation grouping differs + // from the standalone wide (nsg=16) GEMV, so the output is coherent but NOT + // byte-identical to the per-op path. Keep the maxwg query as a further floor + // for any driver that reports < 512. + size_t maxwg = backend_ctx->get_kernel_workgroup_size(kernel); + size_t nsg_y = 8; + while (nsg_y > 1 && 64 * nsg_y > maxwg) { nsg_y >>= 1; } + size_t local_work_size[3] = { 64, nsg_y, 1 }; + size_t global_work_size[3] = { (size_t)CEIL_DIV(M / 2, 64) * 64, nsg_y, 1 }; + + if (getenv("GGML_OPENCL_FUSE_DEBUG")) { + static int dbg = 0; + if (dbg < 3) { fprintf(stderr, "[FUSE_MM_GLU] fired #%d K=%d M=%d glu_op=%d nsg=%zu maxwg=%zu\n", ++dbg, K, M, glu_op, nsg_y, maxwg); fflush(stderr); } + } + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(qg_img)); + CL_CHECK(clReleaseMemObject(qu_img)); + CL_CHECK(clReleaseMemObject(b_img)); + CL_CHECK(clReleaseMemObject(b_sub_buf)); +#else + GGML_UNUSED(backend); + GGML_UNUSED(gate_tensor); + GGML_UNUSED(up_tensor); + GGML_UNUSED(glu_tensor); +#endif +} + + +static void ggml_opencl_op_rms_norm_mul_add_fused(ggml_backend_t backend, ggml_tensor * rms_norm_tensor, ggml_tensor * mul_tensor, ggml_tensor * add_tensor) { + GGML_ASSERT(rms_norm_tensor && mul_tensor && add_tensor); + + const ggml_tensor * src0 = rms_norm_tensor->src[0]; + const ggml_tensor * src1 = mul_tensor->src[0] == rms_norm_tensor ? mul_tensor->src[1] : mul_tensor->src[0]; + const ggml_tensor * src2 = add_tensor->src[0] == mul_tensor ? add_tensor->src[1] : add_tensor->src[0]; + const ggml_tensor * dst = add_tensor; + + GGML_ASSERT(src0 && src0->extra); + GGML_ASSERT(src1 && src1->extra); + GGML_ASSERT(src2 && src2->extra); + GGML_ASSERT(dst && dst->extra); + + ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *)src0->extra; + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *)src1->extra; + ggml_tensor_extra_cl * extra2 = (ggml_tensor_extra_cl *)src2->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + + cl_ulong offset0 = extra0->offset + src0->view_offs; + cl_ulong offset1 = extra1->offset + src1->view_offs; + cl_ulong offset2 = extra2->offset + src2->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + float eps; + memcpy(&eps, rms_norm_tensor->op_params, sizeof(float)); + + const int ne00 = src0->ne[0], ne01 = src0->ne[1], ne02 = src0->ne[2], ne03 = src0->ne[3]; + const cl_ulong nb01 = src0->nb[1], nb02 = src0->nb[2], nb03 = src0->nb[3]; + const int ne10 = src1->ne[0], ne11 = src1->ne[1], ne12 = src1->ne[2], ne13 = src1->ne[3]; + const cl_ulong nb11 = src1->nb[1], nb12 = src1->nb[2], nb13 = src1->nb[3]; + const int ne20 = src2->ne[0], ne21 = src2->ne[1], ne22 = src2->ne[2], ne23 = src2->ne[3]; + const cl_ulong nb21 = src2->nb[1], nb22 = src2->nb[2], nb23 = src2->nb[3]; + const cl_ulong nb1 = dst->nb[1], nb2 = dst->nb[2], nb3 = dst->nb[3]; + + GGML_ASSERT(ne00 % 4 == 0); + + size_t sgs; + if (backend_ctx->gpu_family == ADRENO) sgs = 64; + else if (backend_ctx->gpu_family == INTEL) sgs = 32; + else GGML_ASSERT(false && "Unsupported GPU"); + + cl_kernel kernel = backend_ctx->kernel_rms_norm_mul_add; + + int nth = sgs; + int max_workgroup_size = backend_ctx->get_kernel_workgroup_size(kernel); + while (nth < ne00 && nth < max_workgroup_size) nth *= 2; + nth = MIN(nth, max_workgroup_size); + nth = MIN(nth, ne00); + + size_t global_work_size[] = {(size_t)ne01*nth, (size_t)ne02, (size_t)ne03}; + size_t local_work_size[] = {(size_t)nth, 1, 1}; + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra1->data_device)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &extra2->data_device)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &offset2)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(int), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(int), &ne02)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(int), &ne03)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_ulong), &nb01)); + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_ulong), &nb02)); + CL_CHECK(clSetKernelArg(kernel, 14, sizeof(cl_ulong), &nb03)); + CL_CHECK(clSetKernelArg(kernel, 15, sizeof(int), &ne10)); + CL_CHECK(clSetKernelArg(kernel, 16, sizeof(int), &ne11)); + CL_CHECK(clSetKernelArg(kernel, 17, sizeof(int), &ne12)); + CL_CHECK(clSetKernelArg(kernel, 18, sizeof(int), &ne13)); + CL_CHECK(clSetKernelArg(kernel, 19, sizeof(cl_ulong), &nb11)); + CL_CHECK(clSetKernelArg(kernel, 20, sizeof(cl_ulong), &nb12)); + CL_CHECK(clSetKernelArg(kernel, 21, sizeof(cl_ulong), &nb13)); + CL_CHECK(clSetKernelArg(kernel, 22, sizeof(int), &ne20)); + CL_CHECK(clSetKernelArg(kernel, 23, sizeof(int), &ne21)); + CL_CHECK(clSetKernelArg(kernel, 24, sizeof(int), &ne22)); + CL_CHECK(clSetKernelArg(kernel, 25, sizeof(int), &ne23)); + CL_CHECK(clSetKernelArg(kernel, 26, sizeof(cl_ulong), &nb21)); + CL_CHECK(clSetKernelArg(kernel, 27, sizeof(cl_ulong), &nb22)); + CL_CHECK(clSetKernelArg(kernel, 28, sizeof(cl_ulong), &nb23)); + CL_CHECK(clSetKernelArg(kernel, 29, sizeof(cl_ulong), &nb1)); + CL_CHECK(clSetKernelArg(kernel, 30, sizeof(cl_ulong), &nb2)); + CL_CHECK(clSetKernelArg(kernel, 31, sizeof(cl_ulong), &nb3)); + CL_CHECK(clSetKernelArg(kernel, 32, sizeof(float), &eps)); + CL_CHECK(clSetKernelArg(kernel, 33, sizeof(float)*sgs, NULL)); + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); +} + static ggml_status ggml_backend_opencl_graph_compute(ggml_backend_t backend, ggml_cgraph * cgraph) { ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; @@ -6981,18 +8350,43 @@ static ggml_status ggml_backend_opencl_graph_compute(ggml_backend_t backend, ggm continue; } - if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD })) { + if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(backend_ctx, cgraph, i, { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD })) { ggml_opencl_op_norm_fused(backend, node, cgraph->nodes[i+1], cgraph->nodes[i+2]); i += 2; continue; } - if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(cgraph, i, { GGML_OP_GROUP_NORM, GGML_OP_MUL, GGML_OP_ADD })) { + if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(backend_ctx, cgraph, i, { GGML_OP_GROUP_NORM, GGML_OP_MUL, GGML_OP_ADD })) { ggml_opencl_op_group_norm_fused(backend, node, cgraph->nodes[i+1], cgraph->nodes[i+2]); i += 2; continue; } // Fuse the MoE combine: router-weight mul + cross-expert add chain -> // one weighted-sum-across-experts kernel. + // Fold the gpt-oss MoE bias epilogue: add_id(gate_bias) + add_id(up_bias) + + // glu(swiglu_oai) -> one kernel, leaving the two matmuls as their own dispatches. + // Both add_ids are in-place passes over a tensor the GLU reads again, so this + // drops two full read+write passes per layer. Opt out GGML_OPENCL_FUSE_MOE_BIAS_GLU=0. + if (backend_ctx->fuse_moe_bias_glu && !backend_ctx->disable_fusion && + ggml_opencl_can_fuse_moe_bias_glu(cgraph, i)) { + ggml_cl_moe_bias_glu_fused(backend, node, cgraph->nodes[i+1], cgraph->nodes[i+2], + cgraph->nodes[i+3], cgraph->nodes[i+4]); + i += 4; + continue; + } + + // Fold the MoE down-projection bias into the combine: add_id(down_bias) + the whole + // combine subgraph -> one kernel. Checked before the plain combine arm so the longer + // pattern wins. Opt out GGML_OPENCL_FUSE_MOE_BIAS_COMBINE=0. + if (backend_ctx->fuse_moe_bias_combine && backend_ctx->fuse_moe_combine && + !backend_ctx->disable_fusion) { + const ggml_tensor * bias_combine_out = nullptr; + if (ggml_opencl_can_fuse_moe_bias_combine(cgraph, i, &bias_combine_out)) { + ggml_cl_moe_bias_combine_fused(backend, node, cgraph->nodes[i+1], bias_combine_out); + i += 2 * (int)node->ne[1]; // ADD_ID + MUL + k VIEWs + (k-1) ADDs + continue; + } + } + if (backend_ctx->fuse_moe_combine && !backend_ctx->disable_fusion) { const ggml_tensor * combine_out = nullptr; if (ggml_opencl_can_fuse_moe_combine(cgraph, i, &combine_out)) { @@ -7002,11 +8396,35 @@ static ggml_status ggml_backend_opencl_graph_compute(ggml_backend_t backend, ggm } } - if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) { + // Fuse rms_norm + mul(weight) + add(residual). Checked before the + // rms_norm+mul fuse so the 3-op pattern wins over its 2-op prefix. + // Default on, opt-out GGML_OPENCL_FUSE_RMS_ADD=0. + if (!backend_ctx->disable_fusion && backend_ctx->fuse_rms_add && + ggml_opencl_can_fuse(backend_ctx, cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD })) { + ggml_opencl_op_rms_norm_mul_add_fused(backend, node, cgraph->nodes[i+1], cgraph->nodes[i+2]); + i += 2; + continue; + } + if (!backend_ctx->disable_fusion && ggml_opencl_can_fuse(backend_ctx, cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) { ggml_opencl_op_rms_norm_fused(backend, node, cgraph->nodes[i+1]); i++; continue; } + // Fuse mul_mat(Wg,x) + mul_mat(Wu,x) + glu — fold the FFN's two decode + // GEMVs and the GLU into one dispatch. q4_K only (guarded below); the + // fused kernel uses the same accumulation/reduction order and the same + // scalar GLU formula -> coherent. Default on, opt-out GGML_OPENCL_FUSE_MM_GLU=0. +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // The fused executor (ggml_cl_mul_mat_q4_k_glu_fused) is image-path / + // Adreno-only (GGML_ABORT on the non-Adreno #else); gate the dispatch to + // match so the FFN GLU subgraph stays dormant on Intel/other drivers. + if (backend_ctx->fuse_mm_glu && !backend_ctx->disable_fusion && + ggml_opencl_can_fuse(backend_ctx, cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_MUL_MAT, GGML_OP_GLU })) { + ggml_cl_mul_mat_q4_k_glu_fused(backend, node, cgraph->nodes[i+1], cgraph->nodes[i+2]); + i += 2; + continue; + } +#endif bool ok = ggml_cl_compute_forward(backend, node); if (!ok) { @@ -7031,9 +8449,20 @@ inline bool use_adreno_kernels(const ggml_backend_opencl_context *backend_ctx, c bool threashold_ok = tensor->ne[0] >= threshold_ne0 && tensor->ne[1] >= threshold_ne1 && tensor->ne[2] == 1 && tensor->ne[3] == 1; - // q6_K adreno kernels requires ne1 is multiple of 128 - if (tensor->type == GGML_TYPE_Q6_K) { - return threashold_ok && tensor->ne[1] % 128 == 0; + // The noshuffle layout packs 2 rows per 32-bit texel and the GEMV reads it at an + // ne1/2 texel stride with an exact-cover dispatch, so it is only addressable when + // ne1 is a multiple of 64; an unaligned ne1 truncates the stride and the weight is + // read misaligned. That is a property of the layout, not of one quant -- q4_K, q5_K + // and q8_0 read the same packing as q6_K. The bound is 64, not 128: a q8_0 attention + // weight of ne1 = 2880 is a multiple of 64 but not 128 and is correct. + switch (tensor->type) { + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_Q8_0: + return threashold_ok && tensor->ne[1] % 64 == 0; + default: + break; } return threashold_ok; } @@ -7067,6 +8496,72 @@ inline bool use_adreno_moe_kernels(const ggml_backend_opencl_context *backend_ct return (((strstr(tensor->name, "ffn") != NULL) && (strstr(tensor->name, "exps") != NULL)) || (strstr(tensor->name, "as") != NULL)) && (ne01 % 32 == 0); } +// Device default for the tiled-wide lm_head/embed GEMV layout: ON for X2E and A8X. +// +// These kernels were previously off everywhere on the grounds that they compute +// wrong values at multi-superblock K. They do not: that NMSE ~2 came from the +// backend having no get_tensor restore path for the tiled layout, so +// test-backend-ops (which builds its CPU reference by copying the weights back +// out of the backend) compared a correct GPU result against a reference +// dequantized from tiled bytes. With the restore path added, MUL_MAT passes with +// the tiled kernels on, unmodified, on both devices. +// +// Perf, Qwen3-4B-Q4_K_M (q6_K lm_head 151936x2560), tg128, matched pairs with +// alternating lead, tiled vs o4: +// +// A8X +11.9% 6/6 pairs positive, order bias -0.06% (16.93 vs 15.14 tok/s) +// X2E +6.9% 4/4 pairs positive, order bias -0.03% (35.24 vs 32.87 tok/s) +// +// Measure this one on a COLD device. These kernels are far more clock-sensitive +// than the o4 route they replace: on a heat-soaked A8X (CPU cap at 1.5-1.9 GHz) +// tiled pins at ~14.2 tok/s while o4 still makes ~14.9, which reads as a 4-5% +// LOSS and inverts the ranking. The same box, after a reboot and a gate that +// waits for policy6 to return to 4396800, reports the +11.9% above with no +// order bias. A7X regresses hard on this layout and stays off. +// GGML_OPENCL_{Q4K,Q6K}_GEMV_TILED forces either way (=0 off, any other value on). +inline bool tiled_gemv_default_on(const ggml_backend_opencl_context *backend_ctx) { + return backend_ctx && (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E || + backend_ctx->adreno_gen == ADRENO_GPU_GEN::A8X); +} + +// Tiled-wide q6_K GEMV (default OFF; GGML_OPENCL_Q6K_GEMV_TILED forces either +// way: =0 off everywhere, any other value on everywhere). +// Both the convert (set_tensor) and the GEMV dispatch must agree on this so the +// buffer layout matches the kernel. +inline bool q6k_gemv_tiled_enabled(const ggml_backend_opencl_context *backend_ctx) { + static const char * e = std::getenv("GGML_OPENCL_Q6K_GEMV_TILED"); + if (e && e[0] != '\0') { + return e[0] != '0'; + } + return tiled_gemv_default_on(backend_ctx); +} + +// Only the long-vocab lm_head/embed shapes use the tiled layout; ne01 % 64 == 0 +// is required by the 64-row tiling (no row padding in the buffers). +// use_adreno_kernels is required: only the Adreno GEMV path can read the tiled +// layout, so converting a weight it would decline (e.g. ne00 < 512) leaves the +// generic kernel reading tiled bytes as plain SOA. +inline bool use_q6k_tiled(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { + return q6k_gemv_tiled_enabled(backend_ctx) && tensor->type == GGML_TYPE_Q6_K && + tensor->ne[1] >= 32768 && tensor->ne[1] % 64 == 0 && + use_adreno_kernels(backend_ctx, tensor); +} + +// q4_K analog of the tiled-wide lm_head/embed GEMV (default OFF; +// GGML_OPENCL_Q4K_GEMV_TILED forces either way: =0 off, else on). Same gate. +inline bool q4k_gemv_tiled_enabled(const ggml_backend_opencl_context *backend_ctx) { + static const char * e = std::getenv("GGML_OPENCL_Q4K_GEMV_TILED"); + if (e && e[0] != '\0') { + return e[0] != '0'; + } + return tiled_gemv_default_on(backend_ctx); +} +inline bool use_q4k_tiled(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { + return q4k_gemv_tiled_enabled(backend_ctx) && tensor->type == GGML_TYPE_Q4_K && + tensor->ne[1] >= 32768 && tensor->ne[1] % 64 == 0 && + use_adreno_kernels(backend_ctx, tensor); +} + inline bool enable_adreno_trans_weight(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { bool adreno_kernel = use_adreno_kernels(backend_ctx, tensor); @@ -7089,23 +8584,100 @@ inline bool enable_adreno_trans_weight_q5_K(const ggml_backend_opencl_context *b const size_t elem_num = ggml_nelements(tensor); const size_t q_img_width = elem_num / 8; const size_t qh_img_width = elem_num / 16; + const bool shape_ok = tensor->ne[0] % 32 == 0 && tensor->ne[1] % 4 == 0 && + tensor->ne[2] == 1 && tensor->ne[3] == 1; - return q_img_width <= backend_ctx->image_max_buffer_size && + return shape_ok && q_img_width <= backend_ctx->image_max_buffer_size && qh_img_width <= backend_ctx->image_max_buffer_size; } -static inline bool use_flat_gemv_for_large_m_q4_K(const ggml_tensor *tensor) { +inline bool use_q4_0_bin_kernels(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (!backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32b_trans || + !backend_ctx->kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin) { + return false; + } + return (tensor->ne[0] % 32 == 0) && (tensor->ne[1] % 64 == 0); +#else + GGML_UNUSED(backend_ctx); + GGML_UNUSED(tensor); + return false; +#endif +} + +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static bool use_fa_bin_kernels_prefill(const ggml_backend_opencl_context * backend_ctx, const ggml_tensor * q, const ggml_tensor * k, const ggml_tensor * v) { + if (backend_ctx->fa.kernel_flash_attn_f32_f16_bin == nullptr) { + return false; + } + + const bool is_mixed = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_F16 && v->type == GGML_TYPE_F16; + const bool is_q8_0 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_Q8_0 && v->type == GGML_TYPE_Q8_0; + + const int n_q = q->ne[1]; + const int dk = q->ne[0]; + const int dv = v->ne[0]; + + constexpr bool prefill_only = true; + + return (backend_ctx->gpu_family == GPU_FAMILY::ADRENO && + (is_mixed || is_q8_0) && (dk == dv) + && (dk == 64 || dk == 128 || dk == 256 || dk == 512) + && (!prefill_only || n_q != 1)); +} +#endif + +// The flat-GEMV large-m escape is OPT-IN (GGML_OPENCL_FLAT_LARGE_M=1) because it +// is SLOWER than the route it replaces, not because it is unsafe. It was first +// parked on the theory that it out-of-bounds-writes at vocab-scale shapes; that +// was a misattribution (the test-backend-ops dst sentinel was tripped by the o4 +// GEMV's unguarded tail store, fixed separately - and at the shape it was blamed +// for, k=1536, this predicate returns false anyway, so the flat route never ran). +// +// The escape's original rationale, "gemv_noshuffle perf drops for large M", +// predates the o4 kernel, which now covers the same long-vocab shapes and beats +// this route on every device measured (Qwen3-4B-Q4_K_M, q6_K lm_head +// 151936x2560, tg128, matched pairs vs o4): A8X -10.3% (0/3 pairs), X2E -3.7% +// (0/3). Keep it reachable for shapes o4 declines, but do not default it on. +static inline bool flat_large_m_enabled() { + static const char * e = getenv("GGML_OPENCL_FLAT_LARGE_M"); + static const bool en = e != nullptr && atoi(e) != 0; + return en; +} + +static inline bool use_flat_gemv_for_large_m_q4_K(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { + if (tensor->ne[1] % 4 != 0 && tensor->ne[2] == 1 && tensor->ne[3] == 1) { + return true; + } + + if (!flat_large_m_enabled()) { + return false; + } // gemv_noshuffle variant perf drops for large M, use flat variant for large M. // threshold is well above typical hidden/FFN dims, but below typical vocab sizes. // note that this forces large M weights to use LM GEMM. - return tensor->ne[1] >= 32768 && tensor->ne[2] == 1 && tensor->ne[3] == 1; + // EXCEPT when this branch's tiled-canonical lm_head/embed layout is active: the + // weight is converted to the 64-row tiled layout, which the flat gemv would + // misread as garbage. use_q4k_tiled owns these large-M weights, so defer to it. + return tensor->ne[1] >= 32768 && tensor->ne[2] == 1 && tensor->ne[3] == 1 + && !use_q4k_tiled(backend_ctx, tensor); } static inline bool use_flat_gemv_for_large_m_q6_K(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { + // NOTE on ordering: the ne01 % 128 escape below is a CORRECTNESS guard, not a + // performance one, so it must be reachable regardless of flat_large_m_enabled(). + // The opt-in gate therefore sits after it, and after the tiled deferral. // gemv_noshuffle variant perf drops for large M, use flat variant for large M. // threshold is well above typical hidden/FFN dims, but below typical vocab sizes. // q6_K flat gemv is worse for smaller K; 2048 seems to be a reasonable threshold. // note that this forces large M weights to use LM GEMM. + // When this branch's tiled-canonical lm_head/embed layout is active, the weight is + // converted to the 64-row tiled layout, which the flat gemv would misread as + // garbage. use_q6k_tiled owns these large-M weights (it requires ne01 % 64 == 0, + // so it never claims an odd-vocab weight), so defer to it first. + if (use_q6k_tiled(backend_ctx, tensor)) { + return false; + } // The noshuffle (transposed-weight) layout packs 2 rows per 32-bit texel and the // gemv reads it with a ne01/2 texel stride and an exact-cover dispatch of // ceil(ne01/2 / 64)*64 work-items with no store guard; the gemm uses 4-row tiles. @@ -7120,6 +8692,10 @@ static inline bool use_flat_gemv_for_large_m_q6_K(const ggml_backend_opencl_cont return true; } + if (!flat_large_m_enabled()) { + return false; + } + // The gemv_noshuffle slowdown tracks TOTAL weight size, not ne0 alone; ne0 >= 2048 is a // proxy for "large weight" that misses a narrow-hidden vocab-scale lm_head. // Add a direct size escape so such weights also take the flat path, without changing @@ -7130,6 +8706,36 @@ static inline bool use_flat_gemv_for_large_m_q6_K(const ggml_backend_opencl_cont && tensor->ne[2] == 1 && tensor->ne[3] == 1; } +inline bool use_q6_k_bin_kernels(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (!backend_ctx->kernel_gemv_noshuffle_q6_k_f32_32b_trans || + !backend_ctx->kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin) { + return false; + } + return (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) && + !use_q6k_tiled(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q6_K(backend_ctx, tensor); +#else + GGML_UNUSED(backend_ctx); + GGML_UNUSED(tensor); + return false; +#endif +} + +inline bool use_q4_k_bin_kernels(const ggml_backend_opencl_context *backend_ctx, const ggml_tensor *tensor) { +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (!backend_ctx->kernel_gemv_noshuffle_q4_k_f32_32b_trans || + !backend_ctx->kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin) { + return false; + } + return (tensor->ne[0] % 256 == 0) && (tensor->ne[1] % 64 == 0) && + !use_q4k_tiled(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(backend_ctx, tensor); +#else + GGML_UNUSED(backend_ctx); + GGML_UNUSED(tensor); + return false; +#endif +} + static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_tensor * op) { ggml_backend_opencl_device_context * dev_ctx = (ggml_backend_opencl_device_context *)dev->context; ggml_backend_opencl_context * backend_ctx = dev_ctx->backend_ctx; @@ -7250,6 +8856,15 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te case GGML_UNARY_OP_EXPM1: return op->src[0]->type == GGML_TYPE_F32; case GGML_UNARY_OP_ABS: + case GGML_UNARY_OP_SGN: + case GGML_UNARY_OP_STEP: + case GGML_UNARY_OP_ELU: + case GGML_UNARY_OP_HARDSWISH: + case GGML_UNARY_OP_HARDSIGMOID: + case GGML_UNARY_OP_FLOOR: + case GGML_UNARY_OP_CEIL: + case GGML_UNARY_OP_ROUND: + case GGML_UNARY_OP_TRUNC: return op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16; case GGML_UNARY_OP_SOFTPLUS: return op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16; @@ -7264,6 +8879,7 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return ggml_is_contiguous_1(op->src[0]) && (op->type == GGML_TYPE_F32 || op->type == GGML_TYPE_F16); default: return false; @@ -7301,6 +8917,17 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te (op->src[0]->type == GGML_TYPE_F16 && op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32); case GGML_OP_SSM_CONV: return (op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32); + case GGML_OP_SSM_SCAN: { + if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 || + op->src[1]->type != GGML_TYPE_F32 || op->src[2]->type != GGML_TYPE_F32 || + op->src[3]->type != GGML_TYPE_F32 || op->src[4]->type != GGML_TYPE_F32 || + op->src[5]->type != GGML_TYPE_F32 || op->src[6]->type != GGML_TYPE_I32) { + return false; + } + + const int64_t d_state = op->src[0]->ne[0]; + return d_state >= 1 && d_state <= 256 && (d_state & (d_state - 1)) == 0; + } case GGML_OP_GATED_DELTA_NET: { // Match the Vulkan backend: only F32 -> F32, S_v in {16, 32, 64, 128}. @@ -7311,7 +8938,13 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te return S_v == 16 || S_v == 32 || S_v == 64 || S_v == 128; } case GGML_OP_CONCAT: - return op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32; + { + const ggml_type t = op->src[0]->type; + return op->src[1]->type == t && op->type == t && + !ggml_is_quantized(t) && ggml_blck_size(t) == 1 && + (ggml_type_size(t) == 1 || ggml_type_size(t) == 2 || + ggml_type_size(t) == 4 || ggml_type_size(t) == 8); + } case GGML_OP_TIMESTEP_EMBEDDING: return op->src[0]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32; case GGML_OP_GROUP_NORM: @@ -7335,8 +8968,42 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te op->src[0]->type == GGML_TYPE_Q4_K || op->src[0]->type == GGML_TYPE_Q5_K || op->src[0]->type == GGML_TYPE_Q6_K) { + // The E031.41 compiler (usually with A7x) miscompiles the flat K-quant + // GEMV kernels (kernel_mul_mv_q*_K_f32_flat) and makes lm_head run much + // slower than it should. So, make it fallback to CPU to preserve performance + // for this compiler series. + static const char * a7x_lmhead_env = getenv("GGML_OPENCL_A7X_LMHEAD_CPU"); + static const bool a7x_lmhead_cpu = (a7x_lmhead_env == nullptr || a7x_lmhead_env[0] != '0'); + if (a7x_lmhead_cpu && + backend_ctx->adreno_gen == ADRENO_GPU_GEN::A7X && + (op->src[0]->type == GGML_TYPE_Q4_K || op->src[0]->type == GGML_TYPE_Q5_K || + op->src[0]->type == GGML_TYPE_Q6_K) && + op->src[0]->ne[1] >= 32768) { // vocab-scale weight; no FFN/attn weight is this tall + return false; + } + // The generic mul_mv (GEMV) kernels are wrong for large-batch prefill on + // Adreno. A quant mul_mat only avoids the GEMV when it reaches the Adreno + // trans-weight GEMM, which needs both a GEMM kernel for the type and + // use_adreno_kernels(). Decline the large-N shapes that would otherwise + // fall through to the GEMV. + { + const ggml_type t = op->src[0]->type; + const bool type_has_gemm = (t == GGML_TYPE_Q4_0 || t == GGML_TYPE_Q4_1 || + t == GGML_TYPE_IQ4_NL || t == GGML_TYPE_Q8_0 || + t == GGML_TYPE_Q4_K || t == GGML_TYPE_Q5_K || + t == GGML_TYPE_Q6_K); + const bool uses_gemm = type_has_gemm && use_adreno_kernels(backend_ctx, op->src[0]); + if (!uses_gemm && op->src[1]->ne[1] >= 512) { + return false; + } + } return op->src[1]->type == GGML_TYPE_F32 && ggml_is_contiguous(op->src[0]) && ggml_is_contiguous(op->src[1]); } else if (op->src[0]->type == GGML_TYPE_Q8_0) { + // ggml_cl_mul_mat_q8_0_f32_adreno now honors src1/dst view_offs (the + // activation sub-buffer starts at offset1 and the kernels take offsetd), + // so a broadcast q8_0 matmul (src1 batch > src0 batch, e.g. Qwen3.5-9B-UD + // / Qwen3.6-35B q8_0 GDN ssm_out) runs on GPU via the per-slice broadcast + // iteration in ggml_cl_mul_mat. No special-casing needed. return op->src[1]->type == GGML_TYPE_F32; } return false; @@ -7418,6 +9085,11 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te case GGML_OP_MEAN: return op->src[0]->type == GGML_TYPE_F32; case GGML_OP_FLASH_ATTN_EXT: { +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (use_fa_bin_kernels_prefill(backend_ctx, op->src[0], op->src[1], op->src[2])) { + return true; + } +#endif // The E17 compilers segfault while building FA kernels, skip E17 for now if (adreno_e17_compiler_quirks(backend_ctx)) { return false; @@ -7453,6 +9125,7 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te v->type == GGML_TYPE_F16 && op->type == GGML_TYPE_F16; const bool is_f32_f16 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_F16 && v->type == GGML_TYPE_F16 && op->type == GGML_TYPE_F32; + const bool is_f32_q8_0 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_Q8_0 && v->type == GGML_TYPE_Q8_0 && op->type == GGML_TYPE_F32 && dk % 32 == 0 && dv % 32 == 0; @@ -7460,6 +9133,21 @@ static bool ggml_opencl_supports_op(ggml_backend_dev_t dev, const struct ggml_te v->type == GGML_TYPE_Q4_0 && op->type == GGML_TYPE_F32 && dk % 32 == 0 && dv % 32 == 0; + // A7X (Adreno 740, compiler E031.41) SIGSEGVs inside clBuildProgram + // building the flash_attn programs whose KV path is mixed-type or + // dequantized — f32_f16, q8_0, q4_0 (reproduced at DK=40 and DK=64; it + // is DK-independent). It is a driver crash, not codegen-wrong-output, so + // it cannot be caught in-process (fatal=false only handles clean compile + // errors). The uniform f16_f16 / f32_f32 programs compile fine on this + // compiler, so decline only the KV-convert variants; ggml then runs + // those (f16-KV / quant-KV) attention layers on the CPU backend. + // Negative compiler carve-out, same idiom as the Intel DK=512 decline + // below and the X1E driver-quirk guards. + if (backend_ctx && backend_ctx->adreno_gen == ADRENO_GPU_GEN::A7X && + (is_f32_f16 || is_f32_q8_0 || is_f32_q4_0)) { + return false; + } + // Asymmetric KV: host-dequants both sides to F32, uses f32 kernel. auto is_kv_type_ok = [](ggml_type t) { return t == GGML_TYPE_F16 || t == GGML_TYPE_F32 || @@ -8005,6 +9693,96 @@ static enum ggml_status ggml_backend_opencl_buffer_init_tensor(ggml_backend_buff return GGML_STATUS_SUCCESS; } +// Allocate a temporary upload buffer of `nbytes` and populate it with `data` +// from host. On Adreno X1-85 the device-only pool intermittently fails to +// allocate at hundreds of MB once model weights fragment the heap (observed +// on Qwen3.5-9B output.weight Q6_K at 834 MB). Three-step retry: +// 1. CL_MEM_READ_WRITE alloc + clEnqueueWriteBuffer (normal fast path). +// 2. clFinish + retry (drains in-flight allocs that may be holding heap; +// mirrors the proven pattern at the FD-split partial buffer alloc). +// 3. CL_MEM_ALLOC_HOST_PTR + map(WRITE_INVALIDATE) + memcpy + unmap — +// different memory pool (host-pinned); true zero-copy on Adreno per +// QCOM guidance. (CL_MEM_USE_HOST_PTR is NOT zero-copy on Adreno: the +// driver triggers an internal copy because arbitrary host pages aren't +// guaranteed mappable/coherent, AND it draws from the same exhausted +// device pool — so it doesn't solve the problem.) +// Returns the ready-to-read buffer (caller must clReleaseMemObject) or NULL +// if all three strategies fail. The buffer is opaque to the caller — it can +// be passed as a kernel argument like any normal cl_mem. +static cl_mem ggml_cl_create_temp_upload_buffer( + cl_context context, cl_command_queue queue, + size_t nbytes, const void * data, + const char * tensor_name_for_log) +{ + cl_int err; + cl_mem buf = clCreateBuffer(context, CL_MEM_READ_WRITE, nbytes, NULL, &err); + if (err != CL_SUCCESS) { + clFinish(queue); + buf = clCreateBuffer(context, CL_MEM_READ_WRITE, nbytes, NULL, &err); + } + if (err == CL_SUCCESS) { + const cl_int werr = clEnqueueWriteBuffer(queue, buf, CL_TRUE, 0, nbytes, data, 0, NULL, NULL); + if (werr == CL_SUCCESS) { + return buf; + } + clReleaseMemObject(buf); + } + buf = clCreateBuffer(context, + CL_MEM_READ_ONLY | CL_MEM_ALLOC_HOST_PTR | CL_MEM_HOST_WRITE_ONLY, + nbytes, NULL, &err); + if (err != CL_SUCCESS) { + return NULL; + } + void * mapped = clEnqueueMapBuffer(queue, buf, CL_TRUE, + CL_MAP_WRITE_INVALIDATE_REGION, 0, nbytes, 0, NULL, NULL, &err); + if (err != CL_SUCCESS) { + clReleaseMemObject(buf); + return NULL; + } + memcpy(mapped, data, nbytes); + const cl_int uerr = clEnqueueUnmapMemObject(queue, buf, mapped, 0, NULL, NULL); + if (uerr != CL_SUCCESS) { + clReleaseMemObject(buf); + return NULL; + } + if (tensor_name_for_log) { + GGML_LOG_INFO("ggml_opencl: %s (%.1f MiB) — device alloc failed, using CL_MEM_ALLOC_HOST_PTR fallback\n", + tensor_name_for_log, nbytes / 1024.0 / 1024.0); + } + return buf; +} + +// Allocate a temporary download buffer of `nbytes`. The caller runs a kernel +// that writes into it, then reads it back to host via clEnqueueReadBuffer (or +// equivalent). Mirrors ggml_cl_create_temp_upload_buffer; the host-pinned +// fallback flags are flipped (CL_MEM_WRITE_ONLY | HOST_READ_ONLY) and the +// helper doesn't populate the buffer. +static cl_mem ggml_cl_create_temp_download_buffer( + cl_context context, cl_command_queue queue, + size_t nbytes, const char * tensor_name_for_log) +{ + cl_int err; + cl_mem buf = clCreateBuffer(context, CL_MEM_READ_WRITE, nbytes, NULL, &err); + if (err != CL_SUCCESS) { + clFinish(queue); + buf = clCreateBuffer(context, CL_MEM_READ_WRITE, nbytes, NULL, &err); + } + if (err == CL_SUCCESS) { + return buf; + } + buf = clCreateBuffer(context, + CL_MEM_WRITE_ONLY | CL_MEM_ALLOC_HOST_PTR | CL_MEM_HOST_READ_ONLY, + nbytes, NULL, &err); + if (err != CL_SUCCESS) { + return NULL; + } + if (tensor_name_for_log) { + GGML_LOG_INFO("ggml_opencl: %s download (%.1f MiB) — device alloc failed, using CL_MEM_ALLOC_HOST_PTR fallback\n", + tensor_name_for_log, nbytes / 1024.0 / 1024.0); + } + return buf; +} + static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { ggml_backend_opencl_device_context * dev_ctx = (ggml_backend_opencl_device_context *) buffer->buft->device->context; ggml_backend_opencl_context * backend_ctx = dev_ctx->backend_ctx; @@ -8113,12 +9891,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); // We consider the specified offset arg as always, although For weights // the offset arg should be 0 (we do not assert this). @@ -8230,10 +10004,34 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(K % 32 == 0); - // Transpose q as ushort - transpose_2d_as_16b(backend_ctx, extra->q, extra->q, size_q, K/4, M); - // Transpose d as ushort - transpose_2d_as_16b(backend_ctx, extra->d, extra->d, size_d, K/32, M); + if (use_q4_0_bin_kernels(backend_ctx, tensor)) { + cl_int err; + cl_image_format wimg_fmt; + cl_image_desc wimg_desc; + + // transpose quants as 32-bit words (M-first) + GGML_ASSERT(M % 64 == 0); + transpose_2d_as_32b(backend_ctx, extra->q, extra->q, size_q, K / 8, M); + transpose_2d_as_16b(backend_ctx, extra->d, extra->d, size_d, K / 32, M); + + wimg_fmt = { CL_R, CL_UNSIGNED_INT32 }; + memset(&wimg_desc, 0, sizeof(wimg_desc)); + wimg_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + wimg_desc.image_width = (size_t)M * K / 8; + wimg_desc.buffer = extra->q; + CL_CHECK((extra->q_img = clCreateImage(context, CL_MEM_READ_ONLY, &wimg_fmt, &wimg_desc, NULL, &err), err)); + + wimg_fmt = { CL_R, CL_HALF_FLOAT }; + memset(&wimg_desc, 0, sizeof(wimg_desc)); + wimg_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + wimg_desc.image_width = (size_t)M * K / 32; + wimg_desc.buffer = extra->d; + CL_CHECK((extra->d_img = clCreateImage(context, CL_MEM_READ_ONLY, &wimg_fmt, &wimg_desc, NULL, &err), err)); + } else { + // Transpose q and d as ushort + transpose_2d_as_16b(backend_ctx, extra->q, extra->q, size_q, K/4, M); + transpose_2d_as_16b(backend_ctx, extra->d, extra->d, size_d, K/32, M); + } } #endif // GGML_OPENCL_USE_ADRENO_KERNELS return; @@ -8252,12 +10050,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_m + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -8384,12 +10178,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_qs + size_qh == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -8548,12 +10338,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_m + size_qs + size_qh == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -8701,12 +10487,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_e + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); // The original tensor memory is divided into scales and quants, i.e., // we first store scales, then quants. @@ -8812,12 +10594,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); // The original tensor memory is divided into scales and quants, i.e., // we first store scales, then quants. @@ -8916,12 +10694,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -9000,12 +10774,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, GGML_ASSERT(size_d + size_dm + size_s + size_q == ggml_nbytes(tensor) && "Incorrect tensor size"); cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer( - queue, data_device, CL_TRUE, 0, - ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "q4_K set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -9088,8 +10858,41 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, #endif // GGML_OPENCL_USE_ADRENO_KERNELS #ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // Tiled-wide convert for the long-vocab lm_head/embed (opt-in). The embed/ + // output q4_K weight (token_embd.weight, ne1=vocab) is NOT matched by + // use_adreno_moe_kernels, so it lands here in the general branch. Produce + // the final 64-row-tiled canonical layout directly into q/d/dm/s (buffer + // sizes already match), read back by kernel_gemv_noshuffle_q4_k_f32_tiled. + if (use_q4k_tiled(backend_ctx, tensor)) { + cl_kernel tk = backend_ctx->kernel_convert_block_q4_k_tiled_ns; + + int ne00 = tensor->ne[0]; + int ne01 = tensor->ne[1]; + int ne02 = tensor->ne[2]; + + CL_CHECK(clSetKernelArg(tk, 0, sizeof(cl_mem), &data_device)); + CL_CHECK(clSetKernelArg(tk, 1, sizeof(cl_mem), &extra->q)); + CL_CHECK(clSetKernelArg(tk, 2, sizeof(cl_mem), &extra->d)); + CL_CHECK(clSetKernelArg(tk, 3, sizeof(cl_mem), &extra->dm)); + CL_CHECK(clSetKernelArg(tk, 4, sizeof(cl_mem), &extra->s)); + CL_CHECK(clSetKernelArg(tk, 5, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(tk, 6, sizeof(int), &ne01)); + + size_t gws[] = {static_cast(((ne01 + 63) / 64) * 64), static_cast(ne00 / 256), static_cast(ne02)}; + size_t lws[] = {64, 1, 1}; + + cl_event tevt; + CL_CHECK(clEnqueueNDRangeKernel(queue, tk, 3, NULL, gws, lws, 0, NULL, &tevt)); + CL_CHECK(clWaitForEvents(1, &tevt)); + CL_CHECK(clReleaseMemObject(data_device)); + + extra->q_img = nullptr; + tensor->extra = extra; + return; + } + cl_kernel kernel = backend_ctx->kernel_convert_block_q4_K; - if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(tensor)) { + if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(backend_ctx, tensor)) { kernel = backend_ctx->kernel_convert_block_q4_K_noshuffle; } #else @@ -9117,15 +10920,32 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, tensor->extra = extra; #ifdef GGML_OPENCL_USE_ADRENO_KERNELS - if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(tensor)) { + if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(backend_ctx, tensor)) { int M = tensor->ne[1]; int K = tensor->ne[0]; GGML_ASSERT(K % 32 == 0); - // Transpose q, d, dm as ushort - transpose_2d_as_16b(backend_ctx, extra->q, extra->q, size_q, K/4, M); + if (use_q4_k_bin_kernels(backend_ctx, tensor)) { + cl_int err; + cl_image_format wimg_fmt; + cl_image_desc wimg_desc; + + // transpose quants as 32-bit words (M-first) + GGML_ASSERT(M % 64 == 0); + transpose_2d_as_32b(backend_ctx, extra->q, extra->q, size_q, K/8, M); + + wimg_fmt = { CL_R, CL_UNSIGNED_INT32 }; + memset(&wimg_desc, 0, sizeof(wimg_desc)); + wimg_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + wimg_desc.image_width = (size_t)M * K / 8; + wimg_desc.buffer = extra->q; + CL_CHECK((extra->q_img = clCreateImage(context, CL_MEM_READ_ONLY, &wimg_fmt, &wimg_desc, NULL, &err), err)); + } else { + // Transpose q as ushort + transpose_2d_as_16b(backend_ctx, extra->q, extra->q, size_q, K/4, M); + } transpose_2d_as_16b(backend_ctx, extra->d, extra->d, size_d, K/256, M); transpose_2d_as_16b(backend_ctx, extra->dm, extra->dm, size_dm, K/256, M); @@ -9152,9 +10972,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, "Incorrect tensor size"); cl_int err; - cl_mem data_device; - CL_CHECK((data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, ggml_nbytes(tensor), NULL, &err), err)); - CL_CHECK(clEnqueueWriteBuffer(queue, data_device, CL_TRUE, 0, ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "q5_K set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -9340,9 +11159,8 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, "Incorrect tensor size"); cl_int err; - cl_mem data_device; - CL_CHECK((data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, ggml_nbytes(tensor), NULL, &err), err)); - CL_CHECK(clEnqueueWriteBuffer(queue, data_device, CL_TRUE, 0, ggml_nbytes(tensor), data, 0, NULL, NULL)); + cl_mem data_device = ggml_cl_create_temp_upload_buffer(context, queue, ggml_nbytes(tensor), data, tensor->name); + GGML_ASSERT(data_device != NULL && "q6_K set_tensor: temp upload buffer alloc failed"); cl_buffer_region region; @@ -9443,6 +11261,45 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, CL_CHECK((extra->d = clCreateSubBuffer(extra_orig->data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); previous_origin = region.origin; +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // Tiled-wide convert for the long-vocab lm_head/embed (opt-in). The embed + // /output q6_K weight (e.g. token_embd.weight, ne1=vocab) is NOT matched by + // use_adreno_moe_kernels, so it lands here in the general branch. Produce + // the final 64-row-tiled canonical layout directly into ql/qh/s/d (buffer + // sizes already match), read back by kernel_gemv_noshuffle_q6_K_f32_tiled. + // Bypasses the plain-SOA convert + per-array transpose below. + if (use_q6k_tiled(backend_ctx, tensor)) { + cl_kernel kernel = backend_ctx->kernel_convert_block_q6_k_tiled_ns; + + int ne00 = tensor->ne[0]; + int ne01 = tensor->ne[1]; + int ne02 = tensor->ne[2]; + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra->ql)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra->qh)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra->d)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &extra->s)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne01)); + + size_t global_work_size[] = {static_cast(((ne01 + 63) / 64) * 64), static_cast(ne00 / 256), static_cast(ne02)}; + size_t local_work_size[] = {64, 1, 1}; + + cl_event evt; + CL_CHECK(clEnqueueNDRangeKernel(queue, kernel, 3, NULL, global_work_size, local_work_size, 0, NULL, &evt)); + CL_CHECK(clWaitForEvents(1, &evt)); + CL_CHECK(clReleaseMemObject(data_device)); + + extra->size_ql = size_ql; + extra->size_qh = size_qh; + extra->size_s = size_s; + extra->size_d = size_d; + tensor->extra = extra; + return; + } +#endif // GGML_OPENCL_USE_ADRENO_KERNELS + // Flatten the weights cl_kernel kernel; #ifdef GGML_OPENCL_USE_ADRENO_KERNELS @@ -9484,18 +11341,39 @@ static void ggml_backend_opencl_buffer_set_tensor(ggml_backend_buffer_t buffer, cl_int M = tensor->ne[1]; // ne01 cl_int K = tensor->ne[0]; // ne00 - // Transpose ql as ushort - transpose_2d_as_16b(backend_ctx, - extra->ql, extra->ql, size_ql, K/4, M); - - // Transpose qh as uchar - transpose_2d_as_8b(backend_ctx, - extra->qh, extra->qh, size_qh, K/4, M); + if (use_q6_k_bin_kernels(backend_ctx, tensor)) { + GGML_ASSERT(K % 256 == 0); + GGML_ASSERT(M % 64 == 0); + + transpose_2d_as_32b(backend_ctx, extra->ql, extra->ql, size_ql, K/8, M); + transpose_2d_as_32b(backend_ctx, extra->qh, extra->qh, size_qh, K/16, M); + + cl_image_format wimg_fmt = { CL_R, CL_UNSIGNED_INT32 }; + cl_image_desc wimg_desc; + memset(&wimg_desc, 0, sizeof(wimg_desc)); + wimg_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + wimg_desc.image_width = static_cast(ggml_nelements(tensor) / 8); + wimg_desc.buffer = extra->ql; + CL_CHECK((extra->ql_img = clCreateImage(context, CL_MEM_READ_ONLY, &wimg_fmt, &wimg_desc, NULL, &err), err)); + + memset(&wimg_desc, 0, sizeof(wimg_desc)); + wimg_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + wimg_desc.image_width = static_cast(ggml_nelements(tensor) / 16); + wimg_desc.buffer = extra->qh; + CL_CHECK((extra->qh_img = clCreateImage(context, CL_MEM_READ_ONLY, &wimg_fmt, &wimg_desc, NULL, &err), err)); + } else { + // Transpose ql as ushort + transpose_2d_as_16b(backend_ctx, + extra->ql, extra->ql, size_ql, K/4, M); - // Transpose s as ushort - transpose_2d_as_16b(backend_ctx, - extra->s, extra->s, size_s, K/16/2, M); + // Transpose qh as uchar + transpose_2d_as_8b(backend_ctx, + extra->qh, extra->qh, size_qh, K/4, M); + // Transpose s as ushort + transpose_2d_as_16b(backend_ctx, + extra->s, extra->s, size_s, K/16/2, M); + } // Transpose d as ushort transpose_2d_as_16b(backend_ctx, extra->d, extra->d, size_d, K/256, M); @@ -9641,12 +11519,10 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; cl_kernel kernel = backend_ctx->kernel_restore_block_q4_0_trans4_ns; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); int ne00 = tensor->ne[0]; int ne01 = tensor->ne[1]; @@ -9689,7 +11565,11 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, buf_trans_d.allocate(backend_ctx->context, size_d); buf_unpacked.allocate(backend_ctx->context, ggml_nbytes(tensor)); - transpose_2d_as_16b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K/4); + if (use_q4_0_bin_kernels(backend_ctx, tensor)) { + transpose_2d_as_32b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K / 8); + } else { + transpose_2d_as_16b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K / 4); + } transpose_2d_as_16b(backend_ctx, extra->d, buf_trans_d.buffer, size_d, M, K/32); cl_uchar mask_0F = 0x0F; @@ -9711,10 +11591,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, } #endif - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q4_0; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra->q)); @@ -9739,10 +11617,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q4_1_trans4_ns; int ne00 = tensor->ne[0]; @@ -9814,10 +11690,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, } #endif - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q4_1; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra->q)); @@ -9843,11 +11717,9 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; // TODO: use ggml_cl_buffer to manage this temporary buffer - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q5_0_trans4_ns; @@ -9947,11 +11819,9 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; // TODO: use ggml_cl_buffer to manage this temporary buffer - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q5_1_trans4_ns; @@ -10056,10 +11926,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, if (tensor->type == GGML_TYPE_MXFP4) { ggml_tensor_extra_cl_mxfp4 * extra = (ggml_tensor_extra_cl_mxfp4 *)tensor->extra; - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { @@ -10121,10 +11989,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * extra_src = tensor->view_src != nullptr ? tensor->view_src : tensor; ggml_tensor_extra_cl_q8_0 * extra = (ggml_tensor_extra_cl_q8_0 *)extra_src->extra; - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (enable_adreno_trans_weight(backend_ctx, tensor)) { @@ -10177,10 +12043,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, if (tensor->type == GGML_TYPE_IQ4_NL) { ggml_tensor_extra_cl_iq4_nl * extra = (ggml_tensor_extra_cl_iq4_nl *)tensor->extra; - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_kernels(backend_ctx, tensor)) { @@ -10249,20 +12113,64 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, if (tensor->type == GGML_TYPE_Q4_K) { ggml_tensor_extra_cl_q4_K * extra = (ggml_tensor_extra_cl_q4_K *)tensor->extra; - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_uchar mask_0F = 0x0F; cl_uchar mask_F0 = 0xF0; #ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // Undo the 64-row-tiled canonical pack (kernel_convert_block_q4_k_tiled_ns). + // Without this, a read-back of a tiled weight returns the tiled bytes + // reinterpreted as block_q4_K -- which is how test-backend-ops builds its + // CPU reference (ggml_backend_graph_copy -> tensor_get), so the tiled path + // "failed" the suite while computing the correct product. + if (use_q4k_tiled(backend_ctx, tensor)) { + const int ne00v = tensor->ne[0]; + const int ne01v = tensor->ne[1]; + const int nbv = ne00v / 256; + const size_t n_blk = (size_t)nbv * ne01v; + + std::vector tq(n_blk*32); + std::vector td(n_blk), tdm(n_blk); + std::vector ts(n_blk*12); + CL_CHECK(clEnqueueReadBuffer(queue, extra->q, CL_TRUE, 0, tq.size()*4, tq.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->d, CL_TRUE, 0, td.size()*2, td.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->dm, CL_TRUE, 0, tdm.size()*2, tdm.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->s, CL_TRUE, 0, ts.size(), ts.data(), 0, NULL, NULL)); + + std::vector rebuilt(ggml_nbytes(tensor), 0); + for (int i01 = 0; i01 < ne01v; ++i01) { + const int rt = i01/64, rit = i01%64; + for (int i00 = 0; i00 < nbv; ++i00) { + uint8_t * b = rebuilt.data() + ((size_t)i00 + (size_t)i01*nbv)*144; + const int tb = rt*nbv + i00; + const size_t si = (size_t)tb*64 + rit; + + memcpy(b + 0, &td [si], 2); + memcpy(b + 2, &tdm[si], 2); + memcpy(b + 4, &ts[si*12], 12); + + uint32_t qw[32]; + for (int gr = 0; gr < 8; ++gr) { + const size_t base = ((size_t)tb*8 + gr)*64 + rit; + for (int j = 0; j < 4; ++j) qw[gr*4 + j] = tq[base*4 + j]; + } + uint8_t * q = b + 16; + for (int e = 0; e < 256; ++e) { + const int g = e>>6, w = e&63, h = w>>5, l = w&31; + const uint32_t code = (qw[e>>3] >> ((e&7)*4)) & 0xF; + q[g*32 + l] |= (uint8_t)(h ? (code << 4) : code); + } + } + } + memcpy(data, rebuilt.data() + offset, size); + CL_CHECK(clReleaseMemObject(data_device)); + return; + } if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q4_k_trans4_ns; @@ -10292,7 +12200,7 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, CL_CHECK(clReleaseMemObject(data_device)); return; } - if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(tensor)) { + if (use_adreno_kernels(backend_ctx, tensor) && !use_flat_gemv_for_large_m_q4_K(backend_ctx, tensor)) { int M = tensor->ne[1]; int K = tensor->ne[0]; @@ -10312,7 +12220,11 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, buf_trans_s.allocate(backend_ctx->context, size_s); // Transpose q, d, dm, s back - transpose_2d_as_16b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K/4); + if (use_q4_k_bin_kernels(backend_ctx, tensor)) { + transpose_2d_as_32b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K/8); + } else { + transpose_2d_as_16b(backend_ctx, extra->q, buf_trans_q.buffer, size_q, M, K/4); + } transpose_2d_as_16b(backend_ctx, extra->d, buf_trans_d.buffer, size_d, M, K/256); transpose_2d_as_16b(backend_ctx, extra->dm, buf_trans_dm.buffer, size_dm, M, K/256); transpose_2d_as_8b (backend_ctx, extra->s, buf_trans_s.buffer, size_s, M, K/256*12, true, true); @@ -10363,20 +12275,16 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, if (tensor->type == GGML_TYPE_Q5_K) { ggml_tensor_extra_cl_q5_K * extra = (ggml_tensor_extra_cl_q5_K *)tensor->extra; - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_uchar mask_0F = 0x0F; cl_uchar mask_F0 = 0xF0; #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q5_k_trans4_ns; int ne00 = tensor->ne[0]; @@ -10480,11 +12388,64 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, ggml_tensor_extra_cl_q6_K * extra = (ggml_tensor_extra_cl_q6_K *)tensor->extra; #ifdef GGML_OPENCL_USE_ADRENO_KERNELS + // Undo the 64-row-tiled canonical pack (kernel_convert_block_q6_k_tiled_ns). + // See the q4_K tiled restore above for why a read-back path is required. + if (use_q6k_tiled(backend_ctx, tensor)) { + const int ne00v = tensor->ne[0]; + const int ne01v = tensor->ne[1]; + const int nbv = ne00v / 256; + const size_t n_blk = (size_t)nbv * ne01v; + + std::vector tql(n_blk*32), tqh(n_blk*16); + std::vector ts(n_blk*16); + std::vector td(n_blk); + CL_CHECK(clEnqueueReadBuffer(queue, extra->ql, CL_TRUE, 0, tql.size()*4, tql.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->qh, CL_TRUE, 0, tqh.size()*4, tqh.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->s, CL_TRUE, 0, ts.size(), ts.data(), 0, NULL, NULL)); + CL_CHECK(clEnqueueReadBuffer(queue, extra->d, CL_TRUE, 0, td.size()*2, td.data(), 0, NULL, NULL)); + + std::vector rebuilt(ggml_nbytes(tensor), 0); + for (int i01 = 0; i01 < ne01v; ++i01) { + const int rt = i01/64, rit = i01%64; + for (int i00 = 0; i00 < nbv; ++i00) { + uint8_t * b = rebuilt.data() + ((size_t)i00 + (size_t)i01*nbv)*210; + const int tb = rt*nbv + i00; + const size_t si = (size_t)tb*64 + rit; + + uint32_t qlw[32], qhw[16]; + for (int g = 0; g < 8; ++g) { + const size_t base = ((size_t)tb*8 + g)*64 + rit; + for (int j = 0; j < 4; ++j) qlw[g*4 + j] = tql[base*4 + j]; + } + for (int g = 0; g < 4; ++g) { + const size_t base = ((size_t)tb*4 + g)*64 + rit; + for (int j = 0; j < 4; ++j) qhw[g*4 + j] = tqh[base*4 + j]; + } + + uint8_t * ql = b; + uint8_t * qh = b + 128; + for (int e = 0; e < 256; ++e) { + const int n = (e >= 128) ? 1 : 0; + const int within = e - n*128, q = within/32, l = within%32; + const int off_ql = n*64, off_qh = n*32; + const uint8_t low4 = (qlw[e>>3] >> ((e&7)*4)) & 0xF; + const uint8_t hi2 = (qhw[e>>4] >> ((e&15)*2)) & 0x3; + if (q == 0) ql[off_ql + l] |= low4; + else if (q == 1) ql[off_ql + l + 32] |= low4; + else if (q == 2) ql[off_ql + l] |= (uint8_t)(low4 << 4); + else ql[off_ql + l + 32] |= (uint8_t)(low4 << 4); + qh[off_qh + l] |= (uint8_t)(hi2 << (q*2)); + } + memcpy(b + 192, &ts[si*16], 16); + memcpy(b + 208, &td[si], 2); + } + } + memcpy(data, rebuilt.data() + offset, size); + return; + } if (use_adreno_moe_kernels(backend_ctx, tensor)) { - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_kernel kernel = backend_ctx->kernel_restore_block_q6_k_trans4_ns; @@ -10537,15 +12498,24 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, buf_trans_ql.allocate(backend_ctx->context, size_ql); buf_trans_qh.allocate(backend_ctx->context, size_qh); - buf_trans_s.allocate(backend_ctx->context, size_s); buf_trans_d.allocate(backend_ctx->context, size_d); buf_unpacked.allocate(backend_ctx->context, ggml_nbytes(tensor)); - // transpose ql, qh, s and d back - transpose_2d_as_16b(backend_ctx, extra->ql, buf_trans_ql.buffer, size_ql, M, K/4); - transpose_2d_as_8b(backend_ctx, extra->qh, buf_trans_qh.buffer, size_qh, M, K/4); - transpose_2d_as_16b(backend_ctx, extra->s, buf_trans_s.buffer, size_s, M, K/16/2); - transpose_2d_as_16b(backend_ctx, extra->d, buf_trans_d.buffer, size_d, M, K/256); + cl_mem s_buffer; + if (use_q6_k_bin_kernels(backend_ctx, tensor)) { + transpose_2d_as_32b(backend_ctx, extra->ql, buf_trans_ql.buffer, size_ql, M, K/8); + transpose_2d_as_32b(backend_ctx, extra->qh, buf_trans_qh.buffer, size_qh, M, K/16); + // s is left row-major, untransposed, for the binary layout. + s_buffer = extra->s; + } else { + // transpose ql, qh, s and d back + buf_trans_s.allocate(backend_ctx->context, size_s); + transpose_2d_as_16b(backend_ctx, extra->ql, buf_trans_ql.buffer, size_ql, M, K/4); + transpose_2d_as_8b(backend_ctx, extra->qh, buf_trans_qh.buffer, size_qh, M, K/4); + transpose_2d_as_16b(backend_ctx, extra->s, buf_trans_s.buffer, size_s, M, K/16/2); + s_buffer = buf_trans_s.buffer; + } + transpose_2d_as_16b(backend_ctx, extra->d, buf_trans_d.buffer, size_d, M, K/256); // unpack cl_uchar mask = 0xFF; @@ -10553,7 +12523,7 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, cl_kernel kernel = backend_ctx->kernel_restore_block_q6_K_noshuffle; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &buf_trans_ql.buffer)); CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &buf_trans_qh.buffer)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &buf_trans_s.buffer)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s_buffer)); CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &buf_trans_d.buffer)); CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &buf_unpacked.buffer)); CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_uchar), &mask)); @@ -10571,10 +12541,8 @@ static void ggml_backend_opencl_buffer_get_tensor(ggml_backend_buffer_t buffer, } #endif // GGML_OPENCL_USE_ADRENO_KERNELS - cl_int err; - cl_mem data_device = clCreateBuffer(context, CL_MEM_READ_WRITE, - ggml_nbytes(tensor), NULL, &err); - CL_CHECK(err); + cl_mem data_device = ggml_cl_create_temp_download_buffer(context, queue, ggml_nbytes(tensor), tensor->name); + GGML_ASSERT(data_device != NULL && "get_tensor: temp download buffer alloc failed"); cl_uchar mask = 0xFF; cl_ulong n_blk = ggml_nelements(tensor)/ggml_blck_size(tensor->type); @@ -10702,6 +12670,21 @@ static ggml_backend_buffer_t ggml_backend_opencl_buffer_type_alloc_buffer(ggml_b cl_int err; cl_mem mem = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, size, NULL, &err); + // On Adreno X1-85 the device pool intermittently fails at hundreds of MB + // once the heap fragments (e.g. graph-allocator compute-buffer reserve + // after model load). Four-step retry: + // 1. normal alloc (fast path) + // 2. clFinish + retry (drains in-flight allocs) + // 3. cl_qcom_large_buffer (X2-class driver only, OpenCL 3.0 only) + // 4. ALLOC_HOST_PTR (host-pinned pool) — last-resort fallback. This + // buffer backs compute scratch read/written by every kernel in the + // graph, so kernel accesses fall to host memory and runtime perf + // degrades meaningfully. Better than failing to load, but the user + // should see the warning and consider -ngl reduction. + if (err != CL_SUCCESS) { + clFinish(backend_ctx->queue); + mem = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, size, NULL, &err); + } #if GGML_OPENCL_TARGET_VERSION >= 300 // clCreateBufferWithProperties and cl_mem_properties are OpenCL 3.0. Drivers older than // that do not export the symbol, so a build targeting them fails to link. The large @@ -10712,9 +12695,20 @@ static ggml_backend_buffer_t ggml_backend_opencl_buffer_type_alloc_buffer(ggml_b mem = clCreateBufferWithProperties(backend_ctx->context, props, CL_MEM_READ_WRITE, size, NULL, &err); } #endif + if (err != CL_SUCCESS) { + mem = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE | CL_MEM_ALLOC_HOST_PTR, size, NULL, &err); + if (err == CL_SUCCESS) { + GGML_LOG_WARN("%s: %.2f MiB allocated via CL_MEM_ALLOC_HOST_PTR fallback — " + "device pool exhausted; runtime perf will be degraded. " + "Consider lowering -ngl or context size.\n", + __func__, size / 1024.0 / 1024.0); + } + } if (err != CL_SUCCESS) { - GGML_LOG_INFO("%s: failed to allocate %.2f MiB\n", __func__, size / 1024.0 / 1024.0); + GGML_LOG_ERROR("%s: failed to allocate %.2f MiB (err=%d). " + "Consider reducing -ngl, lowering -c / -ub, or using quantized KV cache.\n", + __func__, size / 1024.0 / 1024.0, err); return nullptr; } @@ -11469,6 +13463,7 @@ static void ggml_cl_set_rows(ggml_backend_t backend, const ggml_tensor * src0, c (size_t)ne03}; size_t local_work_size[] = {(size_t)nth0, (size_t)rows_per_workgroup, 1}; + // ne01 == 0 makes global_work_size[0] zero here; enqueue_ndrange_kernel drops the empty range. backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); } @@ -12257,6 +14252,146 @@ static void ggml_cl_mean(ggml_backend_t backend, const ggml_tensor * src0, const backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); } +static void ggml_cl_ssm_scan(ggml_backend_t backend, ggml_tensor * dst) { + GGML_ASSERT(dst); + GGML_ASSERT(dst->extra); + GGML_ASSERT(dst->src[0]); + GGML_ASSERT(dst->src[0]->extra); + GGML_ASSERT(dst->src[1]); + GGML_ASSERT(dst->src[1]->extra); + GGML_ASSERT(dst->src[2]); + GGML_ASSERT(dst->src[2]->extra); + GGML_ASSERT(dst->src[3]); + GGML_ASSERT(dst->src[3]->extra); + GGML_ASSERT(dst->src[4]); + GGML_ASSERT(dst->src[4]->extra); + GGML_ASSERT(dst->src[5]); + GGML_ASSERT(dst->src[5]->extra); + GGML_ASSERT(dst->src[6]); + GGML_ASSERT(dst->src[6]->extra); + + ggml_backend_opencl_context * backend_ctx = (ggml_backend_opencl_context *) backend->context; + + ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *) dst->src[0]->extra; + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *) dst->src[1]->extra; + ggml_tensor_extra_cl * extra2 = (ggml_tensor_extra_cl *) dst->src[2]->extra; + ggml_tensor_extra_cl * extra3 = (ggml_tensor_extra_cl *) dst->src[3]->extra; + ggml_tensor_extra_cl * extra4 = (ggml_tensor_extra_cl *) dst->src[4]->extra; + ggml_tensor_extra_cl * extra5 = (ggml_tensor_extra_cl *) dst->src[5]->extra; + ggml_tensor_extra_cl * extra6 = (ggml_tensor_extra_cl *) dst->src[6]->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *) dst->extra; + + const cl_ulong offset0 = extra0->offset + dst->src[0]->view_offs; + const cl_ulong offset1 = extra1->offset + dst->src[1]->view_offs; + const cl_ulong offset2 = extra2->offset + dst->src[2]->view_offs; + const cl_ulong offset3 = extra3->offset + dst->src[3]->view_offs; + const cl_ulong offset4 = extra4->offset + dst->src[4]->view_offs; + const cl_ulong offset5 = extra5->offset + dst->src[5]->view_offs; + const cl_ulong offset6 = extra6->offset + dst->src[6]->view_offs; + const cl_ulong offsetd = extrad->offset + dst->view_offs; + + const ggml_tensor * s = dst->src[0]; + const ggml_tensor * x = dst->src[1]; + const ggml_tensor * dt = dst->src[2]; + const ggml_tensor * A = dst->src[3]; + const ggml_tensor * B = dst->src[4]; + const ggml_tensor * C = dst->src[5]; + + const cl_ulong s_nb1 = s->nb[1]; + const cl_ulong s_nb2 = s->nb[2]; + const cl_ulong s_nb3 = s->nb[3]; + const cl_ulong x_nb1 = x->nb[1]; + const cl_ulong x_nb2 = x->nb[2]; + const cl_ulong x_nb3 = x->nb[3]; + const cl_ulong dt_nb1 = dt->nb[1]; + const cl_ulong dt_nb2 = dt->nb[2]; + const cl_ulong A_nb1 = A->nb[1]; + const cl_ulong B_nb1 = B->nb[1]; + const cl_ulong B_nb2 = B->nb[2]; + const cl_ulong B_nb3 = B->nb[3]; + const cl_ulong C_nb1 = C->nb[1]; + const cl_ulong C_nb2 = C->nb[2]; + const cl_ulong C_nb3 = C->nb[3]; + + const cl_uint A_ne0 = A->ne[0]; + const cl_uint d_state = s->ne[0]; + const cl_int head_dim = x->ne[0]; + const cl_int n_head = x->ne[1]; + const cl_int n_group = B->ne[1]; + const cl_int n_tokens = x->ne[2]; + const cl_uint n_seqs = x->ne[3]; + const cl_uint K = ggml_get_op_params_i32(dst, 0); + const cl_ulong s_off_bytes = (cl_ulong) ggml_nelements(x) * sizeof(float); + + cl_kernel kernel = backend_ctx->kernel_ssm_scan_f32; + size_t nth = d_state; + if (A_ne0 == 1 && K == 1) { + cl_kernel kernel_mamba2 = nullptr; + if (d_state == 128) { + kernel_mamba2 = backend_ctx->kernel_ssm_scan_f32_mamba2_d128; + } else if (d_state == 256) { + kernel_mamba2 = backend_ctx->kernel_ssm_scan_f32_mamba2_d256; + } + if (kernel_mamba2 != nullptr) { + kernel = kernel_mamba2; + nth = 64; + } + } + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra1->data_device)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &extra2->data_device)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &offset2)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_mem), &extra3->data_device)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_ulong), &offset3)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_mem), &extra4->data_device)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &offset4)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_mem), &extra5->data_device)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_ulong), &offset5)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_mem), &extra6->data_device)); + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_ulong), &offset6)); + CL_CHECK(clSetKernelArg(kernel, 14, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 15, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 16, sizeof(cl_ulong), &s_nb2)); + CL_CHECK(clSetKernelArg(kernel, 17, sizeof(cl_ulong), &s_nb3)); + CL_CHECK(clSetKernelArg(kernel, 18, sizeof(cl_ulong), &x_nb2)); + CL_CHECK(clSetKernelArg(kernel, 19, sizeof(cl_ulong), &x_nb3)); + CL_CHECK(clSetKernelArg(kernel, 20, sizeof(cl_ulong), &dt_nb1)); + CL_CHECK(clSetKernelArg(kernel, 21, sizeof(cl_ulong), &dt_nb2)); + CL_CHECK(clSetKernelArg(kernel, 22, sizeof(cl_ulong), &A_nb1)); + CL_CHECK(clSetKernelArg(kernel, 23, sizeof(cl_ulong), &B_nb2)); + CL_CHECK(clSetKernelArg(kernel, 24, sizeof(cl_ulong), &B_nb3)); + CL_CHECK(clSetKernelArg(kernel, 25, sizeof(cl_ulong), &C_nb2)); + CL_CHECK(clSetKernelArg(kernel, 26, sizeof(cl_ulong), &C_nb3)); + CL_CHECK(clSetKernelArg(kernel, 27, sizeof(cl_ulong), &s_off_bytes)); + CL_CHECK(clSetKernelArg(kernel, 28, sizeof(cl_int), &head_dim)); + CL_CHECK(clSetKernelArg(kernel, 29, sizeof(cl_int), &n_head)); + CL_CHECK(clSetKernelArg(kernel, 30, sizeof(cl_int), &n_group)); + CL_CHECK(clSetKernelArg(kernel, 31, sizeof(cl_int), &n_tokens)); + + if (kernel == backend_ctx->kernel_ssm_scan_f32) { + CL_CHECK(clSetKernelArg(kernel, 32, sizeof(cl_ulong), &s_nb1)); + CL_CHECK(clSetKernelArg(kernel, 33, sizeof(cl_ulong), &x_nb1)); + CL_CHECK(clSetKernelArg(kernel, 34, sizeof(cl_ulong), &B_nb1)); + CL_CHECK(clSetKernelArg(kernel, 35, sizeof(cl_ulong), &C_nb1)); + CL_CHECK(clSetKernelArg(kernel, 36, sizeof(cl_uint), &A_ne0)); + CL_CHECK(clSetKernelArg(kernel, 37, sizeof(cl_uint), &d_state)); + CL_CHECK(clSetKernelArg(kernel, 38, sizeof(cl_uint), &n_seqs)); + CL_CHECK(clSetKernelArg(kernel, 39, sizeof(cl_uint), &K)); + CL_CHECK(clSetKernelArg(kernel, 40, d_state * sizeof(float), nullptr)); + } + + size_t global_work_size[] = { + (size_t) head_dim * (size_t) n_head * nth, + (size_t) n_seqs, + }; + size_t local_work_size[] = { nth, 1 }; + + backend_ctx->enqueue_ndrange_kernel(kernel, 2, global_work_size, local_work_size, dst); +} + static void ggml_cl_ssm_conv(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { GGML_ASSERT(src0); GGML_ASSERT(src0->extra); @@ -12690,7 +14825,10 @@ static void ggml_cl_norm(ggml_backend_t backend, const ggml_tensor * src0, const GGML_TENSOR_LOCALS(int, ne0, src0, ne); GGML_TENSOR_LOCALS(cl_ulong, nb0, src0, nb); - const int nth = MIN(64, ne00); + int nth = 1; + while (nth < ne00 && nth < 64) { + nth *= 2; + } cl_kernel kernel = backend_ctx->kernel_norm; @@ -13595,14 +15733,17 @@ static void ggml_cl_abs(ggml_backend_t backend, const ggml_tensor * src0, const } } -static void ggml_cl_softplus(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { +// Shared driver for the extended unary ops (unary_ext.cl), same selection as +// ggml_cl_abs: contiguous picks the vec4 kernel when the element count is a +// multiple of 4 (else scalar); non-contiguous uses the stride-addressed kernel. +static void ggml_cl_unary_ext(ggml_backend_t backend, const ggml_tensor * src0, ggml_tensor * dst, + cl_kernel k_f32, cl_kernel k_f32_4, cl_kernel k_f32_nc, + cl_kernel k_f16, cl_kernel k_f16_4, cl_kernel k_f16_nc) { GGML_ASSERT(src0); GGML_ASSERT(src0->extra); GGML_ASSERT(dst); GGML_ASSERT(dst->extra); - UNUSED(src1); - ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *)src0->extra; @@ -13611,39 +15752,20 @@ static void ggml_cl_softplus(ggml_backend_t backend, const ggml_tensor * src0, c cl_ulong offset0 = extra0->offset + src0->view_offs; cl_ulong offsetd = extrad->offset + dst->view_offs; - const int ne00 = src0->ne[0]; - const int ne01 = src0->ne[1]; - const int ne02 = src0->ne[2]; - const int ne03 = src0->ne[3]; - - const cl_ulong nb00 = src0->nb[0]; - const cl_ulong nb01 = src0->nb[1]; - const cl_ulong nb02 = src0->nb[2]; - const cl_ulong nb03 = src0->nb[3]; - - const cl_ulong nb0 = dst->nb[0]; - const cl_ulong nb1 = dst->nb[1]; - const cl_ulong nb2 = dst->nb[2]; - const cl_ulong nb3 = dst->nb[3]; + const int ne00 = src0->ne[0], ne01 = src0->ne[1], ne02 = src0->ne[2], ne03 = src0->ne[3]; + const cl_ulong nb00 = src0->nb[0], nb01 = src0->nb[1], nb02 = src0->nb[2], nb03 = src0->nb[3]; + const cl_ulong nb0 = dst->nb[0], nb1 = dst->nb[1], nb2 = dst->nb[2], nb3 = dst->nb[3]; + const bool is_f16 = (src0->type == GGML_TYPE_F16); cl_kernel kernel; if (ggml_is_contiguous(src0)) { - // Handle contiguous input int n = ggml_nelements(dst); if (n % 4 == 0) { - if (src0->type == GGML_TYPE_F32) { - kernel = backend_ctx->kernel_softplus_f32_4; - } else { - kernel = backend_ctx->kernel_softplus_f16_4; - } + kernel = is_f16 ? k_f16_4 : k_f32_4; n /= 4; } else { - if (src0->type == GGML_TYPE_F32) { - kernel = backend_ctx->kernel_softplus_f32; - } else { - kernel = backend_ctx->kernel_softplus_f16; - } + kernel = is_f16 ? k_f16 : k_f32; } CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); @@ -13652,21 +15774,14 @@ static void ggml_cl_softplus(ggml_backend_t backend, const ggml_tensor * src0, c CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offsetd)); size_t global_work_size[] = {(size_t)n, 1, 1}; - size_t local_work_size[] = {64, 1, 1}; - + size_t local_work_size[] = {64, 1, 1}; size_t * local_work_size_ptr = local_work_size; if (n % 64 != 0 && !backend_ctx->non_uniform_workgroups) { local_work_size_ptr = nullptr; } - backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size_ptr, dst); } else { - // Handle non-contiguous input - if (src0->type == GGML_TYPE_F32) { - kernel = backend_ctx->kernel_softplus_f32_nc; - } else { - kernel = backend_ctx->kernel_softplus_f16_nc; - } + kernel = is_f16 ? k_f16_nc : k_f32_nc; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); @@ -13683,30 +15798,47 @@ static void ggml_cl_softplus(ggml_backend_t backend, const ggml_tensor * src0, c CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_ulong), &nb3)); int nth = 64; - size_t global_work_size[] = {(size_t)ne01*nth, (size_t)ne02, (size_t)ne03}; - size_t local_work_size[] = {(size_t)nth, 1, 1}; - + size_t local_work_size[] = {(size_t)nth, 1, 1}; backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); } } -static void ggml_cl_repeat(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1_shape_def, ggml_tensor * dst) { +#define GGML_CL_UNARY_EXT_WRAP(FN, OP) \ +static void FN(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { \ + UNUSED(src1); \ + ggml_backend_opencl_context *c = (ggml_backend_opencl_context *)backend->context; \ + ggml_cl_unary_ext(backend, src0, dst, c->kernel_##OP##_f32, c->kernel_##OP##_f32_4, c->kernel_##OP##_f32_nc, \ + c->kernel_##OP##_f16, c->kernel_##OP##_f16_4, c->kernel_##OP##_f16_nc); \ +} + +GGML_CL_UNARY_EXT_WRAP(ggml_cl_sgn, sgn) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_step, step) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_elu, elu) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_hardswish, hardswish) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_hardsigmoid, hardsigmoid) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_floor, floor) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_ceil, ceil) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_round, round) +GGML_CL_UNARY_EXT_WRAP(ggml_cl_trunc, trunc) + +#undef GGML_CL_UNARY_EXT_WRAP + +static void ggml_cl_softplus(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { GGML_ASSERT(src0); GGML_ASSERT(src0->extra); GGML_ASSERT(dst); GGML_ASSERT(dst->extra); - GGML_ASSERT(dst->type == src0->type); - UNUSED(src1_shape_def); + UNUSED(src1); ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *)src0->extra; - ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; cl_ulong offset0 = extra0->offset + src0->view_offs; - cl_ulong offsetd = extrad->offset + dst->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; const int ne00 = src0->ne[0]; const int ne01 = src0->ne[1]; @@ -13718,9 +15850,106 @@ static void ggml_cl_repeat(ggml_backend_t backend, const ggml_tensor * src0, con const cl_ulong nb02 = src0->nb[2]; const cl_ulong nb03 = src0->nb[3]; - const int ne0 = dst->ne[0]; - const int ne1 = dst->ne[1]; - const int ne2 = dst->ne[2]; + const cl_ulong nb0 = dst->nb[0]; + const cl_ulong nb1 = dst->nb[1]; + const cl_ulong nb2 = dst->nb[2]; + const cl_ulong nb3 = dst->nb[3]; + + cl_kernel kernel; + + if (ggml_is_contiguous(src0)) { + // Handle contiguous input + int n = ggml_nelements(dst); + if (n % 4 == 0) { + if (src0->type == GGML_TYPE_F32) { + kernel = backend_ctx->kernel_softplus_f32_4; + } else { + kernel = backend_ctx->kernel_softplus_f16_4; + } + n /= 4; + } else { + if (src0->type == GGML_TYPE_F32) { + kernel = backend_ctx->kernel_softplus_f32; + } else { + kernel = backend_ctx->kernel_softplus_f16; + } + } + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offsetd)); + + size_t global_work_size[] = {(size_t)n, 1, 1}; + size_t local_work_size[] = {64, 1, 1}; + + size_t * local_work_size_ptr = local_work_size; + if (n % 64 != 0 && !backend_ctx->non_uniform_workgroups) { + local_work_size_ptr = nullptr; + } + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size_ptr, dst); + } else { + // Handle non-contiguous input + if (src0->type == GGML_TYPE_F32) { + kernel = backend_ctx->kernel_softplus_f32_nc; + } else { + kernel = backend_ctx->kernel_softplus_f16_nc; + } + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &nb00)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &nb01)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_ulong), &nb02)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &nb03)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &nb0)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &nb1)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_ulong), &nb2)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_ulong), &nb3)); + + int nth = 64; + + size_t global_work_size[] = {(size_t)ne01*nth, (size_t)ne02, (size_t)ne03}; + size_t local_work_size[] = {(size_t)nth, 1, 1}; + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + } +} + +static void ggml_cl_repeat(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1_shape_def, ggml_tensor * dst) { + GGML_ASSERT(src0); + GGML_ASSERT(src0->extra); + GGML_ASSERT(dst); + GGML_ASSERT(dst->extra); + GGML_ASSERT(dst->type == src0->type); + + UNUSED(src1_shape_def); + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *)src0->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + + cl_ulong offset0 = extra0->offset + src0->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + const int ne00 = src0->ne[0]; + const int ne01 = src0->ne[1]; + const int ne02 = src0->ne[2]; + const int ne03 = src0->ne[3]; + + const cl_ulong nb00 = src0->nb[0]; + const cl_ulong nb01 = src0->nb[1]; + const cl_ulong nb02 = src0->nb[2]; + const cl_ulong nb03 = src0->nb[3]; + + const int ne0 = dst->ne[0]; + const int ne1 = dst->ne[1]; + const int ne2 = dst->ne[2]; const int ne3 = dst->ne[3]; const cl_ulong nb0 = dst->nb[0]; @@ -13972,9 +16201,8 @@ static void ggml_cl_concat(ggml_backend_t backend, const ggml_tensor * src0, con GGML_ASSERT(src1->extra); GGML_ASSERT(dst); GGML_ASSERT(dst->extra); - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(src0->type == src1->type); + GGML_ASSERT(src0->type == dst->type); ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; @@ -14016,9 +16244,21 @@ static void ggml_cl_concat(ggml_backend_t backend, const ggml_tensor * src0, con int nth = MIN(64, ne0); - const bool concat_pack = (dim == 0 && ne0 < 32); - cl_kernel kernel = concat_pack ? backend_ctx->kernel_concat_f32_pack - : backend_ctx->kernel_concat_f32; + const size_t ts = ggml_type_size(dst->type); + // the pack kernel copies 4-byte elements, so it is only valid for those. + const bool concat_pack = (dim == 0 && ne0 < 32 && ts == 4); + cl_kernel kernel; + if (concat_pack) { + kernel = backend_ctx->kernel_concat_b4_pack; + } else { + switch (ts) { + case 1: kernel = backend_ctx->kernel_concat_b1; break; + case 2: kernel = backend_ctx->kernel_concat_b2; break; + case 4: kernel = backend_ctx->kernel_concat_b4; break; + case 8: kernel = backend_ctx->kernel_concat_b8; break; + default: GGML_ABORT("unsupported concat element size: %zu", ts); + } + } CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0->data_device)); CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset0)); @@ -14390,98 +16630,1074 @@ static bool ggml_cl_flash_attn_prepare_quantized_tensor( cl_int err; temp.data = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, buffer_size, NULL, &err); CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer(backend_ctx->queue, temp.data, CL_TRUE, 0, buffer_size, host_linear.data(), 0, NULL, NULL)); - - data_device = temp.data; - offset = 0; - nb1 = (cl_ulong) (tensor->ne[0] * bytes_per_elem); - nb2 = (cl_ulong) (tensor->ne[1] * nb1); - nb3 = (cl_ulong) (tensor->ne[2] * nb2); + CL_CHECK(clEnqueueWriteBuffer(backend_ctx->queue, temp.data, CL_TRUE, 0, buffer_size, host_linear.data(), 0, NULL, NULL)); + + data_device = temp.data; + offset = 0; + nb1 = (cl_ulong) (tensor->ne[0] * bytes_per_elem); + nb2 = (cl_ulong) (tensor->ne[1] * nb1); + nb3 = (cl_ulong) (tensor->ne[2] * nb2); + + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("ggml_opencl: OpenCL flash attention dequantizes GPU-resident quantized KV cache into temporary linear buffers; performance may be poor\n"); + warned = true; + } + + return true; +} + +// Host-side F16 -> F32 for the asymmetric-KV F32 fallback path. +static bool ggml_cl_flash_attn_convert_f16_to_f32( + ggml_backend_opencl_context * backend_ctx, + const ggml_tensor * tensor, + ggml_cl_flash_attn_temp_buffer & temp, + cl_mem & data_device, + cl_ulong & offset, + cl_ulong & nb1, + cl_ulong & nb2, + cl_ulong & nb3 +) { + if (tensor->type != GGML_TYPE_F16) { + return false; + } + + cl_mem src_buffer = data_device; + cl_ulong src_offset = offset; + cl_ulong src_nb1 = nb1; + cl_ulong src_nb2 = nb2; + cl_ulong src_nb3 = nb3; + ggml_cl_flash_attn_resolve_src(tensor, src_buffer, src_offset, src_nb1, src_nb2, src_nb3); + + const int64_t n = ggml_nelements(tensor); + const size_t row_bytes = (size_t) tensor->ne[0] * sizeof(ggml_fp16_t); + const size_t total_bytes = (size_t) n * sizeof(ggml_fp16_t); + std::vector host_f16(total_bytes); + + sync_with_other_backends(backend_ctx); + ggml_cl_flash_attn_read_tensor_host(backend_ctx, tensor, src_buffer, src_offset, + src_nb1, src_nb2, src_nb3, + row_bytes, host_f16.data(), total_bytes); + + std::vector host_f32(n); + ggml_fp16_to_fp32_row((const ggml_fp16_t *) host_f16.data(), host_f32.data(), n); + + const size_t f32_bytes = (size_t) n * sizeof(float); + cl_int err; + temp.data = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, f32_bytes, NULL, &err); + CL_CHECK(err); + CL_CHECK(clEnqueueWriteBuffer(backend_ctx->queue, temp.data, CL_TRUE, 0, + f32_bytes, host_f32.data(), 0, NULL, NULL)); + + data_device = temp.data; + offset = 0; + nb1 = (cl_ulong) (tensor->ne[0] * sizeof(float)); + nb2 = (cl_ulong) (tensor->ne[1] * nb1); + nb3 = (cl_ulong) (tensor->ne[2] * nb2); + + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("ggml_opencl: OpenCL flash attention asymmetric KV converts an F16 cache to F32 host-side; performance may be poor\n"); + warned = true; + } + + return true; +} + +// Flash-Decoding (K-split) dispatch thresholds. FD fires for non-causal +// attention with n_kv >= FD_MIN_N_KV and d_head <= FD_MAX_DK; the KV range is +// split into ~n_kv/FD_KV_PER_SPLIT partials, clamped to [FD_MIN_SPLITS, +// FD_MAX_SPLITS]. Multi-query FD is restricted to small heads +// (d_head <= FD_MAX_DK_MULTI) and capped at FD_MAX_N_Q_MULTI queries. +static constexpr int FD_MIN_N_KV = 2048; +static constexpr int FD_KV_PER_SPLIT = 2048; +// f16 KV decode wants more splits than the 2048 default; quantized KV keeps 2048. +static constexpr int FD_KV_PER_SPLIT_F16 = 512; +static constexpr int FD_MIN_SPLITS = 2; +static constexpr int FD_MAX_SPLITS = 16; +static constexpr int FD_MAX_DK = 128; +static constexpr int FD_MAX_DK_MULTI = 64; +static constexpr int FD_MAX_N_Q_MULTI = 8; +// MQ FD split-groups have few subgroups (MQ_NSG_SPLIT), so use a smaller +// kv_per_split to keep the softmax recurrence short; non-MQ keeps FD_KV_PER_SPLIT. +static constexpr int FD_MQ_KV_PER_SPLIT = 256; +static constexpr int FD_MQ_MAX_SPLITS = 128; + +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +struct ggml_cl_adreno_xmem_attn_schedule { + int qk_lws0 = 256; + int qk_lws2 = 1; + int softmax_reduce_lws0 = 256; + int softmax_apply_lws0 = 64; + int softmax_apply_lws2 = 4; + int pv_lws0 = 64; + int pv_lws2 = 4; +}; + +static inline size_t ggml_cl_round_up(size_t x, size_t a) { + return ((x + a - 1) / a) * a; +} + +static inline int ggml_cl_round_up_div(int x, int y) { + return (x + y - 1) / y; +} + +static inline void ggml_cl_set_arg_int4(cl_kernel kernel, cl_uint index, int x, int y, int z, int w) { + struct { int x, y, z, w; } value { x, y, z, w }; + CL_CHECK(clSetKernelArg(kernel, index, sizeof(value), &value)); +} + +static cl_mem ggml_cl_make_image2d_half4(cl_context context, cl_mem_flags flags, size_t width, size_t height) { + cl_int err = CL_SUCCESS; + cl_image_format format = { CL_RGBA, CL_HALF_FLOAT }; + cl_image_desc desc = {}; + desc.image_type = CL_MEM_OBJECT_IMAGE2D; + desc.image_width = width; + desc.image_height = height; + cl_mem image = clCreateImage(context, flags, &format, &desc, nullptr, &err); + CL_CHECK(err); + return image; +} + +static cl_mem ggml_cl_make_image1d_buffer_half4(cl_context context, cl_mem_flags flags, size_t width, cl_mem backing_buffer) { + cl_int err = CL_SUCCESS; + cl_image_format format = { CL_RGBA, CL_HALF_FLOAT }; + cl_image_desc desc = {}; + desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + desc.image_width = width; + desc.buffer = backing_buffer; + cl_mem image = clCreateImage(context, flags, &format, &desc, nullptr, &err); + CL_CHECK(err); + return image; +} + +static void ggml_cl_release_mem(cl_mem & mem) { + if (mem != nullptr) { + CL_CHECK(clReleaseMemObject(mem)); + mem = nullptr; + } +} + +static void ggml_cl_adreno_xmem_attn_release_scratch(ggml_backend_opencl_context * backend_ctx) { + auto & s = backend_ctx->adreno_xmem_attn.scratch; + ggml_cl_release_mem(s.q_img); + ggml_cl_release_mem(s.k_img); + ggml_cl_release_mem(s.v_img); + ggml_cl_release_mem(s.out_img); + ggml_cl_release_mem(s.k_transpose_img1d); + ggml_cl_release_mem(s.k_transpose_buf); + ggml_cl_release_mem(s.k_packed_buf); + ggml_cl_release_mem(s.v_packed_buf); + ggml_cl_release_mem(s.score_img1d); + ggml_cl_release_mem(s.prob_img1d); + ggml_cl_release_mem(s.score_buf); + ggml_cl_release_mem(s.prob_buf); + ggml_cl_release_mem(s.softmax_stats_img2d); + ggml_cl_release_mem(s.xmem_qk); + ggml_cl_release_mem(s.xmem_pv); + s = {}; +} + +static ggml_cl_adreno_xmem_attn_schedule ggml_cl_adreno_xmem_attn_select_schedule( + const ggml_backend_opencl_context * backend_ctx, + int n_q, + int n_kv, + int heads_total, + int q_width, + int gqa_ratio) { + const bool big_h = heads_total >= 8; + ggml_cl_adreno_xmem_attn_schedule sched; + + if (gqa_ratio == 1) { + if (n_q >= 512) { sched.qk_lws0 = 512; } + else if (n_q >= 256) { sched.qk_lws0 = 128; } + else { sched.qk_lws0 = 64; } + sched.qk_lws2 = (big_h && n_q >= 512) ? 2 : 1; + } else { + if (q_width >= 2048) { sched.qk_lws0 = 512; } + else if (q_width >= 256) { sched.qk_lws0 = 128; } + else { sched.qk_lws0 = 64; } + sched.qk_lws2 = MIN(8, (int) backend_ctx->max_workgroup_size / sched.qk_lws0); + } + + if (n_kv >= 2048) { sched.softmax_reduce_lws0 = 1024; } + else if (n_kv >= 512) { sched.softmax_reduce_lws0 = big_h ? 256 : 512; } + else { sched.softmax_reduce_lws0 = 256; } + + if (n_kv < 256) { sched.softmax_apply_lws0 = 64; } + else { sched.softmax_apply_lws0 = big_h ? 128 : 64; } + sched.softmax_apply_lws2 = n_kv >= 512 ? 8 : 4; + + if (n_q < 256) { sched.pv_lws0 = 64; } + else { sched.pv_lws0 = big_h ? 128 : 64; } + sched.pv_lws2 = big_h ? 8 : (n_q <= 256 ? 8 : 4); + + const int max_wg = (int) backend_ctx->max_workgroup_size; + auto fix = [&](int & l0, int & l2) { + while (l0 * l2 > max_wg) { + if (l2 > 1) { l2 /= 2; } + else if (l0 > 32) { l0 /= 2; } + else { break; } + } + }; + fix(sched.qk_lws0, sched.qk_lws2); + fix(sched.softmax_apply_lws0, sched.softmax_apply_lws2); + fix(sched.pv_lws0, sched.pv_lws2); + while (sched.softmax_reduce_lws0 > max_wg) { + sched.softmax_reduce_lws0 /= 2; + } + + return sched; +} + +static bool ggml_cl_adreno_xmem_attn_prepare( + ggml_backend_opencl_context * backend_ctx, + int n_q, + int n_kv, + int d_head_q, + int d_head_v, + int n_head, + int n_head_kv, + int n_batch) { + auto & s = backend_ctx->adreno_xmem_attn.scratch; + const int gqa_ratio = n_head / n_head_kv; + const int q_width = n_q * gqa_ratio; + const int kv_heads_total = n_head_kv * n_batch; + const int n_kv_padded = (int) ggml_cl_round_up((size_t) n_kv, 32); + if (s.q_img != nullptr && + s.n_q == n_q && + s.n_kv == n_kv && + s.n_kv_padded == n_kv_padded && + s.d_head_q == d_head_q && + s.d_head_v == d_head_v && + s.q_width == q_width && + s.kv_heads_total == kv_heads_total) { + return true; + } + + ggml_cl_adreno_xmem_attn_release_scratch(backend_ctx); + + const int qpack = d_head_q / 4; + const int vpack = d_head_v / 4; + const int npack = n_kv_padded / 4; + const size_t q_img_h = (size_t) kv_heads_total * qpack; + const size_t v_img_h = (size_t) kv_heads_total * vpack; + + s.q_img = ggml_cl_make_image2d_half4(backend_ctx->context, CL_MEM_READ_WRITE, (size_t) q_width, q_img_h); + s.k_img = ggml_cl_make_image2d_half4(backend_ctx->context, CL_MEM_READ_WRITE, (size_t) n_kv_padded, q_img_h); + s.v_img = ggml_cl_make_image2d_half4(backend_ctx->context, CL_MEM_READ_WRITE, (size_t) n_kv_padded, v_img_h); + s.out_img = ggml_cl_make_image2d_half4(backend_ctx->context, CL_MEM_READ_WRITE, (size_t) q_width, v_img_h); + + const size_t k_transpose_half4_elems = (size_t) npack * kv_heads_total * d_head_q; + s.k_transpose_buf = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, k_transpose_half4_elems * sizeof(uint16_t) * 4, nullptr, nullptr); + GGML_ASSERT(s.k_transpose_buf != nullptr); + s.k_transpose_img1d = ggml_cl_make_image1d_buffer_half4(backend_ctx->context, CL_MEM_READ_ONLY, k_transpose_half4_elems, s.k_transpose_buf); + + const size_t k_groups16 = (size_t) ggml_cl_round_up_div(kv_heads_total * d_head_q, 16); + const size_t v_groups16 = (size_t) ggml_cl_round_up_div(kv_heads_total * d_head_v, 16); + const size_t k_packed_half4_elems = (size_t) n_kv_padded * k_groups16 * 4; + const size_t v_packed_half4_elems = (size_t) n_kv_padded * v_groups16 * 4; + s.k_packed_buf = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, k_packed_half4_elems * sizeof(uint16_t) * 4, nullptr, nullptr); + s.v_packed_buf = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, v_packed_half4_elems * sizeof(uint16_t) * 4, nullptr, nullptr); + GGML_ASSERT(s.k_packed_buf != nullptr && s.v_packed_buf != nullptr); + + const size_t score_half4_elems = (size_t) npack * kv_heads_total * q_width; + const size_t score_bytes = score_half4_elems * sizeof(uint16_t) * 4; + s.score_buf = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, score_bytes, nullptr, nullptr); + s.prob_buf = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, score_bytes, nullptr, nullptr); + GGML_ASSERT(s.score_buf != nullptr && s.prob_buf != nullptr); + s.score_img1d = ggml_cl_make_image1d_buffer_half4(backend_ctx->context, CL_MEM_READ_ONLY, score_half4_elems, s.score_buf); + s.prob_img1d = ggml_cl_make_image1d_buffer_half4(backend_ctx->context, CL_MEM_READ_ONLY, score_half4_elems, s.prob_buf); + s.softmax_stats_img2d = ggml_cl_make_image2d_half4(backend_ctx->context, CL_MEM_READ_WRITE, + (size_t) q_width, (size_t) kv_heads_total); + s.xmem_qk = clCreateBuffer(backend_ctx->context, CL_MEM_READ_ONLY, 6144, nullptr, nullptr); + s.xmem_pv = clCreateBuffer(backend_ctx->context, CL_MEM_READ_ONLY, 6144, nullptr, nullptr); + GGML_ASSERT(s.softmax_stats_img2d != nullptr && s.xmem_qk != nullptr && s.xmem_pv != nullptr); + + s.n_q = n_q; + s.n_kv = n_kv; + s.n_kv_padded = n_kv_padded; + s.d_head_q = d_head_q; + s.d_head_v = d_head_v; + s.q_width = q_width; + s.kv_heads_total = kv_heads_total; + return true; +} + +static bool ggml_cl_adreno_xmem_attn_can_use( + const ggml_backend_opencl_context * backend_ctx, + const ggml_tensor * q, + const ggml_tensor * k, + const ggml_tensor * dst) { + static const char * xmem_sdpa_env = getenv("GGML_OPENCL_XMEM_SDPA"); + if (xmem_sdpa_env == nullptr || xmem_sdpa_env[0] == '0') { + return false; + } + + const ggml_tensor * v = dst->src[2]; + const ggml_tensor * mask = dst->src[3]; + const ggml_tensor * sinks = dst->src[4]; + + if (!backend_ctx->adreno_xmem_attn.compiled || backend_ctx->gpu_family != GPU_FAMILY::ADRENO) { + return false; + } + if (q->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32 || + (k->type != GGML_TYPE_F16 && k->type != GGML_TYPE_F32) || + (v->type != GGML_TYPE_F16 && v->type != GGML_TYPE_F32)) { + return false; + } + if (sinks != nullptr) { + return false; + } + if (q->nb[0] != ggml_type_size(q->type) || k->nb[0] != ggml_type_size(k->type) || + v->nb[0] != ggml_type_size(v->type) || dst->nb[0] != ggml_type_size(dst->type)) { + return false; + } + if (mask != nullptr && (mask->type != GGML_TYPE_F16 || mask->nb[0] != sizeof(ggml_fp16_t))) { + return false; + } + + const int n_q = q->ne[1]; + const int n_kv = k->ne[1]; + const int d_head_q = q->ne[0]; + const int d_head_v = v->ne[0]; + const int n_head = q->ne[2]; + const int n_head_kv = k->ne[2]; + const int n_batch = q->ne[3]; + + if (n_q <= 1 || n_kv <= 0 || n_kv > 8192) { + return false; + } + if (d_head_q != k->ne[0] || d_head_v != v->ne[0] || k->ne[1] != v->ne[1] || k->ne[3] != v->ne[3]) { + return false; + } + if (q->ne[3] != k->ne[3]) { + return false; + } + if (n_head_kv <= 0 || n_head % n_head_kv != 0 || k->ne[2] != v->ne[2]) { + return false; + } + if (dst->ne[0] != d_head_v || dst->ne[1] != n_head || dst->ne[2] != n_q || dst->ne[3] != n_batch) { + return false; + } + if ((d_head_q % 8) != 0 || (d_head_v % 32) != 0) { + return false; + } + if (mask != nullptr && + (mask->ne[0] < n_kv || mask->ne[1] < n_q || mask->ne[2] <= 0 || mask->ne[3] <= 0)) { + return false; + } + + float params[3]; + memcpy(params, dst->op_params, sizeof(params)); + if (params[1] != 0.0f || params[2] != 0.0f) { + return false; + } + + const int gqa_ratio = n_head / n_head_kv; + const int q_width = n_q * gqa_ratio; + const int kv_heads_total = n_head_kv * n_batch; + const int n_kv_padded = (int) ggml_cl_round_up((size_t) n_kv, 32); + const int qpack = d_head_q / 4; + const int vpack = d_head_v / 4; + const int npack = n_kv_padded / 4; + + if ((size_t) q_width > backend_ctx->image2d_max_width || + (size_t) n_kv_padded > backend_ctx->image2d_max_width) { + return false; + } + if ((size_t) kv_heads_total * (size_t) qpack > backend_ctx->image2d_max_height || + (size_t) kv_heads_total * (size_t) vpack > backend_ctx->image2d_max_height) { + return false; + } + if ((size_t) npack * (size_t) kv_heads_total * (size_t) d_head_q > backend_ctx->image_max_buffer_size || + (size_t) npack * (size_t) kv_heads_total * (size_t) q_width > backend_ctx->image_max_buffer_size) { + return false; + } + + return true; +} + +static void ggml_cl_adreno_xmem_attn_run( + ggml_backend_t backend, + const ggml_tensor * q, + const ggml_tensor * k, + ggml_tensor * dst) { + ggml_backend_opencl_context * backend_ctx = (ggml_backend_opencl_context *) backend->context; + auto & xstate = backend_ctx->adreno_xmem_attn; + auto & s = xstate.scratch; + if (!xstate.logged) { + GGML_LOG_INFO("ggml_opencl: using Adreno xmem attention path\n"); + xstate.logged = true; + } + + const ggml_tensor * v = dst->src[2]; + const ggml_tensor * mask = dst->src[3]; + + ggml_tensor_extra_cl * extra_q = (ggml_tensor_extra_cl *) q->extra; + ggml_tensor_extra_cl * extra_k = (ggml_tensor_extra_cl *) k->extra; + ggml_tensor_extra_cl * extra_v = (ggml_tensor_extra_cl *) v->extra; + ggml_tensor_extra_cl * extra_o = (ggml_tensor_extra_cl *) dst->extra; + ggml_tensor_extra_cl * extra_mask = mask ? (ggml_tensor_extra_cl *) mask->extra : nullptr; + + const cl_ulong offset_q = extra_q->offset + q->view_offs; + const cl_ulong offset_k = extra_k->offset + k->view_offs; + const cl_ulong offset_v = extra_v->offset + v->view_offs; + const cl_ulong offset_o = extra_o->offset + dst->view_offs; + const cl_ulong offset_mask = extra_mask ? extra_mask->offset + mask->view_offs : 0; + + const int n_q = q->ne[1]; + const int n_kv = k->ne[1]; + const int d_head_q = q->ne[0]; + const int d_head_v = v->ne[0]; + const int n_head = q->ne[2]; + const int n_head_kv = k->ne[2]; + const int n_batch = q->ne[3]; + const int heads_total = n_head * n_batch; + const int gqa_ratio = n_head / n_head_kv; + const int q_width = n_q * gqa_ratio; + const int kv_heads_total = n_head_kv * n_batch; + const int n_kv_padded = (int) ggml_cl_round_up((size_t) n_kv, 32); + const int qpack = d_head_q / 4; + const int opack = d_head_v / 4; + const int npack = n_kv_padded / 4; + const float scale = ((const float *) dst->op_params)[0]; + + GGML_ASSERT(ggml_cl_adreno_xmem_attn_prepare( + backend_ctx, n_q, n_kv, d_head_q, d_head_v, n_head, n_head_kv, n_batch)); + const ggml_cl_adreno_xmem_attn_schedule sched = + ggml_cl_adreno_xmem_attn_select_schedule( + backend_ctx, n_q, n_kv_padded, heads_total, q_width, gqa_ratio); + + { + size_t gws[3] = {ggml_cl_round_up((size_t) n_q, 8), (size_t) heads_total, (size_t) qpack}; + size_t lws[3] = {8, 1, (size_t) ((qpack <= 32) ? qpack : 1)}; + cl_kernel kernel = xstate.kernel_q_f32_to_img_scaled; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra_q->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset_q)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.q_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(float), &scale)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &d_head_q)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &n_q)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &n_head)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &n_batch)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &q->nb[1])); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &q->nb[2])); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_ulong), &q->nb[3])); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t gws[3] = {(size_t) n_kv_padded, (size_t) kv_heads_total, (size_t) qpack}; + size_t lws[3] = {8, 1, (size_t) ((qpack <= 32) ? qpack : 1)}; + cl_kernel kernel = k->type == GGML_TYPE_F16 ? + xstate.kernel_kv_f16_to_img_gqa : xstate.kernel_kv_f32_to_img_gqa; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra_k->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset_k)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.k_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(int), &d_head_q)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &n_kv)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &n_kv_padded)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &n_batch)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &k->nb[1])); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &k->nb[2])); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &k->nb[3])); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t gws[3] = {(size_t) n_kv_padded, (size_t) kv_heads_total, (size_t) opack}; + size_t lws[3] = {8, 1, (size_t) ((opack <= 32) ? opack : 1)}; + cl_kernel kernel = v->type == GGML_TYPE_F16 ? + xstate.kernel_kv_f16_to_img_gqa : xstate.kernel_kv_f32_to_img_gqa; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra_v->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset_v)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.v_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(int), &d_head_v)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &n_kv)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &n_kv_padded)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &n_batch)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &v->nb[1])); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &v->nb[2])); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &v->nb[3])); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t gws[3] = {(size_t) d_head_q, (size_t) kv_heads_total, (size_t) npack}; + size_t lws[3] = {(size_t) MIN(64, d_head_q), (size_t) (kv_heads_total >= 2 ? 2 : 1), (size_t) MIN(8, npack)}; + if (lws[0] * lws[1] * lws[2] > backend_ctx->max_workgroup_size) { + lws[1] = 1; + } + cl_kernel kernel = xstate.kernel_k_gather; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.k_transpose_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.k_img)); + ggml_cl_set_arg_int4(kernel, 2, n_kv_padded, kv_heads_total, npack, d_head_q); + ggml_cl_set_arg_int4(kernel, 3, qpack, 0, 0, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + { + const size_t groups16 = (size_t) ggml_cl_round_up_div(kv_heads_total * d_head_q, 16); + const size_t packed_linear = (size_t) n_kv_padded * groups16; + const size_t lws0 = MIN((size_t) 1024, backend_ctx->max_workgroup_size); + size_t gws[3] = {ggml_cl_round_up(packed_linear, lws0), 1, 1}; + size_t lws[3] = {lws0, 1, 1}; + cl_kernel kernel = xstate.kernel_pack_k; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.k_packed_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.k_transpose_img1d)); + ggml_cl_set_arg_int4(kernel, 2, 8, (int) packed_linear, qpack, d_head_q); + ggml_cl_set_arg_int4(kernel, 3, kv_heads_total, kv_heads_total, kv_heads_total, npack); + ggml_cl_set_arg_int4(kernel, 4, d_head_q, 0, 0, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t lws[3] = {(size_t) sched.qk_lws0, 1, (size_t) sched.qk_lws2}; + const int slices_per_group = sched.qk_lws2 * 8; + const size_t groups_z = (size_t) ggml_cl_round_up_div(npack, slices_per_group); + const size_t groups_x = (size_t) ggml_cl_round_up_div(q_width, sched.qk_lws0); + size_t gws[3] = { + lws[0] * groups_z, + groups_x, + (size_t) kv_heads_total * lws[2], + }; + + cl_kernel kernel = xstate.kernel_qk_gemm; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.score_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.k_packed_buf)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.xmem_qk)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &s.q_img)); + ggml_cl_set_arg_int4(kernel, 4, kv_heads_total, npack, q_width, 32); + ggml_cl_set_arg_int4(kernel, 5, qpack, 0, 0, kv_heads_total); + ggml_cl_set_arg_int4(kernel, 6, qpack, 1, 1, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + cl_mem softmax_input_img = s.score_img1d; + cl_mem softmax_output_buf = s.prob_buf; + cl_mem pv_prob_img = s.prob_img1d; + + if (mask != nullptr) { + const cl_ulong mask_nb1 = mask->nb[1]; + const cl_ulong mask_nb2 = mask->nb[2]; + const cl_ulong mask_nb3 = mask->nb[3]; + const int mask_ne2 = mask->ne[2]; + const int mask_ne3 = mask->ne[3]; + size_t lws[3] = {(size_t) sched.softmax_apply_lws0, 1, (size_t) sched.softmax_apply_lws2}; + size_t gws[3] = { + ggml_cl_round_up((size_t) q_width, lws[0]), + (size_t) kv_heads_total, + ggml_cl_round_up((size_t) npack, lws[2]), + }; + cl_kernel kernel = xstate.kernel_mask_scores; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.prob_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.score_img1d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra_mask->data_device)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offset_mask)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &q_width)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &n_q)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &n_kv)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &n_kv_padded)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &kv_heads_total)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(int), &n_head)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_ulong), &mask_nb1)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_ulong), &mask_nb2)); + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_ulong), &mask_nb3)); + CL_CHECK(clSetKernelArg(kernel, 14, sizeof(int), &mask_ne2)); + CL_CHECK(clSetKernelArg(kernel, 15, sizeof(int), &mask_ne3)); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + + softmax_input_img = s.prob_img1d; + softmax_output_buf = s.score_buf; + pv_prob_img = s.score_img1d; + } + + { + size_t lws[3] = {(size_t) sched.softmax_reduce_lws0, 1, 1}; + size_t gws[3] = {ggml_cl_round_up((size_t) q_width, lws[0]), (size_t) kv_heads_total, 1}; + cl_kernel kernel = xstate.kernel_softmax_reduce_basic; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &softmax_input_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.softmax_stats_img2d)); + ggml_cl_set_arg_int4(kernel, 2, kv_heads_total, 1, q_width, n_kv); + ggml_cl_set_arg_int4(kernel, 3, kv_heads_total, q_width, 0, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + { + size_t lws[3] = {(size_t) sched.softmax_apply_lws0, 1, (size_t) sched.softmax_apply_lws2}; + size_t gws[3] = { + ggml_cl_round_up((size_t) q_width, lws[0]), + (size_t) kv_heads_total, + ggml_cl_round_up((size_t) npack, lws[2]), + }; + cl_kernel kernel = xstate.kernel_softmax_apply_basic; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &softmax_output_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &softmax_input_img)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.softmax_stats_img2d)); + ggml_cl_set_arg_int4(kernel, 3, kv_heads_total, npack, q_width, 1); + ggml_cl_set_arg_int4(kernel, 4, kv_heads_total, q_width, n_kv, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + { + const size_t groups16 = (size_t) ggml_cl_round_up_div(kv_heads_total * d_head_v, 16); + const size_t packed_linear = (size_t) n_kv_padded * groups16; + const size_t lws0 = MIN((size_t) 1024, backend_ctx->max_workgroup_size); + size_t gws[3] = {ggml_cl_round_up(packed_linear, lws0), 1, 1}; + size_t lws[3] = {lws0, 1, 1}; + cl_kernel kernel = xstate.kernel_pack_v; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.v_packed_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.v_img)); + ggml_cl_set_arg_int4(kernel, 2, 8, (int) packed_linear, npack, n_kv_padded); + ggml_cl_set_arg_int4(kernel, 3, kv_heads_total, kv_heads_total, opack, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t lws[3] = {(size_t) sched.pv_lws0, 1, (size_t) sched.pv_lws2}; + const int blocks = ggml_cl_round_up_div(opack, 8); + const size_t groups_z = (size_t) ggml_cl_round_up_div(blocks, sched.pv_lws2); + const size_t groups_x = (size_t) ggml_cl_round_up_div(q_width, sched.pv_lws0); + size_t gws[3] = { + lws[0] * groups_z, + groups_x, + (size_t) kv_heads_total * lws[2], + }; + + cl_kernel kernel = xstate.kernel_pv_gemm; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &s.v_packed_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &s.xmem_pv)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &pv_prob_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &s.out_img)); + ggml_cl_set_arg_int4(kernel, 4, kv_heads_total, opack, q_width, 32); + ggml_cl_set_arg_int4(kernel, 5, npack, 0, 0, kv_heads_total); + ggml_cl_set_arg_int4(kernel, 6, kv_heads_total * q_width, npack, q_width, 1); + ggml_cl_set_arg_int4(kernel, 7, 1, 0, 0, 0); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } + + { + size_t gws[3] = {ggml_cl_round_up((size_t) n_q, 8), (size_t) heads_total, (size_t) opack}; + size_t lws[3] = {8, 1, (size_t) ((opack <= 32) ? opack : 1)}; + cl_kernel kernel = xstate.kernel_img_to_f32; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra_o->data_device)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_ulong), &offset_o)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &s.out_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(int), &d_head_v)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &n_q)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &n_head)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &n_batch)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &dst->nb[1])); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_ulong), &dst->nb[2])); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_ulong), &dst->nb[3])); + backend_ctx->enqueue_ndrange_kernel(kernel, 3, gws, lws, dst); + } +} + +#endif // GGML_OPENCL_USE_ADRENO_KERNELS + +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static void ggml_cl_flash_attn_prefill_bin(ggml_backend_t backend, const ggml_tensor * q, const ggml_tensor * k, ggml_tensor * dst) { + const ggml_tensor * v = dst->src[2]; + const ggml_tensor * mask = dst->src[3]; + const ggml_tensor * sinks = dst->src[4]; + GGML_ASSERT(q->extra); + GGML_ASSERT(k->extra); + GGML_ASSERT(v->extra); + GGML_ASSERT(dst->extra); + if (mask) { + GGML_ASSERT(mask->extra); + } + if (sinks) { + GGML_ASSERT(sinks->extra); + } + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + cl_context context = backend_ctx->context; + + const int n_q = q->ne[1]; + const int n_kv = k->ne[1]; + const int d_head_q = q->ne[0]; + const int d_head_v = v->ne[0]; + const int n_head = q->ne[2]; + const int n_head_kv = k->ne[2]; + const int n_batch = q->ne[3]; + + const std::pair dk_dv = {d_head_q, d_head_v}; + cl_kernel kernel = backend_ctx->fa.kernel_flash_attn_f32_f16_bin; + GGML_ASSERT(kernel != NULL); + + ggml_tensor_extra_cl * extra_q = (ggml_tensor_extra_cl *)q->extra; + ggml_tensor_extra_cl * extra_k = (ggml_tensor_extra_cl *)k->extra; + ggml_tensor_extra_cl * extra_v = (ggml_tensor_extra_cl *)v->extra; + ggml_tensor_extra_cl * extra_o = (ggml_tensor_extra_cl *)dst->extra; + ggml_tensor_extra_cl * extra_mask = mask ? (ggml_tensor_extra_cl *)mask->extra : NULL; + ggml_tensor_extra_cl * extra_sinks = sinks ? (ggml_tensor_extra_cl *)sinks->extra : NULL; + + cl_ulong offset_q = extra_q->offset + q->view_offs; + cl_ulong offset_o = extra_o->offset + dst->view_offs; + + cl_mem mask_buffer = extra_mask ? extra_mask->data_device : NULL; + cl_ulong offset_mask = extra_mask ? extra_mask->offset + mask->view_offs : 0; + cl_mem sinks_buffer = extra_sinks ? extra_sinks->data_device : NULL; + cl_ulong offset_sinks = extra_sinks ? extra_sinks->offset + sinks->view_offs : 0; + + const cl_ulong q_nb1 = q->nb[1]; + const cl_ulong q_nb2 = q->nb[2]; + const cl_ulong q_nb3 = q->nb[3]; + + cl_mem k_data_device = extra_k->data_device; + cl_ulong offset_k = extra_k->offset + k->view_offs; + cl_ulong k_nb1 = k->nb[1]; + cl_ulong k_nb2 = k->nb[2]; + cl_ulong k_nb3 = k->nb[3]; + + cl_mem v_data_device = extra_v->data_device; + cl_ulong offset_v = extra_v->offset + v->view_offs; + cl_ulong v_nb1 = v->nb[1]; + cl_ulong v_nb2 = v->nb[2]; + cl_ulong v_nb3 = v->nb[3]; + + const cl_ulong o_nb1 = dst->nb[1]; + const cl_ulong o_nb2 = dst->nb[2]; + const cl_ulong o_nb3 = dst->nb[3]; + + const cl_ulong mask_nb1 = mask ? mask->nb[1] : 0; + const cl_ulong mask_nb2 = mask ? mask->nb[2] : 0; + const cl_ulong mask_nb3 = mask ? mask->nb[3] : 0; + const int mask_ne2 = mask ? mask->ne[2] : 0; + const int mask_ne3 = mask ? mask->ne[3] : 0; + + float * params = (float *)dst->op_params; + float scale = params[0]; + float max_bias = params[1]; + float logit_softcap = params[2]; + + const int is_causal = (mask == NULL && n_q > 1 && n_q == n_kv); // redundant n_q > 1 check ? + + const int n_head_log2_val = n_head > 0 ? 1u << (int)floorf(log2f((float)n_head)) : 0; + const float n_head_log2_f = n_head_log2_val > 0 ? (float)n_head_log2_val : 1.0f; + const float m0 = powf(2.0f, -(max_bias) / n_head_log2_f); + const float m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2_f); + + const bool is_q8_0 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_Q8_0 && v->type == GGML_TYPE_Q8_0; + + ggml_cl_flash_attn_temp_buffer temp_k; + ggml_cl_flash_attn_temp_buffer temp_v; + ggml_cl_flash_attn_temp_buffer temp_k_aos; + ggml_cl_flash_attn_temp_buffer temp_v_aos; + + if (is_q8_0) { + ggml_cl_flash_attn_reconstruct_aos( + backend_ctx, k, temp_k_aos, k_data_device, offset_k, k_nb1, k_nb2, k_nb3); + + ggml_cl_flash_attn_reconstruct_aos( + backend_ctx, v, temp_v_aos, v_data_device, offset_v, v_nb1, v_nb2, v_nb3); + + bool k_done = ggml_cl_flash_attn_dequant_kv_gpu( + backend_ctx, k, GGML_TYPE_F16, k_data_device, offset_k, k_nb1, k_nb2, k_nb3, + temp_k, k_data_device, offset_k, k_nb1, k_nb2, k_nb3); + + bool v_done = ggml_cl_flash_attn_dequant_kv_gpu( + backend_ctx, v, GGML_TYPE_F16, v_data_device, offset_v, v_nb1, v_nb2, v_nb3, + temp_v, v_data_device, offset_v, v_nb1, v_nb2, v_nb3); + + GGML_ASSERT(k_done && v_done); + } + + // Allocate input/output memory buffers + cl_mem mem_matrixQ; + cl_mem mem_matrixK; + cl_mem mem_matrixV; + cl_mem mem_matrixO; + cl_buffer_region region; + cl_int err; + + region.origin = offset_q; + region.size = ggml_nbytes(q); + mem_matrixQ = clCreateSubBuffer(extra_q->data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); + + region.origin = offset_k; + region.size = is_q8_0 ? (size_t) k_nb3 * (size_t) k->ne[3] : ggml_nbytes(k); + mem_matrixK = clCreateSubBuffer(k_data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); + + region.origin = offset_v; + region.size = is_q8_0 ? (size_t) v_nb3 * (size_t) v->ne[3] : ggml_nbytes(v); + mem_matrixV = clCreateSubBuffer(v_data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); + + region.origin = offset_o; + region.size = ggml_nbytes(dst); + mem_matrixO = clCreateSubBuffer(extra_o->data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); + + cl_image_format img_fmt_1d = { CL_RGBA, CL_FLOAT}; + cl_image_desc img_desc_1d; + + // use image 1d buffer used as fallback when on mask is applied + cl_mem mem_tex_mask_fallback_1dbuf; + img_fmt_1d = { CL_RGBA, CL_HALF_FLOAT}; + memset(&img_desc_1d, 0, sizeof(img_desc_1d)); + img_desc_1d.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc_1d.image_width = 1; + img_desc_1d.buffer = mem_matrixK; + mem_tex_mask_fallback_1dbuf = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt_1d, &img_desc_1d, NULL, &err); + CL_CHECK(err); + + cl_mem mem_tex_matrixO_1dbuf; + img_fmt_1d = { CL_RGBA, CL_FLOAT}; + memset(&img_desc_1d, 0, sizeof(img_desc_1d)); + img_desc_1d.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc_1d.image_width = ggml_nbytes(dst) / 4 / 4; + img_desc_1d.buffer = mem_matrixO; + mem_tex_matrixO_1dbuf = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt_1d, &img_desc_1d, NULL, &err); + CL_CHECK(err); - static bool warned = false; - if (!warned) { - GGML_LOG_WARN("ggml_opencl: OpenCL flash attention dequantizes GPU-resident quantized KV cache into temporary linear buffers; performance may be poor\n"); - warned = true; - } + // The bin kernel requires 2d (or 3d) buffers packed for data loading/multiplication. + // These repack kernels launch across all buffers to ensure compatibility + cl_mem mem_tex_matrixMask_1dbuf = NULL; + cl_mem mem_matrixMask = NULL; + cl_mem mem_matrixMask_padded = NULL; + cl_ulong mask_nb1_padded = mask_nb1, mask_nb2_padded = mask_nb2, mask_nb3_padded = mask_nb3; + if (extra_mask) { + // allocate mem_matrixMask w/ new padded size + size_t n_kv_padded = GGML_PAD(n_kv, 4); + size_t mask_nb_padded = n_kv_padded * sizeof(cl_half) * mask->ne[1] * mask->ne[2] * mask->ne[3]; + + // apply offset and create subBuffer for mask + region.origin = offset_mask; + region.size = ggml_nbytes(mask); + mem_matrixMask = clCreateSubBuffer(extra_mask->data_device, CL_MEM_READ_WRITE, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); - return true; -} + { + // create padded mask to contain all data + mem_matrixMask_padded = clCreateBuffer(context, CL_MEM_ALLOC_HOST_PTR, mask_nb_padded, NULL, &err); + CL_CHECK(err); -// Host-side F16 -> F32 for the asymmetric-KV F32 fallback path. -static bool ggml_cl_flash_attn_convert_f16_to_f32( - ggml_backend_opencl_context * backend_ctx, - const ggml_tensor * tensor, - ggml_cl_flash_attn_temp_buffer & temp, - cl_mem & data_device, - cl_ulong & offset, - cl_ulong & nb1, - cl_ulong & nb2, - cl_ulong & nb3 -) { - if (tensor->type != GGML_TYPE_F16) { - return false; + // pass extra_mask->data_device, mem_matrixMask to kernel for copying/padding + mask_nb1_padded = (cl_ulong)n_kv_padded * sizeof(cl_half); + mask_nb2_padded = mask_nb1_padded * (cl_ulong)mask->ne[1]; + mask_nb3_padded = mask_nb2_padded * (cl_ulong)mask->ne[2]; + + cl_kernel repack_mask = backend_ctx->fa.kernel_repack_mask_for_wmm; + CL_CHECK(clSetKernelArg(repack_mask, 0, sizeof(cl_mem), &mem_matrixMask)); + CL_CHECK(clSetKernelArg(repack_mask, 1, sizeof(cl_ulong), &mask_nb1)); + CL_CHECK(clSetKernelArg(repack_mask, 2, sizeof(cl_ulong), &mask_nb2)); + CL_CHECK(clSetKernelArg(repack_mask, 3, sizeof(cl_ulong), &mask_nb3)); + CL_CHECK(clSetKernelArg(repack_mask, 4, sizeof(int), &mask_ne2)); + CL_CHECK(clSetKernelArg(repack_mask, 5, sizeof(cl_mem), &mem_matrixMask_padded)); + CL_CHECK(clSetKernelArg(repack_mask, 6, sizeof(cl_ulong), &mask_nb1_padded)); + CL_CHECK(clSetKernelArg(repack_mask, 7, sizeof(cl_ulong), &mask_nb2_padded)); + CL_CHECK(clSetKernelArg(repack_mask, 8, sizeof(cl_ulong), &mask_nb3_padded)); + + size_t repack_mask_gws[3] = {(size_t)n_kv, (size_t)mask->ne[1], (size_t)mask_ne2 * (size_t)mask->ne[3]}; + backend_ctx->enqueue_ndrange_kernel(repack_mask, 3, repack_mask_gws, NULL, dst); + } + + // use image 1d buffer for matrix Mask (padded row stride) + cl_image_format img_fmt_mask_1d = { CL_RGBA, CL_HALF_FLOAT}; + cl_image_desc img_desc_mask_1d; + memset(&img_desc_mask_1d, 0, sizeof(img_desc_mask_1d)); + img_desc_mask_1d.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc_mask_1d.image_width = mask_nb_padded / 2 / 4; + img_desc_mask_1d.buffer = mem_matrixMask_padded; + mem_tex_matrixMask_1dbuf = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt_mask_1d, &img_desc_mask_1d, NULL, &err); + CL_CHECK(err); } - cl_mem src_buffer = data_device; - cl_ulong src_offset = offset; - cl_ulong src_nb1 = nb1; - cl_ulong src_nb2 = nb2; - cl_ulong src_nb3 = nb3; - ggml_cl_flash_attn_resolve_src(tensor, src_buffer, src_offset, src_nb1, src_nb2, src_nb3); - - const int64_t n = ggml_nelements(tensor); - const size_t row_bytes = (size_t) tensor->ne[0] * sizeof(ggml_fp16_t); - const size_t total_bytes = (size_t) n * sizeof(ggml_fp16_t); - std::vector host_f16(total_bytes); + // WMM QK uses repacked 3D images. + // Q image: rows, heads, packed depth. + cl_image_format img_fmt_3d = { CL_RGBA, CL_HALF_FLOAT }; + cl_image_desc img_desc_3d; - sync_with_other_backends(backend_ctx); - ggml_cl_flash_attn_read_tensor_host(backend_ctx, tensor, src_buffer, src_offset, - src_nb1, src_nb2, src_nb3, - row_bytes, host_f16.data(), total_bytes); + memset(&img_desc_3d, 0, sizeof(img_desc_3d)); + img_desc_3d.image_type = CL_MEM_OBJECT_IMAGE3D; + img_desc_3d.image_width = (size_t)n_q; + img_desc_3d.image_height = (size_t)n_batch * (size_t)n_head; + img_desc_3d.image_depth = (size_t)d_head_q / 4; + cl_mem img_q_wmm = NULL; + img_q_wmm = clCreateImage(context, CL_MEM_READ_WRITE, &img_fmt_3d, &img_desc_3d, NULL, &err); + CL_CHECK(err); - std::vector host_f32(n); - ggml_fp16_to_fp32_row((const ggml_fp16_t *) host_f16.data(), host_f32.data(), n); + { + cl_kernel repack_q = backend_ctx->fa.kernel_repack_q_for_wmm; + CL_CHECK(clSetKernelArg(repack_q, 0, sizeof(cl_mem), &mem_matrixQ)); + CL_CHECK(clSetKernelArg(repack_q, 1, sizeof(cl_ulong), &q_nb1)); + CL_CHECK(clSetKernelArg(repack_q, 2, sizeof(cl_ulong), &q_nb2)); + CL_CHECK(clSetKernelArg(repack_q, 3, sizeof(cl_ulong), &q_nb3)); + CL_CHECK(clSetKernelArg(repack_q, 4, sizeof(int), &n_head)); + CL_CHECK(clSetKernelArg(repack_q, 5, sizeof(cl_mem), &img_q_wmm)); + + size_t repack_q_gws[3] = {(size_t)d_head_q / 4, (size_t)n_q, (size_t)n_batch * (size_t)n_head}; + backend_ctx->enqueue_ndrange_kernel(repack_q, 3, repack_q_gws, NULL, dst); + } + + // K image: columns, row groups, KV heads. + const size_t n_kv_row4 = ((size_t)n_kv + 3) / 4; + + memset(&img_desc_3d, 0, sizeof(img_desc_3d)); + img_desc_3d.image_type = CL_MEM_OBJECT_IMAGE3D; + img_desc_3d.image_width = (size_t)d_head_q; + img_desc_3d.image_height = n_kv_row4; + img_desc_3d.image_depth = (size_t)n_batch * (size_t)n_head_kv; + cl_mem img_k_wmm = NULL; + img_k_wmm = clCreateImage(context, CL_MEM_READ_WRITE, &img_fmt_3d, &img_desc_3d, NULL, &err); + CL_CHECK(err); - const size_t f32_bytes = (size_t) n * sizeof(float); - cl_int err; - temp.data = clCreateBuffer(backend_ctx->context, CL_MEM_READ_WRITE, f32_bytes, NULL, &err); + { + cl_kernel repack_k = backend_ctx->fa.kernel_repack_k_for_wmm; + CL_CHECK(clSetKernelArg(repack_k, 0, sizeof(cl_mem), &mem_matrixK)); + CL_CHECK(clSetKernelArg(repack_k, 1, sizeof(cl_ulong), &k_nb1)); + CL_CHECK(clSetKernelArg(repack_k, 2, sizeof(cl_ulong), &k_nb2)); + CL_CHECK(clSetKernelArg(repack_k, 3, sizeof(cl_ulong), &k_nb3)); + CL_CHECK(clSetKernelArg(repack_k, 4, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(repack_k, 5, sizeof(int), &n_kv)); + CL_CHECK(clSetKernelArg(repack_k, 6, sizeof(cl_mem), &img_k_wmm)); + + size_t repack_k_gws[3] = {(size_t)d_head_q, n_kv_row4, (size_t)n_batch * (size_t)n_head_kv}; + backend_ctx->enqueue_ndrange_kernel(repack_k, 3, repack_k_gws, NULL, dst); + } + + // V image: kv-rows (contracted), packed head-dim groups, KV heads. + memset(&img_desc_3d, 0, sizeof(img_desc_3d)); + img_desc_3d.image_type = CL_MEM_OBJECT_IMAGE3D; + img_desc_3d.image_width = (size_t)n_kv; + img_desc_3d.image_height = (size_t)d_head_v / 4; + img_desc_3d.image_depth = (size_t)n_batch * (size_t)n_head_kv; + cl_mem img_v_wmm = NULL; + img_v_wmm = clCreateImage(context, CL_MEM_READ_WRITE, &img_fmt_3d, &img_desc_3d, NULL, &err); CL_CHECK(err); - CL_CHECK(clEnqueueWriteBuffer(backend_ctx->queue, temp.data, CL_TRUE, 0, - f32_bytes, host_f32.data(), 0, NULL, NULL)); - data_device = temp.data; - offset = 0; - nb1 = (cl_ulong) (tensor->ne[0] * sizeof(float)); - nb2 = (cl_ulong) (tensor->ne[1] * nb1); - nb3 = (cl_ulong) (tensor->ne[2] * nb2); + { + cl_kernel repack_v = backend_ctx->fa.kernel_repack_v_for_wmm; + CL_CHECK(clSetKernelArg(repack_v, 0, sizeof(cl_mem), &mem_matrixV)); + CL_CHECK(clSetKernelArg(repack_v, 1, sizeof(cl_ulong), &v_nb1)); + CL_CHECK(clSetKernelArg(repack_v, 2, sizeof(cl_ulong), &v_nb2)); + CL_CHECK(clSetKernelArg(repack_v, 3, sizeof(cl_ulong), &v_nb3)); + CL_CHECK(clSetKernelArg(repack_v, 4, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(repack_v, 5, sizeof(cl_mem), &img_v_wmm)); + + size_t repack_v_gws[3] = {(size_t)d_head_v / 4, (size_t)n_kv, (size_t)n_batch * (size_t)n_head_kv}; + backend_ctx->enqueue_ndrange_kernel(repack_v, 3, repack_v_gws, NULL, dst); + } + + cl_int enable_mask = (extra_mask) ? 1 : 0; + mask_buffer = extra_mask ? mem_tex_matrixMask_1dbuf : mem_tex_mask_fallback_1dbuf; + + cl_mem mem_sinksBuf = NULL; + cl_mem mem_tex_sinks_1dbuf = NULL; + cl_int enable_sinks = (sinks_buffer != NULL) ? 1 : 0; + if (enable_sinks) { + region.origin = offset_sinks; + region.size = ggml_nbytes(sinks); + mem_sinksBuf = clCreateSubBuffer(extra_sinks->data_device, CL_MEM_READ_ONLY, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err); + CL_CHECK(err); - static bool warned = false; - if (!warned) { - GGML_LOG_WARN("ggml_opencl: OpenCL flash attention asymmetric KV converts an F16 cache to F32 host-side; performance may be poor\n"); - warned = true; + cl_image_format img_fmt_sinks_1d = { CL_R, CL_FLOAT }; + cl_image_desc img_desc_sinks_1d; + memset(&img_desc_sinks_1d, 0, sizeof(img_desc_sinks_1d)); + img_desc_sinks_1d.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc_sinks_1d.image_width = (size_t)n_head; + img_desc_sinks_1d.buffer = mem_sinksBuf; + mem_tex_sinks_1dbuf = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt_sinks_1d, &img_desc_sinks_1d, NULL, &err); + CL_CHECK(err); + } else { + // The image obj cannot be null so we back with buffer of size 1 and use matrixK to back because it always exists + cl_image_format img_fmt_sinks_fallback = { CL_R, CL_FLOAT }; + cl_image_desc img_desc_sinks_fallback; + memset(&img_desc_sinks_fallback, 0, sizeof(img_desc_sinks_fallback)); + img_desc_sinks_fallback.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc_sinks_fallback.image_width = 1; + img_desc_sinks_fallback.buffer = mem_matrixK; + mem_tex_sinks_1dbuf = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt_sinks_fallback, &img_desc_sinks_fallback, NULL, &err); + CL_CHECK(err); } - return true; -} + cl_uint arg = 0; + + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &mem_tex_matrixO_1dbuf)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(float), &scale)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &n_q)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &n_kv)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &is_causal)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &n_head)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &q_nb1)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &q_nb2)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &q_nb3)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &k_nb1)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &k_nb2)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &k_nb3)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &v_nb1)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &v_nb2)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &v_nb3)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &o_nb1)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &o_nb2)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &o_nb3)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(float), &max_bias)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(float), &m0)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(float), &m1)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &n_head_log2_val)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(float), &logit_softcap)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &n_head_kv)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &mask_buffer)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &enable_mask)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &mask_nb1_padded)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &mask_nb2_padded)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_ulong), &mask_nb3_padded)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &mask_ne2)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &mask_ne3)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &mem_tex_sinks_1dbuf)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &enable_sinks)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &img_q_wmm)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &img_k_wmm)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(cl_mem), &img_v_wmm)); + CL_CHECK(clSetKernelArg(kernel, arg++, sizeof(int), &d_head_q)); + + size_t global_work_size[3], local_work_size[3]; + + const int n_waves_v = d_head_q / 64; + + local_work_size[0] = 64; + local_work_size[1] = n_waves_v; + local_work_size[2] = 1; + + global_work_size[0] = 64; + global_work_size[1] = ((n_q + 64 - 1) / 64) * n_waves_v; + global_work_size[2] = n_batch * n_head; -// Flash-Decoding (K-split) dispatch thresholds. FD fires for non-causal -// attention with n_kv >= FD_MIN_N_KV and d_head <= FD_MAX_DK; the KV range is -// split into ~n_kv/FD_KV_PER_SPLIT partials, clamped to [FD_MIN_SPLITS, -// FD_MAX_SPLITS]. Multi-query FD is restricted to small heads -// (d_head <= FD_MAX_DK_MULTI) and capped at FD_MAX_N_Q_MULTI queries. -static constexpr int FD_MIN_N_KV = 2048; -static constexpr int FD_KV_PER_SPLIT = 2048; -// f16 KV decode wants more splits than the 2048 default; quantized KV keeps 2048. -static constexpr int FD_KV_PER_SPLIT_F16 = 512; -static constexpr int FD_MIN_SPLITS = 2; -static constexpr int FD_MAX_SPLITS = 16; -static constexpr int FD_MAX_DK = 128; -static constexpr int FD_MAX_DK_MULTI = 64; -static constexpr int FD_MAX_N_Q_MULTI = 8; -// MQ FD split-groups have few subgroups (MQ_NSG_SPLIT), so use a smaller -// kv_per_split to keep the softmax recurrence short; non-MQ keeps FD_KV_PER_SPLIT. -static constexpr int FD_MQ_KV_PER_SPLIT = 256; -static constexpr int FD_MQ_MAX_SPLITS = 128; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(mem_tex_matrixO_1dbuf)); + CL_CHECK(clReleaseMemObject(img_q_wmm)); + CL_CHECK(clReleaseMemObject(img_k_wmm)); + CL_CHECK(clReleaseMemObject(img_v_wmm)); + + if (mem_tex_matrixMask_1dbuf) { + CL_CHECK(clReleaseMemObject(mem_tex_matrixMask_1dbuf)); + } + if (mem_matrixMask) { + CL_CHECK(clReleaseMemObject(mem_matrixMask)); + } + if (mem_matrixMask_padded) { + CL_CHECK(clReleaseMemObject(mem_matrixMask_padded)); + } + if (mem_tex_sinks_1dbuf) { + CL_CHECK(clReleaseMemObject(mem_tex_sinks_1dbuf)); + } + if (mem_sinksBuf) { + CL_CHECK(clReleaseMemObject(mem_sinksBuf)); + } + CL_CHECK(clReleaseMemObject(mem_matrixQ)); + CL_CHECK(clReleaseMemObject(mem_matrixK)); + CL_CHECK(clReleaseMemObject(mem_matrixV)); + CL_CHECK(clReleaseMemObject(mem_matrixO)); +} +#endif // GGML_OPENCL_USE_ADRENO_KERNELS static void ggml_cl_flash_attn(ggml_backend_t backend, const ggml_tensor * q, const ggml_tensor * k, ggml_tensor * dst) { const ggml_tensor * v = dst->src[2]; @@ -14510,6 +17726,13 @@ static void ggml_cl_flash_attn(ggml_backend_t backend, const ggml_tensor * q, co const int n_head_kv = k->ne[2]; const int n_batch = q->ne[3]; +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (ggml_cl_adreno_xmem_attn_can_use(backend_ctx, q, k, dst)) { + ggml_cl_adreno_xmem_attn_run(backend, q, k, dst); + return; + } +#endif + // DK=512 (Gemma-4 global layers) runs decode-only (q1 / q1_split) on // Adreno - it never uses the BM-tile path, and the prepass + split-tile // programs OOM the compiler at DK=512; supports_op only admits @@ -14531,6 +17754,14 @@ static void ggml_cl_flash_attn(ggml_backend_t backend, const ggml_tensor * q, co const bool is_q8_0 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_Q8_0 && v->type == GGML_TYPE_Q8_0; const bool is_q4_0 = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_Q4_0 && v->type == GGML_TYPE_Q4_0; +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS + if (use_fa_bin_kernels_prefill(backend_ctx, q, k, v)) { + // We support the prefill path of flash attn with a specialized d_head = 64/128/256 + ggml_cl_flash_attn_prefill_bin(backend, q, k, dst); + return; + } +#endif + if (is_f16) { ggml_opencl_ensure_fa_variant(backend_ctx, d_head_q, d_head_v, FA_VARIANT_F16); } else if (is_mixed) { @@ -15604,16 +18835,34 @@ static void ggml_cl_conv_2d(ggml_backend_t backend, const ggml_tensor * src0, co cl_ulong offset1 = extra1->offset + src1->view_offs; cl_ulong offsetd = extrad->offset + dst->view_offs; - const cl_uint Cout = ne03; const cl_uint Cin = ne02; const cl_uint N = ne13; - const cl_uint KW = ne00; const cl_uint KH = ne01; const cl_uint W = ne10; const cl_uint H = ne11; const cl_uint OW = ne0; const cl_uint OH = ne1; - - const cl_uint s0 = dst->op_params[0]; const cl_uint s1 = dst->op_params[1]; - const cl_uint p0 = dst->op_params[2]; const cl_uint p1 = dst->op_params[3]; - const cl_uint d0 = dst->op_params[4]; const cl_uint d1 = dst->op_params[5]; - - const cl_uint cl_nb01 = nb01/ggml_type_size(src0->type); const cl_uint cl_nb02 = nb02/ggml_type_size(src0->type); const cl_uint cl_nb03 = nb03/ggml_type_size(src0->type); - const cl_uint cl_nb11 = nb11/ggml_type_size(src1->type); const cl_uint cl_nb12 = nb12/ggml_type_size(src1->type); const cl_uint cl_nb13 = nb13/ggml_type_size(src1->type); - const cl_uint cl_nb1 = nb1/ggml_type_size(dst->type); const cl_uint cl_nb2 = nb2/ggml_type_size(dst->type); const cl_uint cl_nb3 = nb3/ggml_type_size(dst->type); + const cl_uint Cout = ne03; + const cl_uint Cin = ne02; + const cl_uint N = ne13; + const cl_uint KW = ne00; + const cl_uint KH = ne01; + const cl_uint W = ne10; + const cl_uint H = ne11; + const cl_uint OW = ne0; + const cl_uint OH = ne1; + + const cl_uint s0 = dst->op_params[0]; + const cl_uint s1 = dst->op_params[1]; + const cl_uint p0 = dst->op_params[2]; + const cl_uint p1 = dst->op_params[3]; + const cl_uint d0 = dst->op_params[4]; + const cl_uint d1 = dst->op_params[5]; + + const cl_uint cl_nb00 = nb00/ggml_type_size(src0->type); + const cl_uint cl_nb01 = nb01/ggml_type_size(src0->type); + const cl_uint cl_nb02 = nb02/ggml_type_size(src0->type); + const cl_uint cl_nb03 = nb03/ggml_type_size(src0->type); + const cl_uint cl_nb10 = nb10/ggml_type_size(src1->type); + const cl_uint cl_nb11 = nb11/ggml_type_size(src1->type); + const cl_uint cl_nb12 = nb12/ggml_type_size(src1->type); + const cl_uint cl_nb13 = nb13/ggml_type_size(src1->type); + const cl_uint cl_nb1 = nb1/ggml_type_size(dst->type); + const cl_uint cl_nb2 = nb2/ggml_type_size(dst->type); + const cl_uint cl_nb3 = nb3/ggml_type_size(dst->type); const int64_t NPQ = (int64_t)N * OW * OH; @@ -15649,18 +18898,39 @@ static void ggml_cl_conv_2d(ggml_backend_t backend, const ggml_tensor * src0, co } cl_uint idx = 0; - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extra0->data_device)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offset0)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extra1->data_device)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offset1)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extrad->data_device)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extra1->data_device)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_ulong), &offsetd)); CL_CHECK(clSetKernelArg(kernel, idx++, shmem_size, NULL)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &Cout)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &Cin)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &N)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &KW)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &KH)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &W)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &H)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &OW)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &OH)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &s0)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &s1)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &p0)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &p1)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &d0)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &d1)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb01)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb02)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb03)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb11)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb12)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb13)); - CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb1)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb2)); CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb3)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &Cout)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &Cin)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &N)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &KW)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &KH)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &W)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &H)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &OW)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &OH)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &s0)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &s1)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &p0)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &p1)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &d0)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &d1)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb00)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb01)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb02)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb03)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb10)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb11)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb12)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb13)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb1)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb2)); + CL_CHECK(clSetKernelArg(kernel, idx++, sizeof(cl_uint), &cl_nb3)); size_t global_work_size[] = { (size_t)NB_K * WG_K, (size_t)NB_NPQ * WG_NPQ, 1 }; size_t local_work_size[] = { (size_t)WG_K, (size_t)WG_NPQ, 1 }; @@ -15668,7 +18938,13 @@ static void ggml_cl_conv_2d(ggml_backend_t backend, const ggml_tensor * src0, co backend_ctx->enqueue_ndrange_kernel(kernel, 2, global_work_size, local_work_size, dst); } -static void ggml_cl_mul_mat_kq_kqv_adreno(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { +// is_kq selects which of the two products this call is, and it is decided by the +// CALLER -- the two admission arms in ggml_cl_mul_mat, each of which knows which +// one it matched. It used to be re-derived here from nb01 > nb02, i.e. "K is +// head-major, V^T is not". That discriminator COLLAPSES at n_head_kv == 1, where +// the two strides are equal because there is only one head to order, so nothing +// here could tell a KQ from a KQV. Pass it in rather than infer it. +static void ggml_cl_mul_mat_kq_kqv_adreno(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst, bool is_kq) { ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; ggml_tensor_extra_cl * extra0 = (ggml_tensor_extra_cl *)src0->extra; @@ -15710,19 +18986,14 @@ static void ggml_cl_mul_mat_kq_kqv_adreno(ggml_backend_t backend, const ggml_ten int N = ne1; int K = ne00; - if (nb01 > nb02) { - // KQ - kernel = backend_ctx->kernel_mul_mm_f16_f32_kq; - } else { - // KQV - kernel = backend_ctx->kernel_mul_mm_f16_f32_kqv; - } + kernel = is_kq ? backend_ctx->kernel_mul_mm_f16_f32_kq + : backend_ctx->kernel_mul_mm_f16_f32_kqv; // create sub-buffer for A // <--------------------------------------------> // extra0 = src0->view_src ? (ggml_tensor_extra_cl *)src0->view_src->extra : (ggml_tensor_extra_cl *)src0->extra; region.origin = (extra0->offset + src0->view_offs); - if (nb01 > nb02) { + if (is_kq) { // KQ region.size = nb01 * ne01; } else { @@ -15746,7 +19017,7 @@ static void ggml_cl_mul_mat_kq_kqv_adreno(ggml_backend_t backend, const ggml_ten img_fmt_1d = {CL_RGBA, CL_FLOAT}; memset(&img_desc_1d, 0, sizeof(img_desc_1d)); img_desc_1d.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; - if (nb01 > nb02) { + if (is_kq) { img_desc_1d.image_width = (nb01 * ne01 / 4)/4; } else { @@ -15940,70 +19211,296 @@ static void ggml_cl_mul_mat_q1_0_f32_adreno(ggml_backend_t backend, const ggml_t padding = 8 - extra_elements; } - // subbuffer for transposed activations - region.origin = 0; - region.size = K * (N + padding) * sizeof(float)/2; - backend_ctx->prealloc_act_trans.allocate(context, region.size); - CL_CHECK((b_sub_buf_trans = clCreateSubBuffer(backend_ctx->prealloc_act_trans.buffer, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + // subbuffer for transposed activations + region.origin = 0; + region.size = K * (N + padding) * sizeof(float)/2; + backend_ctx->prealloc_act_trans.allocate(context, region.size); + CL_CHECK((b_sub_buf_trans = clCreateSubBuffer(backend_ctx->prealloc_act_trans.buffer, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + // image for transposed activations + img_fmt = {CL_RGBA, CL_HALF_FLOAT}; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = K * (N + padding) / 4; + img_desc.buffer = b_sub_buf_trans; + CL_CHECK((b_img_trans = clCreateImage(context, 0, &img_fmt, &img_desc, NULL, &err), err)); + + // transpose activations + int height_B = N/4; + if (height_B == 0) { + height_B = 1; + } + int width_B = K/4; + int padded_height_B = (N + padding)/4; + + kernel = backend_ctx->kernel_transpose_32_16; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &b_img_trans)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(int), &height_B)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(int), &width_B)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &padded_height_B)); + + size_t local_work_size_t[2] = { 1, 16 }; + size_t global_work_size_t[2] = { (size_t)width_B, (size_t)padded_height_B }; + backend_ctx->enqueue_ndrange_kernel(kernel, 2, global_work_size_t, local_work_size_t, dst); + + // gemm + kernel = backend_ctx->kernel_gemm_noshuffle_q1_0_f32; + int padded_N = N + padding; + + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q1_0->q)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q1_0->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img_trans)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &K)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &M)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &padded_N)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &N)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &offsetd)); + + size_t global_work_size[] = { (size_t)CEIL_DIV(N, 8), (size_t)CEIL_DIV(M, 4), 1 }; + size_t local_work_size[] = { 2, 128, 1 }; + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_img_trans)); + CL_CHECK(clReleaseMemObject(b_sub_buf_trans)); + CL_CHECK(clReleaseMemObject(b_img)); + CL_CHECK(clReleaseMemObject(b_sub_buf)); + } +#else + GGML_UNUSED(backend); + GGML_UNUSED(src0); + GGML_UNUSED(src1); + GGML_UNUSED(dst); +#endif +} + +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static void ggml_cl_mul_mat_q4_0_f32_adreno_ila(ggml_backend_t backend, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst) { + GGML_ASSERT(src0); + GGML_ASSERT(src0->extra); + GGML_ASSERT(src1); + GGML_ASSERT(src1->extra); + GGML_ASSERT(dst); + GGML_ASSERT(dst->extra); + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *)src1->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + ggml_tensor_extra_cl_q4_0 * extra0_q4_0 = (ggml_tensor_extra_cl_q4_0 *)src0->extra; + + cl_ulong offset1 = extra1->offset + src1->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + const int ne00 = src0->ne[0]; + const int ne01 = src0->ne[1]; + + const int ne1 = dst->ne[1]; + + GGML_ASSERT(ne00 % ggml_blck_size(src0->type) == 0); + + cl_context context = backend_ctx->context; + cl_kernel kernel; + + cl_int err; + cl_image_format img_fmt; + cl_image_desc img_desc; + cl_buffer_region region; + + int M = ne01; + int N = ne1; + int K = ne00; + + if (ne1 == 1) { + cl_mem b_sub_buf = nullptr; + cl_mem b_img = nullptr; + + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt = { CL_RGBA, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N / 4; + img_desc.buffer = b_sub_buf; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32b_trans; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_0->q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_0->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_int), &K)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_int), &M)); + + size_t wavesize = backend_ctx->adreno_wave_size; + size_t local_work_size[3] = { wavesize, 4, 1 }; + size_t global_work_size[3] = { (size_t)CEIL_DIV(M, 64) * 64, 4, 1 }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_sub_buf)); + CL_CHECK(clReleaseMemObject(b_img)); + } else { + const int gemm_tile_n = 64; + int N_pad = (N + gemm_tile_n - 1) & ~(gemm_tile_n - 1); + + cl_mem a_img = extra0_q4_0->q_img; + cl_mem s_img = extra0_q4_0->d_img; + GGML_ASSERT(a_img && s_img && "ILA Q4_0 weight images missing; set_tensor should have built them"); + + static const char * q4_0_bin_dp4a_env = getenv("GGML_OPENCL_Q4_0_BIN_DP4A"); + bool q4_0_bin_dp4a_on = q4_0_bin_dp4a_env + ? (atoi(q4_0_bin_dp4a_env) != 0) + : true; + // dot prod has to be available + q4_0_bin_dp4a_on = backend_ctx->has_integer_dot && q4_0_bin_dp4a_on; + + if (q4_0_bin_dp4a_on && backend_ctx->kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8_bin) { + const int dp4a_N_pad = CEIL_DIV(N, 32) * 32; + const size_t n_blocks = (size_t)dp4a_N_pad * (K / 32); + + backend_ctx->prealloc_moe_qa.allocate(context, (size_t)dp4a_N_pad * K * sizeof(cl_char)); + backend_ctx->prealloc_moe_da.allocate(context, n_blocks * sizeof(cl_half)); + backend_ctx->prealloc_moe_sa.allocate(context, n_blocks * sizeof(cl_half)); + + cl_mem b_sub = nullptr; + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + cl_int tb = (cl_int)((size_t)N * (K / 32)); + cl_kernel qk = backend_ctx->kernel_quant_a_q8_1; + CL_CHECK(clSetKernelArg(qk, 0, sizeof(cl_mem), &b_sub)); + CL_CHECK(clSetKernelArg(qk, 1, sizeof(cl_mem), &backend_ctx->prealloc_moe_qa.buffer)); + CL_CHECK(clSetKernelArg(qk, 2, sizeof(cl_mem), &backend_ctx->prealloc_moe_da.buffer)); + CL_CHECK(clSetKernelArg(qk, 3, sizeof(cl_mem), &backend_ctx->prealloc_moe_sa.buffer)); + CL_CHECK(clSetKernelArg(qk, 4, sizeof(cl_int), &tb)); + size_t q_local[1] = { 64 }; + size_t q_global[1] = { (size_t)CEIL_DIV(tb, 64) * 64 }; + backend_ctx->enqueue_ndrange_kernel(qk, 1, q_global, q_local, dst); + + cl_mem d_sub = nullptr; + cl_mem d_img = nullptr; + region.origin = offsetd; + region.size = (size_t)M * N * sizeof(float); + CL_CHECK((d_sub = clCreateSubBuffer(extrad->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); - // image for transposed activations - img_fmt = {CL_RGBA, CL_HALF_FLOAT}; - memset(&img_desc, 0, sizeof(img_desc)); - img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; - img_desc.image_width = K * (N + padding) / 4; - img_desc.buffer = b_sub_buf_trans; - CL_CHECK((b_img_trans = clCreateImage(context, 0, &img_fmt, &img_desc, NULL, &err), err)); + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * N; + img_desc.buffer = d_sub; + CL_CHECK((d_img = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - // transpose activations - int height_B = N/4; - if (height_B == 0) { - height_B = 1; - } - int width_B = K/4; - int padded_height_B = (N + padding)/4; + kernel = backend_ctx->kernel_gemm_noshuffle_q4_0_q8_1_dp4a_ila_a8_bin; - kernel = backend_ctx->kernel_transpose_32_16; - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &b_img)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &b_img_trans)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(int), &height_B)); - CL_CHECK(clSetKernelArg(kernel, 3, sizeof(int), &width_B)); - CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &padded_height_B)); + cl_uint k_arg = 0; + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &a_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &extra0_q4_0->d)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &backend_ctx->prealloc_moe_qa.buffer)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &backend_ctx->prealloc_moe_da.buffer)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &d_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &K)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &M)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &N)); - size_t local_work_size_t[2] = { 1, 16 }; - size_t global_work_size_t[2] = { (size_t)width_B, (size_t)padded_height_B }; - backend_ctx->enqueue_ndrange_kernel(kernel, 2, global_work_size_t, local_work_size_t, dst); + size_t local_work_size[3] = { 64, 1, 1 }; + size_t global_work_size[3] = { 64, (size_t)(M / 64), (size_t)(dp4a_N_pad / 32) }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); - // gemm - kernel = backend_ctx->kernel_gemm_noshuffle_q1_0_f32; - int padded_N = N + padding; + CL_CHECK(clReleaseMemObject(b_sub)); + CL_CHECK(clReleaseMemObject(d_img)); + CL_CHECK(clReleaseMemObject(d_sub)); + return; + } - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q1_0->q)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q1_0->d)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img_trans)); - CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extrad->data_device)); - CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &K)); - CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &M)); - CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &padded_N)); - CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &N)); - CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_ulong), &offsetd)); + // Pad B through a zero-filled scratch buffer when N needs + // padding, since the GEMM kernel always reads a full N-tile. + const bool need_pad = N_pad > N; + cl_mem b_sub_buf = nullptr; + cl_mem b_padded = nullptr; + if (need_pad) { + CL_CHECK((b_padded = clCreateBuffer(context, CL_MEM_READ_WRITE, + (size_t)K * N_pad * sizeof(float), NULL, &err), err)); + const float zero = 0.0f; + CL_CHECK(clEnqueueFillBuffer(backend_ctx->queue, b_padded, &zero, sizeof(zero), + 0, (size_t)K * N_pad * sizeof(float), 0, NULL, NULL)); + CL_CHECK(clEnqueueCopyBuffer(backend_ctx->queue, extra1->data_device, b_padded, + offset1, 0, (size_t)K * N * sizeof(float), 0, NULL, NULL)); + } else { + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + } - size_t global_work_size[] = { (size_t)CEIL_DIV(N, 8), (size_t)CEIL_DIV(M, 4), 1 }; - size_t local_work_size[] = { 2, 128, 1 }; + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = need_pad ? (size_t)K * N_pad : (size_t)K * N; + img_desc.buffer = need_pad ? b_padded : b_sub_buf; + cl_mem b_img; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + region.origin = offsetd; + region.size = (size_t)M * N * sizeof(float); + cl_mem d_sub_buf; + CL_CHECK((d_sub_buf = clCreateSubBuffer(extrad->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * N; + img_desc.buffer = d_sub_buf; + cl_mem d_img; + CL_CHECK((d_img = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + int line_stride_matrix_A_in_bytes = M * 4; + int line_stride_matrix_S_in_bytes = M * 2; + int line_stride_matrix_B_in_bytes = K * 4; + int line_stride_matrix_C_in_bytes = M * 4; + + int c_offset_for_kernel = 0; + int b_offset_for_kernel = 0; + + kernel = backend_ctx->kernel_gemm_noshuffle_q4_0_f32_32b_trans_ila_a8_bin; + + cl_uint k_arg = 0; + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &a_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &s_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &b_offset_for_kernel)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &d_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &c_offset_for_kernel)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &K)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &line_stride_matrix_A_in_bytes)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &line_stride_matrix_S_in_bytes)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &line_stride_matrix_B_in_bytes)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &line_stride_matrix_C_in_bytes)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &M)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(int), &N)); + + size_t local_work_size[3] = { 64, 2, 2 }; + size_t m_tiles = (size_t)CEIL_DIV(M, 64); + size_t global_work_size[3] = { 64, m_tiles, (size_t)CEIL_DIV(N_pad, gemm_tile_n) }; backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); - CL_CHECK(clReleaseMemObject(b_img_trans)); - CL_CHECK(clReleaseMemObject(b_sub_buf_trans)); CL_CHECK(clReleaseMemObject(b_img)); - CL_CHECK(clReleaseMemObject(b_sub_buf)); + if (b_sub_buf) { + CL_CHECK(clReleaseMemObject(b_sub_buf)); + } + if (b_padded) { + CL_CHECK(clReleaseMemObject(b_padded)); + } + CL_CHECK(clReleaseMemObject(d_img)); + CL_CHECK(clReleaseMemObject(d_sub_buf)); } -#else - GGML_UNUSED(backend); - GGML_UNUSED(src0); - GGML_UNUSED(src1); - GGML_UNUSED(dst); -#endif } +#endif // GGML_OPENCL_USE_ADRENO_KERNELS static void ggml_cl_mul_mat_q4_0_f32_adreno(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { #ifdef GGML_OPENCL_USE_ADRENO_KERNELS @@ -16047,7 +19544,31 @@ static void ggml_cl_mul_mat_q4_0_f32_adreno(ggml_backend_t backend, const ggml_t int N = ne1; int K = ne00; - if (ne1 == 1) { + // Multi-column (N=3) verify GEMV for q4_0: route the spec/MTP verify batch + // (ne1==3) onto the efficient GEMV path instead of the transposed-GEMM dead- + // zone (gemm_noshuffle_q4_0 is ~50% of MTP decode on a Q4_0 model since q4_0 + // weights have no cok/mc3, unlike q4_K). Reuses the ne1==1 GEMV image setup + // (activation image already sized by N=ne1). Byte-identical. Opt-in via + // GGML_OPENCL_Q40_MC3=1. Per-layer only (ne01 < 32768); q4_0 lm_head doesn't + // occur (token_embd/output stay Q6_K), guard kept for parity with q4_K mc3. + static const bool q40_mc3 = (getenv("GGML_OPENCL_Q40_MC3") != nullptr); + const bool use_q40_mc3 = q40_mc3 && (ne1 >= 2 && ne1 <= 4) && (ne01 < 32768); + + const bool use_bin = use_q4_0_bin_kernels(backend_ctx, src0); + + if (use_bin) { + if (use_q40_mc3) { + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("ggml_opencl: GGML_OPENCL_Q40_MC3 is bypassed by Q4_0 binary kernels\n"); + warned = true; + } + } + ggml_cl_mul_mat_q4_0_f32_adreno_ila(backend, src0, src1, dst); + return; + } + + if (ne1 == 1 || use_q40_mc3) { cl_mem q_img = nullptr; cl_mem b_sub_buf = nullptr; cl_mem b_img = nullptr; @@ -16073,38 +19594,56 @@ static void ggml_cl_mul_mat_q4_0_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buf; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32; - if (M == 4096 && K == 4096) { - kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_4096_1_4096; - } else if (M == 4096 && K == 11008) { - kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_4096_1_11008; - } else if (M == 11008 && K == 4096) { - kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_11008_1_4096; - } else if (M == 32000 && K == 4096) { - kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32000_1_4096; - } + if (use_q40_mc3) { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_mc3; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_0->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &ne1)); + } else { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32; + if (M == 4096 && K == 4096) { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_4096_1_4096; + } else if (M == 4096 && K == 11008) { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_4096_1_11008; + } else if (M == 11008 && K == 4096) { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_11008_1_4096; + } else if (M == 32000 && K == 4096) { + kernel = backend_ctx->kernel_gemv_noshuffle_q4_0_f32_32000_1_4096; + } - int r2 = 1; - int r3 = 1; + int r2 = 1; + int r3 = 1; - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_0->d)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img)); - CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offset1)); - CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &extrad->data_device)); - CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &offsetd)); - CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne00)); - CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &ne01)); - CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &ne02)); - CL_CHECK(clSetKernelArg(kernel, 9, sizeof(int), &ne10)); - CL_CHECK(clSetKernelArg(kernel, 10, sizeof(int), &ne12)); - CL_CHECK(clSetKernelArg(kernel, 11, sizeof(int), &ne0)); - CL_CHECK(clSetKernelArg(kernel, 12, sizeof(int), &ne1)); - CL_CHECK(clSetKernelArg(kernel, 13, sizeof(int), &r2)); - CL_CHECK(clSetKernelArg(kernel, 14, sizeof(int), &r3)); + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_0->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &ne02)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(int), &ne10)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(int), &ne12)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(int), &ne0)); + CL_CHECK(clSetKernelArg(kernel, 12, sizeof(int), &ne1)); + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(int), &r2)); + CL_CHECK(clSetKernelArg(kernel, 14, sizeof(int), &r3)); + } - size_t local_work_size[3] = {64, 4, 1}; - size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, 4, 1}; + // Small-M mc3 verify is occupancy/latency-bound (too few WGs at small M, so + // its bandwidth falls well short of the FFN matmuls'). Use 8 subgroups (512-WI WGs, half the + // per-lane K-walk) for small M. Layout stride is fixed (4 uints/block), so only + // the K-split count changes; the mc3 kernel reads it via get_local_size(1). The + // ne1==1 base kernel hardcodes N_SIMDGROUP=4, so it always stays at 4. + const int mc3_nsg = (use_q40_mc3 && ne01 < 4096) ? 8 : 4; + size_t local_work_size[3] = {64, (size_t)mc3_nsg, 1}; + size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, (size_t)mc3_nsg, 1}; backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); @@ -16322,7 +19861,14 @@ static void ggml_cl_mul_mat_q4_1_f32_adreno(ggml_backend_t backend, const ggml_t int N = ne1; int K = ne00; - if (ne1 == 1) { + // Multi-column (N=3) verify GEMV for q4_1: route the spec/MTP verify batch + // (ne1==3) onto the efficient GEMV path instead of the transposed-GEMM dead- + // zone (gemm_noshuffle_q4_1). Reuses the ne1==1 GEMV image setup. Opt-in via + // GGML_OPENCL_Q41_MC3=1. Per-layer only (ne01 < 32768). + static const bool q41_mc3 = (getenv("GGML_OPENCL_Q41_MC3") != nullptr); + const bool use_q41_mc3 = q41_mc3 && (ne1 >= 2 && ne1 <= 4) && (ne01 < 32768); + + if (ne1 == 1 || use_q41_mc3) { cl_mem q_img = nullptr; cl_mem b_sub_buf = nullptr; cl_mem b_img = nullptr; @@ -16348,7 +19894,8 @@ static void ggml_cl_mul_mat_q4_1_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buf; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - kernel = backend_ctx->kernel_gemv_noshuffle_q4_1_f32; + kernel = use_q41_mc3 ? backend_ctx->kernel_gemv_noshuffle_q4_1_f32_mc3 + : backend_ctx->kernel_gemv_noshuffle_q4_1_f32; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_1->d)); @@ -16358,6 +19905,9 @@ static void ggml_cl_mul_mat_q4_1_f32_adreno(ggml_backend_t backend, const ggml_t CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_ulong), &offsetd)); CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_int), &ne00)); CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne01)); + if (use_q41_mc3) { + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne1)); // n_cols + } size_t local_work_size[3] = {64, 4, 1}; size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, 4, 1}; @@ -17201,6 +20751,66 @@ static void ggml_cl_mul_mat_q8_0_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buf; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + // Split-K for small-M decode GEMVs. The base kernel puts one output row + // per lane and splits K only inside one workgroup, so M is the sole source + // of workgroup parallelism: gpt-oss's K and V projections are M=512 = 8 + // workgroups on a 16-CU X2, and the kernel measures 48 GB/s where the + // M=2880/4096 projections in the same decode graph reach 122-123. Mirrors + // the q4_0/q4_K split-K above and reuses their reduce kernel. + // + // Enabled where it is measured to win, like the q4_K gate: X2-90 +2.8% + // tg32 @d4096 on gpt-oss; Adreno 840 (12 CU) NEUTRAL on Llama-3.2-3B-Q8_0 + // (0.0% @d4096 -- its K/V proj is M=1024 = 16 workgroups, which already + // fills 12 CUs). Unmeasured on X1E/A7X/A6X and the q4_K split-K measured + // -0.7% on X1E, so the default is not widened on absence of evidence. + static const bool q8_splitk_env_set = []{ + const char * e = std::getenv("GGML_OPENCL_Q8_GEMV_SPLITK"); + return e && e[0] != '\0'; + }(); + static const bool q8_splitk_env_on = []{ + const char * e = std::getenv("GGML_OPENCL_Q8_GEMV_SPLITK"); + return !(e && e[0] == '0'); + }(); + const bool q8_splitk_on = q8_splitk_env_set + ? q8_splitk_env_on + : (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E); + if (q8_splitk_on && backend_ctx->kernel_gemv_noshuffle_q8_0_f32_splitk && + ne01 <= 1024 && ne01 % 64 == 0) { + const int nsg = 8; + const int ksplit = 8; // -> 8 * M/64 workgroups + const size_t gx = (size_t) CEIL_DIV(ne01, 64) * 64; + + backend_ctx->prealloc_splitk_partial.allocate( + backend_ctx->context, (size_t) ksplit * ne01 * sizeof(float)); + cl_mem partial = backend_ctx->prealloc_splitk_partial.buffer; + + cl_kernel ks = backend_ctx->kernel_gemv_noshuffle_q8_0_f32_splitk; + CL_CHECK(clSetKernelArg(ks, 0, sizeof(cl_mem), &q_img)); + CL_CHECK(clSetKernelArg(ks, 1, sizeof(cl_mem), &extra0_q8_0->d)); + CL_CHECK(clSetKernelArg(ks, 2, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(ks, 3, sizeof(cl_mem), &partial)); + CL_CHECK(clSetKernelArg(ks, 4, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(ks, 5, sizeof(cl_int), &ne01)); + size_t lsk[3] = { 64, (size_t) nsg, 1 }; + size_t gsk[3] = { gx, (size_t) (nsg * ksplit), 1 }; + backend_ctx->enqueue_ndrange_kernel(ks, 3, gsk, lsk, dst); + + cl_kernel kr = backend_ctx->kernel_gemv_splitk_reduce_f32; + CL_CHECK(clSetKernelArg(kr, 0, sizeof(cl_mem), &partial)); + CL_CHECK(clSetKernelArg(kr, 1, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kr, 2, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kr, 3, sizeof(cl_int), &ne01)); + CL_CHECK(clSetKernelArg(kr, 4, sizeof(cl_int), &ksplit)); + size_t lr[3] = { 64, 1, 1 }; + size_t gr[3] = { (size_t) CEIL_DIV(ne01, 64) * 64, 1, 1 }; + backend_ctx->enqueue_ndrange_kernel(kr, 3, gr, lr, dst); + + CL_CHECK(clReleaseMemObject(q_img)); + CL_CHECK(clReleaseMemObject(b_img)); + CL_CHECK(clReleaseMemObject(b_sub_buf)); + return; + } + kernel = backend_ctx->kernel_gemv_noshuffle_q8_0_f32; int r2 = 1; @@ -17501,6 +21111,214 @@ static void ggml_cl_mul_mat_q8_0_f32_adreno(ggml_backend_t backend, const ggml_t #endif } +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static void ggml_cl_mul_mat_q4_k_f32_adreno_ila(ggml_backend_t backend, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst) { + GGML_ASSERT(src0); + GGML_ASSERT(src0->extra); + GGML_ASSERT(src1); + GGML_ASSERT(src1->extra); + GGML_ASSERT(dst); + GGML_ASSERT(dst->extra); + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *)src1->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + ggml_tensor_extra_cl_q4_K * extra0_q4_k = (ggml_tensor_extra_cl_q4_K *)src0->extra; + + cl_ulong offset1 = extra1->offset + src1->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + const int ne00 = src0->ne[0]; + const int ne01 = src0->ne[1]; + + const int ne1 = dst->ne[1]; + + GGML_ASSERT(ne00 % ggml_blck_size(src0->type) == 0); + + cl_context context = backend_ctx->context; + cl_kernel kernel; + + cl_int err; + cl_image_format img_fmt; + cl_image_desc img_desc; + cl_buffer_region region; + + int M = ne01; + int N = ne1; + int K = ne00; + + if (ne1 == 1) { + cl_mem b_sub_buf = nullptr; + cl_mem b_img = nullptr; + + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt = { CL_RGBA, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N / 4; + img_desc.buffer = b_sub_buf; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemv_noshuffle_q4_k_f32_32b_trans; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_k->q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); + + size_t local_work_size[3] = { 64, 8, 1 }; + size_t global_work_size[3] = { (size_t)ne01, 8, 1 }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_sub_buf)); + CL_CHECK(clReleaseMemObject(b_img)); + } else { + const int gemm_tile_n = 64; + int N_pad = CEIL_DIV(N, gemm_tile_n) * gemm_tile_n; + + static const char * q4_k_bin_dp4a_env = getenv("GGML_OPENCL_Q4_K_BIN_DP4A"); + bool q4_k_bin_dp4a_on = q4_k_bin_dp4a_env + ? (atoi(q4_k_bin_dp4a_env) != 0) + : true; + // dot prod has to be available + q4_k_bin_dp4a_on = backend_ctx->has_integer_dot && q4_k_bin_dp4a_on; + + if (q4_k_bin_dp4a_on && backend_ctx->kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8_bin) { + const int dp4a_N_pad = CEIL_DIV(N, 32) * 32; + const size_t n_blocks = (size_t)dp4a_N_pad * (K / 32); + + backend_ctx->prealloc_moe_qa.allocate(context, (size_t)dp4a_N_pad * K * sizeof(cl_char)); + backend_ctx->prealloc_moe_da.allocate(context, n_blocks * sizeof(cl_half)); + backend_ctx->prealloc_moe_sa.allocate(context, n_blocks * sizeof(cl_half)); + + cl_mem b_sub = nullptr; + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + cl_int tb = (cl_int)((size_t)N * (K / 32)); + cl_kernel qk = backend_ctx->kernel_quant_a_q8_1; + CL_CHECK(clSetKernelArg(qk, 0, sizeof(cl_mem), &b_sub)); + CL_CHECK(clSetKernelArg(qk, 1, sizeof(cl_mem), &backend_ctx->prealloc_moe_qa.buffer)); + CL_CHECK(clSetKernelArg(qk, 2, sizeof(cl_mem), &backend_ctx->prealloc_moe_da.buffer)); + CL_CHECK(clSetKernelArg(qk, 3, sizeof(cl_mem), &backend_ctx->prealloc_moe_sa.buffer)); + CL_CHECK(clSetKernelArg(qk, 4, sizeof(cl_int), &tb)); + size_t q_local[1] = { 64 }; + size_t q_global[1] = { (size_t)CEIL_DIV(tb, 64) * 64 }; + backend_ctx->enqueue_ndrange_kernel(qk, 1, q_global, q_local, dst); + + cl_mem d_sub = nullptr; + cl_mem d_img = nullptr; + region.origin = offsetd; + region.size = (size_t)M * N * sizeof(float); + CL_CHECK((d_sub = clCreateSubBuffer(extrad->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * N; + img_desc.buffer = d_sub; + CL_CHECK((d_img = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemm_noshuffle_q4_k_q8_1_dp4a_ila_a8_bin; + + cl_uint k_arg = 0; + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &extra0_q4_k->q_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &backend_ctx->prealloc_moe_qa.buffer)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &backend_ctx->prealloc_moe_da.buffer)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &backend_ctx->prealloc_moe_sa.buffer)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_mem), &d_img)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_uint), &ne00)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_uint), &ne01)); + CL_CHECK(clSetKernelArg(kernel, k_arg++, sizeof(cl_int), &N)); + + size_t local_work_size[3] = { 64, 1, 1 }; + size_t global_work_size[3] = { 64, (size_t)(M / 64), (size_t)(dp4a_N_pad / 32) }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_sub)); + CL_CHECK(clReleaseMemObject(d_img)); + CL_CHECK(clReleaseMemObject(d_sub)); + return; + } + + cl_mem b_sub_buf = nullptr; + cl_mem b_padded = nullptr; + cl_mem b_buf = nullptr; + if (N_pad == N) { + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + b_buf = b_sub_buf; + } else { + CL_CHECK((b_padded = clCreateBuffer(context, CL_MEM_READ_WRITE, (size_t)K * N_pad * sizeof(float), NULL, &err), err)); + const float zero = 0.0f; + CL_CHECK(clEnqueueFillBuffer(backend_ctx->queue, b_padded, &zero, sizeof(zero), 0, (size_t)K * N_pad * sizeof(float), 0, NULL, NULL)); + CL_CHECK(clEnqueueCopyBuffer(backend_ctx->queue, extra1->data_device, b_padded, offset1, 0, (size_t)K * N * sizeof(float), 0, NULL, NULL)); + b_buf = b_padded; + } + + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N_pad; + img_desc.buffer = b_buf; + cl_mem b_img; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + region.origin = offsetd; + region.size = (size_t)M * N * sizeof(float); + cl_mem d_sub_buf; + CL_CHECK((d_sub_buf = clCreateSubBuffer(extrad->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * N; + img_desc.buffer = d_sub_buf; + cl_mem d_img; + CL_CHECK((d_img = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemm_noshuffle_q4_k_f32_32b_trans_ila_a8_bin; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_k->q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &d_img)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_uint), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_uint), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &N)); + + size_t local_work_size[3] = { 64, 2, 2 }; + size_t m_tiles = (size_t)CEIL_DIV(M, 64); + size_t global_work_size[3] = { 64, m_tiles, (size_t)CEIL_DIV(N_pad, gemm_tile_n) }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_img)); + if (b_sub_buf) { + CL_CHECK(clReleaseMemObject(b_sub_buf)); + } + if (b_padded) { + CL_CHECK(clReleaseMemObject(b_padded)); + } + CL_CHECK(clReleaseMemObject(d_img)); + CL_CHECK(clReleaseMemObject(d_sub_buf)); + } +} +#endif // GGML_OPENCL_USE_ADRENO_KERNELS + static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { #ifdef GGML_OPENCL_USE_ADRENO_KERNELS GGML_ASSERT(src0); @@ -17542,18 +21360,47 @@ static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_t cl_uchar mask_d4 = 0x0F; cl_uchar mask_hi2 = 0xC0; - if (ne1 == 1) { + // Multi-column verify GEMV: route the spec/MTP verify batch (ne1==3 = 2 + // drafts + 1 bonus) onto the efficient GEMV path (subgroup-broadcast, no + // transpose) instead of the transposed-GEMM dead-zone. Reuses the ne1==1 + // GEMV setup (the activation image is already sized by N=ne1). Byte- + // identical. Opt-in via GGML_OPENCL_Q4K_MC3=1 while validating. + static const bool q4k_mc3 = (getenv("GGML_OPENCL_Q4K_MC3") != nullptr); + // Per-layer only (ne01 < 32768): the batched large-vocab lm_head at ne1==3 + // is left to the existing routing (corrupts on the Adreno GEMV path; x2- + // unified routes batched Q6_K lm_head to CPU). Per-layer mc3 is byte-identical. + const bool use_mc3 = q4k_mc3 && (ne1 == 3) && (ne01 < 32768); + + const bool use_bin = use_q4_k_bin_kernels(backend_ctx, src0); + + if (use_bin) { + if (use_mc3) { + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("ggml_opencl: GGML_OPENCL_Q4K_MC3 is bypassed by Q4_K binary kernels\n"); + warned = true; + } + } + ggml_cl_mul_mat_q4_k_f32_adreno_ila(backend, src0, src1, dst); + return; + } + + if (ne1 == 1 || use_mc3) { cl_mem q_img = nullptr; cl_mem b_sub_buf = nullptr; cl_mem b_img = nullptr; - // image for q - img_fmt = { CL_R, CL_UNSIGNED_INT32}; - memset(&img_desc, 0, sizeof(img_desc)); - img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; - img_desc.image_width = M * K / 2 / 4; - img_desc.buffer = extra0_q4_k->q; - CL_CHECK((q_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + const bool use_tiled = !use_mc3 && use_q4k_tiled(backend_ctx, src0); + + // image for q (not needed for the tiled path, which reads __global) + if (!use_tiled) { + img_fmt = { CL_R, CL_UNSIGNED_INT32}; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = M * K / 2 / 4; + img_desc.buffer = extra0_q4_k->q; + CL_CHECK((q_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + } // subbuffer for activations region.origin = offset1; @@ -17568,27 +21415,173 @@ static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buf; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - kernel = backend_ctx->kernel_gemv_noshuffle_q4_k_f32; - - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->d)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->dm)); - CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->s)); - CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); - CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extrad->data_device)); - CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &offsetd)); - CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); - CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); - CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_uchar), &mask_d6)); - CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_uchar), &mask_d4)); - CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_uchar), &mask_hi2)); + // 4-output-per-WI o4 variant for the long-vocab lm_head/embed GEMV + // (ne01 = vocab ~256K on Gemma): shares one activation read across 4 + // output rows. Gated to large ne01 (lm_head/embed). Default on; opt-out + // GGML_OPENCL_Q4K_GEMV_O4=0. (Skipped when mc3 handles the ne1==3 verify.) + static const bool q4k_o4_env = []{ + const char * e = std::getenv("GGML_OPENCL_Q4K_GEMV_O4"); + return !e || e[0] == '\0' || e[0] != '0'; + }(); + const bool use_q4k_o4 = !use_tiled && !use_mc3 && q4k_o4_env && (ne01 % 4 == 0) && (ne01 >= 32768); + // Split-K across workgroups for small-M decode GEMVs. A single-token GEMV + // makes only CEIL_DIV(M/2,64) workgroups; even with the wide intra-WG split + // (16 subgroups) those all land on ONE CU, so small-M matmuls under-fill the + // 16 CUs and their bandwidth falls well short of what the large-M FFN matmuls + // reach. Adding a `ksplit` second grid dim that spreads K across WGs (+ a + // reduce pass) fills the CUs. Gate is M<=2560: the tiny M<=1024 ones only + // break even (the reduce dispatch eats the kernel win), but the big-K M=2560 + // cases (ffn_down, attn_output) make the per-call win dwarf the reduce, and + // are byte-identical. ffn_gate/up (large M) fill the CUs already and are excluded. + // + // DEVICE-GATED. Split-K buys GPU time by spending an extra kernel LAUNCH (the + // reduce), so it only pays where launches are cheap. That is a per-device + // property and it does not travel from the X2-90 this was tuned on. Measured + // with one binary, env A/B (tg32, GGML_OPENCL_Q4K_GEMV_SPLITK=0/1): + // + // X2-90 +3.36% gemma-4 E4B (the number this gate was built on) + // 840 -1.3% Qwen3.5-4B-Q4_K_M 14.00 -> 13.85 + // 850 -20.0% Qwen3-1.7B-Q4_K_M 6.97 -> 5.58 (6 interleaved reps) + // + // The kernel is not the problem. On the 850 split-K makes the GPU strictly + // faster -- total busy 537 -> 485 ms, this GEMV 43.7 -> 34.0 us/call (-22%) -- + // and still costs a fifth of decode, because the +3696 reduce dispatches cost + // ~550 us of HOST round-trip each against 2.7 us of GPU work (~200x; that part + // is ~95% host-bound at decode). The 840 pays the same tax at ~42 us/dispatch. + // Break-even needs launch cost below the ~9.7 us/call the split actually saves, + // so this is not a "the 850 is slow" adjustment that a faster part would fix -- + // the 840 is 13x cheaper per launch and still loses. + // + // Enabled where it is measured to win, i.e. X2E only. The X1-85 was measured + // afterwards and is NOT a win either: Qwen3.5-4B-Q4_K_M tg32, split-K off + // 17.98/18.10/18.19 vs on 18.03/17.91/17.97 = -0.7%, so X1E stays excluded on + // evidence rather than on absence of it. Do not widen this without a NEW + // measurement. The env still forces either way so every device stays measurable. + static const bool splitk_env_set = []{ + const char * e = std::getenv("GGML_OPENCL_Q4K_GEMV_SPLITK"); + return e && e[0] != '\0'; + }(); + static const bool splitk_env_on = []{ + const char * e = std::getenv("GGML_OPENCL_Q4K_GEMV_SPLITK"); + return !(e && e[0] == '0'); + }(); + const bool splitk_wg_env = splitk_env_set + ? splitk_env_on + : (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E); + // Gate: small-M decode GEMVs that under-fill the 16 CUs even with the wide + // intra-WG split (all 16 subgroups land on one CU). M<=2560 covers Kcur/Vcur + // (M=1024), Qcur (2048), attn_output + ffn_down (2560). The tiny ones + // (M<=1024) only break even (reduce dispatch eats the kernel win), but the + // big-K M=2560 cases (ffn_down K=10240 @182us, attn_output @42us) have a + // large per-call win that dwarfs the ~5us reduce, so extending to 2560 nets + // positive end-to-end. ffn_gate/up (M=10240) already fill the CUs -> excluded. + const bool use_splitk = splitk_wg_env && !use_tiled && !use_q4k_o4 && !use_mc3 && ne01 <= 2560; + + if (use_splitk) { + const int nsg = 8; + const int ksplit = (ne01 <= 512) ? 8 : 4; // -> ~32 total WGs + const size_t gx = (size_t)CEIL_DIV(ne01/2, 64) * 64; + + backend_ctx->prealloc_splitk_partial.allocate( + backend_ctx->context, (size_t)ksplit * ne01 * sizeof(float)); + cl_mem partial = backend_ctx->prealloc_splitk_partial.buffer; + + cl_kernel ks = backend_ctx->kernel_gemv_noshuffle_q4_k_f32_splitk; + CL_CHECK(clSetKernelArg(ks, 0, sizeof(cl_mem), &q_img)); + CL_CHECK(clSetKernelArg(ks, 1, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(ks, 2, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(ks, 3, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(ks, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(ks, 5, sizeof(cl_mem), &partial)); + CL_CHECK(clSetKernelArg(ks, 6, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(ks, 7, sizeof(cl_int), &ne01)); + CL_CHECK(clSetKernelArg(ks, 8, sizeof(cl_uchar), &mask_d6)); + CL_CHECK(clSetKernelArg(ks, 9, sizeof(cl_uchar), &mask_d4)); + CL_CHECK(clSetKernelArg(ks, 10, sizeof(cl_uchar), &mask_hi2)); + size_t lsk[3] = {64, (size_t)nsg, 1}; + size_t gsk[3] = {gx, (size_t)(nsg * ksplit), 1}; + backend_ctx->enqueue_ndrange_kernel(ks, 3, gsk, lsk, dst); + + cl_kernel kr = backend_ctx->kernel_gemv_splitk_reduce_f32; + CL_CHECK(clSetKernelArg(kr, 0, sizeof(cl_mem), &partial)); + CL_CHECK(clSetKernelArg(kr, 1, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kr, 2, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kr, 3, sizeof(cl_int), &ne01)); + CL_CHECK(clSetKernelArg(kr, 4, sizeof(cl_int), &ksplit)); + size_t lr[3] = {64, 1, 1}; + size_t gr[3] = {(size_t)CEIL_DIV(ne01, 64) * 64, 1, 1}; + backend_ctx->enqueue_ndrange_kernel(kr, 3, gr, lr, dst); + + if (q_img) CL_CHECK(clReleaseMemObject(q_img)); + CL_CHECK(clReleaseMemObject(b_sub_buf)); + CL_CHECK(clReleaseMemObject(b_img)); + return; + } - size_t local_work_size[3] = {64, 4, 1}; - size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, 4, 1}; + kernel = use_mc3 ? backend_ctx->kernel_gemv_noshuffle_q4_k_f32_mc3 + : use_tiled ? backend_ctx->kernel_gemv_noshuffle_q4_k_f32_tiled + : use_q4k_o4 ? backend_ctx->kernel_gemv_noshuffle_q4_k_f32_o4 + : backend_ctx->kernel_gemv_noshuffle_q4_k_f32; + + if (use_tiled) { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_k->q)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); + } else { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->d)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->dm)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->s)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 9, sizeof(cl_uchar), &mask_d6)); + CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_uchar), &mask_d4)); + CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_uchar), &mask_hi2)); + } + + // Wide K-split for the decode GEMV: the default 4-subgroup K-split leaves + // each Adreno SP with only ~4 waves, too few to hide LPDDR weight-load + // latency, so even the large FFN matmuls run well below the achievable + // bandwidth. Widen to 16 subgroups/WG (= the 1024-lane Adreno WG max) so + // each SP holds enough in-flight memory requests. Prefill is unaffected (the + // GEMM path is separate) and coherence-identical (greedy output unchanged). + // Applies to the plain base + // GEMV only; tiled/o4/mc3 keep 4 (their reductions are hard-coded to 4). + // Layout-safe: the base kernel derives its K-split from get_local_size(1) + // and the packed block stride is a physical constant (independent of it). + // Opt-out: GGML_OPENCL_Q4K_GEMV_WIDE=0. + static const bool splitk_wide_env = []{ + const char * e = std::getenv("GGML_OPENCL_Q4K_GEMV_WIDE"); + return !e || e[0] == '\0' || e[0] != '0'; + }(); + const bool splitk_wide = splitk_wide_env && !use_tiled && !use_q4k_o4 && !use_mc3; + size_t nsg_y = splitk_wide ? 16 : 4; + // Cap the wide K-split by the kernel's real max WG. X1-class drivers cap + // this GEMV at 768 (< 64*16 = 1024), so an uncapped lws aborts the + // dispatch with CL_INVALID_WORK_GROUP_SIZE (-54) and breaks ALL q4_K + // decode for M>2560. nsg_y is a pure K-split (the base kernel reads it + // from get_local_size(1); the packed block stride is a physical constant), + // so halving it stays coherent — just a narrower split. X2 keeps 16 + // (maxwg 1024); X1 falls to 8. + if (splitk_wide) { + const size_t maxwg = backend_ctx->get_kernel_workgroup_size(kernel); + while (nsg_y > 4 && 64 * nsg_y > maxwg) { nsg_y >>= 1; } + } + size_t local_work_size[3] = {64, nsg_y, 1}; + size_t global_work_size[3] = {(size_t)CEIL_DIV(use_tiled ? ne01 : (use_q4k_o4 ? ne01/4 : ne01/2), 64)*64, nsg_y, 1}; backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); - CL_CHECK(clReleaseMemObject(q_img)); + if (q_img) CL_CHECK(clReleaseMemObject(q_img)); CL_CHECK(clReleaseMemObject(b_sub_buf)); CL_CHECK(clReleaseMemObject(b_img)); } else { @@ -17748,10 +21741,44 @@ static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_t } // gemm - kernel = backend_ctx->kernel_gemm_noshuffle_q4_k_f32; + // Small-batch (medium n_q) occupancy fix: at ne1<=8 the 2x8 grid is + // (1, ceil(M/2)) -> ~M/256 workgroups, which under-occupies the SP and + // makes the GEMM much slower than the ne1==1 GEMV at the same weight + // traffic. The _r1 (1-row) kernel doubles the M-axis workgroup count + // and removes the accumulator spill. Opt-in via env while validating. + static const bool q4k_gemm_r1 = (getenv("GGML_OPENCL_Q4K_GEMM_R1") != nullptr); + static const bool q4k_gemm_kimg = (getenv("GGML_OPENCL_Q4K_GEMM_KIMG") != nullptr); + // Cooperative-K (intra-WG K-split + reduction) for the small-batch + // (n_q in [2..8]) path: DEFAULT ON, opt out with GGML_OPENCL_Q4K_GEMM_COK=0. + // Byte-identical greedy output; large-batch (ne1>8) untouched. + static const char * q4k_cok_env = getenv("GGML_OPENCL_Q4K_GEMM_COK"); + static const bool q4k_gemm_cok = (q4k_cok_env == nullptr) || (atoi(q4k_cok_env) != 0); + const bool use_cok = q4k_gemm_cok && (ne1 <= 8); + const bool use_r1 = !use_cok && q4k_gemm_r1 && (ne1 <= 8); + // Weights-as-image (L1/TPL1) for the small-batch weight-read-bound path. + const bool use_kimg = !use_cok && !use_r1 && q4k_gemm_kimg && (ne1 <= 8); + + cl_mem q_img = nullptr; + if (use_kimg) { + img_fmt = { CL_R, CL_UNSIGNED_INT32 }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = M * K / 2 / 4; + img_desc.buffer = extra0_q4_k->q; + CL_CHECK((q_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + } + + kernel = use_cok ? backend_ctx->kernel_gemm_noshuffle_q4_k_f32_cok + : use_r1 ? backend_ctx->kernel_gemm_noshuffle_q4_k_f32_r1 + : use_kimg ? backend_ctx->kernel_gemm_noshuffle_q4_k_f32_kimg + : backend_ctx->kernel_gemm_noshuffle_q4_k_f32; int padded_N = N + padding; - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_k->q)); + if (use_kimg) { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); + } else { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q4_k->q)); + } CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q4_k->s)); CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q4_k->d)); CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q4_k->dm)); @@ -17766,10 +21793,45 @@ static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_t CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_uchar), &mask_d4)); CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_uchar), &mask_hi2)); - size_t global_work_size[3] = {(size_t)CEIL_DIV(ne1, 8), (size_t)CEIL_DIV(ne01, 4), 1}; - size_t local_work_size[3] = {1, 128, 1}; + size_t global_work_size[3]; + size_t local_work_size[3]; + if (use_cok) { + // (COK_SG lanes x COK_NSG subgroups): one row per lane, K split + // across the COK_NSG subgroups. ne01 is a multiple of 64. + global_work_size[0] = (size_t)ne01; // rows + global_work_size[1] = 8; // COK_NSG + global_work_size[2] = 1; + local_work_size[0] = 64; // COK_SG + local_work_size[1] = 8; // COK_NSG + local_work_size[2] = 1; + } else if (use_r1) { + // 1 row per WI (opt-in occupancy experiment). + global_work_size[0] = (size_t)CEIL_DIV(ne1, 8); + global_work_size[1] = (size_t)ne01; + global_work_size[2] = 1; + local_work_size[0] = 1; + local_work_size[1] = 128; + local_work_size[2] = 1; + } else if (use_kimg) { + // kimg is a 2-row tile (opt-in weights-as-image experiment). + global_work_size[0] = (size_t)CEIL_DIV(ne1, 8); + global_work_size[1] = (size_t)CEIL_DIV(ne01, 2); + global_work_size[2] = 1; + local_work_size[0] = 1; + local_work_size[1] = 128; + local_work_size[2] = 1; + } else { + // Default: x2-unified base kernel is the 4-row (gx<<2) tile. + global_work_size[0] = (size_t)CEIL_DIV(ne1, 8); + global_work_size[1] = (size_t)CEIL_DIV(ne01, 4); + global_work_size[2] = 1; + local_work_size[0] = 1; + local_work_size[1] = 128; + local_work_size[2] = 1; + } backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + if (q_img) CL_CHECK(clReleaseMemObject(q_img)); CL_CHECK(clReleaseMemObject(b_sub_buf)); CL_CHECK(clReleaseMemObject(b_sub_buf_trans)); CL_CHECK(clReleaseMemObject(b_img)); @@ -17783,6 +21845,145 @@ static void ggml_cl_mul_mat_q4_k_f32_adreno(ggml_backend_t backend, const ggml_t #endif } +#ifdef GGML_OPENCL_USE_ADRENO_KERNELS +static void ggml_cl_mul_mat_q6_K_f32_adreno_ila(ggml_backend_t backend, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst) { + GGML_ASSERT(src0); + GGML_ASSERT(src0->extra); + GGML_ASSERT(src1); + GGML_ASSERT(src1->extra); + GGML_ASSERT(dst); + GGML_ASSERT(dst->extra); + + ggml_backend_opencl_context *backend_ctx = (ggml_backend_opencl_context *)backend->context; + + ggml_tensor_extra_cl_q6_K * extra0_q6_K = (ggml_tensor_extra_cl_q6_K *)src0->extra; + ggml_tensor_extra_cl * extra1 = (ggml_tensor_extra_cl *)src1->extra; + ggml_tensor_extra_cl * extrad = (ggml_tensor_extra_cl *)dst->extra; + + cl_ulong offset1 = extra1->offset + src1->view_offs; + cl_ulong offsetd = extrad->offset + dst->view_offs; + + const int ne00 = src0->ne[0]; + const int ne01 = src0->ne[1]; + + const int ne1 = dst->ne[1]; + + GGML_ASSERT(ne00 % ggml_blck_size(src0->type) == 0); + + cl_context context = backend_ctx->context; + cl_kernel kernel; + + cl_int err; + cl_buffer_region region; + cl_image_format img_fmt; + cl_image_desc img_desc; + + const int M = ne01; + const int N = ne1; + const int K = ne00; + + if (ne1 == 1) { + cl_mem b_sub_buf = nullptr; + cl_mem b_img = nullptr; + + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt = { CL_RGBA, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N / 4; + img_desc.buffer = b_sub_buf; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemv_noshuffle_q6_k_f32_32b_trans; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q6_K->ql_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q6_K->qh_img)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q6_K->s)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q6_K->d)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); + + size_t local_work_size[3] = { 64, 8, 1 }; + size_t global_work_size[3] = { (size_t)ne01, 8, 1 }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_img)); + CL_CHECK(clReleaseMemObject(b_sub_buf)); + } else { + const int gemm_tile_n = 64; + int N_pad = CEIL_DIV(N, gemm_tile_n) * gemm_tile_n; + + cl_mem b_sub_buf = nullptr; + cl_mem b_padded = nullptr; + cl_mem b_buf = nullptr; + if (N_pad == N) { + region.origin = offset1; + region.size = (size_t)K * N * sizeof(float); + CL_CHECK((b_sub_buf = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + b_buf = b_sub_buf; + } else { + CL_CHECK((b_padded = clCreateBuffer(context, CL_MEM_READ_WRITE, (size_t)K * N_pad * sizeof(float), NULL, &err), err)); + const float zero = 0.0f; + CL_CHECK(clEnqueueFillBuffer(backend_ctx->queue, b_padded, &zero, sizeof(zero), 0, (size_t)K * N_pad * sizeof(float), 0, NULL, NULL)); + CL_CHECK(clEnqueueCopyBuffer(backend_ctx->queue, extra1->data_device, b_padded, offset1, 0, (size_t)K * N * sizeof(float), 0, NULL, NULL)); + b_buf = b_padded; + } + + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)K * N_pad; + img_desc.buffer = b_buf; + cl_mem b_img; + CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + region.origin = offsetd; + region.size = (size_t)M * N * sizeof(float); + cl_mem d_sub_buf; + CL_CHECK((d_sub_buf = clCreateSubBuffer(extrad->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + img_fmt = { CL_R, CL_FLOAT }; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = (size_t)M * N; + img_desc.buffer = d_sub_buf; + cl_mem d_img; + CL_CHECK((d_img = clCreateImage(context, CL_MEM_WRITE_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + kernel = backend_ctx->kernel_gemm_noshuffle_q6_k_f32_32b_trans_ila_a8_bin; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q6_K->ql_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q6_K->qh)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q6_K->s)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q6_K->d)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(cl_mem), &d_img)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(cl_uint), &ne00)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_uint), &ne01)); + CL_CHECK(clSetKernelArg(kernel, 8, sizeof(int), &N)); + + size_t local_work_size[3] = { 64, 2, 2 }; + size_t m_tiles = (size_t)CEIL_DIV(M, 64); + size_t global_work_size[3] = { 64, m_tiles, (size_t)CEIL_DIV(N_pad, gemm_tile_n) }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_img)); + if (b_sub_buf) { + CL_CHECK(clReleaseMemObject(b_sub_buf)); + } + if (b_padded) { + CL_CHECK(clReleaseMemObject(b_padded)); + } + CL_CHECK(clReleaseMemObject(d_img)); + CL_CHECK(clReleaseMemObject(d_sub_buf)); + } +} +#endif // GGML_OPENCL_USE_ADRENO_KERNELS + static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { #ifdef GGML_OPENCL_USE_ADRENO_KERNELS GGML_ASSERT(src0); @@ -17817,29 +22018,79 @@ static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_t cl_image_desc img_desc; // subbuffer and image for activation - if (ne1 == 1) { + // Multi-column verify GEMV: route the spec/MTP verify q6_K matmuls (ne1==3) + // onto the efficient GEMV path instead of the transposed-GEMM dead-zone. + // Reuses the ne1==1 image setup (activation image sized by N=ne1). Byte- + // identical. Opt-in via GGML_OPENCL_Q6K_MC3=1 while validating. + static const bool q6k_mc3 = (getenv("GGML_OPENCL_Q6K_MC3") != nullptr); + // Per-layer only (ne01 < 32768): batched large-vocab lm_head stays on the + // existing path (x2-unified routes batched Q6_K lm_head to CPU; the Adreno + // GEMV corrupts it). Per-layer mc3 is byte-identical. + const bool use_q6k_mc3 = q6k_mc3 && (ne1 == 3) && (ne01 < 32768); + // Batched verify lm_head/embed (ne1==3, tiled layout): multi-column tiled + // GEMV — streams the large lm_head weight once across the 3 verify columns + // (the #1 MTP bottleneck; mc3 above can't, it reads the noshuffle layout). + const bool use_q6k_tiled_mc = q6k_mc3 && (ne1 == 3) && (ne01 >= 32768) && use_q6k_tiled(backend_ctx, src0); + + const bool use_bin = use_q6_k_bin_kernels(backend_ctx, src0); + + if (use_bin) { + if (use_q6k_mc3 || use_q6k_tiled_mc) { + static bool warned = false; + if (!warned) { + GGML_LOG_WARN("ggml_opencl: GGML_OPENCL_Q6K_MC3 is bypassed by Q6_K binary kernels\n"); + warned = true; + } + } + ggml_cl_mul_mat_q6_K_f32_adreno_ila(backend, src0, src1, dst); + return; + } + + if (ne1 == 1 || use_q6k_mc3 || use_q6k_tiled_mc) { cl_mem ql_img = nullptr; cl_mem qh_img = nullptr; cl_mem b_sub_buffer = nullptr; cl_mem b_img = nullptr; - // image for ql - img_fmt.image_channel_order = CL_R; - img_fmt.image_channel_data_type = CL_FLOAT; - memset(&img_desc, 0, sizeof(img_desc)); - img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; - img_desc.image_width = ne01 * ne00 / 8; - img_desc.buffer = extra0_q6_K->ql; - CL_CHECK((ql_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + // o4 = 4-output-per-WI variant for long-vocab lm_head/embed; gated to + // ne01 >= 32768 so per-layer q6_K (ne01=hidden 2-8K) keeps the 2-output + // kernel (o4 regresses there). o4_global reads the weights from __global + // coalesced instead of image1d_buffer -- the texture cache caps the + // read-once-per-token lm_head bandwidth, while __global reaches the higher + // rate the rest of the model gets. Both default ON; opt out via + // GGML_OPENCL_Q6K_GEMV_O4 / GGML_OPENCL_Q6K_GEMV_O4_GLOBAL = 0. + static const bool gemv_o4_env = []{ + const char * e = std::getenv("GGML_OPENCL_Q6K_GEMV_O4"); + return !e || e[0] == '\0' || e[0] != '0'; + }(); + static const bool o4_global_env = []{ + const char * e = std::getenv("GGML_OPENCL_Q6K_GEMV_O4_GLOBAL"); + return !e || e[0] == '\0' || e[0] != '0'; + }(); + const bool use_tiled = !use_q6k_mc3 && use_q6k_tiled(backend_ctx, src0); + const bool use_o4 = !use_tiled && !use_q6k_mc3 && gemv_o4_env && (ne01 % 4 == 0) && (ne01 >= 32768); + const bool use_o4_global = use_o4 && o4_global_env; + + // ql/qh image views are only needed when NOT reading weights from global. + if (!use_o4_global && !use_tiled) { + // image for ql + img_fmt.image_channel_order = CL_R; + img_fmt.image_channel_data_type = CL_FLOAT; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = ne01 * ne00 / 8; + img_desc.buffer = extra0_q6_K->ql; + CL_CHECK((ql_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - // image for qh - img_fmt.image_channel_order = CL_R; - img_fmt.image_channel_data_type = CL_HALF_FLOAT; - memset(&img_desc, 0, sizeof(img_desc)); - img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; - img_desc.image_width = ne01 * ne00 / 8; - img_desc.buffer = extra0_q6_K->qh; - CL_CHECK((qh_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + // image for qh + img_fmt.image_channel_order = CL_R; + img_fmt.image_channel_data_type = CL_HALF_FLOAT; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = ne01 * ne00 / 8; + img_desc.buffer = extra0_q6_K->qh; + CL_CHECK((qh_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + } region.origin = offset1; region.size = ne00 * ne1 * sizeof(float); @@ -17853,10 +22104,20 @@ static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buffer; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - kernel = backend_ctx->kernel_gemv_noshuffle_q6_K_f32; + kernel = use_q6k_mc3 ? backend_ctx->kernel_gemv_noshuffle_q6_K_f32_mc3 + : use_q6k_tiled_mc ? backend_ctx->kernel_gemv_noshuffle_q6_K_f32_tiled_mc3 + : use_tiled ? backend_ctx->kernel_gemv_noshuffle_q6_K_f32_tiled + : use_o4_global ? backend_ctx->kernel_gemv_noshuffle_q6_K_f32_o4_global + : use_o4 ? backend_ctx->kernel_gemv_noshuffle_q6_K_f32_o4 + : backend_ctx->kernel_gemv_noshuffle_q6_K_f32; - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &ql_img)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &qh_img)); + if (use_o4_global || use_tiled) { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &extra0_q6_K->ql)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &extra0_q6_K->qh)); + } else { + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &ql_img)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &qh_img)); + } CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &extra0_q6_K->s)); CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &extra0_q6_K->d)); CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &b_img)); @@ -17865,16 +22126,67 @@ static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_t CL_CHECK(clSetKernelArg(kernel, 7, sizeof(cl_int), &ne00)); CL_CHECK(clSetKernelArg(kernel, 8, sizeof(cl_int), &ne01)); - size_t local_work_size[3] = {64, 4, 1}; - size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, 4, 1}; + const size_t gws_x = use_tiled + ? (size_t) CEIL_DIV(ne01, 64) * 64 + : use_o4 + ? (size_t) CEIL_DIV(ne01/4, 64) * 64 + : (size_t) CEIL_DIV(ne01/2, 64) * 64; + size_t local_work_size[3] = {64, 4, 1}; + size_t global_work_size[3] = {gws_x, 4, 1}; backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); - CL_CHECK(clReleaseMemObject(ql_img)); - CL_CHECK(clReleaseMemObject(qh_img)); + if (ql_img) CL_CHECK(clReleaseMemObject(ql_img)); + if (qh_img) CL_CHECK(clReleaseMemObject(qh_img)); CL_CHECK(clReleaseMemObject(b_sub_buffer)); CL_CHECK(clReleaseMemObject(b_img)); } else { + // Tiled-layout batched GEMM. When the weight was converted to the 64-row + // tiled canonical layout (use_q6k_tiled — the default for lm_head/embed), + // the plain noshuffle GEMM below reads it as plain-transposed and produces + // garbage. Use the batched GEMM that matches the decode tiled GEMV's + // layout; it reads the f32 activation directly (column-major, no transpose). + if (use_q6k_tiled(backend_ctx, src0)) { + cl_mem b_sub_buf_t = nullptr; + cl_mem b_img_t = nullptr; + + region.origin = offset1; + region.size = ne00 * ne1 * sizeof(float); + CL_CHECK((b_sub_buf_t = clCreateSubBuffer(extra1->data_device, 0, CL_BUFFER_CREATE_TYPE_REGION, ®ion, &err), err)); + + img_fmt.image_channel_order = CL_RGBA; + img_fmt.image_channel_data_type = CL_FLOAT; + memset(&img_desc, 0, sizeof(img_desc)); + img_desc.image_type = CL_MEM_OBJECT_IMAGE1D_BUFFER; + img_desc.image_width = ne00 * ne1 / 4; + img_desc.buffer = b_sub_buf_t; + CL_CHECK((b_img_t = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); + + cl_kernel kt = backend_ctx->kernel_gemm_noshuffle_q6_K_f32_tiled; + CL_CHECK(clSetKernelArg(kt, 0, sizeof(cl_mem), &extra0_q6_K->ql)); + CL_CHECK(clSetKernelArg(kt, 1, sizeof(cl_mem), &extra0_q6_K->qh)); + CL_CHECK(clSetKernelArg(kt, 2, sizeof(cl_mem), &extra0_q6_K->s)); + CL_CHECK(clSetKernelArg(kt, 3, sizeof(cl_mem), &extra0_q6_K->d)); + CL_CHECK(clSetKernelArg(kt, 4, sizeof(cl_mem), &b_img_t)); + CL_CHECK(clSetKernelArg(kt, 5, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kt, 6, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kt, 7, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kt, 8, sizeof(int), &ne01)); + CL_CHECK(clSetKernelArg(kt, 9, sizeof(int), &ne1)); + + // Must match the kernel: NTILES=4 64-row tiles per work-group (256 rows), + // BN=8 output columns per work-group. + const int BN_T = 16; + const int WROWS = 4 * 64; // NTILES * TILE_ROWS + size_t local_work_size[3] = {64, 4, 1}; + size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01, WROWS) * 64, 4, (size_t)CEIL_DIV(ne1, BN_T)}; + backend_ctx->enqueue_ndrange_kernel(kt, 3, global_work_size, local_work_size, dst); + + CL_CHECK(clReleaseMemObject(b_img_t)); + CL_CHECK(clReleaseMemObject(b_sub_buf_t)); + return; + } + cl_mem b_sub_buf; cl_mem b_buf_trans; cl_mem b_img; @@ -17986,7 +22298,19 @@ static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_t backend_ctx->enqueue_ndrange_kernel(kernel, 2, global_size_t, local_size_t, dst); // gemm - kernel = backend_ctx->kernel_gemm_noshuffle_q6_K_f32; + // Cooperative-K small-batch (n_q in [2..8]) path: intra-WG K-split, + // mirrors the q4_K _cok path (batched serving). OPT-IN + // (GGML_OPENCL_Q6K_GEMM_COK=1), DEFAULT OFF: q6_K is the tied lm_head/ + // output projection, so the K-reassociation perturbs final logits and + // greedy is NOT byte-identical (op-tests pass, output coherent, but not + // bit-exact). It is also NEUTRAL on end-to-end MTP (q4_K cok already + // captured that; the MTP bottleneck moved off the GEMMs). Keep opt-in + // for batched serving until PPL-validated on a non-GDN q6_K model. + static const char * q6k_cok_env = getenv("GGML_OPENCL_Q6K_GEMM_COK"); + static const bool q6k_gemm_cok = (q6k_cok_env != nullptr) && (atoi(q6k_cok_env) != 0); + const bool use_q6k_cok = q6k_gemm_cok && (ne1 <= 8); + kernel = use_q6k_cok ? backend_ctx->kernel_gemm_noshuffle_q6_K_f32_cok + : backend_ctx->kernel_gemm_noshuffle_q6_K_f32; int padded_N = ne1 + padding; cl_ushort mask_f000 = 0xF000; @@ -18006,8 +22330,23 @@ static void ggml_cl_mul_mat_q6_K_f32_adreno(ggml_backend_t backend, const ggml_t CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_ushort),&mask_f000)); CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_uchar), &mask_c0)); - size_t global_work_size[3] = {(size_t)CEIL_DIV(ne1, 8), (size_t)CEIL_DIV(ne01, 4), 1}; - size_t local_work_size[3] = {2, 128, 1}; + size_t global_work_size[3]; + size_t local_work_size[3]; + if (use_q6k_cok) { + global_work_size[0] = (size_t)ne01; // rows (1 per lane) + global_work_size[1] = 8; // COK_NSG + global_work_size[2] = 1; + local_work_size[0] = 64; // COK_SG + local_work_size[1] = 8; // COK_NSG + local_work_size[2] = 1; + } else { + global_work_size[0] = (size_t)CEIL_DIV(ne1, 8); + global_work_size[1] = (size_t)CEIL_DIV(ne01, 4); + global_work_size[2] = 1; + local_work_size[0] = 2; + local_work_size[1] = 128; + local_work_size[2] = 1; + } backend_ctx->enqueue_ndrange_kernel(kernel, 3, global_work_size, local_work_size, dst); CL_CHECK(clReleaseMemObject(b_sub_buf)); @@ -18063,7 +22402,15 @@ static void ggml_cl_mul_mat_q5_K_f32_adreno(ggml_backend_t backend, const ggml_t cl_uchar mask_d4 = 0x0F; cl_uchar mask_hi2 = 0xC0; - if (ne1 == 1) { + // Multi-column (N=3) verify GEMV for q5_K: route the spec/MTP verify batch + // (ne1==3) onto the efficient GEMV path instead of the transposed-GEMM dead- + // zone (gemm_noshuffle_q5_k, the #2 chunk of MTP decode on a Q4_0-mix model + // after q4_0 mc3). Reuses the ne1==1 GEMV image setup (q + qh + activations). + // Opt-in via GGML_OPENCL_Q5K_MC3=1. Per-layer only (ne01 < 32768). + static const bool q5k_mc3 = (getenv("GGML_OPENCL_Q5K_MC3") != nullptr); + const bool use_q5k_mc3 = q5k_mc3 && (ne1 >= 2 && ne1 <= 4) && (ne01 < 32768); + + if (ne1 == 1 || use_q5k_mc3) { cl_mem q_img = nullptr; cl_mem qh_img = nullptr; cl_mem b_sub_buf = nullptr; @@ -18098,7 +22445,8 @@ static void ggml_cl_mul_mat_q5_K_f32_adreno(ggml_backend_t backend, const ggml_t img_desc.buffer = b_sub_buf; CL_CHECK((b_img = clCreateImage(context, CL_MEM_READ_ONLY, &img_fmt, &img_desc, NULL, &err), err)); - kernel = backend_ctx->kernel_gemv_noshuffle_q5_k_f32; + kernel = use_q5k_mc3 ? backend_ctx->kernel_gemv_noshuffle_q5_k_f32_mc3 + : backend_ctx->kernel_gemv_noshuffle_q5_k_f32; CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &q_img)); CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &qh_img)); @@ -18113,6 +22461,9 @@ static void ggml_cl_mul_mat_q5_K_f32_adreno(ggml_backend_t backend, const ggml_t CL_CHECK(clSetKernelArg(kernel, 10, sizeof(cl_uchar), &mask_d6)); CL_CHECK(clSetKernelArg(kernel, 11, sizeof(cl_uchar), &mask_d4)); CL_CHECK(clSetKernelArg(kernel, 12, sizeof(cl_uchar), &mask_hi2)); + if (use_q5k_mc3) { + CL_CHECK(clSetKernelArg(kernel, 13, sizeof(cl_int), &ne1)); // n_cols + } size_t local_work_size[3] = {64, 4, 1}; size_t global_work_size[3] = {(size_t)CEIL_DIV(ne01/2, 64)*64, 4, 1}; @@ -18636,13 +22987,61 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co #ifdef GGML_OPENCL_USE_ADRENO_KERNELS if(src0t == GGML_TYPE_F16 && src1t == GGML_TYPE_F32){ - if (ne01 >= 64 && ne1 >= 32 && ne00 >= 16 && (ne12 % ne02) == 0 && + // Two tiling assumptions these kernels make but nothing enforced: + // + // ne00 % TILESIZE_K(16): the K loop has no tail, so a K that does not + // divide folds 1-15 rows of whatever follows the operands into every + // output. + // + // ne01 % TILESIZE_M(64): mm_store_c_N guards the n direction with its + // `mask` argument but nothing guards m -- the store walks all 64 rows + // of the tile at a stride of M. When M does not divide, the last tile + // does not run off the end of the buffer, it writes 64 - (M % 64) + // values ON TOP OF the next column, so the result is silently wrong. + // Reachable on the KQV side for any head size >= 64 that is not a + // multiple of it (80, 96, 112). + // + // Attention shapes in the graph satisfy both -- head sizes are multiples + // of 64 and n_kv is padded -- which is why this has stayed latent. + // Declining leaves the odd shapes on the generic GEMM, which handles them. + if (ne01 >= 64 && ne1 >= 32 && ne00 >= 16 && + (ne00 % 16) == 0 && (ne01 % 64) == 0 && (ne12 % ne02) == 0 && // the KQ/KQV image kernels do not handle dim 3 (multi-stream batches) ne03 == 1 && ne13 == 1 && // dst is wrapped with image1d_buffer, the size limit applies, also src0 (ne0 * ne1 * dst->ne[2] * dst->nb[0] / 4 <= backend_ctx->image_max_buffer_size)) { - // For KQ - if (ggml_is_permuted(src0) && ggml_is_permuted(src1) && + // For KQ. + // + // Layout admission, mirroring the KQV arm below. The KQ kernel takes + // no stride arguments for A or B: it derives them as K*D_A*2 and + // K*D_B*4, i.e. it assumes both operands pack exactly D heads of K + // elements per row. Every real KV-cache view and permuted-Q view + // does, but a view spanning part of a wider allocation does not, and + // the kernel then walks the wrong rows with nothing to range-check + // it. Gate on the packed layout itself rather than on the stride + // ORDERING, which a wider parent satisfies just as well. + const bool kq_packed_a = (nb01 == (cl_ulong)ne00 * ne02 * ggml_type_size(src0t)) && + (nb02 == (cl_ulong)ne00 * ggml_type_size(src0t)); + const bool kq_packed_b = (nb11 == (cl_ulong)ne10 * ne12 * ggml_type_size(src1t)) && + (nb12 == (cl_ulong)ne10 * ggml_type_size(src1t)); + // + // ggml_is_permuted(src0) stands in for "K is head-major", but it is + // only a proxy and it COLLAPSES at n_head_kv == 1: with a single + // head there is no head stride to be out of order, so nb01 == nb02 + // and the view reports itself unpermuted. Such a KQ was declined + // here and fell through to the generic GEMM (gemma-4 E2B, and any + // other multi-query model). The packed check above is the contract + // the kernel actually needs -- it pins both strides exactly -- so + // require permutedness only where there is more than one head for + // it to mean anything. + // + // Default on; GGML_OPENCL_KQ_NHEAD_KV1=0 restores the old proxy so + // the two routings can be compared in one binary. + static const char * kq_nhkv1_env = getenv("GGML_OPENCL_KQ_NHEAD_KV1"); + static const bool kq_nhkv1_on = + (kq_nhkv1_env == nullptr || kq_nhkv1_env[0] != '0'); + if ((ggml_is_permuted(src0) || (ne02 == 1 && kq_nhkv1_on)) && ggml_is_permuted(src1) && + kq_packed_a && kq_packed_b && ((nb01 * ne01 / 4)/4 <= backend_ctx->image_max_buffer_size) && nb00 <= nb02 && nb02 <= nb01 && @@ -18650,13 +23049,15 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co nb10 <= nb12 && nb12 <= nb11 && nb11 <= nb13) { - ggml_cl_mul_mat_kq_kqv_adreno(backend, src0, src1, dst); + ggml_cl_mul_mat_kq_kqv_adreno(backend, src0, src1, dst, /*is_kq =*/ true); return; } - // For KQV + // For KQV. Reaching this arm is what makes the op a KQV; the callee + // is told so explicitly rather than re-deriving it from the strides + // the arm above has already ruled on. if (!ggml_is_contiguous(src0) && ggml_is_contiguous(src1) && ((nb02 * ne02 / 4)/4 <= backend_ctx->image_max_buffer_size)) { - ggml_cl_mul_mat_kq_kqv_adreno(backend, src0, src1, dst); + ggml_cl_mul_mat_kq_kqv_adreno(backend, src0, src1, dst, /*is_kq =*/ false); return; } } @@ -18934,7 +23335,7 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co } // q4_k x fp32 - if (src0t == GGML_TYPE_Q4_K && src1t == GGML_TYPE_F32 && !use_flat_gemv_for_large_m_q4_K(src0)) { + if (src0t == GGML_TYPE_Q4_K && src1t == GGML_TYPE_F32 && !use_flat_gemv_for_large_m_q4_K(backend_ctx, src0)) { ggml_cl_mul_mat_q4_k_f32_adreno(backend, src0, src1, dst); return; } @@ -18956,11 +23357,49 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co // GEMM using local memory // Current BK = 16, so ne00 % 16 == 0 + // + // Certain A7X compiler (E031.41) executes kernel_mul_mm_f32_f32_l4_lm poorly; + // matrices with ne11 <= 8 appears OK. + // Fallback to the MV style kernels for A7x and ne11 > 8. + // Override with GGML_OPENCL_A7X_F32_LM_BYPASS=0. + static const char * a7x_f32lm_env = getenv("GGML_OPENCL_A7X_F32_LM_BYPASS"); + static const bool a7x_f32lm_bypass = (a7x_f32lm_env == nullptr || a7x_f32lm_env[0] != '0'); if (src1t == GGML_TYPE_F32 && ne00 % 16 == 0 && - ne11 > 1) { + ne11 > 1 && + !(a7x_f32lm_bypass && src0t == GGML_TYPE_F32 && ne11 > 8 && + backend_ctx->adreno_gen == ADRENO_GPU_GEN::A7X)) { switch(src0t) { case GGML_TYPE_F32: { + // Small-N f32 GEMV for the spec/MTP verify batch: the tiled GEMM + // below always computes a full 64x64 tile, so at ne11=3 with a + // skinny f32 weight (GDN ssm_alpha/ssm_beta, M=32) it launches one + // under-occupied WG at ~2.3% tile utilization. Route to a per-output + // (m,n) GEMV (64-thread WG, K-split + __local reduce) instead. + // Opt-in GGML_OPENCL_F32_MC=1; 2D contiguous, small N + skinny M only. + static const bool f32_mc = (getenv("GGML_OPENCL_F32_MC") != nullptr); + if (f32_mc && ne11 >= 2 && ne11 <= 8 && ne01 <= 512 && (ne00 % 4 == 0) && + ne02 == 1 && ne12 == 1 && ne13 == 1 && + ggml_is_contiguous(src0) && ggml_is_contiguous(src1)) { + cl_kernel kmc = backend_ctx->kernel_gemv_f32_f32_mc; + int stride_a = ne00, stride_b = ne00, stride_d = ne01; + CL_CHECK(clSetKernelArg(kmc, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(kmc, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(kmc, 2, sizeof(cl_mem), &extra1->data_device)); + CL_CHECK(clSetKernelArg(kmc, 3, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(kmc, 4, sizeof(cl_mem), &extrad->data_device)); + CL_CHECK(clSetKernelArg(kmc, 5, sizeof(cl_ulong), &offsetd)); + CL_CHECK(clSetKernelArg(kmc, 6, sizeof(int), &ne00)); + CL_CHECK(clSetKernelArg(kmc, 7, sizeof(int), &ne01)); + CL_CHECK(clSetKernelArg(kmc, 8, sizeof(int), &ne11)); + CL_CHECK(clSetKernelArg(kmc, 9, sizeof(int), &stride_a)); + CL_CHECK(clSetKernelArg(kmc, 10, sizeof(int), &stride_b)); + CL_CHECK(clSetKernelArg(kmc, 11, sizeof(int), &stride_d)); + size_t gws[3] = {64, (size_t)ne01 * (size_t)ne11, 1}; + size_t lws[3] = {64, 1, 1}; + backend_ctx->enqueue_ndrange_kernel(kmc, 3, gws, lws, dst); + return; + } kernel = backend_ctx->kernel_mul_mm_f32_f32_l4_lm; nth0 = 128; // calculated as (BM*BN)/(TM*TN) @@ -19407,7 +23846,8 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co } kernel = backend_ctx->kernel_mul_mm_q4_k_f32_l4_lm; - nth0 = 128; // calculated as (BM*BN)/(TM*TN) + // (BM*BN)/(TM*TN): Intel uses an 8x8 microtile (WG=64), others 4x8 (WG=128) + nth0 = (backend_ctx->gpu_family == INTEL) ? 64 : 128; int batch_stride_a = ne00*ne01; int batch_stride_b = ne10*ne11; @@ -19451,7 +23891,7 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co } kernel = backend_ctx->kernel_mul_mm_q5_k_f32_l4_lm; - nth0 = 128; // calculated as (BM*BN)/(TM*TN) + nth0 = (backend_ctx->gpu_family == INTEL) ? 64 : 128; // Intel 8x8 microtile int batch_stride_a = ne00*ne01; int batch_stride_b = ne10*ne11; @@ -19616,6 +24056,7 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co } // use custom matrix x vector kernel + bool use_f16_mrow = false; switch (src0t) { case GGML_TYPE_F32: //GGML_ASSERT(ne02 == ne12); @@ -19681,7 +24122,46 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co (ne12 % r2) == 0; if (ne11 * ne12 < 4) { - kernel = backend_ctx->kernel_mul_mat_f16_f32_1row; + // Decode (single token): the legacy _1row runs one 64-lane + // subgroup per WG (one output row), under-utilizing BW. Route the + // wide f16 weight matmuls (attn proj + lm_head) to the multi-row + // variant: MROW rows per WG -> more loads in flight + activation + // staged once in __local. ne00<=8192 bounds the LDS. The mrow WG + // is 64 x MROW = 1024 work-items (> Intel's 512 max) and reduces + // within a 64-wide subgroup, so skip on Intel. + if (backend_ctx->f16_mrow && backend_ctx->gpu_family != INTEL && + backend_ctx->kernel_mul_mat_f16_f32_mrow != nullptr && + ne00 >= 128 && ne01 >= 8 && ne00 % 4 == 0 && ne00 <= 8192) { + // The register-blocked / half8 variants cast the src0 row pointer to + // half4 / half8 (8- and 16-byte loads) with no scalar fallback inside + // the kernel. ne00 % 4 == 0 constrains the element count per row, NOT + // the byte stride between rows: a permuted or strided src0 (or a view + // at an odd offset) can leave nb01/nb02/nb03 unaligned. Only take them + // when every row this dispatch touches is aligned; the base mrow kernel + // re-checks per row and falls back to its scalar loop. + const cl_ulong row_addr_bits = offset0 | nb01 | nb02 | nb03; + const bool aligned8 = (row_addr_bits & 7) == 0; + const bool aligned16 = (row_addr_bits & 15) == 0; + + // Register-blocked variants: each subgroup does RPT rows (more + // weight loads in flight per lane). 8/16 use half8 (128-bit) + // loads, gated on ne00 % 8 == 0. + const int rpt = backend_ctx->f16_mrow_rpt; + if (rpt == 16 && ne00 % 8 == 0 && aligned16 && backend_ctx->kernel_mul_mat_f16_f32_mrow_h8r2 != nullptr) { + kernel = backend_ctx->kernel_mul_mat_f16_f32_mrow_h8r2; + } else if (rpt == 8 && ne00 % 8 == 0 && aligned16 && backend_ctx->kernel_mul_mat_f16_f32_mrow_h8 != nullptr) { + kernel = backend_ctx->kernel_mul_mat_f16_f32_mrow_h8; + } else if (rpt == 4 && aligned8 && backend_ctx->kernel_mul_mat_f16_f32_mrow_r4 != nullptr) { + kernel = backend_ctx->kernel_mul_mat_f16_f32_mrow_r4; + } else if (rpt == 2 && aligned8 && backend_ctx->kernel_mul_mat_f16_f32_mrow_r2 != nullptr) { + kernel = backend_ctx->kernel_mul_mat_f16_f32_mrow_r2; + } else { + kernel = backend_ctx->kernel_mul_mat_f16_f32_mrow; + } + use_f16_mrow = true; + } else { + kernel = backend_ctx->kernel_mul_mat_f16_f32_1row; + } } else if (adreno_use_lane_split && ne00 >= 64 && ne00 <= 128) { kernel = backend_ctx->kernel_mul_mat_f16_f32_l4_dr_lq; nrows = 1; @@ -19767,6 +24247,23 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co CL_CHECK(clSetKernelArg(kernel, 21, sizeof(int), &ne1)); CL_CHECK(clSetKernelArg(kernel, 22, sizeof(int), &r2)); CL_CHECK(clSetKernelArg(kernel, 23, sizeof(int), &r3)); + if (use_f16_mrow) { + const int MROW = 16; // must match MROW in mul_mv_f16_f32_mrow.cl + // rows-per-subgroup multiplier for the selected variant: + // 1/2/4 -> half4 register blocking; 8 -> half8(1 row); 16 -> half8(2 rows) + const int rpt = backend_ctx->f16_mrow_rpt; + int rmul; + if (rpt == 16) rmul = (ne00 % 8 == 0) ? 2 : 1; + else if (rpt == 8) rmul = 1; + else rmul = rpt; // 1,2,4 + const int rows_per_wg = MROW * rmul; + // __local activation buffer: ne00 floats, rounded up for float4 access + CL_CHECK(clSetKernelArg(kernel, 24, sizeof(float) * ((ne00 + 3) / 4 * 4), nullptr)); + size_t mrow_global[] = { (size_t)((ne01 + rows_per_wg - 1) / rows_per_wg) * 64, (size_t)ne11 * MROW, (size_t)ne12 * ne13 }; + size_t mrow_local[] = { 64, (size_t)MROW, 1 }; + backend_ctx->enqueue_ndrange_kernel(kernel, 3, mrow_global, mrow_local, dst); + return; + } break; case GGML_TYPE_Q1_0: { #ifdef GGML_OPENCL_SOA_Q @@ -20265,7 +24762,7 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co if (backend_ctx->gpu_family == INTEL) { nth0 = 16; nth1 = 1; - ndst = 4; + ndst = 16; // 8->16 rows per subgroup — matches N_DST in mul_mv_q4_k_f32_flat.cl (32 spills) } else if (backend_ctx->gpu_family == ADRENO) { nth0 = 64; nth1 = 2; @@ -20339,7 +24836,7 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co if (backend_ctx->gpu_family == INTEL) { nth0 = 16; nth1 = 1; - ndst = 4; + ndst = 8; // 4->8 rows per subgroup (2x activation reuse) } else if (backend_ctx->gpu_family == ADRENO) { nth0 = 64; nth1 = 2; @@ -20440,6 +24937,12 @@ static void ggml_cl_mul_mat(ggml_backend_t backend, const ggml_tensor * src0, co CL_CHECK(clSetKernelArg(kernel, 14, sizeof(int), &ne1)); CL_CHECK(clSetKernelArg(kernel, 15, sizeof(int), &r2)); CL_CHECK(clSetKernelArg(kernel, 16, sizeof(int), &r3)); + // The optimizer-barrier arg exists only in the ADRENO_OLD_COMPILER build of + // this kernel; conformant compilers get the original 17-arg signature. + if (backend_ctx->q6_k_flat_old_compiler) { + cl_uchar q6k_mask = 0xFF; // never 0xFE in prod; see the kernel note + CL_CHECK(clSetKernelArg(kernel, 17, sizeof(cl_uchar), &q6k_mask)); + } #else kernel = backend_ctx->kernel_mul_mv_q6_K_f32; @@ -20725,18 +25228,42 @@ static void moe_router_reoerder(ggml_backend_t backend, const ggml_tensor * src, size_t fill_local_size[] = {64, 1, 1}; backend_ctx->enqueue_ndrange_kernel(kernel, 3, fill_global_size, fill_local_size, src); - // Scatter - kernel = backend_ctx->kernel_moe_scatter; - CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &original_router_buf)); - CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &post_router_buf)); - CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &emap_buf)); - CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &tile_offset_buf)); - CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &slot_counter_buf)); - CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &ne21)); - CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne20)); - CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &ne02)); + // Scatter. The deterministic variant is the default: kernel_moe_scatter derives + // each token's slot from an atomic counter, so the packing inside an expert - and + // with it the output of the ragged prefill GEMM - changes from run to run. Set + // GGML_OPENCL_MOE_STABLE_SCATTER=0 to restore the atomic version. + static const bool stable_scatter = []{ + const char * e = getenv("GGML_OPENCL_MOE_STABLE_SCATTER"); + return !e || e[0] == '\0' || e[0] != '0'; + }(); - backend_ctx->enqueue_ndrange_kernel(kernel, 3, histogram_global_size, histogram_local_size, src); + if (stable_scatter) { + kernel = backend_ctx->kernel_moe_scatter_stable; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &original_router_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &post_router_buf)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &emap_buf)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &tile_offset_buf)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(int), &ne21)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &ne20)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne02)); + + // one workgroup (one wave) per expert; each ranks its own tokens + size_t scatter_global_size[] = {64, (size_t)ne02}; + size_t scatter_local_size[] = {64, 1}; + backend_ctx->enqueue_ndrange_kernel(kernel, 2, scatter_global_size, scatter_local_size, src); + } else { + kernel = backend_ctx->kernel_moe_scatter; + CL_CHECK(clSetKernelArg(kernel, 0, sizeof(cl_mem), &original_router_buf)); + CL_CHECK(clSetKernelArg(kernel, 1, sizeof(cl_mem), &post_router_buf)); + CL_CHECK(clSetKernelArg(kernel, 2, sizeof(cl_mem), &emap_buf)); + CL_CHECK(clSetKernelArg(kernel, 3, sizeof(cl_mem), &tile_offset_buf)); + CL_CHECK(clSetKernelArg(kernel, 4, sizeof(cl_mem), &slot_counter_buf)); + CL_CHECK(clSetKernelArg(kernel, 5, sizeof(int), &ne21)); + CL_CHECK(clSetKernelArg(kernel, 6, sizeof(int), &ne20)); + CL_CHECK(clSetKernelArg(kernel, 7, sizeof(int), &ne02)); + + backend_ctx->enqueue_ndrange_kernel(kernel, 3, histogram_global_size, histogram_local_size, src); + } // [MOE_TILES] env-gated padding probe: read back total_tiles (= Sum_e // ceil(k_e/n_tile_size)) and compare to the ideal tile count for the real @@ -20925,10 +25452,33 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, CL_CHECK(clReleaseMemObject(buf_src2)); } else { // for gemm - kernel = backend_ctx->kernel_gemm_moe_q4_0_f32_ns; - if (backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin) { - kernel = backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin; - } + // dp4a (int8) prefill GEMM variant + static const char * q4_0_moe_dp4a_env = getenv("GGML_OPENCL_Q4_0_MOE_DP4A"); + + // It turns out that the prebuilt kernel only outperforms the dp4a variant (on X2-90) + // at very large routing counts, so we gate its use accordingly using moe_bin_min, + // which can be overridden via the GGML_OPENCL_MOE_BIN_MIN_ROUTINGS environment variable. + // The routing count is ne20 * ne21 (n_expert_used * n_tokens). + static const char * moe_bin_min_env = getenv("GGML_OPENCL_MOE_BIN_MIN_ROUTINGS"); + const int moe_bin_min = moe_bin_min_env ? atoi(moe_bin_min_env) : 4096; + + // whether bin kernels are available + const bool bin_available = backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin != nullptr; + const bool dp4a_bin_available = backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a_bin != nullptr; + + bool use_moe_dp4a = q4_0_moe_dp4a_env + ? (atoi(q4_0_moe_dp4a_env) != 0) + : (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E + && (dp4a_bin_available || !bin_available + || (int)(ne20 * ne21) < moe_bin_min)); + // dot prod has to be available + use_moe_dp4a = backend_ctx->has_integer_dot && use_moe_dp4a; + + const bool use_bin_kernel = bin_available && !use_moe_dp4a; + + kernel = use_bin_kernel + ? backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin + : backend_ctx->kernel_gemm_moe_q4_0_f32_ns; // Reorder router if called from test-backend-ops or when new router is generated. // Otherwise reuse the reordered result from previous mul_mat_id call. @@ -20941,16 +25491,6 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, cl_mem buf_src1_reordered = nullptr, image_src1_reordered = nullptr; cl_mem buf_src2, buf_src2_emap; - // dp4a (int8) prefill GEMM variant - static const char * q4_0_moe_dp4a_env = getenv("GGML_OPENCL_Q4_0_MOE_DP4A"); - bool use_moe_dp4a = q4_0_moe_dp4a_env - ? (atoi(q4_0_moe_dp4a_env) != 0) - : (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E); - // dot prod has to be available - use_moe_dp4a = backend_ctx->has_integer_dot && use_moe_dp4a; - // bin kernel takes precedence - use_moe_dp4a = use_moe_dp4a && backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin == nullptr; - cl_buffer_region region; region.origin = 0; region.size = sizeof(int) * max_post_router_tile * n_tile_size; @@ -20989,7 +25529,7 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, cl_image_desc image_desc_buf_src1; image_format_buf_src1 = {CL_RGBA, CL_FLOAT}; image_desc_buf_src1 = {CL_MEM_OBJECT_IMAGE1D_BUFFER, static_cast(ne00 * max_post_router_tile * n_tile_size / 4), 0,0,0,0,0,0,0, {buf_src1_reordered}}; - if (backend_ctx->kernel_gemm_moe_q4_0_f32_ns_bin) { + if (use_bin_kernel) { // bin kernel uses slightly different image format image_format_buf_src1 = {CL_R, CL_FLOAT}; image_desc_buf_src1.image_width = static_cast(ne00 * max_post_router_tile * n_tile_size); @@ -21055,6 +25595,10 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, // dp4a GEMM cl_kernel dk = backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a; + if (backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a_bin) { + dk = backend_ctx->kernel_gemm_moe_q4_0_q8_1_dp4a_bin; + } + int aidx = 0; CL_CHECK(clSetKernelArg(dk, aidx++, sizeof(cl_mem), &extra0_q4_0->q_img)); CL_CHECK(clSetKernelArg(dk, aidx++, sizeof(cl_mem), &extra0_q4_0->d)); @@ -22893,8 +27437,10 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, : (backend_ctx->adreno_gen == ADRENO_GPU_GEN::X2E); // dot prod has to be available use_moe_dp4a = backend_ctx->has_integer_dot && use_moe_dp4a; - // bin kernel takes precedence - use_moe_dp4a = use_moe_dp4a && backend_ctx->kernel_gemm_moe_mxfp4_f32_ns_bin == nullptr; + // bin kernel takes precedence, dp4a bin kernel has higher priority than normal bin kernel + if (backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a_bin == nullptr) { + use_moe_dp4a = use_moe_dp4a && backend_ctx->kernel_gemm_moe_mxfp4_f32_ns_bin == nullptr; + } cl_buffer_region region; region.origin = 0; @@ -23003,6 +27549,10 @@ static void ggml_cl_mul_mat_id(ggml_backend_t backend, const ggml_tensor * src0, // dp4a GEMM cl_kernel dk = backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a; + if (backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a_bin) { + dk = backend_ctx->kernel_gemm_moe_mxfp4_q8_1_dp4a_bin; + } + int aidx = 0; CL_CHECK(clSetKernelArg(dk, aidx++, sizeof(cl_mem), &extra0_mxfp4->q_img)); CL_CHECK(clSetKernelArg(dk, aidx++, sizeof(cl_mem), &extra0_mxfp4->e)); @@ -23240,6 +27790,38 @@ static void ggml_cl_cpy(ggml_backend_t backend, const ggml_tensor * src0, const cl_ulong offset0 = extra0->offset + src0->view_offs; cl_ulong offset1 = extra1->offset + src1->view_offs; + // A contiguous f32 -> f32 copy is a linear move. The kernel below maps one workgroup to + // each row, so a tensor with few long rows runs on a single compute unit; dispatch those + // over the whole device instead. GGML_OPENCL_CPY_FLAT=0 restores the row-mapped path. + static const bool cpy_flat_on = []{ + const char * e = getenv("GGML_OPENCL_CPY_FLAT"); + return !(e && e[0] == '0'); + }(); + if (cpy_flat_on && backend_ctx->kernel_cpy_f32_f32_flat != nullptr && + src0t == GGML_TYPE_F32 && src1t == GGML_TYPE_F32 && + ggml_is_contiguous(src0) && ggml_is_contiguous(src1) && + ggml_nelements(src0) == ggml_nelements(src1)) { + cl_kernel k = backend_ctx->kernel_cpy_f32_f32_flat; + const cl_ulong nelem = (cl_ulong) ggml_nelements(src0); + const cl_ulong n4 = nelem / 4; + + CL_CHECK(clSetKernelArg(k, 0, sizeof(cl_mem), &extra0->data_device)); + CL_CHECK(clSetKernelArg(k, 1, sizeof(cl_ulong), &offset0)); + CL_CHECK(clSetKernelArg(k, 2, sizeof(cl_mem), &extra1->data_device)); + CL_CHECK(clSetKernelArg(k, 3, sizeof(cl_ulong), &offset1)); + CL_CHECK(clSetKernelArg(k, 4, sizeof(cl_ulong), &nelem)); + CL_CHECK(clSetKernelArg(k, 5, sizeof(cl_ulong), &n4)); + + // one work item per float4, plus one for the trailing scalars + const size_t items = (size_t) n4 + ((nelem % 4) ? 1 : 0); + const size_t lsz = MIN((size_t) 64, backend_ctx->max_workgroup_size); + size_t global_work_size[] = { ((items + lsz - 1) / lsz) * lsz, 1, 1 }; + size_t local_work_size[] = { lsz, 1, 1 }; + + backend_ctx->enqueue_ndrange_kernel(k, 1, global_work_size, local_work_size, src1); + return; + } + cl_kernel kernel; switch (src0t) { @@ -23703,6 +28285,7 @@ static void ggml_cl_rope(ggml_backend_t backend, const ggml_tensor * src0, const const int n_dims = ((int *) dst->op_params)[1]; const int mode = ((int *) dst->op_params)[2]; const int n_ctx_orig = ((int32_t *) dst->op_params)[4]; + const int n_offs = ((int32_t *) dst->op_params)[15]; float freq_base; float freq_scale; @@ -23731,6 +28314,7 @@ static void ggml_cl_rope(ggml_backend_t backend, const ggml_tensor * src0, const if (is_vision) { GGML_ASSERT(n_dims == ne00/2); + GGML_ASSERT(n_offs == 0); // offset not supported for vision, as the rotated pairs span the whole row } cl_kernel kernel; @@ -23822,6 +28406,12 @@ static void ggml_cl_rope(ggml_backend_t backend, const ggml_tensor * src0, const if (is_mrope && !is_vision) { CL_CHECK(clSetKernelArg(kernel, 34, sizeof(int), &is_imrope)); } + // norm and neox have n_offs after beta_slow, mrope has it after is_imrope + if (!is_mrope && !is_vision) { + CL_CHECK(clSetKernelArg(kernel, 33, sizeof(int), &n_offs)); + } else if (is_mrope && !is_vision) { + CL_CHECK(clSetKernelArg(kernel, 35, sizeof(int), &n_offs)); + } size_t global_work_size[] = {(size_t)ne01*nth, (size_t)ne02, (size_t)ne03}; size_t local_work_size[] = {(size_t)nth, 1, 1}; @@ -24245,6 +28835,13 @@ static void ggml_cl_glu(ggml_backend_t backend, const ggml_tensor * src0, const case GGML_GLU_OP_SWIGLU_OAI: kernel = backend_ctx->kernel_swiglu_oai; break; + case GGML_GLU_OP_SWIGLU_CLAMP: + if (dst->type == GGML_TYPE_F32) { + kernel = backend_ctx->kernel_swiglu_clamp; + } else { + kernel = backend_ctx->kernel_swiglu_clamp_f16; + } + break; case GGML_GLU_OP_GEGLU_ERF: if (dst->type == GGML_TYPE_F32) { kernel = backend_ctx->kernel_geglu_erf; @@ -24300,8 +28897,10 @@ static void ggml_cl_glu(ggml_backend_t backend, const ggml_tensor * src0, const CL_CHECK(clSetKernelArg(kernel, 10, sizeof(int), &ne00_off)); CL_CHECK(clSetKernelArg(kernel, 11, sizeof(int), &ne10_off)); - if (ggml_get_glu_op(dst) == GGML_GLU_OP_SWIGLU_OAI) { + if (ggml_get_glu_op(dst) == GGML_GLU_OP_SWIGLU_OAI || ggml_get_glu_op(dst) == GGML_GLU_OP_SWIGLU_CLAMP) { CL_CHECK(clSetKernelArg(kernel, 12, sizeof(float), &limit)); + } + if (ggml_get_glu_op(dst) == GGML_GLU_OP_SWIGLU_OAI) { CL_CHECK(clSetKernelArg(kernel, 13, sizeof(float), &alpha)); } @@ -24656,6 +29255,42 @@ bool ggml_cl_compute_forward(ggml_backend_t backend, struct ggml_tensor * tensor } func = ggml_cl_abs; break; + case GGML_UNARY_OP_SGN: + if (!any_on_device) { return false; } + func = ggml_cl_sgn; + break; + case GGML_UNARY_OP_STEP: + if (!any_on_device) { return false; } + func = ggml_cl_step; + break; + case GGML_UNARY_OP_ELU: + if (!any_on_device) { return false; } + func = ggml_cl_elu; + break; + case GGML_UNARY_OP_HARDSWISH: + if (!any_on_device) { return false; } + func = ggml_cl_hardswish; + break; + case GGML_UNARY_OP_HARDSIGMOID: + if (!any_on_device) { return false; } + func = ggml_cl_hardsigmoid; + break; + case GGML_UNARY_OP_FLOOR: + if (!any_on_device) { return false; } + func = ggml_cl_floor; + break; + case GGML_UNARY_OP_CEIL: + if (!any_on_device) { return false; } + func = ggml_cl_ceil; + break; + case GGML_UNARY_OP_ROUND: + if (!any_on_device) { return false; } + func = ggml_cl_round; + break; + case GGML_UNARY_OP_TRUNC: + if (!any_on_device) { return false; } + func = ggml_cl_trunc; + break; case GGML_UNARY_OP_SOFTPLUS: if (!any_on_device) { return false; @@ -24743,6 +29378,14 @@ bool ggml_cl_compute_forward(ggml_backend_t backend, struct ggml_tensor * tensor } func = ggml_cl_ssm_conv; break; + case GGML_OP_SSM_SCAN: + if (!any_on_device) { + return false; + } + // SSM_SCAN has 7 source tensors, so it cannot use the standard + // (src0, src1, dst) func signature. Dispatch directly and return. + ggml_cl_ssm_scan(backend, tensor); + return true; case GGML_OP_GATED_DELTA_NET: if (!any_on_device) { return false; diff --git a/ggml/src/ggml-opencl/kernels/concat.cl b/ggml/src/ggml-opencl/kernels/concat.cl index 2fbd7851..8ecf7466 100644 --- a/ggml/src/ggml-opencl/kernels/concat.cl +++ b/ggml/src/ggml-opencl/kernels/concat.cl @@ -1,56 +1,66 @@ -kernel void kernel_concat_f32( - global const char * src0, - ulong offset0, - global const char * src1, - ulong offset1, - global char * dst, - ulong offsetd, - int ne00, - int ne01, - int ne02, - int ne03, - ulong nb00, - ulong nb01, - ulong nb02, - ulong nb03, - ulong nb10, - ulong nb11, - ulong nb12, - ulong nb13, - int ne0, - ulong nb0, - ulong nb1, - ulong nb2, - ulong nb3, - int dim -) { - src0 = src0 + offset0; - src1 = src1 + offset1; - dst = dst + offsetd; - - const int i3 = get_group_id(2); - const int i2 = get_group_id(1); - const int i1 = get_group_id(0); - - int o[4] = {0, 0, 0, 0}; - o[dim] = dim == 0 ? ne00 : (dim == 1 ? ne01 : (dim == 2 ? ne02 : ne03)); - - global const float * x; - - for (int i0 = get_local_id(0); i0 < ne0; i0 += get_local_size(0)) { - if (i0 < ne00 && i1 < ne01 && i2 < ne02 && i3 < ne03) { - x = (global const float *)(src0 + (i3 )*nb03 + (i2 )*nb02 + (i1 )*nb01 + (i0 )*nb00); - } else { - x = (global const float *)(src1 + (i3 - o[3])*nb13 + (i2 - o[2])*nb12 + (i1 - o[1])*nb11 + (i0 - o[0])*nb10); - } - - global float * y = (global float *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); +// concat is a pure copy, so the kernels are keyed by element byte size +// (1/2/4/8) rather than logical type, matching the CUDA backend. - *y = *x; - } +#define KERNEL_CONCAT(SUFFIX, T) \ +kernel void kernel_concat_##SUFFIX( \ + global const char * src0, \ + ulong offset0, \ + global const char * src1, \ + ulong offset1, \ + global char * dst, \ + ulong offsetd, \ + int ne00, \ + int ne01, \ + int ne02, \ + int ne03, \ + ulong nb00, \ + ulong nb01, \ + ulong nb02, \ + ulong nb03, \ + ulong nb10, \ + ulong nb11, \ + ulong nb12, \ + ulong nb13, \ + int ne0, \ + ulong nb0, \ + ulong nb1, \ + ulong nb2, \ + ulong nb3, \ + int dim \ +) { \ + src0 = src0 + offset0; \ + src1 = src1 + offset1; \ + dst = dst + offsetd; \ + \ + const int i3 = get_group_id(2); \ + const int i2 = get_group_id(1); \ + const int i1 = get_group_id(0); \ + \ + int o[4] = {0, 0, 0, 0}; \ + o[dim] = dim == 0 ? ne00 : (dim == 1 ? ne01 : (dim == 2 ? ne02 : ne03)); \ + \ + global const T * x; \ + \ + for (int i0 = get_local_id(0); i0 < ne0; i0 += get_local_size(0)) { \ + if (i0 < ne00 && i1 < ne01 && i2 < ne02 && i3 < ne03) { \ + x = (global const T *)(src0 + (i3 )*nb03 + (i2 )*nb02 + (i1 )*nb01 + (i0 )*nb00); \ + } else { \ + x = (global const T *)(src1 + (i3 - o[3])*nb13 + (i2 - o[2])*nb12 + (i1 - o[1])*nb11 + (i0 - o[0])*nb10); \ + } \ + \ + global T * y = (global T *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); \ + \ + *y = *x; \ + } \ } -kernel void kernel_concat_f32_pack( +KERNEL_CONCAT(b1, char) +KERNEL_CONCAT(b2, short) +KERNEL_CONCAT(b4, int) +KERNEL_CONCAT(b8, long) + +// packed variant for the common dim==0, small-ne0 case (4-byte elements only). +kernel void kernel_concat_b4_pack( global const char * src0, ulong offset0, global const char * src1, @@ -104,14 +114,14 @@ kernel void kernel_concat_f32_pack( o[dim] = dim == 0 ? ne00 : (dim == 1 ? ne01 : (dim == 2 ? ne02 : ne03)); for (int i0 = lane; i0 < ne0; i0 += tpr) { - global const float * x; + global const int * x; if (i0 < ne00 && i1 < ne01 && i2 < ne02 && i3 < ne03) { - x = (global const float *)(src0 + (i3 )*nb03 + (i2 )*nb02 + (i1 )*nb01 + (i0 )*nb00); + x = (global const int *)(src0 + (i3 )*nb03 + (i2 )*nb02 + (i1 )*nb01 + (i0 )*nb00); } else { - x = (global const float *)(src1 + (i3 - o[3])*nb13 + (i2 - o[2])*nb12 + (i1 - o[1])*nb11 + (i0 - o[0])*nb10); + x = (global const int *)(src1 + (i3 - o[3])*nb13 + (i2 - o[2])*nb12 + (i1 - o[1])*nb11 + (i0 - o[0])*nb10); } - global float * y = (global float *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); + global int * y = (global int *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); *y = *x; } diff --git a/ggml/src/ggml-opencl/kernels/conv2d.cl b/ggml/src/ggml-opencl/kernels/conv2d.cl index e339c90c..8a04c2e5 100644 --- a/ggml/src/ggml-opencl/kernels/conv2d.cl +++ b/ggml/src/ggml-opencl/kernels/conv2d.cl @@ -48,8 +48,8 @@ kernel void kernel_conv_2d( uint Cout, uint Cin, uint N, uint KW, uint KH, uint W, uint H, uint OW, uint OH, uint s0, uint s1, uint p0, uint p1, uint d0, uint d1, - uint nb01, uint nb02, uint nb03, - uint nb11, uint nb12, uint nb13, + uint nb00, uint nb01, uint nb02, uint nb03, + uint nb10, uint nb11, uint nb12, uint nb13, uint nb1, uint nb2, uint nb3 ) { global T_FLOAT* knl_data = (global T_FLOAT*) ((global char*)p_knl + off_knl); @@ -95,7 +95,7 @@ kernel void kernel_conv_2d( const uint Cin_idx = crs_g / (KW*KH); const uint KH_idx = (crs_g - Cin_idx*KW*KH) / KW; const uint KW_idx = crs_g - Cin_idx*KW*KH - KH_idx*KW; - const uint knl_idx = KW_idx + KH_idx*nb01 + Cin_idx*nb02 + k_g*nb03; + const uint knl_idx = KW_idx*nb00 + KH_idx*nb01 + Cin_idx*nb02 + k_g*nb03; Ash[k_l * BS_CRS + crs_l] = knl_data[knl_idx]; } else { Ash[k_l * BS_CRS + crs_l] = (T_FLOAT)0.0f; @@ -123,7 +123,7 @@ kernel void kernel_conv_2d( const int W_idx = (int)(OW_idx * s0 + KW_idx * d0 - p0); if (H_idx >= 0 && H_idx < H && W_idx >= 0 && W_idx < W) { - const uint src_idx = W_idx + H_idx * nb11 + Cin_idx * nb12 + N_idx * nb13; + const uint src_idx = W_idx * nb10 + H_idx * nb11 + Cin_idx * nb12 + N_idx * nb13; ((T_FLOAT*)&val)[v] = src_data[src_idx]; } } diff --git a/ggml/src/ggml-opencl/kernels/conv2d_f16_f32.cl b/ggml/src/ggml-opencl/kernels/conv2d_f16_f32.cl index cb05637f..94788e7e 100644 --- a/ggml/src/ggml-opencl/kernels/conv2d_f16_f32.cl +++ b/ggml/src/ggml-opencl/kernels/conv2d_f16_f32.cl @@ -39,8 +39,8 @@ kernel void kernel_conv_2d( uint Cout, uint Cin, uint N, uint KW, uint KH, uint W, uint H, uint OW, uint OH, uint s0, uint s1, uint p0, uint p1, uint d0, uint d1, - uint nb01, uint nb02, uint nb03, - uint nb11, uint nb12, uint nb13, + uint nb00, uint nb01, uint nb02, uint nb03, + uint nb10, uint nb11, uint nb12, uint nb13, uint nb1, uint nb2, uint nb3 ) { global half* knl_data = (global half*) ((global char*)p_knl + off_knl); @@ -86,7 +86,7 @@ kernel void kernel_conv_2d( const uint Cin_idx = crs_g / (KW*KH); const uint KH_idx = (crs_g - Cin_idx*KW*KH) / KW; const uint KW_idx = crs_g - Cin_idx*KW*KH - KH_idx*KW; - const uint knl_idx = KW_idx + KH_idx*nb01 + Cin_idx*nb02 + k_g*nb03; + const uint knl_idx = KW_idx*nb00 + KH_idx*nb01 + Cin_idx*nb02 + k_g*nb03; Ash[k_l * BS_CRS + crs_l] = knl_data[knl_idx]; } else { Ash[k_l * BS_CRS + crs_l] = (half)0.0f; @@ -114,7 +114,7 @@ kernel void kernel_conv_2d( const int W_idx = (int)(OW_idx * s0 + KW_idx * d0 - p0); if (H_idx >= 0 && H_idx < H && W_idx >= 0 && W_idx < W) { - const uint src_idx = W_idx + H_idx * nb11 + Cin_idx * nb12 + N_idx * nb13; + const uint src_idx = W_idx * nb10 + H_idx * nb11 + Cin_idx * nb12 + N_idx * nb13; ((float*)&val)[v] = src_data[src_idx]; } } diff --git a/ggml/src/ggml-opencl/kernels/cpy.cl b/ggml/src/ggml-opencl/kernels/cpy.cl index adbd2e76..e875bfaf 100644 --- a/ggml/src/ggml-opencl/kernels/cpy.cl +++ b/ggml/src/ggml-opencl/kernels/cpy.cl @@ -286,3 +286,28 @@ kernel void kernel_cpy_i32_i32( dst_data[i00] = src[0]; } } + +// Contiguous f32 copy, one work item per float4 over the whole tensor. The kernels above map +// one workgroup to each row, which leaves a tensor with few long rows on a single compute unit. +// vload4/vstore4 rather than a float4 cast: these buffers carry an arbitrary 4-byte view offset. +kernel void kernel_cpy_f32_f32_flat( + global float * src0, + ulong offset0, + global float * dst, + ulong offsetd, + ulong ne, + ulong n4 +) { + src0 = (global float*)((global char*)src0 + offset0); + dst = (global float*)((global char*)dst + offsetd); + + const ulong i = get_global_id(0); + + if (i < n4) { + vstore4(vload4(i, src0), i, dst); + } else if (i == n4) { + for (ulong t = n4 * 4; t < ne; ++t) { + dst[t] = src0[t]; + } + } +} diff --git a/ggml/src/ggml-opencl/kernels/cvt.cl b/ggml/src/ggml-opencl/kernels/cvt.cl index 3d6cff7c..acc8f980 100644 --- a/ggml/src/ggml-opencl/kernels/cvt.cl +++ b/ggml/src/ggml-opencl/kernels/cvt.cl @@ -1110,6 +1110,78 @@ kernel void kernel_restore_block_q4_k_trans4_ns( } } +//------------------------------------------------------------------------------ +// kernel_convert_block_q4_k_tiled_ns +// +// Tiled-wide layout for the long-vocab q4_K lm_head/embed GEMV (decode path). +// Mirror of kernel_convert_block_q6_k_tiled_ns: recovers each weight's 4-bit +// code in CANONICAL ggml element order (e in [0,256)) and re-packs into 32 uints +// (8 codes/uint), stored TILED by 64 output rows so the matching GEMV +// (gemv_noshuffle_q4_k_f32_tiled) coalesces every weight load. The 12-byte +// packed scale block `s` and d/dm are stored per (row, K-block) tiled; the GEMV +// re-derives the 8 (scale,min) pairs via get_scale_min_k4, exactly like the o4 +// kernel. Both ends owned here -> correct by construction vs the reference q4_K +// dequant. Requires ne01 % 64 == 0 (gated host-side). Buffer sizes identical to +// the trans4_ns layout. +// +// q uint4 granule g of (row r, K-block sb): idx = ((rt*ne00_blk+sb)*8 + g)*64 + rit +// s (12 bytes) of (r, sb): idx = (rt*ne00_blk+sb)*64 + rit, *12 +// d/dm (half) of (r, sb): idx = (rt*ne00_blk+sb)*64 + rit +// where rt = r/64, rit = r%64. +//------------------------------------------------------------------------------ +kernel void kernel_convert_block_q4_k_tiled_ns( + __global struct block_q4_K * src0, + __global uint * dst_q, // 32 uints / superblock (4-bit codes, 8 codes/uint) + __global half * dst_d, // 1 half / superblock + __global half * dst_dm, // 1 half / superblock + __global uchar * dst_s, // K_SCALE_SIZE (12) bytes / superblock + uint ne00, + uint ne01 +) { + uint i00 = get_global_id(1); // K-block index (superblock along ne00) + uint i01 = get_global_id(0); // output row index (along ne01) + uint i02 = get_global_id(2); // batch + + uint ne00_blk = ne00 / QK_K; + + uint src_blk_offset = i00 + i01 * ne00_blk + i02 * ne00_blk * ne01; + __global struct block_q4_K * b = src0 + src_blk_offset; + + uint rt = i01 / 64; + uint rit = i01 % 64; + uint tile_blk = (i02 * (ne01 / 64) + rt) * ne00_blk + i00; + + // --- recover canonical 4-bit codes in e-order, pack 8 codes/uint --- + uint qw[32] = {0}; + for (uint e = 0; e < 256; ++e) { + uint g = e >> 6; // group 0..3 (q advances 32 bytes/group) + uint within = e & 63u; + uint hlf = within >> 5; // 0 = low nibble, 1 = high nibble + uint l = within & 31u; // 0..31 + uchar byte = b->q[g * 32u + l]; + uint code = (hlf == 0u) ? (uint)(byte & 0x0F) : (uint)(byte >> 4); + qw[e >> 3] |= code << ((e & 7u) * 4u); + } + + for (uint gr = 0; gr < 8; ++gr) { + uint base = (tile_blk * 8u + gr) * 64u + rit; // uint4 index + dst_q[base * 4u + 0u] = qw[gr * 4u + 0u]; + dst_q[base * 4u + 1u] = qw[gr * 4u + 1u]; + dst_q[base * 4u + 2u] = qw[gr * 4u + 2u]; + dst_q[base * 4u + 3u] = qw[gr * 4u + 3u]; + } + + // packed scales (12 bytes), tiled per (row, block) + __global uchar * s_dst = dst_s + (tile_blk * 64u + rit) * K_SCALE_SIZE; + #pragma unroll + for (int i = 0; i < K_SCALE_SIZE; ++i) { + s_dst[i] = b->s[i]; + } + + dst_d [tile_blk * 64u + rit] = b->d; + dst_dm[tile_blk * 64u + rit] = b->dm; +} + kernel void kernel_convert_block_q5_k_trans4_ns( __global struct block_q5_K * src0, __global uint * dst_qs, @@ -1494,6 +1566,105 @@ kernel void kernel_restore_block_mxfp4_trans( b->e = src_e[src_blk_offset]; } +//------------------------------------------------------------------------------ +// kernel_convert_block_q6_k_tiled_ns +// +// Tiled-wide layout for the long-vocab q6_K lm_head/embed GEMV (decode path). +// Unlike *_trans4_ns (which mirrors the bit-interleave the legacy 2-output GEMV +// consumes), this kernel is correct-by-construction against the CANONICAL ggml +// q6_K dequant: it recovers each weight's 6-bit code in element order e in +// [0,256), then re-packs low-4-bits into 32 uints (8 codes/uint) and high-2-bits +// into 16 uints (16 codes/uint). The matching GEMV (gemv_noshuffle_q6_k_f32_tiled) +// unpacks the same order, so both ends are owned here. +// +// Storage is TILED by 64 output rows so the GEMV's 64-thread tile coalesces: +// ql uint4 granule g of (row r, K-block sb): idx = ((rt*ne00_blk + sb)*8 + g)*64 + rit +// qh uint4 granule g: idx = ((rt*ne00_blk + sb)*4 + g)*64 + rit +// scales (char16) of (r, sb): idx = (rt*ne00_blk + sb)*64 + rit +// d (half) of (r, sb): idx = (rt*ne00_blk + sb)*64 + rit +// where rt = r/64, rit = r%64. Requires ne01 % 64 == 0 (gated host-side). +// Buffer sizes are byte-identical to the trans4_ns layout. +//------------------------------------------------------------------------------ +kernel void kernel_convert_block_q6_k_tiled_ns( + __global struct block_q6_K * src0, + __global uint * dst_ql, // 32 uints / superblock (low 4 bits, 8 codes/uint) + __global uint * dst_qh, // 16 uints / superblock (high 2 bits, 16 codes/uint) + __global half * dst_d, // 1 half / superblock + __global char * dst_s, // 16 chars/ superblock + uint ne00, + uint ne01 +) { + uint i00 = get_global_id(1); // K-block index (superblock along ne00) + uint i01 = get_global_id(0); // output row index (along ne01) + uint i02 = get_global_id(2); // batch + + uint ne00_blk = ne00 / QK_K; + + // Source block: row-major over (i02, i01, i00). + uint src_blk_offset = i00 + i01 * ne00_blk + i02 * ne00_blk * ne01; + __global struct block_q6_K * b = src0 + src_blk_offset; + + uint rt = i01 / 64; + uint rit = i01 % 64; + uint tile_blk = (i02 * (ne01 / 64) + rt) * ne00_blk + i00; // tile-major (row-tile, K-block) + + // --- recover canonical 6-bit codes, pack into ql (4b) + qh (2b) in e-order --- + // 32 ql-uints (8 low-nibbles each) + 16 qh-uints (16 2-bit slots each). + uint qlw[32] = {0}; + uint qhw[16] = {0}; + + for (uint e = 0; e < 256; ++e) { + uint n = (e >= 128) ? 1u : 0u; // which 128-half + uint within = e - n * 128u; + uint q = within / 32u; // quadrant 0..3 + uint l = within % 32u; // 0..31 + + uint off_ql = n * 64u; // raw ql byte base for this half + uint off_qh = n * 32u; // raw qh byte base for this half + + uchar low4; + uchar qlb0 = b->ql[off_ql + l]; + uchar qlb1 = b->ql[off_ql + l + 32]; + if (q == 0) low4 = qlb0 & 0x0F; + else if (q == 1) low4 = qlb1 & 0x0F; + else if (q == 2) low4 = (qlb0 >> 4) & 0x0F; + else low4 = (qlb1 >> 4) & 0x0F; + + uchar hi2 = (b->qh[off_qh + l] >> (q * 2u)) & 0x03; + + // pack low4 (e-order): uint e/8, nibble (e%8) + qlw[e >> 3] |= ((uint)low4) << ((e & 7u) * 4u); + // pack hi2 (e-order): uint e/16, 2-bit slot (e%16) + qhw[e >> 4] |= ((uint)hi2) << ((e & 15u) * 2u); + } + + // --- write tiled --- + for (uint g = 0; g < 8; ++g) { + uint base = (tile_blk * 8u + g) * 64u + rit; // uint4 index + dst_ql[base * 4u + 0u] = qlw[g * 4u + 0u]; + dst_ql[base * 4u + 1u] = qlw[g * 4u + 1u]; + dst_ql[base * 4u + 2u] = qlw[g * 4u + 2u]; + dst_ql[base * 4u + 3u] = qlw[g * 4u + 3u]; + } + for (uint g = 0; g < 4; ++g) { + uint base = (tile_blk * 4u + g) * 64u + rit; // uint4 index + dst_qh[base * 4u + 0u] = qhw[g * 4u + 0u]; + dst_qh[base * 4u + 1u] = qhw[g * 4u + 1u]; + dst_qh[base * 4u + 2u] = qhw[g * 4u + 2u]; + dst_qh[base * 4u + 3u] = qhw[g * 4u + 3u]; + } + + // scales: 16 chars contiguous per (row, block), tiled + __global char * s_dst = dst_s + (tile_blk * 64u + rit) * 16u; + #pragma unroll + for (int i = 0; i < 16; ++i) { + s_dst[i] = b->scales[i]; + } + + // super-block scale + dst_d[tile_blk * 64u + rit] = b->d; +} + kernel void kernel_convert_block_mxfp4_trans4_ns( global struct block_mxfp4 * src0, __global uint * dst_q, diff --git a/ggml/src/ggml-opencl/kernels/flash_attn_f16.cl b/ggml/src/ggml-opencl/kernels/flash_attn_f16.cl index fc58a22e..f9797d34 100644 --- a/ggml/src/ggml-opencl/kernels/flash_attn_f16.cl +++ b/ggml/src/ggml-opencl/kernels/flash_attn_f16.cl @@ -118,6 +118,17 @@ __kernel void flash_attn_f16( __local DATA_TYPE4 l_v[BLOCK_N][DV_VEC]; for (int k_start = 0; k_start < n_kv; k_start += BLOCK_N) { +#if WG_SIZE > FA_SG + // WAR on l_k/l_v: a thread that finishes the compute below early — either + // it skipped it (my_query_row >= n_q, the continue) or its subgroup simply + // ran ahead — wraps around and reloads the tiles while another subgroup is + // still reading them. Any WG that is exactly one lockstep subgroup + // (WG_SIZE == FA_SG) cannot diverge and hides this; a WG spanning multiple + // subgroups (Intel sg=32, or BLOCK_M > 64 on Adreno) corrupts the result. + // All threads reach this each iteration (no-op on the first), so it does + // not diverge with the continue. Compiled out when WG == one subgroup. + barrier(CLK_LOCAL_MEM_FENCE); +#endif for (int i = tid; i < BLOCK_N * DK_VEC; i += WG_SIZE) { const int row = i / DK_VEC; const int col = i % DK_VEC; diff --git a/ggml/src/ggml-opencl/kernels/flash_attn_f32.cl b/ggml/src/ggml-opencl/kernels/flash_attn_f32.cl index 599877bd..5911524e 100644 --- a/ggml/src/ggml-opencl/kernels/flash_attn_f32.cl +++ b/ggml/src/ggml-opencl/kernels/flash_attn_f32.cl @@ -119,13 +119,15 @@ __kernel void flash_attn_f32( __local DATA_TYPE4 l_v[BLOCK_N][DV_VEC]; for (int k_start = 0; k_start < n_kv; k_start += BLOCK_N) { -#if FA_SG < 64 - // WAR on l_k/l_v: threads with my_query_row >= n_q skip the compute below - // (continue) and would race ahead to reload the tiles while active threads - // still read them. A single 64-wide Adreno subgroup (WG == sg) runs lockstep - // and hides this; a WG that spans multiple narrower subgroups (Intel sg=32) - // corrupts the result. All threads reach this each iteration (no-op on the - // first), so it does not diverge with the continue. Compiled out at sg=64. +#if WG_SIZE > FA_SG + // WAR on l_k/l_v: a thread that finishes the compute below early — either + // it skipped it (my_query_row >= n_q, the continue) or its subgroup simply + // ran ahead — wraps around and reloads the tiles while another subgroup is + // still reading them. Any WG that is exactly one lockstep subgroup + // (WG_SIZE == FA_SG) cannot diverge and hides this; a WG spanning multiple + // subgroups (Intel sg=32, or BLOCK_M > 64 on Adreno) corrupts the result. + // All threads reach this each iteration (no-op on the first), so it does + // not diverge with the continue. Compiled out when WG == one subgroup. barrier(CLK_LOCAL_MEM_FENCE); #endif for (int i = tid; i < BLOCK_N * DK_VEC; i += WG_SIZE) { diff --git a/ggml/src/ggml-opencl/kernels/flash_attn_repack.cl b/ggml/src/ggml-opencl/kernels/flash_attn_repack.cl new file mode 100644 index 00000000..db78d563 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/flash_attn_repack.cl @@ -0,0 +1,92 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +__kernel void kernel_repack_mask_for_wmm( + const global half* mask_buf, + const ulong mask_nb1, + const ulong mask_nb2, + const ulong mask_nb3, + const int mask_ne2, + global half* mask_buf_padded, + const ulong mask_nb1_padded, + const ulong mask_nb2_padded, + const ulong mask_nb3_padded +) { + int col = get_global_id(0); // 0 .. n_kv + int row = get_global_id(1); // 0 .. n_q + int slice = get_global_id(2); // 0 .. (n_head * n_batch) + + int head_idx = slice % mask_ne2; + int batch_idx = slice / mask_ne2; + + ulong src_off = (ulong)batch_idx * mask_nb3 + (ulong)head_idx * mask_nb2 + (ulong)row * mask_nb1; + ulong dst_off = (ulong)batch_idx * mask_nb3_padded + (ulong)head_idx * mask_nb2_padded + (ulong)row * mask_nb1_padded; + + mask_buf_padded[dst_off / 2 + col] = mask_buf[src_off / 2 + col]; +} + +__kernel void kernel_repack_q_for_wmm( + const global float* q_buf, + const ulong q_nb1, + const ulong q_nb2, + const ulong q_nb3, + const int n_head, + __write_only image3d_t img_q_wmm +) { + int k4 = get_global_id(0); + int row = get_global_id(1); + int slice = get_global_id(2); + int batch_idx = slice / n_head; + int head_idx = slice % n_head; + + + ulong elem_off = (batch_idx * q_nb3 + head_idx * q_nb2 + row * q_nb1) / 4 + (ulong)k4 * 4; + float4 v = vload4(elem_off / 4, q_buf); + + write_imageh(img_q_wmm, (int4)(row, slice, k4, 0), convert_half4(v)); +} + +__kernel void kernel_repack_k_for_wmm( + const global half* k_buf, + const ulong k_nb1, + const ulong k_nb2, + const ulong k_nb3, + const int n_head_kv, + const int n_kv, + __write_only image3d_t img_k_wmm +) { + int kk = get_global_id(0); + int row4 = get_global_id(1); + int slice = get_global_id(2); + int batch_idx = slice / n_head_kv; + int head_kv_idx = slice % n_head_kv; + + ulong base = batch_idx * k_nb3 + head_kv_idx * k_nb2; + int row0 = row4 * 4; + half4 v; + v.x = (row0 + 0 < n_kv) ? k_buf[(base + (ulong)(row0 + 0) * k_nb1) / 2 + kk] : (half)0; + v.y = (row0 + 1 < n_kv) ? k_buf[(base + (ulong)(row0 + 1) * k_nb1) / 2 + kk] : (half)0; + v.z = (row0 + 2 < n_kv) ? k_buf[(base + (ulong)(row0 + 2) * k_nb1) / 2 + kk] : (half)0; + v.w = (row0 + 3 < n_kv) ? k_buf[(base + (ulong)(row0 + 3) * k_nb1) / 2 + kk] : (half)0; + + write_imageh(img_k_wmm, (int4)(kk, row4, slice, 0), v); +} + +__kernel void kernel_repack_v_for_wmm( + const global half* v_buf, + const ulong v_nb1, + const ulong v_nb2, + const ulong v_nb3, + const int n_head_kv, + __write_only image3d_t img_v_wmm +) { + int hdim4 = get_global_id(0); // now fastest — walks contiguous memory + int row = get_global_id(1); + int slice = get_global_id(2); + int batch_idx = slice / n_head_kv; + int head_kv_idx = slice % n_head_kv; + + ulong row_off = batch_idx * v_nb3 + head_kv_idx * v_nb2 + (ulong)row * v_nb1; + half4 v = vload4((row_off / 2 + (ulong)hdim4 * 4) / 4, v_buf); + + write_imageh(img_v_wmm, (int4)(row, hdim4, slice, 0), v); +} diff --git a/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl index 22b4e911..c379a9a3 100644 --- a/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl @@ -4,6 +4,7 @@ #pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable #define ADRENO_GPU 1 #define REQD_SUBGROUP_SIZE_128 __attribute__((qcom_reqd_sub_group_size("full"))) +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) #endif #define QK_K 256 #define K_SCALE_SIZE 12 @@ -171,3 +172,319 @@ kernel void kernel_gemm_noshuffle_q4_k_f32( vstore4((float4)(c0.s7, c1.s7, c2.s7, c3.s7), 0, dst + idx); } } + +// 1x8 per-WI tile (1 output row x 8 output cols). For the small-batch +// (medium n_q, e.g. MTP/spec verify) path where the 2x8 kernel is starved: +// at ne1<=8 the grid is (1, ceil(M/2)) -> only ~M/256 workgroups, leaving +// the SP under-occupied. 1 row per WI doubles the M-axis workgroup count +// (ceil(M/1)/128 vs ceil(M/2)/128) AND collapses the accumulators to a +// single half8 (16 regs, no spill), so more waves co-reside. Same weight +// traffic as 2x8 (rows never share weights); the win is pure occupancy. +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_128 +#endif +kernel void kernel_gemm_noshuffle_q4_k_f32_r1( + global const ushort * src0_q, + global const uchar * src0_s, + global const half * src0_d, + global const half * src0_dm, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int m, + int n, + int k, + int n_no_padding, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2 +) { + dst = (global float *)((global char *)dst + offsetd); + int n_4 = n >> 2; + int gy = get_global_id(0); + int gx = get_global_id(1); // 1 row per WI + + half8 c0 = 0; + half8 B; + half dq; + + int num_blocks_K = k / QK_K; + + global const ushort * weight_ptr = src0_q + gx; + global const half * d_ptr = src0_d + gx; + global const half * dm_ptr = src0_dm + gx; + + for (int i = 0; i < k; i += 32) { + int sb_idx = i / QK_K; + int sub_idx = (i / 32) % 8; + + half dd = d_ptr [sb_idx * m]; + half dmm = dm_ptr[sb_idx * m]; + + global const uchar * sc0 = src0_s + sb_idx * K_SCALE_SIZE * m + gx; + + uchar sv0, mn0; + get_scale_min_k4(sub_idx, sc0, m, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + + half scale = convert_half(convert_float(dd) * (float)sv0); + half mval = convert_half(convert_float(dmm) * (float)mn0); + + for (int l = 0; l < 32; l += 4) { + int ki = i + l; + ushort bits = weight_ptr[(ki/4) * m]; + + B.s0123 = read_imageh(src1, gy*2 + (ki+0) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+0) * n_4); + dq = (bits & 0x000F) * scale - mval; + c0 += B * dq; + + B.s0123 = read_imageh(src1, gy*2 + (ki+1) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+1) * n_4); + dq = ((bits & 0x00F0) >> 4) * scale - mval; + c0 += B * dq; + + B.s0123 = read_imageh(src1, gy*2 + (ki+2) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+2) * n_4); + dq = ((bits & 0x0F00) >> 8) * scale - mval; + c0 += B * dq; + + B.s0123 = read_imageh(src1, gy*2 + (ki+3) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+3) * n_4); + dq = ((bits & 0xF000) >> 12) * scale - mval; + c0 += B * dq; + } + } + + // Output: 8 cols, 1 row per col-step. Scalar store, coalesced across + // neighbouring WIs (consecutive gx -> consecutive dst addresses). + int idx = (gy<<3)*m + gx; + if (idx < m*n_no_padding) { dst[idx] = c0.s0; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s1; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s2; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s3; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s4; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s5; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s6; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = c0.s7; } +} + +// 2x8 tile, but weights read through an image1d_buffer (CL_R/UINT32 over the +// same packed-q buffer) instead of a plain global buffer. The ne1==1 GEMV +// already does this and is much faster per weight byte than this GEMM at +// small n_q; the structural difference is the image path hits the dedicated +// TPL1 weight cache (L1) while the global path only reaches L2. At small n_q +// the forward is weight-read-bound, so L1-cached weights is the lever. +// The 2 adjacent rows the 2x8 tile reads as a ushort2 are exactly one uint32, +// so the vload2 becomes a single read_imageui at index gx + (ki/4)*(m/2). +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_128 +#endif +kernel void kernel_gemm_noshuffle_q4_k_f32_kimg( + read_only image1d_buffer_t src0_q_img, + global const uchar * src0_s, + global const half * src0_d, + global const half * src0_dm, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int m, + int n, + int k, + int n_no_padding, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2 +) { + dst = (global float *)((global char *)dst + offsetd); + int n_4 = n >> 2; + int m_2 = m >> 1; + int gy = get_global_id(0); + int gx = get_global_id(1); + int gx_2 = gx << 1; + + half8 c0 = 0, c1 = 0; + half8 B; + half2 dequantized_weights; + + int num_blocks_K = k / QK_K; + + global const half * d_ptr = src0_d + gx_2; + global const half * dm_ptr = src0_dm + gx_2; + + for (int i = 0; i < k; i += 32) { + int sb_idx = i / QK_K; + int sub_idx = (i / 32) % 8; + + half2 d = vload2(0, d_ptr + sb_idx * m); + half2 dm = vload2(0, dm_ptr + sb_idx * m); + + global const uchar * sc0 = src0_s + sb_idx * K_SCALE_SIZE * m + (gx_2+0); + global const uchar * sc1 = sc0 + 1; + + uchar sv0, mn0, sv1, mn1; + get_scale_min_k4(sub_idx, sc0, m, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(sub_idx, sc1, m, &sv1, &mn1, mask_d6, mask_d4, mask_hi2); + + half2 scale = convert_half2(convert_float2(d) * convert_float2((uchar2)(sv0, sv1))); + half2 mval = convert_half2(convert_float2(dm) * convert_float2((uchar2)(mn0, mn1))); + + for (int l = 0; l < 32; l += 4) { + int ki = i + l; + uint wpacked = read_imageui(src0_q_img, gx + (ki/4) * m_2).x; + ushort2 bits2 = (ushort2)((ushort)(wpacked & 0xFFFFu), (ushort)(wpacked >> 16)); + + // j=0 + B.s0123 = read_imageh(src1, gy*2 + (ki+0) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+0) * n_4); + dequantized_weights.s0 = (bits2.s0 & 0x000F) * scale.s0 - mval.s0; + dequantized_weights.s1 = (bits2.s1 & 0x000F) * scale.s1 - mval.s1; + c0 += B * dequantized_weights.s0; + c1 += B * dequantized_weights.s1; + + // j=1 + B.s0123 = read_imageh(src1, gy*2 + (ki+1) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+1) * n_4); + dequantized_weights.s0 = ((bits2.s0 & 0x00F0) >> 4) * scale.s0 - mval.s0; + dequantized_weights.s1 = ((bits2.s1 & 0x00F0) >> 4) * scale.s1 - mval.s1; + c0 += B * dequantized_weights.s0; + c1 += B * dequantized_weights.s1; + + // j=2 + B.s0123 = read_imageh(src1, gy*2 + (ki+2) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+2) * n_4); + dequantized_weights.s0 = ((bits2.s0 & 0x0F00) >> 8) * scale.s0 - mval.s0; + dequantized_weights.s1 = ((bits2.s1 & 0x0F00) >> 8) * scale.s1 - mval.s1; + c0 += B * dequantized_weights.s0; + c1 += B * dequantized_weights.s1; + + // j=3 + B.s0123 = read_imageh(src1, gy*2 + (ki+3) * n_4); + B.s4567 = read_imageh(src1, gy*2+1 + (ki+3) * n_4); + dequantized_weights.s0 = ((bits2.s0 & 0xF000) >> 12) * scale.s0 - mval.s0; + dequantized_weights.s1 = ((bits2.s1 & 0xF000) >> 12) * scale.s1 - mval.s1; + c0 += B * dequantized_weights.s0; + c1 += B * dequantized_weights.s1; + } + } + + int idx = (gy<<3)*m + (gx<<1); + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s0, c1.s0), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s1, c1.s1), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s2, c1.s2), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s3, c1.s3), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s4, c1.s4), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s5, c1.s5), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s6, c1.s6), 0, dst + idx); idx += m; } + if (idx+1 < m*n_no_padding) { vstore2((float2)(c0.s7, c1.s7), 0, dst + idx); } +} + +// Cooperative-K GEMM for the small-batch (n_q in [2..8]) path. Mirrors the +// ne1==1 GEMV's structure: a WG is (COK_SG lanes x COK_NSG subgroups); each +// lane owns ONE output row and computes its 8 (padded) columns, and the +// COK_NSG subgroups SPLIT the K reduction round-robin, combining via a +// __local reduction. This is the thing the per-WI GEMM lacked — at small n_q +// the old kernel had ~M/256 workgroups each walking all of K serially; this +// has M/64 workgroups AND COK_NSG-way K parallelism. Uses REQD_SUBGROUP_SIZE_64 +// + barrier (same safe reduction pattern as the GEMV; never sub_group_reduce +// at full width on X2 per the GDN miscompile note). +#define COK_NSG 8 +#define COK_SG 64 +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemm_noshuffle_q4_k_f32_cok( + global const ushort * src0_q, + global const uchar * src0_s, + global const half * src0_d, + global const half * src0_dm, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int m, + int n, + int k, + int n_no_padding, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2 +) { + dst = (global float *)((global char *)dst + offsetd); + int n_4 = n >> 2; + int gx = get_global_id(0); // output row + int sg = get_local_id(1); // subgroup index (K-split lane) + int lane = get_local_id(0); // lane within subgroup (0..COK_SG-1) + + int num_blocks_K = k / QK_K; + int num_32blk = k / 32; + + global const ushort * weight_ptr = src0_q + gx; + global const half * d_ptr = src0_d + gx; + global const half * dm_ptr = src0_dm + gx; + + half8 acc = 0; + half8 B; + half dq; + + for (int blk = sg; blk < num_32blk; blk += COK_NSG) { + int i = blk << 5; // blk * 32 + int sb_idx = blk >> 3; // (blk*32) / QK_K (QK_K = 256 = 32*8) + int sub_idx = blk & 7; // (i/32) % 8 + + half dd = d_ptr [sb_idx * m]; + half dmm = dm_ptr[sb_idx * m]; + + global const uchar * sc0 = src0_s + sb_idx * K_SCALE_SIZE * m + gx; + uchar sv0, mn0; + get_scale_min_k4(sub_idx, sc0, m, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + half scale = convert_half(convert_float(dd) * (float)sv0); + half mval = convert_half(convert_float(dmm) * (float)mn0); + + for (int l = 0; l < 32; l += 4) { + int ki = i + l; + ushort bits = weight_ptr[(ki>>2) * m]; + + B.s0123 = read_imageh(src1, (ki+0) * n_4); + B.s4567 = read_imageh(src1, 1 + (ki+0) * n_4); + dq = (bits & 0x000F) * scale - mval; + acc += B * dq; + + B.s0123 = read_imageh(src1, (ki+1) * n_4); + B.s4567 = read_imageh(src1, 1 + (ki+1) * n_4); + dq = ((bits & 0x00F0) >> 4) * scale - mval; + acc += B * dq; + + B.s0123 = read_imageh(src1, (ki+2) * n_4); + B.s4567 = read_imageh(src1, 1 + (ki+2) * n_4); + dq = ((bits & 0x0F00) >> 8) * scale - mval; + acc += B * dq; + + B.s0123 = read_imageh(src1, (ki+3) * n_4); + B.s4567 = read_imageh(src1, 1 + (ki+3) * n_4); + dq = ((bits & 0xF000) >> 12) * scale - mval; + acc += B * dq; + } + } + + // cross-subgroup reduction over the K-split (float for accuracy) + local float8 reduceLM[COK_SG * (COK_NSG - 1)]; + if (sg > 0) { + reduceLM[(sg - 1) * COK_SG + lane] = convert_float8(acc); + } + barrier(CLK_LOCAL_MEM_FENCE); + + if (sg == 0) { + float8 sum = convert_float8(acc); + for (int s = 0; s < COK_NSG - 1; s++) { + sum += reduceLM[s * COK_SG + lane]; + } + int idx = gx; + if (idx < m*n_no_padding) { dst[idx] = sum.s0; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s1; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s2; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s3; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s4; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s5; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s6; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s7; } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl index 3a9c6245..141f6a2f 100644 --- a/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl @@ -5,6 +5,7 @@ #pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable #define ADRENO_GPU 1 #define REQD_SUBGROUP_SIZE_128 __attribute__((qcom_reqd_sub_group_size("full"))) +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) #endif #ifdef ADRENO_GPU @@ -138,3 +139,107 @@ kernel void kernel_gemm_noshuffle_q6_K_f32( vstore4((float4)(c0.s7, c1.s7, c2.s7, c3.s7), 0, dst + idx); } } + +// Cooperative-K q6_K GEMM for the small-batch (n_q in [2..8]) path. Same idea +// as the q4_K _cok kernel: WG = (COK_SG lanes x COK_NSG subgroups), each lane +// owns ONE output row (half8 over the 8 padded cols), and the COK_NSG +// subgroups split the K iterations round-robin and combine via a __local +// reduction. Replaces the default 4-row-per-WI tile that walked all of K alone +// (~M/512 WGs + serial reduction) at small n_q. REQD_SUBGROUP_SIZE_64 + +// barrier (never sub_group_reduce at full width on X2). +#define COK_NSG 8 +#define COK_SG 64 +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemm_noshuffle_q6_K_f32_cok( + global const ushort * src0_ql, + global const uchar * src0_qh, + global const ushort * src0_s, + global const half * src0_d, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int m, + int n, + int k, + int n_no_padding, + ushort mask_f000, + uchar mask_c0 +) { + dst = (global float *)( (global char *)dst + offsetd ); + + int n_4 = n >> 2; + int gx = get_global_id(0); // output row + int sg = get_local_id(1); // subgroup index (K-split) + int lane = get_local_id(0); // lane within subgroup + + global const ushort * ptr_ql = src0_ql + gx; + global const uchar * ptr_qh = src0_qh + gx; + global const ushort * ptr_s = src0_s + gx; + global const half * ptr_d = src0_d + gx; + + half8 acc = 0; + half8 B; + half dq; + + int num_iter = k >> 2; // k/4 iterations, 4 k-values each + + for (int ib = sg; ib < num_iter; ib += COK_NSG) { + int i = ib << 2; // ib * 4 + + ushort bits4 = ptr_ql[ib * m]; // ql for row gx at this 4-block + uchar bits2 = ptr_qh[ib * m]; // qh + + ushort s_packed = ptr_s[(i >> 5) * m]; // (i/16/2) = i/32 + char2 sc2 = as_char2(s_packed); + char scale_s = (((i >> 4) & 1) == 0) ? sc2.s0 : sc2.s1; // (i/16)%2 + half scale_d = ptr_d[(i >> 8) * m]; // i/256 + + // j=0 + B.s0123 = read_imageh(src1, (i + 0)*n_4 + 0); + B.s4567 = read_imageh(src1, (i + 0)*n_4 + 1); + dq = (convert_half((bits4 & 0x000F) | ((bits2 & 0x03) << 4)) - 32.f) * scale_s * scale_d; + acc += B * dq; + + // j=1 + B.s0123 = read_imageh(src1, (i + 1)*n_4 + 0); + B.s4567 = read_imageh(src1, (i + 1)*n_4 + 1); + dq = (convert_half(((bits4 & 0x00F0) >> 4) | ((bits2 & 0x0C) << 2)) - 32.f) * scale_s * scale_d; + acc += B * dq; + + // j=2 + B.s0123 = read_imageh(src1, (i + 2)*n_4 + 0); + B.s4567 = read_imageh(src1, (i + 2)*n_4 + 1); + dq = (convert_half(((bits4 & 0x0F00) >> 8) | (bits2 & 0x30)) - 32.f) * scale_s * scale_d; + acc += B * dq; + + // j=3 + B.s0123 = read_imageh(src1, (i + 3)*n_4 + 0); + B.s4567 = read_imageh(src1, (i + 3)*n_4 + 1); + dq = (convert_half(((bits4 & mask_f000) >> 12) | ((bits2 & mask_c0) >> 2)) - 32.f) * scale_s * scale_d; + acc += B * dq; + } + + local float8 reduceLM[COK_SG * (COK_NSG - 1)]; + if (sg > 0) { + reduceLM[(sg - 1) * COK_SG + lane] = convert_float8(acc); + } + barrier(CLK_LOCAL_MEM_FENCE); + + if (sg == 0) { + float8 sum = convert_float8(acc); + for (int s = 0; s < COK_NSG - 1; s++) { + sum += reduceLM[s * COK_SG + lane]; + } + int idx = gx; + if (idx < m*n_no_padding) { dst[idx] = sum.s0; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s1; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s2; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s3; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s4; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s5; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s6; idx += m; } + if (idx < m*n_no_padding) { dst[idx] = sum.s7; } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32_tiled.cl b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32_tiled.cl new file mode 100644 index 00000000..ffd943a2 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32_tiled.cl @@ -0,0 +1,136 @@ +// Batched (N>1) q6_K GEMM over the 64-row-TILED canonical layout produced by +// kernel_convert_block_q6_k_tiled_ns (cvt.cl). Companion to the decode kernel +// kernel_gemv_noshuffle_q6_K_f32_tiled: SAME pack, SAME canonical e-order +// dequant (correct by construction vs reference ggml q6_K), extended to N output +// columns. Makes the batched lm_head/embed (perplexity, spec-decode verify, +// batched serving) correct on GPU while keeping the tiled convert the fast decode +// GEMV depends on. +// +// One work-item owns one output ROW for a block of BN columns. A work-group is +// {64 lanes, NTILES subgroups} = NTILES*64 rows; the global z dimension tiles the +// N columns by BN. Each work-item computes its row's FULL K (no K-split, so no +// cross-subgroup reduction), which lets the whole work-group share one staged +// activation block: +// +// __local activation staging — the BN columns of the current superblock (BN*256 +// floats) are loaded into __local once per superblock, cooperatively by all +// NTILES*64 work-items, then every row reads its activation from __local. This +// removes the ~Nrows-fold redundant image reads of the first version (each lane +// re-read the activation), which made the batched GEMM ~2x slower than the plain +// noshuffle GEMM. +// +// Weights are read from __global (coalesced) — matching the decode kernel; the +// lm_head weight is streamed with little reuse where coalesced global beats the +// Adreno texture cache. + +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +#ifdef cl_qcom_reqd_sub_group_size +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#endif + +#define NTILES 4 // 64-row tiles per work-group (NTILES*64 = 256 rows) +#define TILE_ROWS 64 +#define BN 16 // output columns handled per work-group (global z step) +#define WG_THREADS (NTILES * TILE_ROWS) + +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemm_noshuffle_q6_K_f32_tiled( + __global uint4 * src0_ql, // tiled: 8 uint4 granules / superblock + __global uint4 * src0_qh, // tiled: 4 uint4 granules / superblock + __global char * src0_s, // tiled: 16 chars / superblock + __global half * src0_d, // tiled: 1 half / superblock + read_only image1d_buffer_t src1, // activation [ne00, ne11] f32 (RGBA), column-major + global float * dst, + ulong offsetd, + int ne00, + int ne01, + int ne11 +) { + int rit = get_local_id(0); // 0..63 (lane within a tile; coalesces weight loads) + int sg = get_local_id(1); // 0..NTILES-1 + int lid = sg * TILE_ROWS + rit; // 0..WG_THREADS-1 (flat local id) + int row = get_group_id(0) * WG_THREADS + lid; + int rt = row / TILE_ROWS; // global 64-row tile index + int col0 = get_global_id(2) * BN; // first output column of this block + + int nb = ne00 / 256; // superblocks per row + int act_col_stride = ne00 / 4; // activation float4 pixels per column + + const bool row_ok = row < ne01; + + // staged activation: BN columns x 256 elements for the current superblock + __local float lact[BN * 256]; + + float acc[BN]; + #pragma unroll + for (int j = 0; j < BN; ++j) acc[j] = 0.0f; + + for (int sb = 0; sb < nb; ++sb) { + // cooperatively stage BN columns' 256 activation elements (= BN*64 float4) + for (int p = lid; p < BN * 64; p += WG_THREADS) { + int j = p >> 6; // column within the BN block (p / 64) + int e4 = p & 63; // element-quad within the column (p % 64) + int c = col0 + j; + float4 v = (c < ne11) + ? read_imagef(src1, c * act_col_stride + sb * 64 + e4) + : (float4)(0.0f); + lact[p * 4 + 0] = v.x; + lact[p * 4 + 1] = v.y; + lact[p * 4 + 2] = v.z; + lact[p * 4 + 3] = v.w; // lact[j*256 + e], e = e4*4 + t + } + barrier(CLK_LOCAL_MEM_FENCE); + + if (row_ok) { + int tile_blk = rt * nb + sb; // ne02 == 1 for lm_head/embed + + float dval = (float)src0_d[tile_blk * TILE_ROWS + rit]; + __global char * sc = src0_s + (tile_blk * TILE_ROWS + rit) * 16; + + uint ql[32]; + uint qh[16]; + #pragma unroll + for (int g = 0; g < 8; ++g) { + uint4 v = src0_ql[(tile_blk * 8 + g) * TILE_ROWS + rit]; + ql[g*4+0] = v.x; ql[g*4+1] = v.y; ql[g*4+2] = v.z; ql[g*4+3] = v.w; + } + #pragma unroll + for (int g = 0; g < 4; ++g) { + uint4 v = src0_qh[(tile_blk * 4 + g) * TILE_ROWS + rit]; + qh[g*4+0] = v.x; qh[g*4+1] = v.y; qh[g*4+2] = v.z; qh[g*4+3] = v.w; + } + + // NOTE: the e loop (256) is deliberately NOT unrolled. Fully unrolling + // 256*BN MACs overflows the in-process Adreno compiler (host stack + // overflow at clBuildProgram, same class as the FA DK=512 OOM). + for (int e = 0; e < 256; ++e) { + uint low4 = (ql[e >> 3] >> ((e & 7) * 4)) & 0xF; + uint hi2 = (qh[e >> 4] >> ((e & 15) * 2)) & 0x3; + int code = (int)(low4 | (hi2 << 4)) - 32; + int sidx = ((e >> 7) << 3) + (((e >> 5) & 3) << 1) + ((e >> 4) & 1); + float cs = (float)code * (float)sc[sidx] * dval; + #pragma unroll + for (int j = 0; j < BN; ++j) { + acc[j] += cs * lact[j * 256 + e]; + } + } + } + barrier(CLK_LOCAL_MEM_FENCE); + } + + if (row_ok) { + dst = (global float*)((global char*)dst + offsetd); + #pragma unroll + for (int j = 0; j < BN; ++j) { + int c = col0 + j; + if (c < ne11) { + dst[(ulong)c * ne01 + row] = acc[j]; + } + } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl index 8de0de1c..023e848f 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32.cl @@ -277,3 +277,107 @@ __kernel void kernel_gemv_noshuffle_q4_0_f32( } } + +// Multi-column (N in [2..4]) variant of the q4_0 decode GEMV, for the speculative +// / MTP verify batch (n_cols = 2..4 = drafted + bonus positions). Routes the small- +// batch verify OFF the transposed-GEMM dead-zone (gemm_noshuffle_q4_0) onto the +// efficient GEMV path. Each K-block's weights (regA hi+lo) are loaded ONCE and +// reused across the n_cols activation columns. Per-column accumulation is +// independent and identical to n_cols standalone GEMVs. n_cols==3 is byte-identical +// to the original mc3 (col3 disabled, slots 6/7 stay zero). Kept the _mc3 name. +#ifdef VECTOR_SUB_GROUP_BROADCAST +#define MC_DQ_HI dequantizeBlockAccum_ns_sgbroadcast_8_hi +#define MC_DQ_LO dequantizeBlockAccum_ns_sgbroadcast_8_lo +#else +#define MC_DQ_HI dequantizeBlockAccum_ns_sgbroadcast_1_hi +#define MC_DQ_LO dequantizeBlockAccum_ns_sgbroadcast_1_lo +#endif +// One column c: load this column's activation (own brace scope so the macros' +// `shared_y` decl is re-scoped), then dequant (hi+lo) against the shared weights. +#define MC_COL_Q40(ts, c) \ + { if (slid < 4) { regB.s0123 = read_imagef(src1, (c)*COL_STRIDE + slid*2 + k*8); \ + regB.s4567 = read_imagef(src1, (c)*COL_STRIDE + 1 + slid*2 + k*8); } \ + MC_DQ_HI(ts, as_ushort8(regA_hi), regS, regB); \ + MC_DQ_LO(ts, as_ushort8(regA_lo), regS, regB); } + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +__kernel void kernel_gemv_noshuffle_q4_0_f32_mc3( + __read_only image1d_buffer_t src0_q, // quantized A + global half2 * src0_d, // A scales + __read_only image1d_buffer_t src1, // B (n_cols columns, col-major image) + global float * dst, // C (column-major [M x n_cols]) + ulong offsetd, + int ne00, // K + int ne01, // M + int n_cols) // N (2..4) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + // BLOCK_STRIDE_A is the LAYOUT stride between consecutive K-blocks = 4 uints + // per q4_0 block * M (set by the trans4_ns convert). The "4" is uints/block, NOT + // the subgroup count — keep it fixed so the K-split count (nsg) can vary. + uint BLOCK_STRIDE_A = N_SIMDGROUP * M; // = 4 * M (N_SIMDGROUP is the #define 4) + uint COL_STRIDE = K / 4; // float4 pixels per activation column + uint nsg = get_local_size(1); // runtime K-split (4 default, 8 small-M) + + __private uint4 regA_hi, regA_lo; + __private half2 regS; + __private float8 regB; + + __private float2 ts0 = (float2)(0.0f); + __private float2 ts1 = (float2)(0.0f); + __private float2 ts2 = (float2)(0.0f); + __private float2 ts3 = (float2)(0.0f); + + for (uint k = groupId; k < (K / QK4_0); k += nsg) { + regS = src0_d[gid + k * LINE_STRIDE_A]; + + // weights loaded ONCE, reused across the columns + regA_hi.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA_hi.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA_hi.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA_hi.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA_lo.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA_lo.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA_lo.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA_lo.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + + MC_COL_Q40(ts0, 0); + MC_COL_Q40(ts1, 1); + if (n_cols > 2) MC_COL_Q40(ts2, 2); + if (n_cols > 3) MC_COL_Q40(ts3, 3); + } + + // cross-subgroup reduce over nsg subgroups: pack the (up to 4) columns' float2 + // into a float8. Generalized to runtime nsg (4 default, 8 for small-M). Each + // subgroup writes its partial; subgroup 0 sums the rest into its own acc. At + // nsg==4 this is byte-identical to the original (sums subgroups 1,2,3 in order). + __local float8 reduceLM[SIMDGROUP_WIDTH * 8]; + float8 acc = (float8)(ts0.s0, ts0.s1, ts1.s0, ts1.s1, ts2.s0, ts2.s1, ts3.s0, ts3.s1); + reduceLM[groupId * SIMDGROUP_WIDTH + slid] = acc; + + barrier(CLK_LOCAL_MEM_FENCE); + + if (groupId == 0) { + for (uint g = 1; g < nsg; g++) { + acc += reduceLM[g * SIMDGROUP_WIDTH + slid]; + } + dst = (global float*)((global char*)dst + offsetd); + // dst is column-major [M rows x n_cols cols]: (row, col) at col*M + row + vstore2((float2)(acc.s0, acc.s1), 0, &(dst[0 * M + gid * 2])); + vstore2((float2)(acc.s2, acc.s3), 0, &(dst[1 * M + gid * 2])); + if (n_cols > 2) vstore2((float2)(acc.s4, acc.s5), 0, &(dst[2 * M + gid * 2])); + if (n_cols > 3) vstore2((float2)(acc.s6, acc.s7), 0, &(dst[3 * M + gid * 2])); + } +} +#undef MC_COL_Q40 +#undef MC_DQ_HI +#undef MC_DQ_LO diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32_32b_trans.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32_32b_trans.cl new file mode 100644 index 00000000..565285f4 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_0_f32_32b_trans.cl @@ -0,0 +1,137 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_khr_subgroups : enable + +#ifdef cl_qcom_reqd_sub_group_size +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#endif + +#define QK4_0 32 +#define N_SIMDGROUP 4 + +#define dequantizeBlockAccum_ila_1row_hi(total_sum, bits4, scale, y) \ + float shared_y; \ + shared_y = sub_group_broadcast(y.s0, 0); \ + total_sum += ((bits4.s0 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 0); \ + total_sum += (((bits4.s0 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 0); \ + total_sum += (((bits4.s0 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 0); \ + total_sum += (((bits4.s0 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 0); \ + total_sum += ((bits4.s1 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 0); \ + total_sum += (((bits4.s1 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 0); \ + total_sum += (((bits4.s1 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 0); \ + total_sum += (((bits4.s1 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 1); \ + total_sum += ((bits4.s2 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 1); \ + total_sum += (((bits4.s2 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 1); \ + total_sum += (((bits4.s2 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 1); \ + total_sum += (((bits4.s2 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 1); \ + total_sum += ((bits4.s3 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 1); \ + total_sum += (((bits4.s3 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 1); \ + total_sum += (((bits4.s3 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 1); \ + total_sum += (((bits4.s3 & 0xF000) >> 12) - 8) * scale * shared_y; + +#define dequantizeBlockAccum_ila_1row_lo(total_sum, bits4, scale, y) \ + shared_y = sub_group_broadcast(y.s0, 2); \ + total_sum += ((bits4.s4 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 2); \ + total_sum += (((bits4.s4 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 2); \ + total_sum += (((bits4.s4 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 2); \ + total_sum += (((bits4.s4 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 2); \ + total_sum += ((bits4.s5 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 2); \ + total_sum += (((bits4.s5 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 2); \ + total_sum += (((bits4.s5 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 2); \ + total_sum += (((bits4.s5 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 3); \ + total_sum += ((bits4.s6 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 3); \ + total_sum += (((bits4.s6 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 3); \ + total_sum += (((bits4.s6 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 3); \ + total_sum += (((bits4.s6 & 0xF000) >> 12) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 3); \ + total_sum += ((bits4.s7 & 0x000F) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 3); \ + total_sum += (((bits4.s7 & 0x00F0) >> 4) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 3); \ + total_sum += (((bits4.s7 & 0x0F00) >> 8) - 8) * scale * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 3); \ + total_sum += (((bits4.s7 & 0xF000) >> 12) - 8) * scale * shared_y; + + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +__kernel void kernel_gemv_noshuffle_q4_0_f32_32b_trans( + __read_only image1d_buffer_t src0_q, + global half * src0_d, + __read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + uint K = ne00; + uint M = ne01; + + __private uint4 regA; + __private half regS; + __private float8 regB; + __private float totalSum = 0.0f; + + for (uint k = groupId; k < (K / QK4_0); k += N_SIMDGROUP) { + regS = src0_d[k * M + gid]; + if (slid < 4) { + regB.s0123 = read_imagef(src1, (slid * 2 + k * 8)); + regB.s4567 = read_imagef(src1, (1 + slid * 2 + k * 8)); + } + regA.s0 = read_imageui(src0_q, ((k * 4 + 0) * M + gid)).x; + regA.s1 = read_imageui(src0_q, ((k * 4 + 1) * M + gid)).x; + regA.s2 = read_imageui(src0_q, ((k * 4 + 2) * M + gid)).x; + regA.s3 = read_imageui(src0_q, ((k * 4 + 3) * M + gid)).x; + + dequantizeBlockAccum_ila_1row_hi(totalSum, as_ushort8(regA), regS, regB); + dequantizeBlockAccum_ila_1row_lo(totalSum, as_ushort8(regA), regS, regB); + } + + __local float reduceLM[SIMDGROUP_WIDTH * 3]; + if (groupId == 1) reduceLM[SIMDGROUP_WIDTH * 0 + slid] = totalSum; + if (groupId == 2) reduceLM[SIMDGROUP_WIDTH * 1 + slid] = totalSum; + if (groupId == 3) reduceLM[SIMDGROUP_WIDTH * 2 + slid] = totalSum; + barrier(CLK_LOCAL_MEM_FENCE); + if (groupId == 0) totalSum += reduceLM[SIMDGROUP_WIDTH * 0 + slid]; + if (groupId == 0) totalSum += reduceLM[SIMDGROUP_WIDTH * 1 + slid]; + if (groupId == 0) totalSum += reduceLM[SIMDGROUP_WIDTH * 2 + slid]; + + if (groupId == 0) { + dst = (global float*)((global char*)dst + offsetd); + if (gid < M) { + dst[gid] = totalSum; + } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_1_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_1_f32.cl index 5fa31278..2ccf4214 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_1_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_1_f32.cl @@ -286,3 +286,99 @@ kernel void kernel_gemv_noshuffle_q4_1_f32( } } + +// Multi-column (N in [2..4]) variant of the q4_1 decode GEMV (spec/MTP verify) = +// q4_0 mc3 + the q4_1 per-block min (regM; dequant = q*scale + minv). n_cols=2..4; +// routes the small-batch verify OFF the gemm_noshuffle_q4_1 dead-zone. n_cols==3 is +// byte-identical to the original mc3. NB: this file spells the vec-broadcast define +// BROADCAT (no S) — match it so the fast _8 path compiles. +#ifdef VECTOR_SUB_GROUP_BROADCAT +#define MC_DQ1_HI dequantizeBlockAccum_ns_sgbroadcast_8_hi +#define MC_DQ1_LO dequantizeBlockAccum_ns_sgbroadcast_8_lo +#else +#define MC_DQ1_HI dequantizeBlockAccum_ns_sgbroadcast_1_hi +#define MC_DQ1_LO dequantizeBlockAccum_ns_sgbroadcast_1_lo +#endif +#define MC_COL_Q41(ts, c) \ + { if (slid < 4) { regB.s0123 = read_imagef(src1, (c)*COL_STRIDE + slid*2 + k*8); \ + regB.s4567 = read_imagef(src1, (c)*COL_STRIDE + 1 + slid*2 + k*8); } \ + MC_DQ1_HI(ts, as_ushort8(regA_hi), regS, regM, regB); \ + MC_DQ1_LO(ts, as_ushort8(regA_lo), regS, regM, regB); } +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_1_f32_mc3( + read_only image1d_buffer_t src0_q, + global half2 * src0_d, + global half2 * src0_m, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + int n_cols) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = NSUBGROUPS * M; + uint COL_STRIDE = K / 4; // float4 pixels per activation column + + private uint4 regA_hi, regA_lo; + private half2 regS, regM; + private float8 regB; + + private float2 ts0 = (float2)(0.0f); + private float2 ts1 = (float2)(0.0f); + private float2 ts2 = (float2)(0.0f); + private float2 ts3 = (float2)(0.0f); + + for (uint k = groupId; k < (K / QK4_0); k += NSUBGROUPS) { + regS = src0_d[gid + k * LINE_STRIDE_A]; + regM = src0_m[gid + k * LINE_STRIDE_A]; + + // weights loaded ONCE, reused across the columns + regA_hi.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA_hi.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA_hi.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA_hi.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA_lo.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA_lo.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA_lo.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA_lo.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + + MC_COL_Q41(ts0, 0); + MC_COL_Q41(ts1, 1); + if (n_cols > 2) MC_COL_Q41(ts2, 2); + if (n_cols > 3) MC_COL_Q41(ts3, 3); + } + + // cross-subgroup reduce: pack the (up to 4) columns' float2 into a float8. + local float8 reduceLM[SUBGROUP_SIZE * 3]; + float8 acc = (float8)(ts0.s0, ts0.s1, ts1.s0, ts1.s1, ts2.s0, ts2.s1, ts3.s0, ts3.s1); + if (groupId == 1) { reduceLM[SUBGROUP_SIZE * 0 + slid] = acc; } + if (groupId == 2) { reduceLM[SUBGROUP_SIZE * 1 + slid] = acc; } + if (groupId == 3) { reduceLM[SUBGROUP_SIZE * 2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (groupId == 0) { + acc += reduceLM[SUBGROUP_SIZE * 0 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 1 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // dst is column-major [M rows x n_cols cols]: (row, col) at col*M + row + vstore2((float2)(acc.s0, acc.s1), 0, &(dst[0 * M + gid * 2])); + vstore2((float2)(acc.s2, acc.s3), 0, &(dst[1 * M + gid * 2])); + if (n_cols > 2) vstore2((float2)(acc.s4, acc.s5), 0, &(dst[2 * M + gid * 2])); + if (n_cols > 3) vstore2((float2)(acc.s6, acc.s7), 0, &(dst[3 * M + gid * 2])); + } +} +#undef MC_COL_Q41 +#undef MC_DQ1_HI +#undef MC_DQ1_LO diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl index c1829fc3..c0078131 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl @@ -228,12 +228,37 @@ kernel void kernel_gemv_noshuffle_q4_k_f32( uint groupId = get_local_id(1); uint gid = get_global_id(0); ushort slid = get_sub_group_local_id(); + // K-split factor = #subgroups in the WG. Read from the launch (NOT a compile + // constant) so small-M projections (Kcur/Vcur/Qcur) can dispatch a wider + // K-split (more waves/SP -> latency hiding) while large-M keeps 4. The + // physical weight layout stride below is INDEPENDENT of this (see BLOCK_STRIDE_A). + uint nsg = get_local_size(1); uint K = ne00; uint M = ne01; uint LINE_STRIDE_A = M / 2; - uint BLOCK_STRIDE_A = NSUBGROUPS * M; + // Physical per-K-block stride in the packed image: 8 uints/block-row-pair * + // (M/2) row-pairs = 4*M uints. This is a layout constant, not tied to nsg. + uint BLOCK_STRIDE_A = 4 * M; + uint scales_per_row = (K / QK_K) * 12; + + // The x-grid is padded to CEIL_DIV(ne01/2,64)*64, so when ne01 % 128 != 0 the + // tail lanes hold gid >= ne01/2. The output stores below are guarded, but the + // input fetches are not: src0_d and src0_m are raw global half2 pointers, + // src0_s is a raw global uchar pointer, and read_imageui on an + // image1d_buffer_t is UNDEFINED out of range -- an image clamps only for + // SAMPLER reads, which these are not. Those lanes therefore read past the end + // of all three allocations. For a [2816, 2112] weight (2112 % 128 == 64) the + // top tail lane is gid = 1087 while only gid < 1056 is backed, and it runs + // 32 half2 past src0_d/src0_m, 31 uints past the quant image, and 63 bytes + // past src0_s. + // + // Clamp the row used for every fetch. The lanes stay ACTIVE, which the + // sub_group_broadcast in the dequant macros requires, and their results are + // still discarded by the existing output guard. No-op and byte-identical + // whenever ne01 % 128 == 0. + uint gid_s = min(gid, LINE_STRIDE_A - 1); private uint4 regA; private half2 regS; @@ -242,14 +267,14 @@ kernel void kernel_gemv_noshuffle_q4_k_f32( private float2 totalSum = (float2)(0.0f); - for (uint k = groupId; k < (K / 32); k += NSUBGROUPS) { + for (uint k = groupId; k < (K / 32); k += nsg) { uint sb = k / 8; uint j = k % 8; - half2 d = src0_d[gid + sb * LINE_STRIDE_A]; - half2 dm = src0_m[gid + sb * LINE_STRIDE_A]; + half2 d = src0_d[gid_s + sb * LINE_STRIDE_A]; + half2 dm = src0_m[gid_s + sb * LINE_STRIDE_A]; - global const uchar * sc0 = src0_s + sb * 12 * M + 2 * gid; + global const uchar * sc0 = src0_s + sb * 12 * M + 2 * gid_s; global const uchar * sc1 = sc0 + 1; uchar sv0, mn0, sv1, mn1; @@ -265,20 +290,20 @@ kernel void kernel_gemv_noshuffle_q4_k_f32( } // load half weights for two blocks in consecutive rows - regA.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; - regA.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; - regA.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; - regA.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA.s0 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA.s1 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA.s2 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA.s3 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; #ifdef VECTOR_SUB_GROUP_BROADCAST dequantizeBlockAccum_ns_sgbroadcast_8_hi(totalSum, as_ushort8(regA), regS, regM, regB); #else dequantizeBlockAccum_ns_sgbroadcast_1_hi(totalSum, as_ushort8(regA), regS, regM, regB); #endif // VECTOR_SUB_GROUP_BROADCAST - regA.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; - regA.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; - regA.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; - regA.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + regA.s0 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA.s1 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA.s2 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA.s3 = read_imageui(src0_q, (gid_s + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; #ifdef VECTOR_SUB_GROUP_BROADCAST dequantizeBlockAccum_ns_sgbroadcast_8_lo(totalSum, as_ushort8(regA), regS, regM, regB); #else @@ -286,28 +311,21 @@ kernel void kernel_gemv_noshuffle_q4_k_f32( #endif // VECTOR_SUB_GROUP_BROADCAST } - // reduction in local memory, assumes #wave=4 - local float2 reduceLM[SUBGROUP_SIZE * 3]; - if (groupId == 1) { - reduceLM[SUBGROUP_SIZE * 0 + slid] = totalSum; - } - if (groupId == 2) { - reduceLM[SUBGROUP_SIZE * 1 + slid] = totalSum; - } - if (groupId == 3) { - reduceLM[SUBGROUP_SIZE * 2 + slid] = totalSum; + // Cross-subgroup reduction in local memory. Generalized to nsg subgroups + // (was a hard-coded 4-wave unroll). Sized for up to 16 subgroups (the widest + // K-split we dispatch for small M). At nsg==4 the accumulation order is + // identical to the original unroll -> byte-identical for the large-M path. + local float2 reduceLM[SUBGROUP_SIZE * 15]; + if (groupId > 0) { + reduceLM[SUBGROUP_SIZE * (groupId - 1) + slid] = totalSum; } barrier(CLK_LOCAL_MEM_FENCE); if (groupId == 0) { - totalSum += reduceLM[SUBGROUP_SIZE * 0 + slid]; - } - if (groupId == 0) { - totalSum += reduceLM[SUBGROUP_SIZE * 1 + slid]; - } - if (groupId == 0) { - totalSum += reduceLM[SUBGROUP_SIZE * 2 + slid]; + for (uint i = 0; i < nsg - 1; ++i) { + totalSum += reduceLM[SUBGROUP_SIZE * i + slid]; + } } // 2 outputs per fiber in wave 0 @@ -322,3 +340,484 @@ kernel void kernel_gemv_noshuffle_q4_k_f32( } } + +// --- Fused gate+up GEMV + GLU epilogue (FFN) ------------------------------------ +// Folds the FFN's two decode GEMVs (ffn_gate, ffn_up) and the following GLU into a +// SINGLE dispatch: {MUL_MAT(Wg,x), MUL_MAT(Wu,x), GLU}. Both matmuls share the same +// activation x (ffn_norm), so the activation image read is issued ONCE per K-block +// and reused for the gate and up dot products (the per-op path re-reads it twice and +// also materializes the two full ffn-wide intermediates to global, which the GLU +// then re-reads). The gate/up partial sums are accumulated in the SAME per-fiber +// order and reduced in the SAME cross-subgroup order as the standalone GEMV, and the +// GLU formula is the exact scalar expression from kernels/glu.cl, so the output is +// BYTE-IDENTICAL to the per-op matmul+matmul+glu path -> safe to default on. +// glu_op: REGLU=0, GEGLU=1, SWIGLU=2, GEGLU_ERF=4, GEGLU_QUICK=5 (ggml_glu_op). +// Weights: src0g_* = gate (= GLU src[0]); src0u_* = up (= GLU src[1]). +#define GLU_GEGLU_COEF_A 0.044715f +#define GLU_SQRT_2_OVER_PI 0.79788456080286535587989211986876f +#define GLU_SQRT_2_INV 0.70710678118654752440084436210484f +#define GLU_QUICK_COEF -1.702f + +inline float glu_apply(int glu_op, float g, float u) { + float act; + if (glu_op == 1) { // GEGLU (tanh-approx gelu) + act = 0.5f*g*(1.0f + tanh(GLU_SQRT_2_OVER_PI*g*(1.0f + GLU_GEGLU_COEF_A*g*g))); + } else if (glu_op == 2) { // SWIGLU (silu) + act = g / (1.0f + exp(-g)); + } else if (glu_op == 0) { // REGLU + return g*u*(g > 0.0f); + } else if (glu_op == 4) { // GEGLU_ERF + act = 0.5f*g*(1.0f + erf(g*GLU_SQRT_2_INV)); + } else { // GEGLU_QUICK (glu_op == 5) + act = g*(1.0f/(1.0f + exp(GLU_QUICK_COEF*g))); + } + return act*u; +} + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_k_f32_glu( + read_only image1d_buffer_t src0g_q, + global half2 * src0g_d, + global half2 * src0g_m, + global uchar * src0g_s, + read_only image1d_buffer_t src0u_q, + global half2 * src0u_d, + global half2 * src0u_m, + global uchar * src0u_s, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + int glu_op, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + uint nsg = get_local_size(1); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = 4 * M; + + private uint4 regA; + private half2 regS, regM; + private float8 regB; + + private float2 gateSum = (float2)(0.0f); + private float2 upSum = (float2)(0.0f); + + // Two SEQUENTIAL K-loops (gate fully, then up). Keeping only one weight's + // working set live at a time holds the kernel's register footprint at ~the + // base single-weight GEMV's, so its max WG stays 1024 (16 subgroups) and the + // per-subgroup K-split matches the standalone wide GEMV exactly -> the gate + // and up partial sums are BYTE-IDENTICAL to the per-op path. The macro body + // is the base kernel's inner loop verbatim, parameterized by weight source. +#define Q4K_GLU_LOOP(SUM, Q, DD, MM, SS) \ + for (uint k = groupId; k < (K / 32); k += nsg) { \ + uint sb = k / 8; \ + uint j = k % 8; \ + half2 d = DD[gid + sb * LINE_STRIDE_A]; \ + half2 dm = MM[gid + sb * LINE_STRIDE_A]; \ + global const uchar * sc0 = SS + sb * 12 * M + 2 * gid; \ + global const uchar * sc1 = sc0 + 1; \ + uchar sv0, mn0, sv1, mn1; \ + get_scale_min_k4(j, sc0, M, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); \ + get_scale_min_k4(j, sc1, M, &sv1, &mn1, mask_d6, mask_d4, mask_hi2); \ + regS = convert_half2(convert_float2(d) * convert_float2((uchar2)(sv0, sv1))); \ + regM = convert_half2(convert_float2(dm) * convert_float2((uchar2)(mn0, mn1))); \ + if (slid < 4) { \ + regB.s0123 = read_imagef(src1, (slid * 2 + k * 8)); \ + regB.s4567 = read_imagef(src1, (1 + slid * 2 + k * 8)); \ + } \ + regA.s0 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; \ + regA.s1 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; \ + regA.s2 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; \ + regA.s3 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; \ + DEQ_HI(SUM, as_ushort8(regA), regS, regM, regB); \ + regA.s0 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; \ + regA.s1 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; \ + regA.s2 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; \ + regA.s3 = read_imageui(Q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; \ + DEQ_LO(SUM, as_ushort8(regA), regS, regM, regB); \ + } + +#ifdef VECTOR_SUB_GROUP_BROADCAST +#define DEQ_HI dequantizeBlockAccum_ns_sgbroadcast_8_hi +#define DEQ_LO dequantizeBlockAccum_ns_sgbroadcast_8_lo +#else +#define DEQ_HI dequantizeBlockAccum_ns_sgbroadcast_1_hi +#define DEQ_LO dequantizeBlockAccum_ns_sgbroadcast_1_lo +#endif + + Q4K_GLU_LOOP(gateSum, src0g_q, src0g_d, src0g_m, src0g_s) + Q4K_GLU_LOOP(upSum, src0u_q, src0u_d, src0u_m, src0u_s) + +#undef DEQ_HI +#undef DEQ_LO +#undef Q4K_GLU_LOOP + + // Cross-subgroup reduction in local memory. Packs gate (xy) + up (zw) into a + // float4 so both reduce in one pass; summation order matches the base GEMV's + // per-channel loop -> byte-identical partial sums. + local float4 reduceLM[SUBGROUP_SIZE * 15]; + if (groupId > 0) { + reduceLM[SUBGROUP_SIZE * (groupId - 1) + slid] = (float4)(gateSum, upSum); + } + barrier(CLK_LOCAL_MEM_FENCE); + if (groupId == 0) { + for (uint i = 0; i < nsg - 1; ++i) { + float4 p = reduceLM[SUBGROUP_SIZE * i + slid]; + gateSum += p.xy; + upSum += p.zw; + } + dst = (global float*)((global char*)dst + offsetd); + dst[gid * 2 + 0] = glu_apply(glu_op, gateSum.s0, upSum.s0); + dst[gid * 2 + 1] = glu_apply(glu_op, gateSum.s1, upSum.s1); + } +} + +// --- Split-K-across-workgroups decode GEMV (small-M projections) ---------------- +// A single-token GEMV makes only ceil(M/2/64) workgroups; a WG runs on one Adreno +// compute unit, so for small M (Kcur/Vcur, M=512 -> 4 WGs) most of the 16 CUs sit +// idle and the matmul is bandwidth-starved even with a wide intra-WG K-split. This +// variant adds a SECOND grid dimension of `ksplit` workgroups that each reduce a +// disjoint slice of K and write a per-slice partial; kernel_gemv_splitk_reduce_f32 +// then sums the partials into dst. Identical math/layout to the base kernel +// (physical block stride 4*M, get_scale_min_k4) -> coherent. Gated host-side to +// M<=1024 (M>=2048 +// already fills the CUs and the extra reduce dispatch only hurts). +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_k_f32_splitk( + read_only image1d_buffer_t src0_q, + global half2 * src0_d, + global half2 * src0_m, + global uchar * src0_s, + read_only image1d_buffer_t src1, + global float * partial, // [ksplit * M], slice-major + int ne00, + int ne01, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + uint nsg = get_local_size(1); + uint ksplit = get_num_groups(1); + uint kslice = get_group_id(1); + + uint K = ne00; + uint M = ne01; + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = 4 * M; // physical, independent of the K-split + + private uint4 regA; + private half2 regS, regM; + private float8 regB; + private float2 totalSum = (float2)(0.0f); + + // each (kslice, subgroup) pair owns a disjoint set of K-blocks + for (uint k = kslice * nsg + groupId; k < (K / 32); k += ksplit * nsg) { + uint sb = k / 8; + uint j = k % 8; + half2 d = src0_d[gid + sb * LINE_STRIDE_A]; + half2 dm = src0_m[gid + sb * LINE_STRIDE_A]; + global const uchar * sc0 = src0_s + sb * 12 * M + 2 * gid; + global const uchar * sc1 = sc0 + 1; + uchar sv0, mn0, sv1, mn1; + get_scale_min_k4(j, sc0, M, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(j, sc1, M, &sv1, &mn1, mask_d6, mask_d4, mask_hi2); + regS = convert_half2(convert_float2(d) * convert_float2((uchar2)(sv0, sv1))); + regM = convert_half2(convert_float2(dm) * convert_float2((uchar2)(mn0, mn1))); + if (slid < 4) { + regB.s0123 = read_imagef(src1, (slid * 2 + k * 8)); + regB.s4567 = read_imagef(src1, (1 + slid * 2 + k * 8)); + } + regA.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_hi(totalSum, as_ushort8(regA), regS, regM, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_hi(totalSum, as_ushort8(regA), regS, regM, regB); +#endif + regA.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_lo(totalSum, as_ushort8(regA), regS, regM, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_lo(totalSum, as_ushort8(regA), regS, regM, regB); +#endif + } + + local float2 reduceLM[SUBGROUP_SIZE * 15]; + if (groupId > 0) { + reduceLM[SUBGROUP_SIZE * (groupId - 1) + slid] = totalSum; + } + barrier(CLK_LOCAL_MEM_FENCE); + if (groupId == 0) { + for (uint i = 0; i < nsg - 1; ++i) { + totalSum += reduceLM[SUBGROUP_SIZE * i + slid]; + } + vstore2(totalSum, 0, &(partial[kslice * M + gid * 2])); + } +} + +// Sum the per-slice partials [ksplit * M] into dst[M]; applies the dst byte offset. +kernel void kernel_gemv_splitk_reduce_f32( + global float * partial, + global float * dst, + ulong offsetd, + int ne01, // M + int ksplit) +{ + uint r = get_global_id(0); + if (r >= (uint)ne01) return; + float acc = 0.0f; + for (uint s = 0; s < (uint)ksplit; ++s) { + acc += partial[s * (uint)ne01 + r]; + } + dst = (global float*)((global char*)dst + offsetd); + dst[r] = acc; +} + + +// --- Dequant-once macros for the mc3 verify GEMV (Q4K_MC3_DEQUANT_ONCE) --- +// The inline dequantizeBlockAccum_* macros recompute the dequantized weight +// ((code & mask)>>shift)*scale - minv ONCE PER COLUMN (3x), and the flat +// 32-FMA unroll spills ~430 B of temporaries. These macros split the work: +// DEQUANT_Q4K_BLOCK computes the 16 weights/row of one 32-block ONCE into a +// half2[] (row0 in .s0, row1 in .s1) — stored as half, the exact type the +// inline expression yields (int*half-half), so no extra rounding. MAC_Q4K_BLOCK +// then accumulates them against a column's broadcast activation in the SAME +// per-accumulator order as the inline macro. Each weight value and each +// accumulator's add-chain is bit-for-bit identical => byte-identical output, +// while the dequant ALU drops 3x->1x and the live set shrinks. Requires the +// Qualcomm vector sub_group_broadcast (float8); enabled opt-in on Adreno. +#define DEQ_Q4K_HALF2(b0, b1, msk, sh, scale, minv) \ + (half2)( ((b0 & msk) >> sh) * scale.s0 - minv.s0, \ + ((b1 & msk) >> sh) * scale.s1 - minv.s1 ) + +#define DEQUANT_Q4K_BLOCK(wq, bits, scale, minv) \ + wq[0] = DEQ_Q4K_HALF2(bits.s0, bits.s1, 0x000F, 0, scale, minv); \ + wq[1] = DEQ_Q4K_HALF2(bits.s0, bits.s1, 0x00F0, 4, scale, minv); \ + wq[2] = DEQ_Q4K_HALF2(bits.s0, bits.s1, 0x0F00, 8, scale, minv); \ + wq[3] = DEQ_Q4K_HALF2(bits.s0, bits.s1, 0xF000, 12, scale, minv); \ + wq[4] = DEQ_Q4K_HALF2(bits.s2, bits.s3, 0x000F, 0, scale, minv); \ + wq[5] = DEQ_Q4K_HALF2(bits.s2, bits.s3, 0x00F0, 4, scale, minv); \ + wq[6] = DEQ_Q4K_HALF2(bits.s2, bits.s3, 0x0F00, 8, scale, minv); \ + wq[7] = DEQ_Q4K_HALF2(bits.s2, bits.s3, 0xF000, 12, scale, minv); \ + wq[8] = DEQ_Q4K_HALF2(bits.s4, bits.s5, 0x000F, 0, scale, minv); \ + wq[9] = DEQ_Q4K_HALF2(bits.s4, bits.s5, 0x00F0, 4, scale, minv); \ + wq[10] = DEQ_Q4K_HALF2(bits.s4, bits.s5, 0x0F00, 8, scale, minv); \ + wq[11] = DEQ_Q4K_HALF2(bits.s4, bits.s5, 0xF000, 12, scale, minv); \ + wq[12] = DEQ_Q4K_HALF2(bits.s6, bits.s7, 0x000F, 0, scale, minv); \ + wq[13] = DEQ_Q4K_HALF2(bits.s6, bits.s7, 0x00F0, 4, scale, minv); \ + wq[14] = DEQ_Q4K_HALF2(bits.s6, bits.s7, 0x0F00, 8, scale, minv); \ + wq[15] = DEQ_Q4K_HALF2(bits.s6, bits.s7, 0xF000, 12, scale, minv); + +// ln0/ln1 = the two source lanes whose activation float8 this block consumes +// (0,1 for the hi block, 2,3 for the lo block — matching the inline _hi/_lo). +#define MAC_Q4K_BLOCK(ts, wq, y, ln0, ln1) { \ + float8 sy = sub_group_broadcast(y, ln0); \ + ts.s0 += wq[0].s0*sy.s0; ts.s0 += wq[1].s0*sy.s1; ts.s0 += wq[2].s0*sy.s2; ts.s0 += wq[3].s0*sy.s3; \ + ts.s0 += wq[4].s0*sy.s4; ts.s0 += wq[5].s0*sy.s5; ts.s0 += wq[6].s0*sy.s6; ts.s0 += wq[7].s0*sy.s7; \ + ts.s1 += wq[0].s1*sy.s0; ts.s1 += wq[1].s1*sy.s1; ts.s1 += wq[2].s1*sy.s2; ts.s1 += wq[3].s1*sy.s3; \ + ts.s1 += wq[4].s1*sy.s4; ts.s1 += wq[5].s1*sy.s5; ts.s1 += wq[6].s1*sy.s6; ts.s1 += wq[7].s1*sy.s7; \ + sy = sub_group_broadcast(y, ln1); \ + ts.s0 += wq[8].s0*sy.s0; ts.s0 += wq[9].s0*sy.s1; ts.s0 += wq[10].s0*sy.s2; ts.s0 += wq[11].s0*sy.s3; \ + ts.s0 += wq[12].s0*sy.s4; ts.s0 += wq[13].s0*sy.s5; ts.s0 += wq[14].s0*sy.s6; ts.s0 += wq[15].s0*sy.s7; \ + ts.s1 += wq[8].s1*sy.s0; ts.s1 += wq[9].s1*sy.s1; ts.s1 += wq[10].s1*sy.s2; ts.s1 += wq[11].s1*sy.s3; \ + ts.s1 += wq[12].s1*sy.s4; ts.s1 += wq[13].s1*sy.s5; ts.s1 += wq[14].s1*sy.s6; ts.s1 += wq[15].s1*sy.s7; \ +} + +// Multi-column (N=3) variant of the q4_K decode GEMV, for the speculative / +// MTP verify batch (ne1=3 = 2 drafts + 1 bonus). Stays on the efficient GEMV +// path (subgroup-broadcast activation, NSUBGROUPS K-split) instead of the +// transposed-GEMM dead-zone path. Each K-block's weights (regA_hi/regA_lo) are +// loaded ONCE and reused across all 3 activation columns — same weight traffic +// as one decode, ~3x the (cheap) dequant ALU. Per-column accumulation is +// independent and identical to 3 standalone GEMVs => byte-identical, so it does +// NOT perturb the lm_head logits / spec accept rate. +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_k_f32_mc3( + read_only image1d_buffer_t src0_q, + global half2 * src0_d, + global half2 * src0_m, + global uchar * src0_s, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = NSUBGROUPS * M; + uint COL_STRIDE = K / 4; // float4 pixels per activation column + + private uint4 regA_hi, regA_lo; + private half2 regS, regM; + private float8 regB; + + private float2 ts0 = (float2)(0.0f); + private float2 ts1 = (float2)(0.0f); + private float2 ts2 = (float2)(0.0f); + +#ifdef Q4K_MC3_DEQUANT_LDS + // One 16-half2 block buffer per WI (reused hi->lo): forces the dequantized + // weights into LDS instead of private arrays (which spill to slow global on + // Adreno). 64*NSUBGROUPS WIs * 16 half2 = 16 KB; each WI owns its own slot + // range (flat*16) -> no cross-lane sharing, no barrier needed. + local half2 wstage[SUBGROUP_SIZE * NSUBGROUPS * 16]; + local half2 * ws = wstage + (groupId * SUBGROUP_SIZE + slid) * 16; +#endif + + for (uint k = groupId; k < (K / 32); k += NSUBGROUPS) { + uint sb = k / 8; + uint j = k % 8; + + half2 d = src0_d[gid + sb * LINE_STRIDE_A]; + half2 dm = src0_m[gid + sb * LINE_STRIDE_A]; + + global const uchar * sc0 = src0_s + sb * 12 * M + 2 * gid; + global const uchar * sc1 = sc0 + 1; + + uchar sv0, mn0, sv1, mn1; + get_scale_min_k4(j, sc0, M, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(j, sc1, M, &sv1, &mn1, mask_d6, mask_d4, mask_hi2); + + regS = convert_half2(convert_float2(d) * convert_float2((uchar2)(sv0, sv1))); + regM = convert_half2(convert_float2(dm) * convert_float2((uchar2)(mn0, mn1))); + + // weights loaded ONCE, reused across the 3 columns + regA_hi.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA_hi.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA_hi.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA_hi.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA_lo.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA_lo.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA_lo.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA_lo.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + +#ifdef Q4K_MC3_DEQUANT_ONCE + // Dequant the 32 weights/row (16 hi + 16 lo) ONCE into half2[] (byte- + // identical to the inline intermediate), then MAC against each column's + // activation. Drops the dequant ALU 3x->1x and the macro-temp spill. + half2 wq_hi[16], wq_lo[16]; + DEQUANT_Q4K_BLOCK(wq_hi, as_ushort8(regA_hi), regS, regM); + DEQUANT_Q4K_BLOCK(wq_lo, as_ushort8(regA_lo), regS, regM); + { if (slid < 4) { regB.s0123 = read_imagef(src1, 0*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts0, wq_hi, regB, 0, 1); MAC_Q4K_BLOCK(ts0, wq_lo, regB, 2, 3); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 1*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts1, wq_hi, regB, 0, 1); MAC_Q4K_BLOCK(ts1, wq_lo, regB, 2, 3); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 2*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts2, wq_hi, regB, 0, 1); MAC_Q4K_BLOCK(ts2, wq_lo, regB, 2, 3); } +#elif defined(Q4K_MC3_DEQUANT_LDS) + // LDS-staged dequant: dequant a 32-block ONCE into the per-WI LDS slot + // (hi pass then lo pass, overwriting), MAC each column from LDS. ts* + // receive hi-then-lo in the same order as DEQUANT_ONCE -> byte-identical. + // Activations reloaded per pass (cheap, imaged); only one regB + 0 weight + // regs live -> the weight working set lives in LDS, not spilled private. + DEQUANT_Q4K_BLOCK(ws, as_ushort8(regA_hi), regS, regM); + { if (slid < 4) { regB.s0123 = read_imagef(src1, 0*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts0, ws, regB, 0, 1); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 1*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts1, ws, regB, 0, 1); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 2*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts2, ws, regB, 0, 1); } + DEQUANT_Q4K_BLOCK(ws, as_ushort8(regA_lo), regS, regM); + { if (slid < 4) { regB.s0123 = read_imagef(src1, 0*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts0, ws, regB, 2, 3); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 1*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts1, ws, regB, 2, 3); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 2*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + MAC_Q4K_BLOCK(ts2, ws, regB, 2, 3); } +#else + // Per-column: load only this column's activation (single regB live at a + // time -> 1/3 the activation register pressure vs holding all 3) then + // dequant against the shared weights. Cuts the private-mem spill. +#ifdef VECTOR_SUB_GROUP_BROADCAST + { if (slid < 4) { regB.s0123 = read_imagef(src1, 0*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_8_hi(ts0, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_8_lo(ts0, as_ushort8(regA_lo), regS, regM, regB); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 1*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_8_hi(ts1, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_8_lo(ts1, as_ushort8(regA_lo), regS, regM, regB); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 2*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_8_hi(ts2, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_8_lo(ts2, as_ushort8(regA_lo), regS, regM, regB); } +#else + { if (slid < 4) { regB.s0123 = read_imagef(src1, 0*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_1_hi(ts0, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_1_lo(ts0, as_ushort8(regA_lo), regS, regM, regB); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 1*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_1_hi(ts1, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_1_lo(ts1, as_ushort8(regA_lo), regS, regM, regB); } + { if (slid < 4) { regB.s0123 = read_imagef(src1, 2*COL_STRIDE + slid*2 + k*8); + regB.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + dequantizeBlockAccum_ns_sgbroadcast_1_hi(ts2, as_ushort8(regA_hi), regS, regM, regB); + dequantizeBlockAccum_ns_sgbroadcast_1_lo(ts2, as_ushort8(regA_lo), regS, regM, regB); } +#endif +#endif // Q4K_MC3_DEQUANT_ONCE + } + + // cross-subgroup reduce: pack the 3 columns' float2 into a float8 (6 used). + local float8 reduceLM[SUBGROUP_SIZE * 3]; + float8 acc = (float8)(ts0.s0, ts0.s1, ts1.s0, ts1.s1, ts2.s0, ts2.s1, 0.0f, 0.0f); + if (groupId == 1) { reduceLM[SUBGROUP_SIZE * 0 + slid] = acc; } + if (groupId == 2) { reduceLM[SUBGROUP_SIZE * 1 + slid] = acc; } + if (groupId == 3) { reduceLM[SUBGROUP_SIZE * 2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (groupId == 0) { + acc += reduceLM[SUBGROUP_SIZE * 0 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 1 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // dst is column-major [M rows x 3 cols]: (row, col) at col*M + row + vstore2((float2)(acc.s0, acc.s1), 0, &(dst[0 * M + gid * 2])); + vstore2((float2)(acc.s2, acc.s3), 0, &(dst[1 * M + gid * 2])); + vstore2((float2)(acc.s4, acc.s5), 0, &(dst[2 * M + gid * 2])); + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_32b_trans.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_32b_trans.cl new file mode 100644 index 00000000..2dbd943f --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_32b_trans.cl @@ -0,0 +1,134 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable + +#define QK_K 256 +#define K_SCALE_SIZE 12 +#define N_SIMDGROUP 8 +#define SIMDGROUP_WIDTH 64 + +inline void get_scale_min_k4( + int j, + global const uchar * q, + uint stride, + uchar * d, + uchar * m +) { + if (j < 4) { + *d = q[j*stride] & 63; + *m = q[(j+4)*stride] & 63; + } else { + *d = (q[(j+4)*stride] & 0x0F) | ((q[(j-4)*stride] & 0xC0) >> 2); + *m = ((q[(j+4)*stride] >> 4) & 0x0F) | ((q[j*stride] & 0xC0) >> 2); + } +} + +static inline float8 q4_k_to_fp32_packed8(ushort2 q4x8, float scale, float minv) { + float8 fp32x8; + fp32x8.s0 = (q4x8.s0 & 0x000F) * scale - minv; + fp32x8.s1 = ((q4x8.s0 & 0x00F0) >> 4) * scale - minv; + fp32x8.s2 = ((q4x8.s0 & 0x0F00) >> 8) * scale - minv; + fp32x8.s3 = ((q4x8.s0 & 0xF000) >> 12) * scale - minv; + fp32x8.s4 = (q4x8.s1 & 0x000F) * scale - minv; + fp32x8.s5 = ((q4x8.s1 & 0x00F0) >> 4) * scale - minv; + fp32x8.s6 = ((q4x8.s1 & 0x0F00) >> 8) * scale - minv; + fp32x8.s7 = ((q4x8.s1 & 0xF000) >> 12) * scale - minv; + return fp32x8; +} + +__attribute__((qcom_reqd_sub_group_size("half"))) +__kernel void gemv_noshuffle_q4_k_f32_32b_trans( + read_only image1d_buffer_t src0_q, + __global half * src0_d, + __global half * src0_dm, + __global uchar * src0_s, + __read_only image1d_buffer_t src1, + __global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + uint i01 = get_global_id(0); + uint sgid = get_local_id(1); + uint slid = get_sub_group_local_id(); + + int num_subblocks = ne00 / 32; + + __private float sum = 0.0f; + + // Loop over sub-blocks of 32 elements, N_SIMDGROUP sub-blocks per iter + for (uint ib = sgid; ib < num_subblocks; ib += N_SIMDGROUP) { + uint sb = ib / 8; + uint j = ib % 8; + + // Load d and dmin for this super-block + half d_val = src0_d[sb * ne01 + i01]; + half dm_val = src0_dm[sb * ne01 + i01]; + + // Load sub-block scale and min. s is transposed [nb][12][M]; stride ne01 per code. + global const uchar * sc = src0_s + sb * K_SCALE_SIZE * ne01 + i01; + uchar sv, mn; + get_scale_min_k4(j, sc, ne01, &sv, &mn); + + float scale = (float)d_val * (float)sv; + float minv = (float)dm_val * (float)mn; + + // Load 4 uints of quants (32 nibbles = 32 elements), column-major stride ne01 + uint q_base = ib * ne01 * 4 + i01; + + uint4 regQ; + regQ.s0 = read_imageui(src0_q, q_base).x; + regQ.s1 = read_imageui(src0_q, q_base + ne01).x; + regQ.s2 = read_imageui(src0_q, q_base + ne01 * 2).x; + regQ.s3 = read_imageui(src0_q, q_base + ne01 * 3).x; + + // Load activations: 32 floats = 8 float4s + uint y_offset = ib * 8; + + float4 y_local = (slid < 8) ? read_imagef(src1, (y_offset + slid)) : (float4)0.0f; + float4 y0 = sub_group_broadcast(y_local, 0); + float4 y1 = sub_group_broadcast(y_local, 1); + float4 y2 = sub_group_broadcast(y_local, 2); + float4 y3 = sub_group_broadcast(y_local, 3); + float4 y4 = sub_group_broadcast(y_local, 4); + float4 y5 = sub_group_broadcast(y_local, 5); + float4 y6 = sub_group_broadcast(y_local, 6); + float4 y7 = sub_group_broadcast(y_local, 7); + + float8 fp32x8 = q4_k_to_fp32_packed8(as_ushort2(regQ.s0), scale, minv); + float4 acc = y0 * fp32x8.lo; + acc += y1 * fp32x8.hi; + + fp32x8 = q4_k_to_fp32_packed8(as_ushort2(regQ.s1), scale, minv); + acc += y2 * fp32x8.lo; + acc += y3 * fp32x8.hi; + + fp32x8 = q4_k_to_fp32_packed8(as_ushort2(regQ.s2), scale, minv); + acc += y4 * fp32x8.lo; + acc += y5 * fp32x8.hi; + + fp32x8 = q4_k_to_fp32_packed8(as_ushort2(regQ.s3), scale, minv); + acc += y6 * fp32x8.lo; + acc += y7 * fp32x8.hi; + + sum += ((acc.s0 + acc.s1) + (acc.s2 + acc.s3)); + } + + // reduction in local memory over N_SIMDGROUP subgroups + __local float reduceLM[SIMDGROUP_WIDTH * (N_SIMDGROUP - 1)]; + if (sgid > 0) { + reduceLM[SIMDGROUP_WIDTH * (sgid - 1) + slid] = sum; + } + barrier(CLK_LOCAL_MEM_FENCE); + if (sgid == 0) { + for (uint i = 0; i < N_SIMDGROUP - 1; ++i) { + sum += reduceLM[SIMDGROUP_WIDTH * i + slid]; + } + } + + // 1 output per thread in subgroup 0 + if (sgid == 0) { + dst = dst + (offsetd >> 2); + dst[i01] = sum; + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_o4.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_o4.cl new file mode 100644 index 00000000..02916bb9 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_o4.cl @@ -0,0 +1,349 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_khr_subgroups : enable + +#ifdef cl_qcom_reqd_sub_group_size +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#endif + +#define QK_K 256 +#define NSUBGROUPS 4 +#define SUBGROUP_SIZE 64 + +// scales are transposed: consecutive codes of a row are `stride` apart +inline void get_scale_min_k4( + int j, + global const uchar * q, + uint stride, + uchar * d, + uchar * m, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2 +) { + if (j < 4) { + *d = q[j*stride] & mask_d6; + *m = q[(j+4)*stride] & mask_d6; + } else { + *d = (q[(j+4)*stride] & mask_d4) | ((q[(j-4)*stride] & mask_hi2) >> 2); + *m = ((q[(j+4)*stride] >> 4) & mask_d4) | ((q[j*stride] & mask_hi2) >> 2); + } +} + +#define dequantizeBlockAccum_ns_sgbroadcast_1_hi(total_sums, bits4, scale, minv, y) \ + float shared_y; \ + shared_y = sub_group_broadcast(y.s0, 0); \ + total_sums.s0 += ((bits4.s0 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s1 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 0); \ + total_sums.s0 += (((bits4.s0 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 0); \ + total_sums.s0 += (((bits4.s0 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 0); \ + total_sums.s0 += (((bits4.s0 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 0); \ + total_sums.s0 += ((bits4.s2 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s3 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 0); \ + total_sums.s0 += (((bits4.s2 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 0); \ + total_sums.s0 += (((bits4.s2 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 0); \ + total_sums.s0 += (((bits4.s2 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 1); \ + total_sums.s0 += ((bits4.s4 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s5 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 1); \ + total_sums.s0 += (((bits4.s4 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 1); \ + total_sums.s0 += (((bits4.s4 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 1); \ + total_sums.s0 += (((bits4.s4 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 1); \ + total_sums.s0 += ((bits4.s6 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s7 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 1); \ + total_sums.s0 += (((bits4.s6 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 1); \ + total_sums.s0 += (((bits4.s6 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 1); \ + total_sums.s0 += (((bits4.s6 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + + +#define dequantizeBlockAccum_ns_sgbroadcast_1_lo(total_sums, bits4, scale, minv, y) \ + shared_y = sub_group_broadcast(y.s0, 2); \ + total_sums.s0 += ((bits4.s0 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s1 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 2); \ + total_sums.s0 += (((bits4.s0 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 2); \ + total_sums.s0 += (((bits4.s0 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 2); \ + total_sums.s0 += (((bits4.s0 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s1 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 2); \ + total_sums.s0 += ((bits4.s2 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s3 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 2); \ + total_sums.s0 += (((bits4.s2 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 2); \ + total_sums.s0 += (((bits4.s2 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 2); \ + total_sums.s0 += (((bits4.s2 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s3 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 3); \ + total_sums.s0 += ((bits4.s4 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s5 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 3); \ + total_sums.s0 += (((bits4.s4 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 3); \ + total_sums.s0 += (((bits4.s4 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 3); \ + total_sums.s0 += (((bits4.s4 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s5 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 3); \ + total_sums.s0 += ((bits4.s6 & 0x000F) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += ((bits4.s7 & 0x000F) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 3); \ + total_sums.s0 += (((bits4.s6 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 3); \ + total_sums.s0 += (((bits4.s6 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 3); \ + total_sums.s0 += (((bits4.s6 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y; \ + total_sums.s1 += (((bits4.s7 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y; \ + + +#define dequantizeBlockAccum_ns_sgbroadcast_8_hi(total_sums, bits4, scale, minv, y) \ + float8 shared_y; \ + shared_y = sub_group_broadcast(y, 0); \ + total_sums.s0 += ((bits4.s0 & 0x000F) * scale.s0 - minv.s0) * shared_y.s0; \ + total_sums.s0 += (((bits4.s0 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s1; \ + total_sums.s0 += (((bits4.s0 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s2; \ + total_sums.s0 += (((bits4.s0 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s3; \ + total_sums.s0 += ((bits4.s2 & 0x000F) * scale.s0 - minv.s0) * shared_y.s4; \ + total_sums.s0 += (((bits4.s2 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s5; \ + total_sums.s0 += (((bits4.s2 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s6; \ + total_sums.s0 += (((bits4.s2 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s7; \ + total_sums.s1 += ((bits4.s1 & 0x000F) * scale.s1 - minv.s1) * shared_y.s0; \ + total_sums.s1 += (((bits4.s1 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s1; \ + total_sums.s1 += (((bits4.s1 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s2; \ + total_sums.s1 += (((bits4.s1 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s3; \ + total_sums.s1 += ((bits4.s3 & 0x000F) * scale.s1 - minv.s1) * shared_y.s4; \ + total_sums.s1 += (((bits4.s3 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s5; \ + total_sums.s1 += (((bits4.s3 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s6; \ + total_sums.s1 += (((bits4.s3 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s7; \ + shared_y = sub_group_broadcast(y, 1); \ + total_sums.s0 += ((bits4.s4 & 0x000F) * scale.s0 - minv.s0) * shared_y.s0; \ + total_sums.s0 += (((bits4.s4 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s1; \ + total_sums.s0 += (((bits4.s4 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s2; \ + total_sums.s0 += (((bits4.s4 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s3; \ + total_sums.s0 += ((bits4.s6 & 0x000F) * scale.s0 - minv.s0) * shared_y.s4; \ + total_sums.s0 += (((bits4.s6 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s5; \ + total_sums.s0 += (((bits4.s6 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s6; \ + total_sums.s0 += (((bits4.s6 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s7; \ + total_sums.s1 += ((bits4.s5 & 0x000F) * scale.s1 - minv.s1) * shared_y.s0; \ + total_sums.s1 += (((bits4.s5 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s1; \ + total_sums.s1 += (((bits4.s5 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s2; \ + total_sums.s1 += (((bits4.s5 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s3; \ + total_sums.s1 += ((bits4.s7 & 0x000F) * scale.s1 - minv.s1) * shared_y.s4; \ + total_sums.s1 += (((bits4.s7 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s5; \ + total_sums.s1 += (((bits4.s7 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s6; \ + total_sums.s1 += (((bits4.s7 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s7; \ + + +#define dequantizeBlockAccum_ns_sgbroadcast_8_lo(total_sums, bits4, scale, minv, y) \ + shared_y = sub_group_broadcast(y, 2); \ + total_sums.s0 += ((bits4.s0 & 0x000F) * scale.s0 - minv.s0) * shared_y.s0; \ + total_sums.s0 += (((bits4.s0 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s1; \ + total_sums.s0 += (((bits4.s0 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s2; \ + total_sums.s0 += (((bits4.s0 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s3; \ + total_sums.s0 += ((bits4.s2 & 0x000F) * scale.s0 - minv.s0) * shared_y.s4; \ + total_sums.s0 += (((bits4.s2 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s5; \ + total_sums.s0 += (((bits4.s2 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s6; \ + total_sums.s0 += (((bits4.s2 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s7; \ + total_sums.s1 += ((bits4.s1 & 0x000F) * scale.s1 - minv.s1) * shared_y.s0; \ + total_sums.s1 += (((bits4.s1 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s1; \ + total_sums.s1 += (((bits4.s1 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s2; \ + total_sums.s1 += (((bits4.s1 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s3; \ + total_sums.s1 += ((bits4.s3 & 0x000F) * scale.s1 - minv.s1) * shared_y.s4; \ + total_sums.s1 += (((bits4.s3 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s5; \ + total_sums.s1 += (((bits4.s3 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s6; \ + total_sums.s1 += (((bits4.s3 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s7; \ + shared_y = sub_group_broadcast(y, 3); \ + total_sums.s0 += ((bits4.s4 & 0x000F) * scale.s0 - minv.s0) * shared_y.s0; \ + total_sums.s0 += (((bits4.s4 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s1; \ + total_sums.s0 += (((bits4.s4 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s2; \ + total_sums.s0 += (((bits4.s4 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s3; \ + total_sums.s0 += ((bits4.s6 & 0x000F) * scale.s0 - minv.s0) * shared_y.s4; \ + total_sums.s0 += (((bits4.s6 & 0x00F0) >> 4) * scale.s0 - minv.s0) * shared_y.s5; \ + total_sums.s0 += (((bits4.s6 & 0x0F00) >> 8) * scale.s0 - minv.s0) * shared_y.s6; \ + total_sums.s0 += (((bits4.s6 & 0xF000) >> 12) * scale.s0 - minv.s0) * shared_y.s7; \ + total_sums.s1 += ((bits4.s5 & 0x000F) * scale.s1 - minv.s1) * shared_y.s0; \ + total_sums.s1 += (((bits4.s5 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s1; \ + total_sums.s1 += (((bits4.s5 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s2; \ + total_sums.s1 += (((bits4.s5 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s3; \ + total_sums.s1 += ((bits4.s7 & 0x000F) * scale.s1 - minv.s1) * shared_y.s4; \ + total_sums.s1 += (((bits4.s7 & 0x00F0) >> 4) * scale.s1 - minv.s1) * shared_y.s5; \ + total_sums.s1 += (((bits4.s7 & 0x0F00) >> 8) * scale.s1 - minv.s1) * shared_y.s6; \ + total_sums.s1 += (((bits4.s7 & 0xF000) >> 12) * scale.s1 - minv.s1) * shared_y.s7; \ + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_k_f32_o4( + read_only image1d_buffer_t src0_q, + global half2 * src0_d, + global half2 * src0_m, + global uchar * src0_s, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); // 4-output quad index + ushort slid = get_sub_group_local_id(); + + // Two consecutive pair-indices (each the same access pattern the 2-output + // kernel uses); together they cover 4 consecutive output rows. + uint gid_a = gid * 2; + uint gid_b = gid * 2 + 1; + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = NSUBGROUPS * M; + + private uint4 regA; + private half2 regS_a, regS_b; + private half2 regM_a, regM_b; + private float8 regB; + + private float2 totalSum_a = (float2)(0.0f); + private float2 totalSum_b = (float2)(0.0f); + + for (uint k = groupId; k < (K / 32); k += NSUBGROUPS) { + uint sb = k / 8; + uint j = k % 8; + + // pair a scales/mins + half2 d_a = src0_d[gid_a + sb * LINE_STRIDE_A]; + half2 dm_a = src0_m[gid_a + sb * LINE_STRIDE_A]; + global const uchar * sc0a = src0_s + sb * 12 * M + 2 * gid_a; + global const uchar * sc1a = sc0a + 1; + uchar sv0a, mn0a, sv1a, mn1a; + get_scale_min_k4(j, sc0a, M, &sv0a, &mn0a, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(j, sc1a, M, &sv1a, &mn1a, mask_d6, mask_d4, mask_hi2); + regS_a = convert_half2(convert_float2(d_a) * convert_float2((uchar2)(sv0a, sv1a))); + regM_a = convert_half2(convert_float2(dm_a) * convert_float2((uchar2)(mn0a, mn1a))); + + // pair b scales/mins + half2 d_b = src0_d[gid_b + sb * LINE_STRIDE_A]; + half2 dm_b = src0_m[gid_b + sb * LINE_STRIDE_A]; + global const uchar * sc0b = src0_s + sb * 12 * M + 2 * gid_b; + global const uchar * sc1b = sc0b + 1; + uchar sv0b, mn0b, sv1b, mn1b; + get_scale_min_k4(j, sc0b, M, &sv0b, &mn0b, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(j, sc1b, M, &sv1b, &mn1b, mask_d6, mask_d4, mask_hi2); + regS_b = convert_half2(convert_float2(d_b) * convert_float2((uchar2)(sv0b, sv1b))); + regM_b = convert_half2(convert_float2(dm_b) * convert_float2((uchar2)(mn0b, mn1b))); + + // activation: load once, reuse for both pairs + if (slid < 4) { + regB.s0123 = read_imagef(src1, (slid * 2 + k * 8)); + regB.s4567 = read_imagef(src1, (1 + slid * 2 + k * 8)); + } + + // pair a (own block so _lo sees the shared_y declared by _hi) + { + regA.s0 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA.s1 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA.s2 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA.s3 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_hi(totalSum_a, as_ushort8(regA), regS_a, regM_a, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_hi(totalSum_a, as_ushort8(regA), regS_a, regM_a, regB); +#endif + regA.s0 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA.s1 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA.s2 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA.s3 = read_imageui(src0_q, (gid_a + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_lo(totalSum_a, as_ushort8(regA), regS_a, regM_a, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_lo(totalSum_a, as_ushort8(regA), regS_a, regM_a, regB); +#endif + } + + // pair b + { + regA.s0 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA.s1 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA.s2 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA.s3 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_hi(totalSum_b, as_ushort8(regA), regS_b, regM_b, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_hi(totalSum_b, as_ushort8(regA), regS_b, regM_b, regB); +#endif + regA.s0 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA.s1 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA.s2 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA.s3 = read_imageui(src0_q, (gid_b + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; +#ifdef VECTOR_SUB_GROUP_BROADCAST + dequantizeBlockAccum_ns_sgbroadcast_8_lo(totalSum_b, as_ushort8(regA), regS_b, regM_b, regB); +#else + dequantizeBlockAccum_ns_sgbroadcast_1_lo(totalSum_b, as_ushort8(regA), regS_b, regM_b, regB); +#endif + } + } + + // reduce 4 outputs (a.s0, a.s1, b.s0, b.s1) across the 4 subgroups + local float4 reduceLM[SUBGROUP_SIZE * 3]; + float4 acc = (float4)(totalSum_a.s0, totalSum_a.s1, totalSum_b.s0, totalSum_b.s1); + if (groupId == 1) { reduceLM[SUBGROUP_SIZE * 0 + slid] = acc; } + if (groupId == 2) { reduceLM[SUBGROUP_SIZE * 1 + slid] = acc; } + if (groupId == 3) { reduceLM[SUBGROUP_SIZE * 2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (groupId == 0) { + acc += reduceLM[SUBGROUP_SIZE * 0 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 1 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // The dispatch rounds ne01/4 up to the subgroup width, so the tail + // quads past the last row must not store (they wrote 128 rows past + // dst on every ne01 % 256 == 128 vocab, e.g. 151936). + if (gid * 4 + 3 < (uint)ne01) { + vstore4(acc, 0, &(dst[gid * 4])); + } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_tiled.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_tiled.cl new file mode 100644 index 00000000..929538c4 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32_tiled.cl @@ -0,0 +1,118 @@ +// Tiled-wide q4_K GEMV for the long-vocab lm_head/embed (decode path). +// +// Pairs with kernel_convert_block_q4_k_tiled_ns (cvt.cl): the weights are laid +// out CANONICALLY (4-bit code in element order e in [0,256)) and TILED by 64 +// output rows so the 64-thread lane group coalesces every weight load. Both the +// pack (convert) and the unpack (here) are owned by us -> correct by +// construction vs the reference ggml q4_K dequant. Same structure as the q6_K +// tiled GEMV; the only differences are the 4-bit dequant and the q4_K +// scale/min decode (get_scale_min_k4 from the packed 12-byte block). +// +// One work-item produces one output row. WG = {64 lanes, 4 subgroups}: the 64 +// lanes cover the 64 rows of one tile (coalesced uint4 reads), the 4 subgroups +// split the K-blocks and reduce through __local at the end. Weights read from +// __global (lm_head is streamed once per token; texture cache caps it below the +// coalesced-global rate). + +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +#ifdef cl_qcom_reqd_sub_group_size +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#endif + +#define QK_K 256 +#define NSUBGROUPS 4 +#define TILE_ROWS 64 + +// Decode one q4_K sub-block scale + min from the packed 12-byte block. +// Identical to the o4 kernel's helper (masks hard-coded: d6=0x3F, d4=0x0F, hi2=0xC0). +inline void q4k_scale_min(int j, __global const uchar * q, uchar * d, uchar * m) { + if (j < 4) { + *d = q[j] & 0x3F; + *m = q[j+4] & 0x3F; + } else { + *d = (q[j+4] & 0x0F) | ((q[j-4] & 0xC0) >> 2); + *m = ((q[j+4] >> 4) & 0x0F) | ((q[j] & 0xC0) >> 2); + } +} + +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q4_k_f32_tiled( + __global uint4 * src0_q, // tiled: 8 uint4 granules / superblock (4-bit codes) + __global half * src0_d, // tiled: 1 half / superblock + __global half * src0_dm, // tiled: 1 half / superblock + __global uchar * src0_s, // tiled: 12 bytes / superblock (packed scales) + read_only image1d_buffer_t src1, // activation (RGBA f32) + global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + int grp = get_local_id(1); // subgroup index 0..3 (splits K) + int row = get_global_id(0); // output row along ne01 + int rt = row / TILE_ROWS; + int rit = row % TILE_ROWS; + + int nb = ne00 / QK_K; // superblocks per row + + float acc = 0.0f; + + for (int sb = grp; sb < nb; sb += NSUBGROUPS) { + int tile_blk = rt * nb + sb; // ne02 == 1 for lm_head/embed + + float dval = (float)src0_d [tile_blk * TILE_ROWS + rit]; + float dmval = (float)src0_dm[tile_blk * TILE_ROWS + rit]; + + // decode the 8 sub-block (scale, min) pairs + __global uchar * sc = src0_s + (tile_blk * TILE_ROWS + rit) * 12; + float scale[8], minv[8]; + #pragma unroll + for (int is = 0; is < 8; ++is) { + uchar sd, sm; + q4k_scale_min(is, sc, &sd, &sm); + scale[is] = dval * (float)sd; + minv[is] = dmval * (float)sm; + } + + // 32 uints of 4-bit codes (8 codes/uint), e-order + uint q[32]; + #pragma unroll + for (int g = 0; g < 8; ++g) { + uint4 v = src0_q[(tile_blk * 8 + g) * TILE_ROWS + rit]; + q[g*4+0] = v.x; q[g*4+1] = v.y; q[g*4+2] = v.z; q[g*4+3] = v.w; + } + + // dequant 256 codes in canonical e-order, MAC with activation. + int act_base = sb * 64; // activation float4 pixel base (256/4) + #pragma unroll + for (int e4 = 0; e4 < 64; ++e4) { + float4 a = read_imagef(src1, act_base + e4); + #pragma unroll + for (int t = 0; t < 4; ++t) { + int e = e4 * 4 + t; + uint code = (q[e >> 3] >> ((e & 7) * 4)) & 0xF; + int is = e >> 5; // sub-block index = e/32 + float av = (t == 0) ? a.x : (t == 1) ? a.y : (t == 2) ? a.z : a.w; + acc += ((float)code * scale[is] - minv[is]) * av; + } + } + } + + // reduce across the NSUBGROUPS subgroups (same rit, different K-subset) + local float reduce_lm[NSUBGROUPS * TILE_ROWS]; + reduce_lm[grp * TILE_ROWS + rit] = acc; + barrier(CLK_LOCAL_MEM_FENCE); + + if (grp == 0) { + float total = reduce_lm[0 * TILE_ROWS + rit] + + reduce_lm[1 * TILE_ROWS + rit] + + reduce_lm[2 * TILE_ROWS + rit] + + reduce_lm[3 * TILE_ROWS + rit]; + dst = (global float*)((global char*)dst + offsetd); + dst[row] = total; + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl index 446f4653..ae864b19 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl @@ -329,3 +329,125 @@ kernel void kernel_gemv_noshuffle_q5_k_f32( if (gid * 2 + 1 < M) dst[gid * 2 + 1] = totalSum.s1; } } + +// Multi-column (N in [2..4]) variant of the q5_K decode GEMV (spec/MTP verify) = +// q4_K mc3 + the high-bit qh plane (regH). n_cols = 2..4 (drafted + bonus); routes +// the small-batch verify OFF the gemm_noshuffle_q5_k dead-zone. n_cols==3 is byte- +// identical to the original mc3 (col3 disabled, float8 slots 6/7 stay zero). +#ifdef VECTOR_SUB_GROUP_BROADCAST +#define MC_DQ5_HI dequantizeBlockAccum_ns_sgbroadcast_8_hi +#define MC_DQ5_LO dequantizeBlockAccum_ns_sgbroadcast_8_lo +#else +#define MC_DQ5_HI dequantizeBlockAccum_ns_sgbroadcast_1_hi +#define MC_DQ5_LO dequantizeBlockAccum_ns_sgbroadcast_1_lo +#endif +#define MC_COL_Q5K(ts, c) \ + { if (slid < 4) { regB.s0123 = read_imagef(src1, (c)*COL_STRIDE + slid*2 + k*8); \ + regB.s4567 = read_imagef(src1, (c)*COL_STRIDE + 1 + slid*2 + k*8); } \ + MC_DQ5_HI(ts, as_ushort8(regA_hi), as_uchar8(regH), regS, regM, regB); \ + MC_DQ5_LO(ts, as_ushort8(regA_lo), as_uchar8(regH), regS, regM, regB); } +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q5_k_f32_mc3( + read_only image1d_buffer_t src0_q, + read_only image1d_buffer_t src0_qh, + global half2 * src0_d, + global half2 * src0_m, + global uchar * src0_s, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + uchar mask_d6, + uchar mask_d4, + uchar mask_hi2, + int n_cols) +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M / 2; + uint BLOCK_STRIDE_A = NSUBGROUPS * M; + uint LINE_STRIDE_A_QH = M / 2; + uint BLOCK_STRIDE_A_QH = NSUBGROUPS * M / 2; + uint scales_per_row = (K / QK_K) * 12; + uint COL_STRIDE = K / 4; // float4 pixels per activation column + + private uint4 regA_hi, regA_lo; + private ushort4 regH; + private half2 regS, regM; + private float8 regB; + + private float2 ts0 = (float2)(0.0f); + private float2 ts1 = (float2)(0.0f); + private float2 ts2 = (float2)(0.0f); + private float2 ts3 = (float2)(0.0f); + + for (uint k = groupId; k < (K / 32); k += NSUBGROUPS) { + uint sb = k / 8; + uint j = k % 8; + + half2 d = src0_d[gid + sb * LINE_STRIDE_A]; + half2 dm = src0_m[gid + sb * LINE_STRIDE_A]; + + global const uchar * sc0 = src0_s + 2 * gid * scales_per_row + sb * 12; + global const uchar * sc1 = src0_s + (2 * gid + 1) * scales_per_row + sb * 12; + + uchar sv0, mn0, sv1, mn1; + get_scale_min_k4(j, sc0, &sv0, &mn0, mask_d6, mask_d4, mask_hi2); + get_scale_min_k4(j, sc1, &sv1, &mn1, mask_d6, mask_d4, mask_hi2); + + regS = convert_half2(convert_float2(d) * convert_float2((uchar2)(sv0, sv1))); + regM = convert_half2(convert_float2(dm) * convert_float2((uchar2)(mn0, mn1))); + + // high-bit plane + weights loaded ONCE, reused across the columns + regH.s0 = as_ushort(read_imageh(src0_qh, (gid + k * BLOCK_STRIDE_A_QH + LINE_STRIDE_A_QH * 0)).x); + regH.s1 = as_ushort(read_imageh(src0_qh, (gid + k * BLOCK_STRIDE_A_QH + LINE_STRIDE_A_QH * 1)).x); + regH.s2 = as_ushort(read_imageh(src0_qh, (gid + k * BLOCK_STRIDE_A_QH + LINE_STRIDE_A_QH * 2)).x); + regH.s3 = as_ushort(read_imageh(src0_qh, (gid + k * BLOCK_STRIDE_A_QH + LINE_STRIDE_A_QH * 3)).x); + + regA_hi.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA_hi.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA_hi.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA_hi.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA_lo.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA_lo.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA_lo.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA_lo.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + + MC_COL_Q5K(ts0, 0); + MC_COL_Q5K(ts1, 1); + if (n_cols > 2) MC_COL_Q5K(ts2, 2); + if (n_cols > 3) MC_COL_Q5K(ts3, 3); + } + + // cross-subgroup reduce: pack the (up to 4) columns' float2 into a float8. + local float8 reduceLM[SUBGROUP_SIZE * 3]; + float8 acc = (float8)(ts0.s0, ts0.s1, ts1.s0, ts1.s1, ts2.s0, ts2.s1, ts3.s0, ts3.s1); + if (groupId == 1) { reduceLM[SUBGROUP_SIZE * 0 + slid] = acc; } + if (groupId == 2) { reduceLM[SUBGROUP_SIZE * 1 + slid] = acc; } + if (groupId == 3) { reduceLM[SUBGROUP_SIZE * 2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (groupId == 0) { + acc += reduceLM[SUBGROUP_SIZE * 0 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 1 + slid]; + acc += reduceLM[SUBGROUP_SIZE * 2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // dst is column-major [M rows x n_cols cols]: (row, col) at col*M + row + vstore2((float2)(acc.s0, acc.s1), 0, &(dst[0 * M + gid * 2])); + vstore2((float2)(acc.s2, acc.s3), 0, &(dst[1 * M + gid * 2])); + if (n_cols > 2) vstore2((float2)(acc.s4, acc.s5), 0, &(dst[2 * M + gid * 2])); + if (n_cols > 3) vstore2((float2)(acc.s6, acc.s7), 0, &(dst[3 * M + gid * 2])); + } +} +#undef MC_COL_Q5K +#undef MC_DQ5_HI +#undef MC_DQ5_LO diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl index 51682ece..32624ac8 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl @@ -296,3 +296,114 @@ kernel void kernel_gemv_noshuffle_q6_K_f32( if (gid * 2 + 1 < ne01) dst[gid * 2 + 1] = total_sum.s1; } } + +// Multi-column (N=3) q6_K decode GEMV for the spec/MTP verify batch. Same idea +// as the q4_K mc3: stay on the efficient GEMV path (subgroup broadcast, no +// transpose) instead of the transposed-GEMM dead-zone. Each K-block's weights +// (ql/qh, hi+lo) are loaded ONCE and reused across all 3 activation columns. +// Per-column accumulation is independent and identical to 3 standalone GEMVs +// => byte-identical; does NOT perturb the lm_head logits / spec accept rate. +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q6_K_f32_mc3( + read_only image1d_buffer_t src0_ql, + read_only image1d_buffer_t src0_qh, + global half2 * src0_s, + global half2 * src0_d, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + int grp = get_local_id(1); + int gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + + int nb = ne00 / 32; + int line_stride_a = ne01 / 2; + int block_stride_a = NSUBGROUPS * ne01; + int COL_STRIDE = ne00 / 4; // float4 pixels per activation column + + uint4 ql_hi, ql_lo; + ushort4 qh_hi, qh_lo; + half2 reg_d; + char4 reg_s; + float8 reg_b; + + float2 ts0 = 0.0f, ts1 = 0.0f, ts2 = 0.0f; + + for (int k = grp; k < nb; k += NSUBGROUPS) { + reg_d = src0_d[gid + k/8 * line_stride_a]; + reg_s = as_char4(src0_s[gid + k * line_stride_a]); + + // weights loaded ONCE (hi: blocks 0-3, lo: blocks 4-7), reused x3 cols + ql_hi.s0 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*0).x; + ql_hi.s1 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*1).x; + ql_hi.s2 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*2).x; + ql_hi.s3 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*3).x; + qh_hi.s0 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*0).x); + qh_hi.s1 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*1).x); + qh_hi.s2 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*2).x); + qh_hi.s3 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*3).x); + + ql_lo.s0 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*4).x; + ql_lo.s1 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*5).x; + ql_lo.s2 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*6).x; + ql_lo.s3 = read_imageui(src0_ql, gid + k*block_stride_a + line_stride_a*7).x; + qh_lo.s0 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*4).x); + qh_lo.s1 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*5).x); + qh_lo.s2 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*6).x); + qh_lo.s3 = as_ushort(read_imageh(src0_qh, gid + k*block_stride_a + line_stride_a*7).x); + + // Per-column: load only this column's activation (single reg_b live) -> + // 1/3 the activation register pressure, cutting the private-mem spill. +#ifdef VECTOR_SUB_GROUP_BROADCAT + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 0*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_8_hi(ts0, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_8_lo(ts0, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 1*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_8_hi(ts1, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_8_lo(ts1, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 2*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_8_hi(ts2, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_8_lo(ts2, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } +#else + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 0*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 0*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_1_hi(ts0, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_1_lo(ts0, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 1*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 1*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_1_hi(ts1, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_1_lo(ts1, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } + { if (slid < 4) { reg_b.s0123 = read_imagef(src1, 2*COL_STRIDE + 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 2*COL_STRIDE + 1 + slid*2 + k*8); } + dequantize_block_acc_bcast_1_hi(ts2, as_ushort8(ql_hi), as_uchar8(qh_hi), reg_d, reg_s, reg_b); + dequantize_block_acc_bcast_1_lo(ts2, as_ushort8(ql_lo), as_uchar8(qh_lo), reg_d, reg_s, reg_b); } +#endif + } + + local float8 reduce_lm[SUBGROUP_SIZE * 3]; + float8 acc = (float8)(ts0.s0, ts0.s1, ts1.s0, ts1.s1, ts2.s0, ts2.s1, 0.0f, 0.0f); + if (grp == 1) { reduce_lm[SUBGROUP_SIZE*0 + slid] = acc; } + if (grp == 2) { reduce_lm[SUBGROUP_SIZE*1 + slid] = acc; } + if (grp == 3) { reduce_lm[SUBGROUP_SIZE*2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (grp == 0) { + acc += reduce_lm[SUBGROUP_SIZE*0 + slid]; + acc += reduce_lm[SUBGROUP_SIZE*1 + slid]; + acc += reduce_lm[SUBGROUP_SIZE*2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // dst column-major [ne01 rows x 3 cols]: (row, col) at col*ne01 + row + vstore2((float2)(acc.s0, acc.s1), 0, &(dst[0*ne01 + gid*2])); + vstore2((float2)(acc.s2, acc.s3), 0, &(dst[1*ne01 + gid*2])); + vstore2((float2)(acc.s4, acc.s5), 0, &(dst[2*ne01 + gid*2])); + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_32b_trans.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_32b_trans.cl new file mode 100644 index 00000000..2e1e2d76 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_32b_trans.cl @@ -0,0 +1,128 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable + +#define QK_K 256 +#define N_SIMDGROUP 8 +#define SIMDGROUP_WIDTH 64 + +static inline float8 q6_k_to_fp32_packed8(ushort2 ql8, ushort qh8, float d_scale) { + float8 fp32x8; + fp32x8.s0 = ((float)(( ql8.s0 & 0x000F) | ((uint)((qh8 ) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s1 = ((float)((( ql8.s0 >> 4) & 0x000F) | ((uint)((qh8 >> 2) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s2 = ((float)((( ql8.s0 >> 8) & 0x000F) | ((uint)((qh8 >> 4) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s3 = ((float)((( ql8.s0 >> 12)& 0x000F) | ((uint)((qh8 >> 6) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s4 = ((float)(( ql8.s1 & 0x000F) | ((uint)((qh8 >> 8) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s5 = ((float)((( ql8.s1 >> 4) & 0x000F) | ((uint)((qh8 >>10) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s6 = ((float)((( ql8.s1 >> 8) & 0x000F) | ((uint)((qh8 >>12) & 0x3) << 4)) - 32.f) * d_scale; + fp32x8.s7 = ((float)((( ql8.s1 >> 12)& 0x000F) | ((uint)((qh8 >>14) & 0x3) << 4)) - 32.f) * d_scale; + return fp32x8; +} + +__attribute__((qcom_reqd_sub_group_size("half"))) +__kernel void kernel_gemv_noshuffle_q6_k_f32_32b_trans( + __read_only image1d_buffer_t src0_ql, + __read_only image1d_buffer_t src0_qh, + __global char * src0_s, + __global half * src0_d, + __read_only image1d_buffer_t src1, + __global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + uint i01 = get_global_id(0); + uint sgid = get_local_id(1); + uint slid = get_sub_group_local_id(); + + int num_superblocks = ne00 / QK_K; + int num_subblocks = ne00 / 32; // 2 sub-blocks of 16 processed per iter below + int scales_per_row = num_superblocks * 16; + + __private float sum = 0.0f; + + // Loop over 32-element groups (2 sub-blocks of 16 each), N_SIMDGROUP groups per iter. + for (uint ib = sgid; ib < num_subblocks; ib += N_SIMDGROUP) { + uint sb = ib / 8; // super-block index + uint j = ib % 8; // 32-element group within super-block (0..7) + + // Load d for this super-block. + half d_val = src0_d[sb * ne01 + i01]; + + // Load 2 sub-block scales (int8), one per 16 elements. + global const char * sc = src0_s + i01 * scales_per_row + sb * 16; + float scale0 = (float)d_val * (float)sc[j * 2]; + float scale1 = (float)d_val * (float)sc[j * 2 + 1]; + + // Load 4 uints of ql (32 elements, 4-bit each = 128 bits), column-major stride ne01. + uint ql_base = (ib * 4) * ne01 + i01; + uint4 regQL; + regQL.s0 = read_imageui(src0_ql, ql_base).x; + regQL.s1 = read_imageui(src0_ql, ql_base + ne01).x; + regQL.s2 = read_imageui(src0_ql, ql_base + ne01 * 2).x; + regQL.s3 = read_imageui(src0_ql, ql_base + ne01 * 3).x; + + // Load 2 uints of qh (32 elements, 2-bit each = 64 bits), column-major stride ne01. + uint qh_base = (ib * 2) * ne01 + i01; + uint2 regQH; + regQH.s0 = read_imageui(src0_qh, qh_base).x; + regQH.s1 = read_imageui(src0_qh, qh_base + ne01).x; + + // Load activations: 32 floats = 8 float4s. + uint y_offset = ib * 8; + + float4 y_local = (slid < 8) ? read_imagef(src1, (y_offset + slid)) : (float4)0.0f; + float4 y0 = sub_group_broadcast(y_local, 0); + float4 y1 = sub_group_broadcast(y_local, 1); + float4 y2 = sub_group_broadcast(y_local, 2); + float4 y3 = sub_group_broadcast(y_local, 3); + float4 y4v = sub_group_broadcast(y_local, 4); + float4 y5 = sub_group_broadcast(y_local, 5); + float4 y6 = sub_group_broadcast(y_local, 6); + float4 y7 = sub_group_broadcast(y_local, 7); + + // Dequantize elements 0..7 (scale0). + float8 fp32x8 = q6_k_to_fp32_packed8(as_ushort2(regQL.s0), (ushort)(regQH.s0 & 0xFFFF), scale0); + + float4 acc = y0 * fp32x8.lo; + acc += y1 * fp32x8.hi; + + // Dequantize elements 8..15 (scale0). + fp32x8 = q6_k_to_fp32_packed8(as_ushort2(regQL.s1), (ushort)(regQH.s0 >> 16), scale0); + + acc += y2 * fp32x8.lo; + acc += y3 * fp32x8.hi; + + // Dequantize elements 16..23 (scale1). + fp32x8 = q6_k_to_fp32_packed8(as_ushort2(regQL.s2), (ushort)(regQH.s1 & 0xFFFF), scale1); + + acc += y4v * fp32x8.lo; + acc += y5 * fp32x8.hi; + + // Dequantize elements 24..31 (scale1). + fp32x8 = q6_k_to_fp32_packed8(as_ushort2(regQL.s3), (ushort)(regQH.s1 >> 16), scale1); + + acc += y6 * fp32x8.lo; + acc += y7 * fp32x8.hi; + + sum += ((acc.s0 + acc.s1) + (acc.s2 + acc.s3)); + } + + // reduction in local memory, assumes #subgroups=4 + __local float reduceLM[SIMDGROUP_WIDTH * (N_SIMDGROUP - 1)]; + if (sgid > 0) { + reduceLM[SIMDGROUP_WIDTH * (sgid - 1) + slid] = sum; + } + barrier(CLK_LOCAL_MEM_FENCE); + if (sgid == 0) { + for (uint i = 0; i < N_SIMDGROUP - 1; ++i) { + sum += reduceLM[SIMDGROUP_WIDTH * i + slid]; + } + } + + // 1 output per thread in subgroup 0 + if (sgid == 0) { + dst = dst + (offsetd >> 2); + dst[i01] = sum; + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_o4.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_o4.cl new file mode 100644 index 00000000..84447e61 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_o4.cl @@ -0,0 +1,372 @@ +// 4-output-per-WI variant of kernel_gemv_noshuffle_q6_K_f32. +// Each WI now produces 4 consecutive outputs (output quad). The activation +// fetch (reg_b) is shared across all 4 outputs, doubling per-WI ALU per +// activation broadcast and halving the WG count vs the 2-output kernel. +// +// Implementation: each K-block we fetch TWO sets of (scales + ql + qh) +// — one for the low pair (rows 0,1 of the quad) and one for the high pair +// (rows 2,3) — and invoke the existing 2-output dequant macros twice +// against the *same* reg_b. Identical data layout to the 2-output kernel, +// so the host only needs to halve the grid and double the gid-to-output +// mapping. +// +// Opt-in via the host dispatch when GGML_OPENCL_Q6K_GEMV_O4=1. + +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_khr_subgroups : enable + +#ifdef cl_intel_required_subgroup_size +#pragma OPENCL EXTENSION cl_intel_required_subgroup_size : enable +#define INTEL_GPU 1 +#define REQD_SUBGROUP_SIZE_16 __attribute__((intel_reqd_sub_group_size(16))) +#define REQD_SUBGROUP_SIZE_32 __attribute__((intel_reqd_sub_group_size(32))) +#elif defined(cl_qcom_reqd_sub_group_size) +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#define REQD_SUBGROUP_SIZE_128 __attribute__((qcom_reqd_sub_group_size("full"))) +#endif + +#define NSUBGROUPS 4 +#define SUBGROUP_SIZE 64 + +// Macros are identical to the 2-output kernel — they accept `total_sum` as +// a parameter so we can call them twice (once per pair) against different +// accumulators against the same reg_b. +#define dequantize_block_acc_bcast_8_hi(total_sum, bits4, bits2, cs, y) \ + float8 shared_y; \ + shared_y = sub_group_broadcast(y, 0); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x000F) ) | ((bits2.s0 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y.s0; \ + total_sum.s0 += ((float)(((bits4.s0 & 0x00F0) >> 4) | ((bits2.s0 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y.s1; \ + total_sum.s0 += ((float)(((bits4.s0 & 0x0F00) >> 8) | ((bits2.s0 & 0x30) )) - 32.f) * cs.s0 * shared_y.s2; \ + total_sum.s0 += ((float)(((bits4.s0 & 0xF000) >> 12) | ((bits2.s0 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y.s3; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x000F) ) | ((bits2.s2 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y.s4; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x00F0) >> 4) | ((bits2.s2 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y.s5; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x0F00) >> 8) | ((bits2.s2 & 0x30) )) - 32.f) * cs.s0 * shared_y.s6; \ + total_sum.s0 += ((float)(((bits4.s2 & 0xF000) >> 12) | ((bits2.s2 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y.s7; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x000F) ) | ((bits2.s1 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y.s0; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x00F0) >> 4) | ((bits2.s1 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y.s1; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x0F00) >> 8) | ((bits2.s1 & 0x30) )) - 32.f) * cs.s2 * shared_y.s2; \ + total_sum.s1 += ((float)(((bits4.s1 & 0xF000) >> 12) | ((bits2.s1 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y.s3; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x000F) ) | ((bits2.s3 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y.s4; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x00F0) >> 4) | ((bits2.s3 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y.s5; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x0F00) >> 8) | ((bits2.s3 & 0x30) )) - 32.f) * cs.s2 * shared_y.s6; \ + total_sum.s1 += ((float)(((bits4.s3 & 0xF000) >> 12) | ((bits2.s3 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y.s7; \ + shared_y = sub_group_broadcast(y, 1); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x000F) ) | ((bits2.s4 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y.s0; \ + total_sum.s0 += ((float)(((bits4.s4 & 0x00F0) >> 4) | ((bits2.s4 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y.s1; \ + total_sum.s0 += ((float)(((bits4.s4 & 0x0F00) >> 8) | ((bits2.s4 & 0x30) )) - 32.f) * cs.s0 * shared_y.s2; \ + total_sum.s0 += ((float)(((bits4.s4 & 0xF000) >> 12) | ((bits2.s4 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y.s3; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x000F) ) | ((bits2.s6 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y.s4; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x00F0) >> 4) | ((bits2.s6 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y.s5; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x0F00) >> 8) | ((bits2.s6 & 0x30) )) - 32.f) * cs.s0 * shared_y.s6; \ + total_sum.s0 += ((float)(((bits4.s6 & 0xF000) >> 12) | ((bits2.s6 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y.s7; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x000F) ) | ((bits2.s5 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y.s0; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x00F0) >> 4) | ((bits2.s5 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y.s1; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x0F00) >> 8) | ((bits2.s5 & 0x30) )) - 32.f) * cs.s2 * shared_y.s2; \ + total_sum.s1 += ((float)(((bits4.s5 & 0xF000) >> 12) | ((bits2.s5 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y.s3; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x000F) ) | ((bits2.s7 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y.s4; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x00F0) >> 4) | ((bits2.s7 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y.s5; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x0F00) >> 8) | ((bits2.s7 & 0x30) )) - 32.f) * cs.s2 * shared_y.s6; \ + total_sum.s1 += ((float)(((bits4.s7 & 0xF000) >> 12) | ((bits2.s7 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y.s7; \ + +#define dequantize_block_acc_bcast_8_lo(total_sum, bits4, bits2, cs, y) \ + shared_y = sub_group_broadcast(y, 2); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x000F) ) | ((bits2.s0 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y.s0; \ + total_sum.s0 += ((float)(((bits4.s0 & 0x00F0) >> 4) | ((bits2.s0 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y.s1; \ + total_sum.s0 += ((float)(((bits4.s0 & 0x0F00) >> 8) | ((bits2.s0 & 0x30) )) - 32.f) * cs.s1 * shared_y.s2; \ + total_sum.s0 += ((float)(((bits4.s0 & 0xF000) >> 12) | ((bits2.s0 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y.s3; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x000F) ) | ((bits2.s2 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y.s4; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x00F0) >> 4) | ((bits2.s2 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y.s5; \ + total_sum.s0 += ((float)(((bits4.s2 & 0x0F00) >> 8) | ((bits2.s2 & 0x30) )) - 32.f) * cs.s1 * shared_y.s6; \ + total_sum.s0 += ((float)(((bits4.s2 & 0xF000) >> 12) | ((bits2.s2 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y.s7; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x000F) ) | ((bits2.s1 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y.s0; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x00F0) >> 4) | ((bits2.s1 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y.s1; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x0F00) >> 8) | ((bits2.s1 & 0x30) )) - 32.f) * cs.s3 * shared_y.s2; \ + total_sum.s1 += ((float)(((bits4.s1 & 0xF000) >> 12) | ((bits2.s1 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y.s3; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x000F) ) | ((bits2.s3 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y.s4; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x00F0) >> 4) | ((bits2.s3 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y.s5; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x0F00) >> 8) | ((bits2.s3 & 0x30) )) - 32.f) * cs.s3 * shared_y.s6; \ + total_sum.s1 += ((float)(((bits4.s3 & 0xF000) >> 12) | ((bits2.s3 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y.s7; \ + shared_y = sub_group_broadcast(y, 3); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x000F) ) | ((bits2.s4 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y.s0; \ + total_sum.s0 += ((float)(((bits4.s4 & 0x00F0) >> 4) | ((bits2.s4 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y.s1; \ + total_sum.s0 += ((float)(((bits4.s4 & 0x0F00) >> 8) | ((bits2.s4 & 0x30) )) - 32.f) * cs.s1 * shared_y.s2; \ + total_sum.s0 += ((float)(((bits4.s4 & 0xF000) >> 12) | ((bits2.s4 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y.s3; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x000F) ) | ((bits2.s6 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y.s4; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x00F0) >> 4) | ((bits2.s6 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y.s5; \ + total_sum.s0 += ((float)(((bits4.s6 & 0x0F00) >> 8) | ((bits2.s6 & 0x30) )) - 32.f) * cs.s1 * shared_y.s6; \ + total_sum.s0 += ((float)(((bits4.s6 & 0xF000) >> 12) | ((bits2.s6 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y.s7; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x000F) ) | ((bits2.s5 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y.s0; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x00F0) >> 4) | ((bits2.s5 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y.s1; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x0F00) >> 8) | ((bits2.s5 & 0x30) )) - 32.f) * cs.s3 * shared_y.s2; \ + total_sum.s1 += ((float)(((bits4.s5 & 0xF000) >> 12) | ((bits2.s5 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y.s3; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x000F) ) | ((bits2.s7 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y.s4; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x00F0) >> 4) | ((bits2.s7 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y.s5; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x0F00) >> 8) | ((bits2.s7 & 0x30) )) - 32.f) * cs.s3 * shared_y.s6; \ + total_sum.s1 += ((float)(((bits4.s7 & 0xF000) >> 12) | ((bits2.s7 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y.s7; \ + +#define dequantize_block_acc_bcast_1_hi(total_sum, bits4, bits2, cs, y) \ + float shared_y; \ + shared_y = sub_group_broadcast(y.s0, 0); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x000F) ) | ((bits2.s0 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x000F) ) | ((bits2.s1 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 0); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x00F0) >> 4) | ((bits2.s0 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x00F0) >> 4) | ((bits2.s1 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 0); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x0F00) >> 8) | ((bits2.s0 & 0x30) )) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x0F00) >> 8) | ((bits2.s1 & 0x30) )) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 0); \ + total_sum.s0 += ((float)(((bits4.s0 & 0xF000) >> 12) | ((bits2.s0 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0xF000) >> 12) | ((bits2.s1 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 0); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x000F) ) | ((bits2.s2 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x000F) ) | ((bits2.s3 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 0); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x00F0) >> 4) | ((bits2.s2 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x00F0) >> 4) | ((bits2.s3 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 0); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x0F00) >> 8) | ((bits2.s2 & 0x30) )) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x0F00) >> 8) | ((bits2.s3 & 0x30) )) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 0); \ + total_sum.s0 += ((float)(((bits4.s2 & 0xF000) >> 12) | ((bits2.s2 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0xF000) >> 12) | ((bits2.s3 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 1); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x000F) ) | ((bits2.s4 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x000F) ) | ((bits2.s5 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 1); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x00F0) >> 4) | ((bits2.s4 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x00F0) >> 4) | ((bits2.s5 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 1); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x0F00) >> 8) | ((bits2.s4 & 0x30) )) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x0F00) >> 8) | ((bits2.s5 & 0x30) )) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 1); \ + total_sum.s0 += ((float)(((bits4.s4 & 0xF000) >> 12) | ((bits2.s4 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0xF000) >> 12) | ((bits2.s5 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 1); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x000F) ) | ((bits2.s6 & 0x03) << 4)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x000F) ) | ((bits2.s7 & 0x03) << 4)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 1); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x00F0) >> 4) | ((bits2.s6 & 0x0C) << 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x00F0) >> 4) | ((bits2.s7 & 0x0C) << 2)) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 1); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x0F00) >> 8) | ((bits2.s6 & 0x30) )) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x0F00) >> 8) | ((bits2.s7 & 0x30) )) - 32.f) * cs.s2 * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 1); \ + total_sum.s0 += ((float)(((bits4.s6 & 0xF000) >> 12) | ((bits2.s6 & 0xC0) >> 2)) - 32.f) * cs.s0 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0xF000) >> 12) | ((bits2.s7 & 0xC0) >> 2)) - 32.f) * cs.s2 * shared_y; \ + +#define dequantize_block_acc_bcast_1_lo(total_sum, bits4, bits2, cs, y) \ + shared_y = sub_group_broadcast(y.s0, 2); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x000F) ) | ((bits2.s0 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x000F) ) | ((bits2.s1 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 2); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x00F0) >> 4) | ((bits2.s0 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x00F0) >> 4) | ((bits2.s1 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 2); \ + total_sum.s0 += ((float)(((bits4.s0 & 0x0F00) >> 8) | ((bits2.s0 & 0x30) )) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0x0F00) >> 8) | ((bits2.s1 & 0x30) )) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 2); \ + total_sum.s0 += ((float)(((bits4.s0 & 0xF000) >> 12) | ((bits2.s0 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s1 & 0xF000) >> 12) | ((bits2.s1 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 2); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x000F) ) | ((bits2.s2 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x000F) ) | ((bits2.s3 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 2); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x00F0) >> 4) | ((bits2.s2 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x00F0) >> 4) | ((bits2.s3 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 2); \ + total_sum.s0 += ((float)(((bits4.s2 & 0x0F00) >> 8) | ((bits2.s2 & 0x30) )) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0x0F00) >> 8) | ((bits2.s3 & 0x30) )) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 2); \ + total_sum.s0 += ((float)(((bits4.s2 & 0xF000) >> 12) | ((bits2.s2 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s3 & 0xF000) >> 12) | ((bits2.s3 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s0, 3); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x000F) ) | ((bits2.s4 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x000F) ) | ((bits2.s5 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s1, 3); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x00F0) >> 4) | ((bits2.s4 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x00F0) >> 4) | ((bits2.s5 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s2, 3); \ + total_sum.s0 += ((float)(((bits4.s4 & 0x0F00) >> 8) | ((bits2.s4 & 0x30) )) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0x0F00) >> 8) | ((bits2.s5 & 0x30) )) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s3, 3); \ + total_sum.s0 += ((float)(((bits4.s4 & 0xF000) >> 12) | ((bits2.s4 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s5 & 0xF000) >> 12) | ((bits2.s5 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s4, 3); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x000F) ) | ((bits2.s6 & 0x03) << 4)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x000F) ) | ((bits2.s7 & 0x03) << 4)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s5, 3); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x00F0) >> 4) | ((bits2.s6 & 0x0C) << 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x00F0) >> 4) | ((bits2.s7 & 0x0C) << 2)) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s6, 3); \ + total_sum.s0 += ((float)(((bits4.s6 & 0x0F00) >> 8) | ((bits2.s6 & 0x30) )) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0x0F00) >> 8) | ((bits2.s7 & 0x30) )) - 32.f) * cs.s3 * shared_y; \ + shared_y = sub_group_broadcast(y.s7, 3); \ + total_sum.s0 += ((float)(((bits4.s6 & 0xF000) >> 12) | ((bits2.s6 & 0xC0) >> 2)) - 32.f) * cs.s1 * shared_y; \ + total_sum.s1 += ((float)(((bits4.s7 & 0xF000) >> 12) | ((bits2.s7 & 0xC0) >> 2)) - 32.f) * cs.s3 * shared_y; \ + +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +// Q6K_O4_GLOBAL: read the (read-once-per-token, no-reuse) lm_head/embed weights +// from __global coalesced instead of image1d_buffer. The texture cache caps the +// streaming (no-reuse) lm_head read bandwidth; global coalesced reaches the +// higher rate the rest of the model gets. src1 (activation) stays an image (it IS reused via +// the cross-subgroup broadcast). +#ifdef Q6K_O4_GLOBAL +#define Q6K_O4_NAME kernel_gemv_noshuffle_q6_K_f32_o4_global +#define QL_ARG __global uint * src0_ql +#define QH_ARG __global half * src0_qh +#define RD_QL(b,i) (b[i]) +#define RD_QH(b,i) as_ushort(b[i]) +#else +#define Q6K_O4_NAME kernel_gemv_noshuffle_q6_K_f32_o4 +#define QL_ARG read_only image1d_buffer_t src0_ql +#define QH_ARG read_only image1d_buffer_t src0_qh +#define RD_QL(b,i) (read_imageui(b,i).x) +#define RD_QH(b,i) as_ushort(read_imageh(b,i).x) +#endif +kernel void Q6K_O4_NAME( + QL_ARG, + QH_ARG, + global half2 * src0_s, + global half2 * src0_d, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + int grp = get_local_id(1); + int gid = get_global_id(0); // 4-output-quad index + ushort slid = get_sub_group_local_id(); + + // Map quad index to the two pair-indices the existing 2-output access + // pattern uses (consecutive output pairs along ne01). NB: the two pairs are + // kept ADJACENT (gid*2, gid*2+1) on purpose -- a "stride-1" split (pairs + // ne01/4 apart) is slower because two distant cache-line streams have worse + // locality than the adjacent pair whose reads interleave into the same lines + // each iteration. + int gid_a = gid * 2; + int gid_b = gid * 2 + 1; + + int nb = ne00 / 32; + + uint4 reg_a_l_a, reg_a_l_b; + ushort4 reg_a_h_a, reg_a_h_b; + half2 reg_d_a, reg_d_b; + char4 reg_s_a, reg_s_b; + float8 reg_b; + + float2 total_sum_a = 0.0f; + float2 total_sum_b = 0.0f; + + int line_stride_a = ne01 / 2; + int block_stride_a = NSUBGROUPS * ne01; + + for (int k = grp; k < nb; k += NSUBGROUPS) { + reg_d_a = src0_d[gid_a + k/8 * line_stride_a]; + reg_d_b = src0_d[gid_b + k/8 * line_stride_a]; + reg_s_a = as_char4(src0_s[gid_a + k * line_stride_a]); + reg_s_b = as_char4(src0_s[gid_b + k * line_stride_a]); + // Precompute the loop-invariant combined scale (sub-block scale * super-block d) + // once per pair instead of re-multiplying it for every one of the 256 elements. + float4 cs_a = (float4)((float)reg_s_a.s0*(float)reg_d_a.s0, (float)reg_s_a.s1*(float)reg_d_a.s0, + (float)reg_s_a.s2*(float)reg_d_a.s1, (float)reg_s_a.s3*(float)reg_d_a.s1); + float4 cs_b = (float4)((float)reg_s_b.s0*(float)reg_d_b.s0, (float)reg_s_b.s1*(float)reg_d_b.s0, + (float)reg_s_b.s2*(float)reg_d_b.s1, (float)reg_s_b.s3*(float)reg_d_b.s1); + + if (slid < 4) { + reg_b.s0123 = read_imagef(src1, 0 + slid*2 + k*8); + reg_b.s4567 = read_imagef(src1, 1 + slid*2 + k*8); + } + + // Pair a (output rows gid_a*2, gid_a*2+1): read hi+lo then dequant + // both in one block so the `_lo` macro can see the `shared_y` that + // `_hi` declared. Pair b follows in its own block — fresh shared_y. + { + reg_a_l_a.s0 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*0); + reg_a_l_a.s1 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*1); + reg_a_l_a.s2 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*2); + reg_a_l_a.s3 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*3); + reg_a_h_a.s0 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*0); + reg_a_h_a.s1 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*1); + reg_a_h_a.s2 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*2); + reg_a_h_a.s3 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*3); +#ifdef VECTOR_SUB_GROUP_BROADCAT + dequantize_block_acc_bcast_8_hi(total_sum_a, as_ushort8(reg_a_l_a), as_uchar8(reg_a_h_a), cs_a, reg_b); +#else + dequantize_block_acc_bcast_1_hi(total_sum_a, as_ushort8(reg_a_l_a), as_uchar8(reg_a_h_a), cs_a, reg_b); +#endif + + reg_a_l_a.s0 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*4); + reg_a_l_a.s1 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*5); + reg_a_l_a.s2 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*6); + reg_a_l_a.s3 = RD_QL(src0_ql, gid_a + k*block_stride_a + line_stride_a*7); + reg_a_h_a.s0 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*4); + reg_a_h_a.s1 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*5); + reg_a_h_a.s2 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*6); + reg_a_h_a.s3 = RD_QH(src0_qh, gid_a + k*block_stride_a + line_stride_a*7); +#ifdef VECTOR_SUB_GROUP_BROADCAT + dequantize_block_acc_bcast_8_lo(total_sum_a, as_ushort8(reg_a_l_a), as_uchar8(reg_a_h_a), cs_a, reg_b); +#else + dequantize_block_acc_bcast_1_lo(total_sum_a, as_ushort8(reg_a_l_a), as_uchar8(reg_a_h_a), cs_a, reg_b); +#endif + } + + { + reg_a_l_b.s0 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*0); + reg_a_l_b.s1 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*1); + reg_a_l_b.s2 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*2); + reg_a_l_b.s3 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*3); + reg_a_h_b.s0 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*0); + reg_a_h_b.s1 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*1); + reg_a_h_b.s2 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*2); + reg_a_h_b.s3 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*3); +#ifdef VECTOR_SUB_GROUP_BROADCAT + dequantize_block_acc_bcast_8_hi(total_sum_b, as_ushort8(reg_a_l_b), as_uchar8(reg_a_h_b), cs_b, reg_b); +#else + dequantize_block_acc_bcast_1_hi(total_sum_b, as_ushort8(reg_a_l_b), as_uchar8(reg_a_h_b), cs_b, reg_b); +#endif + + reg_a_l_b.s0 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*4); + reg_a_l_b.s1 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*5); + reg_a_l_b.s2 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*6); + reg_a_l_b.s3 = RD_QL(src0_ql, gid_b + k*block_stride_a + line_stride_a*7); + reg_a_h_b.s0 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*4); + reg_a_h_b.s1 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*5); + reg_a_h_b.s2 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*6); + reg_a_h_b.s3 = RD_QH(src0_qh, gid_b + k*block_stride_a + line_stride_a*7); +#ifdef VECTOR_SUB_GROUP_BROADCAT + dequantize_block_acc_bcast_8_lo(total_sum_b, as_ushort8(reg_a_l_b), as_uchar8(reg_a_h_b), cs_b, reg_b); +#else + dequantize_block_acc_bcast_1_lo(total_sum_b, as_ushort8(reg_a_l_b), as_uchar8(reg_a_h_b), cs_b, reg_b); +#endif + } + } + + // Cross-subgroup reduce. Same shape as the 2-output kernel but with the + // pair-a and pair-b accumulators concatenated into a single float4. + local float4 reduce_lm[SUBGROUP_SIZE * 3]; + float4 acc = (float4)(total_sum_a.s0, total_sum_a.s1, total_sum_b.s0, total_sum_b.s1); + if (grp == 1) { reduce_lm[SUBGROUP_SIZE*0 + slid] = acc; } + if (grp == 2) { reduce_lm[SUBGROUP_SIZE*1 + slid] = acc; } + if (grp == 3) { reduce_lm[SUBGROUP_SIZE*2 + slid] = acc; } + + barrier(CLK_LOCAL_MEM_FENCE); + + if (grp == 0) { + acc += reduce_lm[SUBGROUP_SIZE*0 + slid]; + acc += reduce_lm[SUBGROUP_SIZE*1 + slid]; + acc += reduce_lm[SUBGROUP_SIZE*2 + slid]; + dst = (global float*)((global char*)dst + offsetd); + // The dispatch rounds ne01/4 up to the subgroup width, so the tail + // quads past the last row must not store (they wrote 128 rows past + // dst on every ne01 % 256 == 128 vocab, e.g. 151936). + if (gid * 4 + 3 < (uint)ne01) { + vstore4(acc, 0, &(dst[gid * 4])); + } + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_tiled.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_tiled.cl new file mode 100644 index 00000000..c5049f39 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32_tiled.cl @@ -0,0 +1,196 @@ +// Tiled-wide q6_K GEMV for the long-vocab lm_head/embed (decode path). +// +// Pairs with kernel_convert_block_q6_k_tiled_ns (cvt.cl): the weights are laid +// out CANONICALLY (6-bit code in element order e in [0,256)) and TILED by 64 +// output rows so the 64-thread lane group coalesces every weight load. Both the +// pack (convert) and the unpack (here) are owned by us — correct by construction +// against the reference ggml q6_K dequant, no bit-interleave reverse-engineering. +// +// One work-item produces one output row. A work-group is {64 lanes, 4 subgroups}: +// the 64 lanes cover the 64 rows of one tile (coalesced reads), the 4 subgroups +// split the K-blocks and reduce through __local at the end. +// +// Weights are read from __global (coalesced) rather than image1d_buffer: the +// lm_head is read once per token with no reuse, and the Adreno texture cache +// caps such a streaming read well below the coalesced-global rate +// (see opencl_q6k_gemv_o4_shipped / x2-90 roofline notes). + +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +#ifdef cl_qcom_reqd_sub_group_size +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#endif + +#define NSUBGROUPS 4 +#define TILE_ROWS 64 + +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q6_K_f32_tiled( + __global uint4 * src0_ql, // tiled: 8 uint4 granules / superblock + __global uint4 * src0_qh, // tiled: 4 uint4 granules / superblock + __global char * src0_s, // tiled: 16 chars / superblock + __global half * src0_d, // tiled: 1 half / superblock + read_only image1d_buffer_t src1, // activation (RGBA f32) + global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + int grp = get_local_id(1); // subgroup index 0..3 (splits K) + int row = get_global_id(0); // output row along ne01 + int rt = row / TILE_ROWS; + int rit = row % TILE_ROWS; + + int nb = ne00 / 256; // superblocks per row + + float acc = 0.0f; + + for (int sb = grp; sb < nb; sb += NSUBGROUPS) { + int tile_blk = rt * nb + sb; // ne02 == 1 for lm_head/embed + + // d + 16 scales for this (row, superblock) + float dval = (float)src0_d[tile_blk * TILE_ROWS + rit]; + __global char * sc = src0_s + (tile_blk * TILE_ROWS + rit) * 16; + + // 32 ql-uints (8 codes/uint) + 16 qh-uints (16 codes/uint) + uint ql[32]; + uint qh[16]; + #pragma unroll + for (int g = 0; g < 8; ++g) { + uint4 v = src0_ql[(tile_blk * 8 + g) * TILE_ROWS + rit]; + ql[g*4+0] = v.x; ql[g*4+1] = v.y; ql[g*4+2] = v.z; ql[g*4+3] = v.w; + } + #pragma unroll + for (int g = 0; g < 4; ++g) { + uint4 v = src0_qh[(tile_blk * 4 + g) * TILE_ROWS + rit]; + qh[g*4+0] = v.x; qh[g*4+1] = v.y; qh[g*4+2] = v.z; qh[g*4+3] = v.w; + } + + // dequant 256 codes in canonical e-order, MAC with activation. + int act_base = sb * 64; // activation float4 pixel base (256/4) + #pragma unroll + for (int e4 = 0; e4 < 64; ++e4) { + float4 a = read_imagef(src1, act_base + e4); + #pragma unroll + for (int t = 0; t < 4; ++t) { + int e = e4 * 4 + t; + uint low4 = (ql[e >> 3] >> ((e & 7) * 4)) & 0xF; + uint hi2 = (qh[e >> 4] >> ((e & 15) * 2)) & 0x3; + int code = (int)(low4 | (hi2 << 4)) - 32; + int sidx = ((e >> 7) << 3) + (((e >> 5) & 3) << 1) + ((e >> 4) & 1); + float scale = (float)sc[sidx] * dval; + float av = (t == 0) ? a.x : (t == 1) ? a.y : (t == 2) ? a.z : a.w; + acc += (float)code * scale * av; + } + } + } + + // reduce across the NSUBGROUPS subgroups (same rit, different K-subset) + local float reduce_lm[NSUBGROUPS * TILE_ROWS]; + reduce_lm[grp * TILE_ROWS + rit] = acc; + barrier(CLK_LOCAL_MEM_FENCE); + + if (grp == 0) { + float total = reduce_lm[0 * TILE_ROWS + rit] + + reduce_lm[1 * TILE_ROWS + rit] + + reduce_lm[2 * TILE_ROWS + rit] + + reduce_lm[3 * TILE_ROWS + rit]; + dst = (global float*)((global char*)dst + offsetd); + dst[row] = total; + } +} + +// Multi-column (N=3) variant of the tiled q6_K decode GEMV, for the speculative/ +// MTP VERIFY lm_head/embed (ne1=3 = 2 drafts + 1 bonus). Identical tiled weight +// layout + unpack as the ne1=1 kernel above; each WI computes 3 output columns, +// streaming the (large) lm_head weight ONCE per superblock and reusing it across +// the 3 verify activation columns (dequant once per code, MAC into 3 accs). This +// is the lm_head analogue of the per-layer mc3 GEMV; the multiply order matches +// the ne1=1 kernel, so each column is byte-identical to a standalone tiled GEMV. +#if defined(ADRENO_GPU) +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_gemv_noshuffle_q6_K_f32_tiled_mc3( + __global uint4 * src0_ql, + __global uint4 * src0_qh, + __global char * src0_s, + __global half * src0_d, + read_only image1d_buffer_t src1, + global float * dst, + ulong offsetd, + int ne00, + int ne01 +) { + int grp = get_local_id(1); + int row = get_global_id(0); + int rt = row / TILE_ROWS; + int rit = row % TILE_ROWS; + + int nb = ne00 / 256; + int col_stride = ne00 / 4; // activation float4 pixels per column + + float acc0 = 0.0f, acc1 = 0.0f, acc2 = 0.0f; + + for (int sb = grp; sb < nb; sb += NSUBGROUPS) { + int tile_blk = rt * nb + sb; + + float dval = (float)src0_d[tile_blk * TILE_ROWS + rit]; + __global char * sc = src0_s + (tile_blk * TILE_ROWS + rit) * 16; + + uint ql[32]; + uint qh[16]; + #pragma unroll + for (int g = 0; g < 8; ++g) { + uint4 v = src0_ql[(tile_blk * 8 + g) * TILE_ROWS + rit]; + ql[g*4+0] = v.x; ql[g*4+1] = v.y; ql[g*4+2] = v.z; ql[g*4+3] = v.w; + } + #pragma unroll + for (int g = 0; g < 4; ++g) { + uint4 v = src0_qh[(tile_blk * 4 + g) * TILE_ROWS + rit]; + qh[g*4+0] = v.x; qh[g*4+1] = v.y; qh[g*4+2] = v.z; qh[g*4+3] = v.w; + } + + int act_base = sb * 64; + #pragma unroll + for (int e4 = 0; e4 < 64; ++e4) { + float4 a0 = read_imagef(src1, 0*col_stride + act_base + e4); + float4 a1 = read_imagef(src1, 1*col_stride + act_base + e4); + float4 a2 = read_imagef(src1, 2*col_stride + act_base + e4); + #pragma unroll + for (int t = 0; t < 4; ++t) { + int e = e4 * 4 + t; + uint low4 = (ql[e >> 3] >> ((e & 7) * 4)) & 0xF; + uint hi2 = (qh[e >> 4] >> ((e & 15) * 2)) & 0x3; + int code = (int)(low4 | (hi2 << 4)) - 32; + int sidx = ((e >> 7) << 3) + (((e >> 5) & 3) << 1) + ((e >> 4) & 1); + float w = (float)code * ((float)sc[sidx] * dval); // dequant+scale once + float av0 = (t == 0) ? a0.x : (t == 1) ? a0.y : (t == 2) ? a0.z : a0.w; + float av1 = (t == 0) ? a1.x : (t == 1) ? a1.y : (t == 2) ? a1.z : a1.w; + float av2 = (t == 0) ? a2.x : (t == 1) ? a2.y : (t == 2) ? a2.z : a2.w; + acc0 += w * av0; + acc1 += w * av1; + acc2 += w * av2; + } + } + } + + local float4 reduce_lm[NSUBGROUPS * TILE_ROWS]; + reduce_lm[grp * TILE_ROWS + rit] = (float4)(acc0, acc1, acc2, 0.0f); + barrier(CLK_LOCAL_MEM_FENCE); + + if (grp == 0) { + float4 total = reduce_lm[0 * TILE_ROWS + rit] + + reduce_lm[1 * TILE_ROWS + rit] + + reduce_lm[2 * TILE_ROWS + rit] + + reduce_lm[3 * TILE_ROWS + rit]; + dst = (global float*)((global char*)dst + offsetd); + // dst column-major [ne01 rows x 3 cols]: (row, col) at col*ne01 + row + dst[0*ne01 + row] = total.x; + dst[1*ne01 + row] = total.y; + dst[2*ne01 + row] = total.z; + } +} diff --git a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q8_0_f32.cl b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q8_0_f32.cl index 09bae2d5..6f6d7425 100644 --- a/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q8_0_f32.cl +++ b/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q8_0_f32.cl @@ -118,6 +118,87 @@ elem = (char)((bits8.s7 & 0xFF000000) >> 24); \ total_sums += convert_int(elem) * scale * shared_y; \ +// ============================================================================ +// Split-K variant for small-M decode GEMVs. +// ---------------------------------------------------------------------------- +// The base kernel below puts one output row per lane and splits K only across +// the N_SIMDGROUP subgroups of a single workgroup, so M=512 yields M/64 = 8 +// workgroups -- half the compute units on a 16-CU X2 sit idle, and the kernel +// measures ~48 GB/s against the ~122 GB/s the larger projections reach in the +// same graph. Here each (kslice, subgroup) pair reduces a disjoint set of +// K-blocks into partial[kslice * M + row]; kernel_gemv_splitk_reduce_f32 (in +// gemv_noshuffle_q4_k_f32.cl) sums the slices. Same operand order within a +// slice as the base kernel; only the cross-slice grouping differs. +// +// Placed BEFORE the base kernel deliberately: on A6X no kernel may be defined +// after one that uses a subgroup builtin, or it silently miscompiles. +// ============================================================================ +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +__kernel void kernel_gemv_noshuffle_q8_0_f32_splitk( + __read_only image1d_buffer_t src0_q, // quantized A (weights) + global half * src0_d, // A scales + __read_only image1d_buffer_t src1, // B (activations) + global float * partial, // [ksplit * M], slice-major + int ne00, // K + int ne01) // M +{ + uint groupId = get_local_id(1); + uint gid = get_global_id(0); + ushort slid = get_sub_group_local_id(); + uint nsg = get_local_size(1); + uint ksplit = get_num_groups(1); + uint kslice = get_group_id(1); + + uint K = ne00; + uint M = ne01; + + uint LINE_STRIDE_A = M; + uint BLOCK_STRIDE_A = 8 * M; // physical, independent of the K-split + + __private uint8 regA; + __private half regS; + __private float8 regB; + __private float totalSum = (float)(0.0f); + + #pragma unroll 1 + for (uint k = kslice * nsg + groupId; k < (K / QK8_0); k += ksplit * nsg) { + regS = src0_d[gid + k * LINE_STRIDE_A]; + if (slid < 4) { + regB.s0123 = read_imagef(src1, (slid * 2 + k * 8)); + regB.s4567 = read_imagef(src1, (1 + slid * 2 + k * 8)); + } + regA.s0 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 0)).x; + regA.s1 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 1)).x; + regA.s2 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 2)).x; + regA.s3 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 3)).x; + regA.s4 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 4)).x; + regA.s5 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 5)).x; + regA.s6 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 6)).x; + regA.s7 = read_imageui(src0_q, (gid + k * BLOCK_STRIDE_A + LINE_STRIDE_A * 7)).x; + + dequantizeBlockAccum_ns_sgbroadcast_1(totalSum, regA, convert_float(regS), regB); + } + + // Intra-workgroup reduce across this K-slice's subgroups. Sized for + // nsg <= 8; the host never dispatches more. + __local float reduceLM[SIMDGROUP_WIDTH * 7]; + if (groupId > 0) { + reduceLM[SIMDGROUP_WIDTH * (groupId - 1) + slid] = totalSum; + } + barrier(CLK_LOCAL_MEM_FENCE); + if (groupId == 0) { + for (uint i = 0; i < nsg - 1; ++i) { + totalSum += reduceLM[SIMDGROUP_WIDTH * i + slid]; + } + // x-grid is padded to CEIL_DIV(M,wave)*wave; guard the tail rows. + if (gid < M) { + partial[kslice * M + gid] = totalSum; + } + } +} + #ifdef ADRENO_GPU REQD_SUBGROUP_SIZE_64 #endif diff --git a/ggml/src/ggml-opencl/kernels/glu.cl b/ggml/src/ggml-opencl/kernels/glu.cl index 059a4bbf..30bad00f 100644 --- a/ggml/src/ggml-opencl/kernels/glu.cl +++ b/ggml/src/ggml-opencl/kernels/glu.cl @@ -243,6 +243,71 @@ kernel void kernel_swiglu_oai( } } +//------------------------------------------------------------------------------ +// swiglu_clamp +//------------------------------------------------------------------------------ +kernel void kernel_swiglu_clamp( + global char * src0, + ulong offset0, + global char * src1, + ulong offset1, + global char * dst, + ulong offsetd, + ulong nb01, + ulong nb11, + int ne0, + ulong nb1, + int ne00_off, + int ne10_off, + float limit +) { + src0 = (global char*)((global char*)src0 + offset0); + src1 = (global char*)((global char*)src1 + offset1); + dst = (global char*)((global char*)dst + offsetd); + + global float * src0_row = (global float *) ((global char *) src0 + get_group_id(0)*nb01) + ne00_off; + global float * src1_row = (global float *) ((global char *) src1 + get_group_id(0)*nb11) + ne10_off; + global float * dst_row = (global float *) ((global char *) dst + get_group_id(0)*nb1); + + for (int i0 = get_local_id(0); i0 < ne0; i0 += get_local_size(0)) { + const float gate = min(src0_row[i0], limit); + const float up = clamp(src1_row[i0], -limit, limit); + + dst_row[i0] = gate / (1.0f + exp(-gate)) * up; + } +} + +kernel void kernel_swiglu_clamp_f16( + global char * src0, + ulong offset0, + global char * src1, + ulong offset1, + global char * dst, + ulong offsetd, + ulong nb01, + ulong nb11, + int ne0, + ulong nb1, + int ne00_off, + int ne10_off, + float limit +) { + src0 = (global char*)((global char*)src0 + offset0); + src1 = (global char*)((global char*)src1 + offset1); + dst = (global char*)((global char*)dst + offsetd); + + global half * src0_row = (global half *) ((global char *) src0 + get_group_id(0)*nb01) + ne00_off; + global half * src1_row = (global half *) ((global char *) src1 + get_group_id(0)*nb11) + ne10_off; + global half * dst_row = (global half *) ((global char *) dst + get_group_id(0)*nb1); + + for (int i0 = get_local_id(0); i0 < ne0; i0 += get_local_size(0)) { + const float gate = min((float) src0_row[i0], limit); + const float up = clamp((float) src1_row[i0], -limit, limit); + + dst_row[i0] = (half) (gate / (1.0f + exp(-gate)) * up); + } +} + //------------------------------------------------------------------------------ // geglu_erf //------------------------------------------------------------------------------ diff --git a/ggml/src/ggml-opencl/kernels/moe_add_id_glu.cl b/ggml/src/ggml-opencl/kernels/moe_add_id_glu.cl new file mode 100644 index 00000000..6a8e4fb1 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/moe_add_id_glu.cl @@ -0,0 +1,76 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +//------------------------------------------------------------------------------ +// add_id(gate) + add_id(up) + swiglu_oai, fused +// +// gpt-oss-class MoE FFNs run three full passes over the same +// [n_ff, n_expert_used, n_tokens] f32 tensor: a per-expert bias add on the gate +// matmul output, the same on the up matmul output, then swiglu_oai over the +// two. Both bias adds are in-place, so each costs a full read plus a full write +// of a tensor that is only read once more. Folding them into the swiglu pass +// leaves two reads and one write instead of six passes. +// +// Grouping matches kernel_add_id: group 0 = expert slot (i1), group 1 = token +// (i2). For a contiguous destination that addressing is identical to the flat +// row walk kernel_swiglu_oai uses, since row i1 + i2*ne1 sits at +// i1*nb1 + i2*ne1*nb1. +//------------------------------------------------------------------------------ +kernel void kernel_add_id_add_id_swiglu_oai( + global char * src_g, + ulong offset_g, + global char * src_gb, + ulong offset_gb, + global char * src_u, + ulong offset_u, + global char * src_ub, + ulong offset_ub, + global char * src_ids, + ulong offset_ids, + global char * dst, + ulong offsetd, + ulong nb01_g, + ulong nb02_g, + ulong nb01_u, + ulong nb02_u, + ulong nb11_g, + ulong nb11_u, + ulong nb21, + ulong nbd1, + ulong nbd2, + int ne0, + float limit, + float alpha +) { + src_g = (global char *)(src_g + offset_g); + src_gb = (global char *)(src_gb + offset_gb); + src_u = (global char *)(src_u + offset_u); + src_ub = (global char *)(src_ub + offset_ub); + src_ids = (global char *)(src_ids + offset_ids); + dst = (global char *)(dst + offsetd); + + const int i1 = get_group_id(0); + const int i2 = get_group_id(1); + + // The ids tensor is a view into a [n_expert, n_tokens] buffer, so its row + // stride is nb21 and the k selected ids are NOT contiguous per token. + const int i11 = *((global const int *) (src_ids + i1*sizeof(int) + i2*nb21)); + + global const float * g_row = (global const float *)(src_g + i1*nb01_g + i2*nb02_g); + global const float * u_row = (global const float *)(src_u + i1*nb01_u + i2*nb02_u); + global const float * gb_row = (global const float *)(src_gb + i11*nb11_g); + global const float * ub_row = (global const float *)(src_ub + i11*nb11_u); + global float * d_row = (global float *)(dst + i1*nbd1 + i2*nbd2); + + for (int i0 = get_local_id(0); i0 < ne0; i0 += get_local_size(0)) { + float x0 = g_row[i0] + gb_row[i0]; + float x1 = u_row[i0] + ub_row[i0]; + + x0 = min(x0, limit); + x1 = max(min(x1, limit), -limit); + + float out_glu = x0 / (1.0f + exp(-x0 * alpha)); + out_glu = out_glu * (1.0f + x1); + + d_row[i0] = out_glu; + } +} diff --git a/ggml/src/ggml-opencl/kernels/moe_combine.cl b/ggml/src/ggml-opencl/kernels/moe_combine.cl index c195f147..acd08dbe 100644 --- a/ggml/src/ggml-opencl/kernels/moe_combine.cl +++ b/ggml/src/ggml-opencl/kernels/moe_combine.cl @@ -8,6 +8,49 @@ // buffer and the k-1 elementwise add round-trips). Vectorized float4 over rows. // strides e1/e2/w1/w2/d1 are in ELEMENTS (floats). +// Same weighted sum, with the per-expert bias add folded in. +// +// The MoE down projection's bias is applied by an in-place add_id whose only +// consumer is this combine, so it costs a full read plus a full write of a +// tensor that is read once more immediately afterwards. Reading the raw matmul +// output here and adding the bias row while it is already in registers removes +// that pass. Kept as a separate kernel so the unfused path is untouched. +__kernel void kernel_moe_combine_bias_f32( + __global const char * e_buf, ulong off_e, + __global const char * w_buf, ulong off_w, + __global const char * b_buf, ulong off_b, // per-expert bias rows + __global const char * i_buf, ulong off_i, // expert ids + __global char * d_buf, ulong off_d, + int n_embd4, // n_embd / 4 + int k, // n_expert_used + int n_tokens, + uint e1, uint e2, // experts strides (elements): per-expert, per-token + uint w1, uint w2, // weights strides (elements) + uint d1, // dst per-token stride (elements) + ulong nb_b1, // bias row stride (bytes) + ulong nb_i1) // ids row stride (bytes) - ids is a view, not packed +{ + const uint r4 = get_global_id(0); + const uint tok = get_global_id(1); + if (r4 >= (uint)n_embd4 || tok >= (uint)n_tokens) return; + + __global const float * E = (__global const float *)(e_buf + off_e) + tok*e2 + r4*4u; + __global const float * W = (__global const float *)(w_buf + off_w) + tok*w2; + __global const char * B = b_buf + off_b; + __global const char * I = i_buf + off_i + (ulong)tok*nb_i1; + + float4 acc = (float4)(0.0f); + for (int e = 0; e < k; ++e) { + const int i11 = *((__global const int *)(I + (ulong)e*sizeof(int))); + __global const float * Brow = (__global const float *)(B + (ulong)i11*nb_b1) + r4*4u; + const float4 v = vload4(0, E + (uint)e*e1) + vload4(0, Brow); + acc = mad(v, (float4)(W[(uint)e*w1]), acc); + } + + __global float * D = (__global float *)(d_buf + off_d) + tok*d1 + r4*4u; + vstore4(acc, 0, D); +} + __kernel void kernel_moe_combine_f32( __global const char * e_buf, ulong off_e, __global const char * w_buf, ulong off_w, diff --git a/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl b/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl index e6295c81..2f5c110b 100644 --- a/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl +++ b/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl @@ -20,11 +20,13 @@ kernel void kernel_moe_reorder_b( uint router_idx = router[post_router_idx]; - float4 out = (float4)(0); - if (router_idx != 0xFFFFFFFF) { - ushort activation_idx = router_idx / map_ratio; - out = src[activation_idx * K / 4 + k_4]; + // Padded slots need not be written at all. The MoE GEMMs accumulate per output + // column and scatter only the real columns, so whatever sits in a padded slot + // never reaches dst + if (router_idx == 0xFFFFFFFF) { + return; } - dst[post_router_idx * K / 4 + k_4] = out; + ushort activation_idx = router_idx / map_ratio; + dst[post_router_idx * K / 4 + k_4] = src[activation_idx * K / 4 + k_4]; } diff --git a/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl b/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl index d9703429..d52d11aa 100644 --- a/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl +++ b/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl @@ -68,6 +68,79 @@ __kernel void kernel_moe_scatter( emap[tile_idx] = val; } +// Deterministic replacement for kernel_moe_scatter. +// +// kernel_moe_scatter takes each token's slot from atomic_inc(slot_counter[expert]), +// so the token -> slot packing inside an expert depends on which work-item wins the +// atomic and changes from run to run. The ragged prefill GEMM path is sensitive to +// that packing (the non-ragged path is not, since its padded slots alias slot 0 and +// are overwritten last), which makes MoE prompt processing non-reproducible: the same +// binary on the same prompt returns one of several outputs. +// +// Here the slot is the token's rank in flat (n, k) order among the tokens routed to +// the same expert - a fixed function of the routing input. One workgroup per expert +// walks the flat routing list in blocks of 64 and ranks its own tokens with a +// workgroup scan, carrying a running count between blocks. Cost is one pass over the +// routing list per expert; the list is a few KiB and stays in cache. +__kernel void kernel_moe_scatter_stable( + __global const int * input, + __global int * post_router, + __global ushort * emap, + __global const int * tile_offset, + int N, + int topK, + uint n_experts +) { + const int e = get_group_id(1); + const int lid = get_local_id(0); + const int M = N * topK; + + __local int scan[64]; + __local int running; + + if (lid == 0) { + running = 0; + } + barrier(CLK_LOCAL_MEM_FENCE); + + for (int base = 0; base < M; base += 64) { + const int j = base + lid; + + int pred = 0; + if (j < M) { + const int n = j / topK; + const int k = j - n * topK; + pred = (input[n * (int)n_experts + k] == e) ? 1 : 0; + } + + scan[lid] = pred; + barrier(CLK_LOCAL_MEM_FENCE); + + // Hillis-Steele inclusive scan over the 64 lanes + for (int off = 1; off < 64; off <<= 1) { + int add = (lid >= off) ? scan[lid - off] : 0; + barrier(CLK_LOCAL_MEM_FENCE); + scan[lid] += add; + barrier(CLK_LOCAL_MEM_FENCE); + } + + if (pred) { + const int local_slot = running + (scan[lid] - 1); // exclusive rank + const int tile_idx = tile_offset[e] + (local_slot >> 5); + const int lane = local_slot & 31; + + post_router[tile_idx * 32 + lane] = j; + emap[tile_idx] = (ushort)e; + } + + barrier(CLK_LOCAL_MEM_FENCE); + if (lid == 63) { + running += scan[63]; + } + barrier(CLK_LOCAL_MEM_FENCE); + } +} + __kernel void kernel_moe_fill( __global int * post_router, __global int * total_tiles, diff --git a/ggml/src/ggml-opencl/kernels/mul_mm_f32_f32_l4_lm.cl b/ggml/src/ggml-opencl/kernels/mul_mm_f32_f32_l4_lm.cl index d7d5ba64..9dc9862b 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mm_f32_f32_l4_lm.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mm_f32_f32_l4_lm.cl @@ -145,3 +145,52 @@ kernel void kernel_mul_mm_f32_f32_l4_lm( } } } + +// Multi-column f32 GEMV for the small-N (spec/MTP verify) batch. The tiled GEMM +// above always computes a full BM x BN = 64 x 64 output tile, so at ne11=3 with a +// skinny weight (e.g. GDN ssm_alpha/ssm_beta, M=32) it launches ONE under-occupied +// workgroup at ~2.3% tile utilization. This kernel assigns one 64-thread workgroup +// per output element (m,n): the 64 threads split the K reduction (float4) and +// tree-reduce in __local (no subgroup ops -> portable). ne01*ne11 workgroups. +// Weight row is re-read per column (N small -> negligible). Summation order differs +// from the tiled GEMM (lane-strided + tree) -> f32-exact-ish, not bit-identical. +kernel void kernel_gemv_f32_f32_mc( + global float * src0, ulong offset0, // weight: row m at m*stride_a (elements) + global float * src1, ulong offset1, // activations: col n at n*stride_b + global float * dst, ulong offsetd, // dst [M x N] col-major: (m,n) at n*stride_d+m + int ne00, // K + int ne01, // M + int ne11, // N + int stride_a, // weight row stride (elements) = K + int stride_b, // activation col stride (elements) = K + int stride_d) // dst column stride (elements) = M +{ + src0 = (global float*)((global char*)src0 + offset0); + src1 = (global float*)((global char*)src1 + offset1); + dst = (global float*)((global char*)dst + offsetd); + + uint lane = get_local_id(0); // 0..63 + uint out = get_global_id(1); // 0 .. ne01*ne11 - 1 + uint m = out % (uint)ne01; + uint n = out / (uint)ne01; + + global float4 * wrow = (global float4*)(src0 + (ulong)m * (uint)stride_a); + global float4 * xcol = (global float4*)(src1 + (ulong)n * (uint)stride_b); + uint k4 = (uint)ne00 >> 2; + + float acc = 0.0f; + for (uint k = lane; k < k4; k += 64) { + float4 w = wrow[k]; + float4 x = xcol[k]; + acc += w.s0*x.s0 + w.s1*x.s1 + w.s2*x.s2 + w.s3*x.s3; + } + + local float red[64]; + red[lane] = acc; + barrier(CLK_LOCAL_MEM_FENCE); + for (uint s = 32; s > 0; s >>= 1) { + if (lane < s) red[lane] += red[lane + s]; + barrier(CLK_LOCAL_MEM_FENCE); + } + if (lane == 0) dst[(ulong)n * (uint)stride_d + m] = red[0]; +} diff --git a/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl b/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl index 2235b1ae..a9c649a5 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl @@ -1,13 +1,23 @@ #pragma OPENCL EXTENSION cl_khr_fp16 : enable +#ifdef cl_intel_required_subgroup_size +#define INTEL_GPU 1 +#endif + #define LOAD_VEC_A 4 #define LOAD_VEC_B 4 #define BM 64 #define BN 64 #define BK 32 +#ifdef INTEL_GPU +// Intel Xe iGPU: 8x8 microtile (WG = BM*BN/(TM*TN) = 64) — ~+12% pp512 vs 4x8 +#define TM 8 +#define TN 8 +#else #define TM 4 #define TN 8 +#endif kernel void kernel_mul_mm_q4_k_f32_l4_lm( global uchar4 * src0_q, diff --git a/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl b/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl index 8e191f57..a343b5c4 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl @@ -1,13 +1,23 @@ #pragma OPENCL EXTENSION cl_khr_fp16 : enable +#ifdef cl_intel_required_subgroup_size +#define INTEL_GPU 1 +#endif + #define LOAD_VEC_A 4 #define LOAD_VEC_B 4 #define BM 64 #define BN 64 #define BK 32 +#ifdef INTEL_GPU +// Intel Xe iGPU: 8x8 microtile (WG=64) +#define TM 8 +#define TN 8 +#else #define TM 4 #define TN 8 +#endif kernel void kernel_mul_mm_q5_k_f32_l4_lm( global uchar4 * src0_q, diff --git a/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_mrow.cl b/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_mrow.cl new file mode 100644 index 00000000..9a7627cf --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_mrow.cl @@ -0,0 +1,306 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +#ifdef cl_intel_subgroups +#pragma OPENCL EXTENSION cl_intel_subgroups : enable +#else +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif + +#ifdef cl_intel_required_subgroup_size +#pragma OPENCL EXTENSION cl_intel_required_subgroup_size : enable +#define INTEL_GPU 1 +#define REQD_SUBGROUP_SIZE_16 __attribute__((intel_reqd_sub_group_size(16))) +#define REQD_SUBGROUP_SIZE_32 __attribute__((intel_reqd_sub_group_size(32))) +#elif defined(cl_qcom_reqd_sub_group_size) +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define ADRENO_GPU 1 +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#define REQD_SUBGROUP_SIZE_128 __attribute__((qcom_reqd_sub_group_size("full"))) +#endif + +// Multi-row f16xf32 GEMV for the DECODE path (single token, ne11*ne12 small). +// The legacy kernel_mul_mat_f16_f32_1row runs ONE 64-lane subgroup per workgroup = +// one output row per WG, which caps memory-level parallelism at roughly half of +// LPDDR5x peak. This variant packs MROW subgroups per workgroup, each +// computing a distinct output row, so a WG keeps 64*MROW loads in flight. The +// activation column y (shared by every output row) is staged into __local ONCE per +// WG and reused across the MROW rows, cutting redundant activation reads. Used for +// the f16 attention projections (Q/K/V/O) and lm_head, which dominate decode. +// Numerically equivalent to _1row (same f16->f32 widening, same float4 partial sums, +// same subgroup-reduce order), so byte-identical to the per-op path. + +#define MROW 16 + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_mul_mat_f16_f32_mrow( + global char * src0, + ulong offset0, + global char * src1, + ulong offset1, + global float * dst, + ulong offsetd, + int ne00, + int ne01, + int ne02, + ulong nb00, + ulong nb01, + ulong nb02, + ulong nb03, + int ne10, + int ne11, + int ne12, + ulong nb10, + ulong nb11, + ulong nb12, + ulong nb13, + int ne0, + int ne1, + int r2, + int r3, + __local float * ysh +) { + src0 = (global char*)((global char*)src0 + offset0); + src1 = (global char*)((global char*)src1 + offset1); + dst = (global float*)((global char*)dst + offsetd); + + int r0 = get_group_id(0) * MROW + get_local_id(1); // output row + int r1 = get_group_id(1); // token (ne11) + int im = get_group_id(2); + int lid = get_sub_group_local_id(); // 0..63 + int nsg = get_local_size(1); // == MROW + + int i12 = im % ne12; + int i13 = im / ne12; + + ulong offset_src1 = r1*nb11 + (i12)*nb12 + (i13)*nb13; + global float * y = (global float *) (src1 + offset_src1); + + // Cooperatively stage the activation column (ne00 floats) into __local once per + // WG and reuse across the MROW rows. Staging is the actual win here: dropping it + // (each subgroup re-reading y from global) regresses below the 1-row kernel. + for (int i = get_local_id(1)*get_sub_group_size() + lid; i < ne00; i += nsg*get_sub_group_size()) { + ysh[i] = y[i]; + } + barrier(CLK_LOCAL_MEM_FENCE); + + if (r0 >= ne01) { + return; + } + + ulong offset_src0 = r0*nb01 + (i12/r2)*nb02 + (i13/r3)*nb03; + global half * x = (global half *) (src0 + offset_src0); + + // The vector path below casts the row pointer to half4, which must be 8-byte aligned. + // A row address is r0*nb01 + ..., and a permuted or strided src0 leaves nb01/nb02/nb03 + // unconstrained -- ne00 % 4 == 0 bounds the element count per row, not the byte stride + // between rows. Take the vector path only when this work-item's row is actually + // aligned; the scalar loop below has no such requirement. + const bool row_aligned = (((ulong) x) & 7) == 0; + + float sumf = 0.0f; + if (ne00 < 128 || !row_aligned) { + for (int i = lid; i < ne00; i += get_sub_group_size()) { + sumf += (float) x[i] * ysh[i]; + } + float all_sum = sub_group_reduce_add(sumf); + if (lid == 0) { + dst[im*ne1*ne0 + r1*ne0 + r0] = all_sum; + } + } else { + global half4 * x4 = (global half4 *) x; + __local float4 * ysh4 = (__local float4 *) ysh; + for (int i = lid; i < ne00/4; i += get_sub_group_size()) { + float4 yv = ysh4[i]; + sumf += (float) x4[i].s0 * yv.s0; + sumf += (float) x4[i].s1 * yv.s1; + sumf += (float) x4[i].s2 * yv.s2; + sumf += (float) x4[i].s3 * yv.s3; + } + float all_sum = sub_group_reduce_add(sumf); + if (lid == 0) { + for (int i = 4*(ne00/4); i < ne00; ++i) { + all_sum += (float) x[i] * ysh[i]; + } + dst[im*ne1*ne0 + r1*ne0 + r0] = all_sum; + } + } +} + +// Register-blocked variant: each 64-lane subgroup accumulates RPT consecutive +// output rows instead of one. The staged activation is reused across all RPT rows, +// and each lane keeps RPT independent weight loads in flight per column step -> +// more memory-level parallelism on the streaming f16 weight read (the BW limiter), +// plus RPT fewer staging barriers per output row. Per-row reduction order is +// identical to _mrow, so byte-identical to the per-op path. Dispatch guarantees +// ne00 >= 128 and ne00 % 4 == 0, so only the half4 path is needed (no tail). +#define MROW_RB_BODY(RPT) \ + src0 = (global char*)((global char*)src0 + offset0); \ + src1 = (global char*)((global char*)src1 + offset1); \ + dst = (global float*)((global char*)dst + offsetd); \ + int r0b = (get_group_id(0) * get_local_size(1) + get_local_id(1)) * (RPT); \ + int r1 = get_group_id(1); \ + int im = get_group_id(2); \ + int lid = get_sub_group_local_id(); \ + int nsg = get_local_size(1); \ + int i12 = im % ne12; \ + int i13 = im / ne12; \ + ulong off_y = r1*nb11 + i12*nb12 + i13*nb13; \ + global float * y = (global float *) (src1 + off_y); \ + for (int i = get_local_id(1)*get_sub_group_size() + lid; i < ne00; \ + i += nsg*get_sub_group_size()) { \ + ysh[i] = y[i]; \ + } \ + barrier(CLK_LOCAL_MEM_FENCE); \ + __local float4 * ysh4 = (__local float4 *) ysh; \ + global half4 * xr[RPT]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + int row = r0b + rr; \ + if (row > ne01 - 1) row = ne01 - 1; \ + xr[rr] = (global half4 *) (src0 + (ulong)row*nb01 + (i12/r2)*nb02 + (i13/r3)*nb03); \ + } \ + float sumf[RPT]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) sumf[rr] = 0.0f; \ + for (int i = lid; i < ne00/4; i += get_sub_group_size()) { \ + float4 yv = ysh4[i]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + half4 xv = xr[rr][i]; \ + sumf[rr] += (float) xv.s0 * yv.s0 + (float) xv.s1 * yv.s1 \ + + (float) xv.s2 * yv.s2 + (float) xv.s3 * yv.s3; \ + } \ + } \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + float s = sub_group_reduce_add(sumf[rr]); \ + int row = r0b + rr; \ + if (lid == 0 && row < ne01) { \ + dst[im*ne1*ne0 + r1*ne0 + row] = s; \ + } \ + } + +// half8 (128-bit) load variant: Adreno's load/store unit issues 128-bit +// transactions, so half4 (64-bit) loads may leave the load path half-idle. This +// processes 8 weight elements per lane per step via half8. Accumulation groups +// elements in 8s rather than 4s, so it is NOT bit-identical to _1row (float add is +// non-associative) -- experimental BW probe, gate on ne00 % 8 == 0. +#define MROW_H8_BODY(RPT) \ + src0 = (global char*)((global char*)src0 + offset0); \ + src1 = (global char*)((global char*)src1 + offset1); \ + dst = (global float*)((global char*)dst + offsetd); \ + int r0b = (get_group_id(0) * get_local_size(1) + get_local_id(1)) * (RPT); \ + int r1 = get_group_id(1); \ + int im = get_group_id(2); \ + int lid = get_sub_group_local_id(); \ + int nsg = get_local_size(1); \ + int i12 = im % ne12; \ + int i13 = im / ne12; \ + ulong off_y = r1*nb11 + i12*nb12 + i13*nb13; \ + global float * y = (global float *) (src1 + off_y); \ + for (int i = get_local_id(1)*get_sub_group_size() + lid; i < ne00; \ + i += nsg*get_sub_group_size()) { \ + ysh[i] = y[i]; \ + } \ + barrier(CLK_LOCAL_MEM_FENCE); \ + __local float4 * ysh4 = (__local float4 *) ysh; \ + global half8 * xr[RPT]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + int row = r0b + rr; \ + if (row > ne01 - 1) row = ne01 - 1; \ + xr[rr] = (global half8 *) (src0 + (ulong)row*nb01 + (i12/r2)*nb02 + (i13/r3)*nb03); \ + } \ + float sumf[RPT]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) sumf[rr] = 0.0f; \ + for (int i = lid; i < ne00/8; i += get_sub_group_size()) { \ + float4 y0 = ysh4[2*i]; \ + float4 y1 = ysh4[2*i + 1]; \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + half8 xv = xr[rr][i]; \ + sumf[rr] += (float) xv.s0 * y0.s0 + (float) xv.s1 * y0.s1 \ + + (float) xv.s2 * y0.s2 + (float) xv.s3 * y0.s3 \ + + (float) xv.s4 * y1.s0 + (float) xv.s5 * y1.s1 \ + + (float) xv.s6 * y1.s2 + (float) xv.s7 * y1.s3; \ + } \ + } \ + _Pragma("unroll") \ + for (int rr = 0; rr < (RPT); ++rr) { \ + float s = sub_group_reduce_add(sumf[rr]); \ + int row = r0b + rr; \ + if (lid == 0 && row < ne01) { \ + dst[im*ne1*ne0 + r1*ne0 + row] = s; \ + } \ + } + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_mul_mat_f16_f32_mrow_h8( + global char * src0, ulong offset0, + global char * src1, ulong offset1, + global float * dst, ulong offsetd, + int ne00, int ne01, int ne02, + ulong nb00, ulong nb01, ulong nb02, ulong nb03, + int ne10, int ne11, int ne12, + ulong nb10, ulong nb11, ulong nb12, ulong nb13, + int ne0, int ne1, int r2, int r3, + __local float * ysh +) { + MROW_H8_BODY(1) +} + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_mul_mat_f16_f32_mrow_h8r2( + global char * src0, ulong offset0, + global char * src1, ulong offset1, + global float * dst, ulong offsetd, + int ne00, int ne01, int ne02, + ulong nb00, ulong nb01, ulong nb02, ulong nb03, + int ne10, int ne11, int ne12, + ulong nb10, ulong nb11, ulong nb12, ulong nb13, + int ne0, int ne1, int r2, int r3, + __local float * ysh +) { + MROW_H8_BODY(2) +} + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_mul_mat_f16_f32_mrow_r2( + global char * src0, ulong offset0, + global char * src1, ulong offset1, + global float * dst, ulong offsetd, + int ne00, int ne01, int ne02, + ulong nb00, ulong nb01, ulong nb02, ulong nb03, + int ne10, int ne11, int ne12, + ulong nb10, ulong nb11, ulong nb12, ulong nb13, + int ne0, int ne1, int r2, int r3, + __local float * ysh +) { + MROW_RB_BODY(2) +} + +#ifdef ADRENO_GPU +REQD_SUBGROUP_SIZE_64 +#endif +kernel void kernel_mul_mat_f16_f32_mrow_r4( + global char * src0, ulong offset0, + global char * src1, ulong offset1, + global float * dst, ulong offsetd, + int ne00, int ne01, int ne02, + ulong nb00, ulong nb01, ulong nb02, ulong nb03, + int ne10, int ne11, int ne12, + ulong nb10, ulong nb11, ulong nb12, ulong nb13, + int ne0, int ne1, int r2, int r3, + __local float * ysh +) { + MROW_RB_BODY(4) +} diff --git a/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl b/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl index 70391866..5316bd36 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl @@ -40,7 +40,7 @@ typedef struct { #undef N_SIMDWIDTH #ifdef INTEL_GPU -#define N_DST 4 // number of rows each SIMD group works on +#define N_DST 16 // number of rows each SIMD group works on (Intel: 8->16, 2x further activation reuse; 32 spills registers) #define N_SIMDGROUP 1 // number of SIMD groups in a thread group #define N_SIMDWIDTH 16 // SIMD group size #elif defined (ADRENO_GPU) diff --git a/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl b/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl index 6020364b..ab2e1fab 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl @@ -38,7 +38,7 @@ typedef struct { #undef N_SIMDWIDTH #ifdef INTEL_GPU -#define N_DST 4 +#define N_DST 8 // Intel: 4->8 for 2x activation reuse (see mul_mv_q4_k_f32_flat.cl) #define N_SIMDGROUP 1 #define N_SIMDWIDTH 16 #elif defined(ADRENO_GPU) diff --git a/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl b/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl index 57b90c05..2cca5335 100644 --- a/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl +++ b/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl @@ -28,6 +28,13 @@ #define QK_K 256 +// ADRENO_OLD_COMPILER is defined by the host (-D) only for the Adreno E031 +// compilers older than E031.45, which miscompile several constructs this kernel +// used (confirmed on E031.38 and E031.41; E031.45 is clean). Every other +// compiler -- newer E031, E17, DX, Intel, and every non-Adreno device that +// builds this program -- takes the #else branches, which are the original +// source: the workarounds below cost ~13% on the q6_K flat n=1 GEMV where they +// are not needed. inline float block_q_6_K_dot_y_flat( global uchar * blk_ql, global uchar * blk_qh, @@ -37,6 +44,9 @@ inline float block_q_6_K_dot_y_flat( int ip, int is, int l0, +#if defined(ADRENO_OLD_COMPILER) + int dbg, +#endif float4 y0, float4 y1, float4 y2, @@ -48,10 +58,40 @@ inline float block_q_6_K_dot_y_flat( global uchar * q1 = blk_ql + ib*128 + q_offset_l; global uchar * q2 = q1 + QK_K/8; global uchar * qh = blk_qh + ib*64 + q_offset_h; - global char * sc = blk_scales + ib*16 + is; float dall = blk_d[ib]; +#if defined(ADRENO_OLD_COMPILER) + // The vectorized dequant (int4/float4 bit-ops, convert_*4, dot()) and vload4 + // are miscompiled here -> garbage weights. Reconstruct the 6-bit weights and + // take the dot product scalar. q4_K/q5_K flat already use scalar paths, which + // is why q6_K was the only flat GEMV that failed. + // Scales are SIGNED int8; read as uchar and sign-extend arithmetically so the + // result does not depend on whether the compiler treats `char` as signed. + global uchar * sc = (global uchar *)(blk_scales + ib*16 + is); + + int s0 = (int)sc[0] - 256*(sc[0] >> 7); + int s2 = (int)sc[2] - 256*(sc[2] >> 7); + int s4 = (int)sc[4] - 256*(sc[4] >> 7); + int s6 = (int)sc[6] - 256*(sc[6] >> 7); + + // one 6-bit weight: low/high nibble of a ql byte OR'd with a 2-bit qh plane + // (plane p in {0,1,2,3} selects qh bits 2p..2p+1) placed at bits 4-5, minus 32. + #define Q6W(qb, sh, hb, p) ((float)((((int)(qb) >> (sh)) & 15) | ((((int)(hb) >> (2*(p))) & 3) << 4)) - 32.f) + + float d0 = y0.s0*Q6W(q1[0],0,qh[0],0) + y0.s1*Q6W(q1[1],0,qh[1],0) + y0.s2*Q6W(q1[2],0,qh[2],0) + y0.s3*Q6W(q1[3],0,qh[3],0); + float d1 = y1.s0*Q6W(q2[0],0,qh[0],1) + y1.s1*Q6W(q2[1],0,qh[1],1) + y1.s2*Q6W(q2[2],0,qh[2],1) + y1.s3*Q6W(q2[3],0,qh[3],1); + float d2 = y2.s0*Q6W(q1[0],4,qh[0],2) + y2.s1*Q6W(q1[1],4,qh[1],2) + y2.s2*Q6W(q1[2],4,qh[2],2) + y2.s3*Q6W(q1[3],4,qh[3],2); + float d3 = y3.s0*Q6W(q2[0],4,qh[0],3) + y3.s1*Q6W(q2[1],4,qh[1],3) + y3.s2*Q6W(q2[2],4,qh[2],3) + y3.s3*Q6W(q2[3],4,qh[3],3); + #undef Q6W + + if (dbg) printf("HELPER dall=%f s=[%d %d %d %d] d=[%f %f %f %f] ql0=%d qh0=%d y00=%f\n", + dall, s0, s2, s4, s6, d0, d1, d2, d3, (int)q1[0], (int)qh[0], y0.s0); + + return dall * (d0 * s0 + d1 * s2 + d2 * s4 + d3 * s6); +#else + global char * sc = blk_scales + ib*16 + is; + // Vectorized loads: 3 uchar4 weight loads instead of 12 scalar byte reads. // q_offset_l/h are 4-aligned, so these are aligned vector loads. uchar4 q1v = vload4(0, q1); @@ -72,6 +112,7 @@ inline float block_q_6_K_dot_y_flat( return dall * (dot(y0, w0) * sc[0] + dot(y1, w1) * sc[2] + dot(y2, w2) * sc[4] + dot(y3, w3) * sc[6]); +#endif } #undef N_DST @@ -113,6 +154,11 @@ kernel void kernel_mul_mv_q6_K_f32_flat( int ne1, int r2, int r3 +#if defined(ADRENO_OLD_COMPILER) + , + uchar q6k_mask // runtime 0xFF; the host passes it so the compiler cannot + // constant-fold the printf guards below into nothing +#endif ) { src1 = (global float*)((global char*)src1 + offset1); dst = (global float*)((global char*)dst + offsetd); @@ -128,6 +174,22 @@ kernel void kernel_mul_mv_q6_K_f32_flat( int first_row = (N_SIMDGROUP * r0 + get_sub_group_id()) * N_DST; +#if defined(ADRENO_OLD_COMPILER) + // 64-bit `ulong` integer arithmetic is miscompiled here -> the base-pointer byte + // offsets came out wrong, so EVERY weight/scale read hit the wrong address. This + // was the primary cause of the q6_K flat failure (q5_K uses int offsets and is + // unaffected). Compute the block index in `int` and widen to `ulong` only inside + // the pointer expression: the byte offset stays 64-bit, but there is no ulong + // arithmetic chain to miscompile. The int index would overflow past ~2^31 blocks, + // which no realistic weight reaches -- but that is a narrowing, so keep it off the + // conformant path, which retains full ulong arithmetic. + int offset_src0 = first_row*nb + (i12/r2)*(nb*ne01) + (i13/r3)*(nb*ne01*ne02); + + global uchar * blk_ql = (global uchar *) src0_ql + (ulong)offset_src0 * 128; + global uchar * blk_qh = (global uchar *) src0_qh + (ulong)offset_src0 * 64; + global char * blk_scales = (global char *) src0_s + (ulong)offset_src0 * 16; + global half * blk_d = (global half *) src0_d + offset_src0; +#else ulong offset_src0 = first_row*nb + (i12/r2)*(nb*ne01) + (i13/r3)*(nb*ne01*ne02); ulong offset_src0_ql = offset_src0 * 128; ulong offset_src0_qh = offset_src0 * 64; @@ -138,6 +200,7 @@ kernel void kernel_mul_mv_q6_K_f32_flat( global uchar * blk_qh = (global uchar *) src0_qh + offset_src0_qh; global char * blk_scales = (global char *) src0_s + offset_src0_s; global half * blk_d = (global half *) src0_d + offset_src0_d; +#endif global float * yy = (global float *) src1 + r1*ne10 + im*ne00*ne1; int tid = get_sub_group_local_id()%(N_SIMDWIDTH/BLOCK_STRIDE); // within-super-block part, 0..15 @@ -155,24 +218,55 @@ kernel void kernel_mul_mv_q6_K_f32_flat( for (int ib = ix; ib < nb; ib += BLOCK_STRIDE) { global float * y = yy + ib * QK_K + 128*ip + l0; +#if defined(ADRENO_OLD_COMPILER) + // vload4 of f32 is miscompiled here; index the lanes scalar instead. + float4 y0 = (float4)(y[ 0], y[ 1], y[ 2], y[ 3]); + float4 y1 = (float4)(y[32], y[33], y[34], y[35]); + float4 y2 = (float4)(y[64], y[65], y[66], y[67]); + float4 y3 = (float4)(y[96], y[97], y[98], y[99]); +#else float4 y0 = vload4(0, y + 0); float4 y1 = vload4(0, y + 32); float4 y2 = vload4(0, y + 64); float4 y3 = vload4(0, y + 96); +#endif for (int row = 0; row < N_DST; row++) { if (first_row + row < ne01) { +#if defined(ADRENO_OLD_COMPILER) + int dbg = (q6k_mask==0xFE && r0==0 && r1==0 && im==0 && row==0 && ib==0 && + ne00==256 && ne01==16 && get_sub_group_local_id()==0) ? 1 : 0; + sumf[row] += block_q_6_K_dot_y_flat( + blk_ql + row*nb*128, blk_qh + row*nb*64, blk_scales + row*nb*16, blk_d + row*nb, + ib, ip, is, l0, dbg, y0, y1, y2, y3); +#else sumf[row] += block_q_6_K_dot_y_flat( blk_ql + row*nb*128, blk_qh + row*nb*64, blk_scales + row*nb*16, blk_d + row*nb, ib, ip, is, l0, y0, y1, y2, y3); +#endif } } } +#if defined(ADRENO_OLD_COMPILER) + // Optimizer barrier. This compiler drops the sumf partials unless a side effect + // forces them to materialize. q6k_mask is a kernel arg the compiler cannot prove + // is never 0xFE (the host always passes 0xFF), so the printf survives compilation + // but never executes. FRAGILE: the exact set and placement of these guarded + // printfs is load-bearing on E031.41 -- removing any one re-breaks q6_K. + if (q6k_mask==0xFE && r0==0 && r1==0 && im==0 && ne00==256 && ne01==16 && get_sub_group_local_id()<16) { + printf("Q6KLANE lane=%d ip=%d il=%d is=%d l0=%d sumf0=%f\n", + get_sub_group_local_id(), ip, il, is, l0, sumf[0]); + } +#endif for (int row = 0; row < N_DST; row++) { float tot = sub_group_reduce_add(sumf[row]); if (get_sub_group_local_id() == 0 && first_row + row < ne01) { dst[r1*ne0 + im*ne0*ne1 + first_row + row] = tot; +#if defined(ADRENO_OLD_COMPILER) + if (q6k_mask==0xFE && r0==0 && r1==0 && im==0 && row==0 && ne00==256 && ne01==16) + printf("Q6KTOT tot=%f\n", tot); +#endif } } } diff --git a/ggml/src/ggml-opencl/kernels/rms_norm.cl b/ggml/src/ggml-opencl/kernels/rms_norm.cl index 4b18d17d..99085625 100644 --- a/ggml/src/ggml-opencl/kernels/rms_norm.cl +++ b/ggml/src/ggml-opencl/kernels/rms_norm.cl @@ -188,3 +188,182 @@ kernel void kernel_rms_norm_mul( y[i00] = (x[i00] * scale) * f[i00%(ne10/4)]; } } + +//------------------------------------------------------------------------------ +// rms_norm + mul (norm weight) + add (residual), fused. Mirrors +// kernel_rms_norm_mul with an extra residual operand src2: computes +// y = (rmsnorm(x) * w) + g +// in one dispatch, removing one kernel launch + one global round-trip per +// residual block (the dominant per-layer adjacency on Gemma matformers). +//------------------------------------------------------------------------------ +kernel void kernel_rms_norm_mul_add( + global char * src0, + ulong offset0, + global char * src1, + ulong offset1, + global char * src2, + ulong offset2, + global char * dst, + ulong offsetd, + int ne00, + int ne01, + int ne02, + int ne03, + ulong nb01, + ulong nb02, + ulong nb03, + int ne10, + int ne11, + int ne12, + int ne13, + ulong nb11, + ulong nb12, + ulong nb13, + int ne20, + int ne21, + int ne22, + int ne23, + ulong nb21, + ulong nb22, + ulong nb23, + ulong nb1, + ulong nb2, + ulong nb3, + float eps, + local float * sum +) { + src0 = src0 + offset0; + src1 = src1 + offset1; + src2 = src2 + offset2; + dst = dst + offsetd; + + if (get_sub_group_id() == 0) { + sum[get_sub_group_local_id()] = 0.0f; + } + + int i03 = get_group_id(2); + int i02 = get_group_id(1); + int i01 = get_group_id(0); + + global float4 * x = (global float4 *) (src0 + i03*nb03 + i02*nb02 + i01*nb01); + global float4 * f = (global float4 *) (src1 + (i03%ne13)*nb13 + (i02%ne12)*nb12 + (i01%ne11)*nb11); + global float4 * g = (global float4 *) (src2 + (i03%ne23)*nb23 + (i02%ne22)*nb22 + (i01%ne21)*nb21); + + float sumf = 0; + + for (int i00 = get_local_id(0); i00 < ne00/4; i00 += get_local_size(0)) { + sumf += dot(x[i00], x[i00]); + } + sumf = sub_group_reduce_add(sumf); + + barrier(CLK_LOCAL_MEM_FENCE); + + if (get_sub_group_local_id() == 0) { + sum[get_sub_group_id()] = sumf; + } + + barrier(CLK_LOCAL_MEM_FENCE); + + sumf = sum[get_sub_group_local_id()]; + sumf = sub_group_reduce_add(sumf); + + float mean = sumf / ne00; + float scale = 1.0f/sqrt(mean + eps); + + global float4 * y = (global float4 *) (dst + i03*nb3 + i02*nb2 + i01*nb1); + for (int i00 = get_local_id(0); i00 < ne00/4; i00 += get_local_size(0)) { + y[i00] = (x[i00] * scale) * f[i00%(ne10/4)] + g[i00%(ne20/4)]; + } +} + +//------------------------------------------------------------------------------ +// rms_norm + mul(norm weight) + add(residual) + mul(scalar scale), fused. +// Computes y = ((rmsnorm(x) * w) + g) * s, where s is a broadcast SCALAR (e.g. +// Gemma-4 layer_output_scale). Folds the trailing per-layer l_out scale-mul into +// the residual-norm kernel: one extra dispatch + global round-trip saved per +// layer. src3 points at the single scale value. +//------------------------------------------------------------------------------ +kernel void kernel_rms_norm_mul_add_scale( + global char * src0, + ulong offset0, + global char * src1, + ulong offset1, + global char * src2, + ulong offset2, + global char * src3, + ulong offset3, + global char * dst, + ulong offsetd, + int ne00, + int ne01, + int ne02, + int ne03, + ulong nb01, + ulong nb02, + ulong nb03, + int ne10, + int ne11, + int ne12, + int ne13, + ulong nb11, + ulong nb12, + ulong nb13, + int ne20, + int ne21, + int ne22, + int ne23, + ulong nb21, + ulong nb22, + ulong nb23, + ulong nb1, + ulong nb2, + ulong nb3, + float eps, + local float * sum +) { + src0 = src0 + offset0; + src1 = src1 + offset1; + src2 = src2 + offset2; + src3 = src3 + offset3; + dst = dst + offsetd; + + const float sc = *((global float *) src3); + + if (get_sub_group_id() == 0) { + sum[get_sub_group_local_id()] = 0.0f; + } + + int i03 = get_group_id(2); + int i02 = get_group_id(1); + int i01 = get_group_id(0); + + global float4 * x = (global float4 *) (src0 + i03*nb03 + i02*nb02 + i01*nb01); + global float4 * f = (global float4 *) (src1 + (i03%ne13)*nb13 + (i02%ne12)*nb12 + (i01%ne11)*nb11); + global float4 * g = (global float4 *) (src2 + (i03%ne23)*nb23 + (i02%ne22)*nb22 + (i01%ne21)*nb21); + + float sumf = 0; + + for (int i00 = get_local_id(0); i00 < ne00/4; i00 += get_local_size(0)) { + sumf += dot(x[i00], x[i00]); + } + sumf = sub_group_reduce_add(sumf); + + barrier(CLK_LOCAL_MEM_FENCE); + + if (get_sub_group_local_id() == 0) { + sum[get_sub_group_id()] = sumf; + } + + barrier(CLK_LOCAL_MEM_FENCE); + + sumf = sum[get_sub_group_local_id()]; + sumf = sub_group_reduce_add(sumf); + + float mean = sumf / ne00; + float scale = 1.0f/sqrt(mean + eps); + + global float4 * y = (global float4 *) (dst + i03*nb3 + i02*nb2 + i01*nb1); + for (int i00 = get_local_id(0); i00 < ne00/4; i00 += get_local_size(0)) { + y[i00] = ((x[i00] * scale) * f[i00%(ne10/4)] + g[i00%(ne20/4)]) * sc; + } +} diff --git a/ggml/src/ggml-opencl/kernels/rope.cl b/ggml/src/ggml-opencl/kernels/rope.cl index 82f4cd87..27fdbbbc 100644 --- a/ggml/src/ggml-opencl/kernels/rope.cl +++ b/ggml/src/ggml-opencl/kernels/rope.cl @@ -75,7 +75,8 @@ kernel void kernel_rope_norm_f32( float ext_factor, float attn_factor, float beta_fast, - float beta_slow + float beta_slow, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -94,14 +95,15 @@ kernel void kernel_rope_norm_f32( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - float theta = theta_base * pow(freq_base, inv_ndims*i0); + float theta = theta_base * pow(freq_base, inv_ndims*iw); float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); global float * src = (global float *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00); global float * dst_data = (global float *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); @@ -154,7 +156,8 @@ kernel void kernel_rope_norm_f16( float ext_factor, float attn_factor, float beta_fast, - float beta_slow + float beta_slow, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -173,14 +176,15 @@ kernel void kernel_rope_norm_f16( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - float theta = theta_base * pow(freq_base, inv_ndims*i0); + float theta = theta_base * pow(freq_base, inv_ndims*iw); float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); global half * src = (global half *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00); global half * dst_data = (global half *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0); @@ -233,7 +237,8 @@ kernel void kernel_rope_neox_f32( float ext_factor, float attn_factor, float beta_fast, - float beta_slow + float beta_slow, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -252,17 +257,18 @@ kernel void kernel_rope_neox_f32( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - const float theta = theta_base * pow(freq_base, inv_ndims*i0); + const float theta = theta_base * pow(freq_base, inv_ndims*iw); const float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); - global float * src = (global float *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00); - global float * dst_data = (global float *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + ic*nb0); + global float * src = (global float *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + (n_offs + ic)*nb00); + global float * dst_data = (global float *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + (n_offs + ic)*nb0); const float x0 = src[0]; const float x1 = src[n_dims/2]; @@ -312,7 +318,8 @@ kernel void kernel_rope_neox_f16( float ext_factor, float attn_factor, float beta_fast, - float beta_slow + float beta_slow, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -331,17 +338,18 @@ kernel void kernel_rope_neox_f16( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - const float theta = theta_base * pow(freq_base, inv_ndims*i0); + const float theta = theta_base * pow(freq_base, inv_ndims*iw); const float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); - global half * src = (global half *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00); - global half * dst_data = (global half *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + ic*nb0); + global half * src = (global half *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + (n_offs + ic)*nb00); + global half * dst_data = (global half *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + (n_offs + ic)*nb0); const float x0 = src[0]; const float x1 = src[n_dims/2]; @@ -393,7 +401,8 @@ kernel void kernel_rope_multi_f32( float beta_fast, float beta_slow, int4 sections, - int is_imrope + int is_imrope, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -414,10 +423,11 @@ kernel void kernel_rope_multi_f32( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - const int sector = (i0 / 2) % sect_dims; + const int sector = ic % sect_dims; float theta_base = 0.0f; if (is_imrope) { @@ -445,14 +455,14 @@ kernel void kernel_rope_multi_f32( } } - const float theta = theta_base * pow(freq_base, inv_ndims*i0); + const float theta = theta_base * pow(freq_base, inv_ndims*iw); const float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); - global float * src = (global float *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00); - global float * dst_data = (global float *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + ic*nb0); + global float * src = (global float *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + (n_offs + ic)*nb00); + global float * dst_data = (global float *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + (n_offs + ic)*nb0); const float x0 = src[0]; const float x1 = src[n_dims/2]; @@ -504,7 +514,8 @@ kernel void kernel_rope_multi_f16( float beta_fast, float beta_slow, int4 sections, - int is_imrope + int is_imrope, + int n_offs ) { src0 = (global void*)((global char*)src0 + offset0); src1 = (global int*)((global char*)src1 + offset1); @@ -525,10 +536,11 @@ kernel void kernel_rope_multi_f16( float inv_ndims = -1.f/n_dims; for (int i0 = 2*get_local_id(0); i0 < ne0; i0 += 2*get_local_size(0)) { - if (i0 < n_dims) { - int ic = i0/2; + if (i0 >= n_offs && i0 < n_offs + n_dims) { + int iw = i0 - n_offs; // relative idx + int ic = iw/2; - const int sector = (i0 / 2) % sect_dims; + const int sector = ic % sect_dims; float theta_base = 0.0f; if (is_imrope) { @@ -556,14 +568,14 @@ kernel void kernel_rope_multi_f16( } } - const float theta = theta_base * pow(freq_base, inv_ndims*i0); + const float theta = theta_base * pow(freq_base, inv_ndims*iw); const float freq_factor = src2 != src0 ? src2[ic] : 1.0f; - float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, i0, ext_factor, attn_factor); + float2 cos_sin_theta = rope_yarn(theta/freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor); - global half * src = (global half *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + ic*nb00); - global half * dst_data = (global half *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + ic*nb0); + global half * src = (global half *)((global char *) src0 + i3*nb03 + i2*nb02 + i1*nb01 + (n_offs + ic)*nb00); + global half * dst_data = (global half *)((global char *) dst + i3*nb3 + i2*nb2 + i1*nb1 + (n_offs + ic)*nb0); const float x0 = src[0]; const float x1 = src[n_dims/2]; diff --git a/ggml/src/ggml-opencl/kernels/sdpa_xmem_f32_f16_os8.cl b/ggml/src/ggml-opencl/kernels/sdpa_xmem_f32_f16_os8.cl new file mode 100644 index 00000000..26f0fbd5 --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/sdpa_xmem_f32_f16_os8.cl @@ -0,0 +1,871 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable +#pragma OPENCL EXTENSION cl_qcom_subgroup_uniform_load : enable +#pragma OPENCL EXTENSION cl_qcom_subgroup_constant_load : enable + +#define bool2 uchar2 +#define bool3 uchar3 +#define bool4 uchar4 + +__constant sampler_t smp_none = CLK_NORMALIZED_COORDS_FALSE | CLK_ADDRESS_NONE | CLK_FILTER_NEAREST; +__constant sampler_t smp_zero = CLK_NORMALIZED_COORDS_FALSE | CLK_ADDRESS_CLAMP | CLK_FILTER_NEAREST; + +__kernel void adreno_xmem_attn_q_f32_to_img_scaled(const global void * src_void, + ulong src_offset, + write_only image2d_t dst_image2d, + const float scale, + const int d_head, + const int n_q, + const int n_head, + const int n_head_kv, + const int n_batch, + const ulong src_nb1, + const ulong src_nb2, + const ulong src_nb3) { + const int x = get_global_id(0); + const int flat_h = get_global_id(1); + const int d = get_global_id(2); + + const int heads_total = n_head * n_batch; + const int kpack = d_head / 4; + + if (x >= n_q || flat_h >= heads_total || d >= kpack) { + return; + } + + const int batch = flat_h / n_head; + const int head = flat_h % n_head; + const int gqa = n_head / n_head_kv; + const int head_kv = head / gqa; + const int head_group = head - head_kv * gqa; + const int compact_h = batch * n_head_kv + head_kv; + const int compact_x = head_group * n_q + x; + const int c = d * 4; + + const global char * src_base = (const global char *) src_void + src_offset; + const global float * row_ptr = (const global float *) (src_base + batch * src_nb3 + head * src_nb2 + x * src_nb1); + + half4 out = (half4) (0.0h); + out.x = convert_half(row_ptr[c + 0] * scale); + if (c + 1 < d_head) { + out.y = convert_half(row_ptr[c + 1] * scale); + } + if (c + 2 < d_head) { + out.z = convert_half(row_ptr[c + 2] * scale); + } + if (c + 3 < d_head) { + out.w = convert_half(row_ptr[c + 3] * scale); + } + + write_imageh(dst_image2d, (int2) (compact_x, compact_h * kpack + d), out); +} + +__kernel void adreno_xmem_attn_kv_f32_to_img_gqa(const global void * src_void, + ulong src_offset, + write_only image2d_t dst_image2d, + const int d_head, + const int n_kv, + const int n_kv_padded, + const int n_head_kv, + const int n_batch, + const ulong src_nb1, + const ulong src_nb2, + const ulong src_nb3) { + const int x = get_global_id(0); + const int flat_h = get_global_id(1); + const int d = get_global_id(2); + + const int kv_heads_total = n_head_kv * n_batch; + const int kpack = d_head / 4; + + if (x >= n_kv_padded || flat_h >= kv_heads_total || d >= kpack) { + return; + } + + const int batch = flat_h / n_head_kv; + const int head_kv = flat_h % n_head_kv; + const int c = d * 4; + + half4 out = (half4) (0.0h); + if (x < n_kv) { + const global char * src_base = (const global char *) src_void + src_offset; + const global float * row_ptr = + (const global float *) (src_base + batch * src_nb3 + head_kv * src_nb2 + x * src_nb1); + out.x = convert_half(row_ptr[c + 0]); + if (c + 1 < d_head) { + out.y = convert_half(row_ptr[c + 1]); + } + if (c + 2 < d_head) { + out.z = convert_half(row_ptr[c + 2]); + } + if (c + 3 < d_head) { + out.w = convert_half(row_ptr[c + 3]); + } + } + + write_imageh(dst_image2d, (int2) (x, flat_h * kpack + d), out); +} + +__kernel void adreno_xmem_attn_kv_f16_to_img_gqa(const global void * src_void, + ulong src_offset, + write_only image2d_t dst_image2d, + const int d_head, + const int n_kv, + const int n_kv_padded, + const int n_head_kv, + const int n_batch, + const ulong src_nb1, + const ulong src_nb2, + const ulong src_nb3) { + const int x = get_global_id(0); + const int flat_h = get_global_id(1); + const int d = get_global_id(2); + + const int kv_heads_total = n_head_kv * n_batch; + const int kpack = d_head / 4; + + if (x >= n_kv_padded || flat_h >= kv_heads_total || d >= kpack) { + return; + } + + const int batch = flat_h / n_head_kv; + const int head_kv = flat_h % n_head_kv; + const int c = d * 4; + + half4 out = (half4) (0.0h); + if (x < n_kv) { + const global char * src_base = (const global char *) src_void + src_offset; + const global half * row_ptr = + (const global half *) (src_base + batch * src_nb3 + head_kv * src_nb2 + x * src_nb1); + out.x = row_ptr[c + 0]; + if (c + 1 < d_head) { + out.y = row_ptr[c + 1]; + } + if (c + 2 < d_head) { + out.z = row_ptr[c + 2]; + } + if (c + 3 < d_head) { + out.w = row_ptr[c + 3]; + } + } + + write_imageh(dst_image2d, (int2) (x, flat_h * kpack + d), out); +} + +__kernel void adreno_xmem_attn_img_to_f32(global void * dst_void, + ulong dst_offset, + read_only image2d_t src_image2d, + const int d_head, + const int n_q, + const int n_head, + const int n_head_kv, + const int n_batch, + const ulong dst_nb1, + const ulong dst_nb2, + const ulong dst_nb3) { + const int x = get_global_id(0); + const int flat_h = get_global_id(1); + const int d = get_global_id(2); + + const int heads_total = n_head * n_batch; + const int kpack = d_head / 4; + + if (x >= n_q || flat_h >= heads_total || d >= kpack) { + return; + } + + const int batch = flat_h / n_head; + const int head = flat_h % n_head; + const int gqa = n_head / n_head_kv; + const int head_kv = head / gqa; + const int head_group = head - head_kv * gqa; + const int compact_h = batch * n_head_kv + head_kv; + const int compact_x = head_group * n_q + x; + const int c = d * 4; + + global char * dst_base = (global char *) dst_void + dst_offset; + global float * row_ptr = (global float *) (dst_base + batch * dst_nb3 + x * dst_nb2 + head * dst_nb1); + + const half4 in_value = read_imageh(src_image2d, smp_zero, (int2) (compact_x, compact_h * kpack + d)); + row_ptr[c + 0] = convert_float(in_value.x); + if (c + 1 < d_head) { + row_ptr[c + 1] = convert_float(in_value.y); + } + if (c + 2 < d_head) { + row_ptr[c + 2] = convert_float(in_value.z); + } + if (c + 3 < d_head) { + row_ptr[c + 3] = convert_float(in_value.w); + } +} + +__kernel void adreno_xmem_attn_k_gather(global half4 * dst_tensor_buffer, + read_only image2d_t src_tensor_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1) { + int X = get_global_id(0); + int Y = get_global_id(1); + int S = get_global_id(2); + if (X >= shared_int4_0.w || Y >= shared_int4_0.y || S >= shared_int4_0.z) { + return; + } + half temps[4]; + temps[0] = (half) (0.f); + temps[1] = (half) (0.f); + temps[2] = (half) (0.f); + temps[3] = (half) (0.f); + for (int i = 0; i < 4; ++i) { + int dst_channel = S * 4 + i; + if (dst_channel < shared_int4_0.x) { + int s_y = Y; + int s_x = dst_channel; + int s_c = X; + { + int slice_coord_TMP = (s_c) / 4; + int sub_ch_coord_TMP = (s_c) % 4; + half4 src_TMP = read_imageh(src_tensor_image2d, smp_zero, + (int2) ((s_x), ((s_y) *shared_int4_1.x + (slice_coord_TMP)))); + temps[i] = (half[4]){ src_TMP.x, src_TMP.y, src_TMP.z, src_TMP.w }[sub_ch_coord_TMP]; + }; + } + } + half4 result; + result.x = temps[0]; + result.y = temps[1]; + result.z = temps[2]; + result.w = temps[3]; + dst_tensor_buffer[(((S) *shared_int4_0.y + (Y)) * shared_int4_0.w + (X))] = result; +} + +__kernel void adreno_xmem_attn_pack_k(global half4 * dst_tensor_buffer, + read_only image1d_buffer_t src_image_buffer, + const int4 shared_int4_0, + const int4 shared_int4_1, + const int4 shared_int4_2) { + int linear_index = get_global_id(0); + if (linear_index >= shared_int4_0.y) { + return; + } + if (get_global_id(1) != 0) { + return; + } + if (get_global_id(2) != 0) { + return; + } + int dst_o_sp_i_ogroup = linear_index; + int dst_ogroup = dst_o_sp_i_ogroup % shared_int4_0.x; + int dst_o_sp_i = dst_o_sp_i_ogroup / shared_int4_0.x; + int dst_i = dst_o_sp_i % shared_int4_0.z; + int dst_o_sp = dst_o_sp_i / shared_int4_0.z; + int dst_sp = dst_o_sp % shared_int4_1.x; + int dst_o = dst_o_sp / shared_int4_1.x; + int i_slice = dst_i; + int o_slice = dst_o * shared_int4_0.x + dst_ogroup; + int spatial_linear = dst_sp; + int W = spatial_linear % shared_int4_1.y; + int H = spatial_linear / shared_int4_1.y; + half4 w0 = (half4) (0); + half4 w1 = (half4) (0); + half4 w2 = (half4) (0); + half4 w3 = (half4) (0); + + if (i_slice * 4 < shared_int4_0.w && o_slice < shared_int4_1.w) { + w0 = read_imageh(src_image_buffer, (((o_slice) *shared_int4_1.z + (W)) * shared_int4_2.x + (i_slice * 4))); + } + if (i_slice * 4 + 1 < shared_int4_0.w && o_slice < shared_int4_1.w) { + w1 = read_imageh(src_image_buffer, (((o_slice) *shared_int4_1.z + (W)) * shared_int4_2.x + (i_slice * 4 + 1))); + } + if (i_slice * 4 + 2 < shared_int4_0.w && o_slice < shared_int4_1.w) { + w2 = read_imageh(src_image_buffer, (((o_slice) *shared_int4_1.z + (W)) * shared_int4_2.x + (i_slice * 4 + 2))); + } + if (i_slice * 4 + 3 < shared_int4_0.w && o_slice < shared_int4_1.w) { + w3 = read_imageh(src_image_buffer, (((o_slice) *shared_int4_1.z + (W)) * shared_int4_2.x + (i_slice * 4 + 3))); + } + half4 r0 = w0; + half4 r1 = w1; + half4 r2 = w2; + half4 r3 = w3; + dst_tensor_buffer[linear_index * 4 + 0] = r0; + dst_tensor_buffer[linear_index * 4 + 1] = r1; + dst_tensor_buffer[linear_index * 4 + 2] = r2; + dst_tensor_buffer[linear_index * 4 + 3] = r3; +} + +__attribute__((qcom_max_concurrent_subgroups(12))) __kernel void adreno_xmem_attn_qk_gemm( + global half4 * dst_tensor_buffer, + constant half8 * weights_buffer __attribute__((sub_group_uniform)), + constant half8 * xmem_buffer __attribute__((max_constant_size((6144)))), + read_only image2d_t src_tensor_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1, + const int4 shared_int4_2) { + int X = get_group_id(1) * get_local_size(0) + get_local_id(0); + int Y = get_group_id(2) * get_local_size(1) + get_local_id(1); + int Z = get_group_id(0) * get_local_size(2) + get_local_id(2); + if (X >= shared_int4_0.z || Y >= shared_int4_0.x) { + return; + } + if (Z * 8 >= shared_int4_0.y) { + return; + } + + half4 r0 = (half4) (0.f); + half4 r1 = (half4) (0.f); + half4 r2 = (half4) (0.f); + half4 r3 = (half4) (0.f); + half4 r4 = (half4) (0.f); + half4 r5 = (half4) (0.f); + half4 r6 = (half4) (0.f); + half4 r7 = (half4) (0.f); + int x_coord = mad24(X, shared_int4_2.y, shared_int4_1.y); + int y_coord = mad24(Y, shared_int4_2.z, shared_int4_1.z); + int coord_x, coord_y, coord_s; + int f_offset = (Z * shared_int4_1.w + Y) * shared_int4_1.x * 32; + + int subgroup_id = (int) ((0x1F & qcom_get_physical_sub_group_id())); + subgroup_id = subgroup_id % 12; + int c_offset = mul24(subgroup_id, shared_int4_0.w); + __constant half16 * weights_cache = (__constant half16 *) &xmem_buffer[c_offset]; + coord_y = Y; + coord_x = X; + coord_s = 0; + do { + half4 src0 = + read_imageh(src_tensor_image2d, smp_zero, (int2) ((coord_x), ((coord_y) *shared_int4_2.x + (coord_s)))); + coord_s++; + half4 src1 = + read_imageh(src_tensor_image2d, smp_zero, (int2) ((coord_x), ((coord_y) *shared_int4_2.x + (coord_s)))); + coord_s++; + qcom_sub_group_constant_load8(xmem_buffer, weights_buffer, c_offset, f_offset >> 1, 32); + f_offset += 64; + qcom_sub_group_sync(QCOM_CLK_CONST_LOAD_SYNC); + r0 += src0.x * weights_cache[0].s0123; + r0 += src0.y * weights_cache[0].s4567; + r0 += src0.z * weights_cache[0].s89ab; + r0 += src0.w * weights_cache[0].scdef; + r1 += src0.x * weights_cache[1].s0123; + r1 += src0.y * weights_cache[1].s4567; + r1 += src0.z * weights_cache[1].s89ab; + r1 += src0.w * weights_cache[1].scdef; + r2 += src0.x * weights_cache[2].s0123; + r2 += src0.y * weights_cache[2].s4567; + r2 += src0.z * weights_cache[2].s89ab; + r2 += src0.w * weights_cache[2].scdef; + r3 += src0.x * weights_cache[3].s0123; + r3 += src0.y * weights_cache[3].s4567; + r3 += src0.z * weights_cache[3].s89ab; + r3 += src0.w * weights_cache[3].scdef; + r4 += src0.x * weights_cache[4].s0123; + r4 += src0.y * weights_cache[4].s4567; + r4 += src0.z * weights_cache[4].s89ab; + r4 += src0.w * weights_cache[4].scdef; + r5 += src0.x * weights_cache[5].s0123; + r5 += src0.y * weights_cache[5].s4567; + r5 += src0.z * weights_cache[5].s89ab; + r5 += src0.w * weights_cache[5].scdef; + r6 += src0.x * weights_cache[6].s0123; + r6 += src0.y * weights_cache[6].s4567; + r6 += src0.z * weights_cache[6].s89ab; + r6 += src0.w * weights_cache[6].scdef; + r7 += src0.x * weights_cache[7].s0123; + r7 += src0.y * weights_cache[7].s4567; + r7 += src0.z * weights_cache[7].s89ab; + r7 += src0.w * weights_cache[7].scdef; + r0 += src1.x * weights_cache[8].s0123; + r0 += src1.y * weights_cache[8].s4567; + r0 += src1.z * weights_cache[8].s89ab; + r0 += src1.w * weights_cache[8].scdef; + r1 += src1.x * weights_cache[9].s0123; + r1 += src1.y * weights_cache[9].s4567; + r1 += src1.z * weights_cache[9].s89ab; + r1 += src1.w * weights_cache[9].scdef; + r2 += src1.x * weights_cache[10].s0123; + r2 += src1.y * weights_cache[10].s4567; + r2 += src1.z * weights_cache[10].s89ab; + r2 += src1.w * weights_cache[10].scdef; + r3 += src1.x * weights_cache[11].s0123; + r3 += src1.y * weights_cache[11].s4567; + r3 += src1.z * weights_cache[11].s89ab; + r3 += src1.w * weights_cache[11].scdef; + r4 += src1.x * weights_cache[12].s0123; + r4 += src1.y * weights_cache[12].s4567; + r4 += src1.z * weights_cache[12].s89ab; + r4 += src1.w * weights_cache[12].scdef; + r5 += src1.x * weights_cache[13].s0123; + r5 += src1.y * weights_cache[13].s4567; + r5 += src1.z * weights_cache[13].s89ab; + r5 += src1.w * weights_cache[13].scdef; + r6 += src1.x * weights_cache[14].s0123; + r6 += src1.y * weights_cache[14].s4567; + r6 += src1.z * weights_cache[14].s89ab; + r6 += src1.w * weights_cache[14].scdef; + r7 += src1.x * weights_cache[15].s0123; + r7 += src1.y * weights_cache[15].s4567; + r7 += src1.z * weights_cache[15].s89ab; + r7 += src1.w * weights_cache[15].scdef; + } while (coord_s < shared_int4_2.x); + + coord_s = mul24(Z, 8); + coord_x = X; + coord_y = Y; + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r0); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r1); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r2); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r3); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r4); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r5); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r6); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r7); + if (coord_s < 0) { + res += read_imageh(src_tensor_image2d, smp_zero, (int2) ((0), ((0) * shared_int4_2.x + (0)))); + } + dst_tensor_buffer[(((coord_s) *shared_int4_0.x + (coord_y)) * shared_int4_0.z + (coord_x))] = res; + coord_s++; + } +} + +__kernel void adreno_xmem_attn_softmax_reduce_basic(read_only image1d_buffer_t src_tensor_image_buffer, + write_only image2d_t dst_tensor_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1) { + int X = get_global_id(0); + int Y = get_global_id(1); + if (X >= shared_int4_0.z || Y >= shared_int4_0.x) { + return; + } + float sum = 0.0f; + int end_channel = shared_int4_0.w; + int end_slice = (end_channel + 3) / 4; + int start_channel = 0; + int start_slice = start_channel / 4; + bool need_per_channels_check = start_channel % 4 != 0 || end_channel % 4 != 0; + float maximum; + { + int slice_coord_TMP = (start_channel) / 4; + int sub_ch_coord_TMP = (start_channel) % 4; + float4 src_TMP = convert_float4( + read_imageh(src_tensor_image_buffer, ((slice_coord_TMP) *shared_int4_1.x + (Y)) * shared_int4_1.y + (X))); + maximum = (float[4]){ src_TMP.x, src_TMP.y, src_TMP.z, src_TMP.w }[sub_ch_coord_TMP]; + }; + for (int d = start_slice; d < end_slice; d += 1) { + float4 mask_dot = (float4) (1.f); + float4 src = + convert_float4(read_imageh(src_tensor_image_buffer, ((d) *shared_int4_1.x + (Y)) * shared_int4_1.y + (X))); + if (need_per_channels_check && (d == start_slice || d == end_slice - 1)) { + if (d * 4 + 0 < start_channel || d * 4 + 0 >= end_channel) { + mask_dot.x = 0.f; + src.x = maximum; + } + if (d * 4 + 1 < start_channel || d * 4 + 1 >= end_channel) { + mask_dot.y = 0.f; + src.y = maximum; + } + if (d * 4 + 2 < start_channel || d * 4 + 2 >= end_channel) { + mask_dot.z = 0.f; + src.z = maximum; + } + if (d * 4 + 3 < start_channel || d * 4 + 3 >= end_channel) { + mask_dot.w = 0.f; + src.w = maximum; + } + } + float new_max = max(src.x, src.y); + new_max = max(new_max, src.z); + new_max = max(new_max, src.w); + new_max = max(new_max, maximum); + float scale = native_exp(maximum - new_max); + maximum = new_max; + sum *= scale; + float4 exp_res = native_exp(src - maximum); + sum += dot(mask_dot, exp_res); + } + if (!isfinite(maximum) || sum == 0.0f) { + write_imageh(dst_tensor_image2d, (int2) (X, Y), (half4) (0.0h)); + return; + } + write_imageh(dst_tensor_image2d, (int2) (X, Y), + (half4) (convert_half(1.0f / sum), convert_half(maximum), 0.0h, 0.0h)); +} + +__kernel void adreno_xmem_attn_softmax_apply_basic(global half4 * dst_tensor_buffer, + read_only image1d_buffer_t src_tensor_image_buffer, + read_only image2d_t src_tensor_1_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1) { + int X = get_global_id(0); + int Y = get_global_id(1); + int Z = get_global_id(2); + if (X >= shared_int4_0.z || Y >= shared_int4_0.x || Z >= shared_int4_0.y) { + return; + } + half4 src = read_imageh(src_tensor_image_buffer, ((Z) *shared_int4_1.x + (Y)) * shared_int4_1.y + (X)); + { + half4 src_final; + { + { + half4 exp_val = read_imageh(src_tensor_1_image2d, smp_zero, (int2) (X, Y)); + src_final = exp(src - exp_val.y) * exp_val.x; + const int k = Z * 4; + const int n_kv = shared_int4_1.z; + if (k + 0 >= n_kv) { + src_final.x = 0.0h; + } + if (k + 1 >= n_kv) { + src_final.y = 0.0h; + } + if (k + 2 >= n_kv) { + src_final.z = 0.0h; + } + if (k + 3 >= n_kv) { + src_final.w = 0.0h; + } + } + } + dst_tensor_buffer[(((Z) *shared_int4_0.x + (Y)) * shared_int4_0.z + (X))] = src_final; + }; +} + +__kernel void adreno_xmem_attn_mask_scores(global half4 * dst_score_tensor_buffer, + read_only image1d_buffer_t src_score_image_buffer, + const global half * mask, + const ulong mask_offset, + const int q_width, + const int n_q, + const int n_kv, + const int n_kv_padded, + const int kv_heads_total, + const int n_head, + const int n_head_kv, + const ulong mask_nb1, + const ulong mask_nb2, + const ulong mask_nb3, + const int mask_ne2, + const int mask_ne3) { + const int X = get_global_id(0); + const int Y = get_global_id(1); + const int Z = get_global_id(2); + const int npack = n_kv_padded / 4; + if (X >= q_width || Y >= kv_heads_total || Z >= npack) { + return; + } + + const int gqa = n_head / n_head_kv; + const int head_kv = Y % n_head_kv; + const int batch = Y / n_head_kv; + const int head_group = X / n_q; + const int q = X - head_group * n_q; + const int head = head_kv * gqa + head_group; + const int mask_head_idx = head % mask_ne2; + const int mask_batch_idx = batch % mask_ne3; + const global char * mask_base = (const global char *) mask + mask_offset; + const global half * mask_row = (const global half *) (mask_base + mask_batch_idx * mask_nb3 + + mask_head_idx * mask_nb2 + q * mask_nb1); + + const half4 score = read_imageh(src_score_image_buffer, ((Z * kv_heads_total + Y) * q_width + X)); + float vals[4] = { + convert_float(score.x), + convert_float(score.y), + convert_float(score.z), + convert_float(score.w), + }; + + for (int lane = 0; lane < 4; ++lane) { + const int k_idx = Z * 4 + lane; + if (k_idx >= n_kv) { + vals[lane] = -INFINITY; + } else { + vals[lane] += convert_float(mask_row[k_idx]); + } + } + + dst_score_tensor_buffer[((Z * kv_heads_total + Y) * q_width + X)] = + (half4) (convert_half(vals[0]), convert_half(vals[1]), convert_half(vals[2]), convert_half(vals[3])); +} + +__kernel void adreno_xmem_attn_pack_v(global half4 * dst_tensor_buffer, + read_only image2d_t src_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1) { + int linear_index = get_global_id(0); + if (linear_index >= shared_int4_0.y) { + return; + } + if (get_global_id(1) != 0) { + return; + } + if (get_global_id(2) != 0) { + return; + } + int dst_o_sp_i_ogroup = linear_index; + int dst_ogroup = dst_o_sp_i_ogroup % shared_int4_0.x; + int dst_o_sp_i = dst_o_sp_i_ogroup / shared_int4_0.x; + int dst_i = dst_o_sp_i % shared_int4_0.z; + int dst_o_sp = dst_o_sp_i / shared_int4_0.z; + int dst_sp = dst_o_sp % shared_int4_1.x; + int dst_o = dst_o_sp / shared_int4_1.x; + int i_slice = dst_i; + int o_slice = dst_o * shared_int4_0.x + dst_ogroup; + int spatial_linear = dst_sp; + int W = spatial_linear % shared_int4_1.y; + int H = spatial_linear / shared_int4_1.y; + half4 w0 = (half4) (0); + half4 w1 = (half4) (0); + half4 w2 = (half4) (0); + half4 w3 = (half4) (0); + + if (i_slice * 4 < shared_int4_0.w && o_slice < shared_int4_1.z) { + w0 = read_imageh(src_image2d, smp_zero, (int2) ((i_slice * 4), ((W) *shared_int4_1.z + (o_slice)))); + } + if (i_slice * 4 + 1 < shared_int4_0.w && o_slice < shared_int4_1.z) { + w1 = read_imageh(src_image2d, smp_zero, (int2) ((i_slice * 4 + 1), ((W) *shared_int4_1.z + (o_slice)))); + } + if (i_slice * 4 + 2 < shared_int4_0.w && o_slice < shared_int4_1.z) { + w2 = read_imageh(src_image2d, smp_zero, (int2) ((i_slice * 4 + 2), ((W) *shared_int4_1.z + (o_slice)))); + } + if (i_slice * 4 + 3 < shared_int4_0.w && o_slice < shared_int4_1.z) { + w3 = read_imageh(src_image2d, smp_zero, (int2) ((i_slice * 4 + 3), ((W) *shared_int4_1.z + (o_slice)))); + } + half4 r0 = w0; + half4 r1 = w1; + half4 r2 = w2; + half4 r3 = w3; + dst_tensor_buffer[linear_index * 4 + 0] = r0; + dst_tensor_buffer[linear_index * 4 + 1] = r1; + dst_tensor_buffer[linear_index * 4 + 2] = r2; + dst_tensor_buffer[linear_index * 4 + 3] = r3; +} + +__attribute__((qcom_max_concurrent_subgroups(12))) __kernel void adreno_xmem_attn_pv_gemm( + constant half8 * weights_buffer __attribute__((sub_group_uniform)), + constant half8 * xmem_buffer __attribute__((max_constant_size((6144)))), + read_only image1d_buffer_t src_tensor_image_buffer, + write_only image2d_t dst_tensor_image2d, + const int4 shared_int4_0, + const int4 shared_int4_1, + const int4 shared_int4_2, + const int4 shared_int4_3) { + int X = get_group_id(1) * get_local_size(0) + get_local_id(0); + int Y = get_group_id(2) * get_local_size(1) + get_local_id(1); + int Z = get_group_id(0) * get_local_size(2) + get_local_id(2); + if (X >= shared_int4_0.z || Y >= shared_int4_0.x) { + return; + } + if (Z * 8 >= shared_int4_0.y) { + return; + } + + half4 r0 = (half4) (0.f); + half4 r1 = (half4) (0.f); + half4 r2 = (half4) (0.f); + half4 r3 = (half4) (0.f); + half4 r4 = (half4) (0.f); + half4 r5 = (half4) (0.f); + half4 r6 = (half4) (0.f); + half4 r7 = (half4) (0.f); + int x_coord = mad24(X, shared_int4_2.w, shared_int4_1.y); + int y_coord = mad24(Y, shared_int4_3.x, shared_int4_1.z); + int coord_x, coord_y, coord_s; + int f_offset = (Z * shared_int4_1.w + Y) * shared_int4_1.x * 32; + + int subgroup_id = (int) ((0x1F & qcom_get_physical_sub_group_id())); + subgroup_id = subgroup_id % 12; + int c_offset = mul24(subgroup_id, shared_int4_0.w); + __constant half16 * weights_cache = (__constant half16 *) &xmem_buffer[c_offset]; + coord_y = Y; + coord_x = X; + int addr = (((0) * shared_int4_1.w + (coord_y)) * shared_int4_2.z + (coord_x)); + int dz = shared_int4_2.x; + coord_s = 0; + do { + half4 src0 = read_imageh(src_tensor_image_buffer, addr); + addr += dz; + coord_s++; + half4 src1 = read_imageh(src_tensor_image_buffer, addr); + addr += dz; + coord_s++; + qcom_sub_group_constant_load8(xmem_buffer, weights_buffer, c_offset, f_offset >> 1, 32); + f_offset += 64; + qcom_sub_group_sync(QCOM_CLK_CONST_LOAD_SYNC); + r0 += src0.x * weights_cache[0].s0123; + r0 += src0.y * weights_cache[0].s4567; + r0 += src0.z * weights_cache[0].s89ab; + r0 += src0.w * weights_cache[0].scdef; + r1 += src0.x * weights_cache[1].s0123; + r1 += src0.y * weights_cache[1].s4567; + r1 += src0.z * weights_cache[1].s89ab; + r1 += src0.w * weights_cache[1].scdef; + r2 += src0.x * weights_cache[2].s0123; + r2 += src0.y * weights_cache[2].s4567; + r2 += src0.z * weights_cache[2].s89ab; + r2 += src0.w * weights_cache[2].scdef; + r3 += src0.x * weights_cache[3].s0123; + r3 += src0.y * weights_cache[3].s4567; + r3 += src0.z * weights_cache[3].s89ab; + r3 += src0.w * weights_cache[3].scdef; + r4 += src0.x * weights_cache[4].s0123; + r4 += src0.y * weights_cache[4].s4567; + r4 += src0.z * weights_cache[4].s89ab; + r4 += src0.w * weights_cache[4].scdef; + r5 += src0.x * weights_cache[5].s0123; + r5 += src0.y * weights_cache[5].s4567; + r5 += src0.z * weights_cache[5].s89ab; + r5 += src0.w * weights_cache[5].scdef; + r6 += src0.x * weights_cache[6].s0123; + r6 += src0.y * weights_cache[6].s4567; + r6 += src0.z * weights_cache[6].s89ab; + r6 += src0.w * weights_cache[6].scdef; + r7 += src0.x * weights_cache[7].s0123; + r7 += src0.y * weights_cache[7].s4567; + r7 += src0.z * weights_cache[7].s89ab; + r7 += src0.w * weights_cache[7].scdef; + r0 += src1.x * weights_cache[8].s0123; + r0 += src1.y * weights_cache[8].s4567; + r0 += src1.z * weights_cache[8].s89ab; + r0 += src1.w * weights_cache[8].scdef; + r1 += src1.x * weights_cache[9].s0123; + r1 += src1.y * weights_cache[9].s4567; + r1 += src1.z * weights_cache[9].s89ab; + r1 += src1.w * weights_cache[9].scdef; + r2 += src1.x * weights_cache[10].s0123; + r2 += src1.y * weights_cache[10].s4567; + r2 += src1.z * weights_cache[10].s89ab; + r2 += src1.w * weights_cache[10].scdef; + r3 += src1.x * weights_cache[11].s0123; + r3 += src1.y * weights_cache[11].s4567; + r3 += src1.z * weights_cache[11].s89ab; + r3 += src1.w * weights_cache[11].scdef; + r4 += src1.x * weights_cache[12].s0123; + r4 += src1.y * weights_cache[12].s4567; + r4 += src1.z * weights_cache[12].s89ab; + r4 += src1.w * weights_cache[12].scdef; + r5 += src1.x * weights_cache[13].s0123; + r5 += src1.y * weights_cache[13].s4567; + r5 += src1.z * weights_cache[13].s89ab; + r5 += src1.w * weights_cache[13].scdef; + r6 += src1.x * weights_cache[14].s0123; + r6 += src1.y * weights_cache[14].s4567; + r6 += src1.z * weights_cache[14].s89ab; + r6 += src1.w * weights_cache[14].scdef; + r7 += src1.x * weights_cache[15].s0123; + r7 += src1.y * weights_cache[15].s4567; + r7 += src1.z * weights_cache[15].s89ab; + r7 += src1.w * weights_cache[15].scdef; + } while (coord_s < shared_int4_2.y); + + coord_s = mul24(Z, 8); + coord_x = X; + coord_y = Y; + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r0); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r1); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r2); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r3); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r4); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r5); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r6); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } + if (coord_s < shared_int4_0.y) { + half4 res = convert_half4(r7); + if (coord_s < 0) { + res += read_imageh(src_tensor_image_buffer, ((0) * shared_int4_1.w + (0)) * shared_int4_2.z + (0)); + } + write_imageh(dst_tensor_image2d, (int2) ((coord_x), ((coord_y) *shared_int4_0.y + (coord_s))), res); + coord_s++; + } +} diff --git a/ggml/src/ggml-opencl/kernels/ssm_scan.cl b/ggml/src/ggml-opencl/kernels/ssm_scan.cl new file mode 100644 index 00000000..1889b74c --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/ssm_scan.cl @@ -0,0 +1,346 @@ +// Mamba2 fused SSM scan kernel. One workgroup per (head, dim, seq); WG size = +// 64 threads. Each thread owns c_factor = d_state/64 state elements in +// private registers; the state stays resident across the n_tokens t-loop +// +// References: +// ggml/src/ggml-cuda/ssm-scan.cu:117 ssm_scan_f32_group +// ggml/src/ggml-cpu/ops.cpp:9368 ggml_compute_forward_ssm_scan_f32 + +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +#ifdef cl_khr_subgroups +#pragma OPENCL EXTENSION cl_khr_subgroups : enable +#endif + +#if defined(cl_qcom_reqd_sub_group_size) +#pragma OPENCL EXTENSION cl_qcom_reqd_sub_group_size : enable +#define REQD_SUBGROUP_SIZE_64 __attribute__((qcom_reqd_sub_group_size("half"))) +#else +#define REQD_SUBGROUP_SIZE_64 +#endif + +inline float softplus_f32(float x) { + return (x <= 20.0f) ? log(1.0f + exp(x)) : x; +} + +// d_state = 128 (most Mamba-2 models, e.g. mamba2-2.7B, Codestral-Mamba). +// WG = 64 threads, each holds 2 state elements (tid and tid+64). +REQD_SUBGROUP_SIZE_64 +kernel void kernel_ssm_scan_f32_mamba2_d128( + global const char * src0_base, ulong src0_off, + global const char * src1_base, ulong src1_off, + global const char * src2_base, ulong src2_off, + global const char * src3_base, ulong src3_off, + global const char * src4_base, ulong src4_off, + global const char * src5_base, ulong src5_off, + global const char * src6_base, ulong src6_off, + global char * dst_base, ulong dst_off, + ulong s0_nb2, ulong s0_nb3, + ulong x_nb2, ulong x_nb3, + ulong dt_nb1, ulong dt_nb2, + ulong A_nb1, + ulong B_nb2, ulong B_nb3, + ulong C_nb2, ulong C_nb3, + ulong s_off_bytes, + int head_dim, int n_head, int n_group, int n_tokens +) { + const int d_state = 128; + + const int tid = (int) get_local_id(0); + const int wg_x = (int) get_group_id(0); + const int seq_id = (int) get_group_id(1); + + const int head_id = wg_x / head_dim; + const int dim_id = wg_x - head_id * head_dim; + const int g = head_id / (n_head / n_group); + + src0_base += src0_off; + src1_base += src1_off; + src2_base += src2_off; + src3_base += src3_off; + src4_base += src4_off; + src5_base += src5_off; + src6_base += src6_off; + dst_base += dst_off; + + const int seq_slot = ((global const int *) src6_base)[seq_id]; + + const ulong state_base_off = (ulong)seq_slot * s0_nb3 + (ulong)head_id * s0_nb2 + + (ulong)dim_id * d_state * sizeof(float); + global const float * s0_warp = (global const float *)(src0_base + state_base_off); + const ulong state_out_off = (ulong)seq_id * s0_nb3 + (ulong)head_id * s0_nb2 + + (ulong)dim_id * d_state * sizeof(float); + global float * s_warp = (global float *)(dst_base + s_off_bytes + state_out_off); + + global const char * x_seq = src1_base + (ulong)seq_id * x_nb3; + global const char * dt_seq = src2_base + (ulong)seq_id * dt_nb2; + global const char * B_seq = src4_base + (ulong)seq_id * B_nb3 + (ulong)g * d_state * sizeof(float); + global const char * C_seq = src5_base + (ulong)seq_id * C_nb3 + (ulong)g * d_state * sizeof(float); + + const ulong y_dim_total = (ulong)n_head * head_dim; + global float * y_seq = (global float *)dst_base + + (ulong)seq_id * (ulong)n_tokens * y_dim_total; + + const float A_val = ((global const float *)src3_base)[(ulong)head_id * A_nb1 / sizeof(float)]; + + // c_factor = 2: each thread owns 2 state elements (tid and tid+64). + float state0 = s0_warp[tid]; + float state1 = s0_warp[tid + 64]; + + for (int t = 0; t < n_tokens; ++t) { + const float dt_h = ((global const float *)(dt_seq + (ulong)t * dt_nb1))[head_id]; + const float dt_softplus = softplus_f32(dt_h); + const float dA = exp(dt_softplus * A_val); + const float x_val = ((global const float *)(x_seq + (ulong)t * x_nb2))[(ulong)head_id * head_dim + dim_id]; + const float x_dt = x_val * dt_softplus; + + const float B0 = ((global const float *)(B_seq + (ulong)t * B_nb2))[tid]; + const float B1 = ((global const float *)(B_seq + (ulong)t * B_nb2))[tid + 64]; + const float C0 = ((global const float *)(C_seq + (ulong)t * C_nb2))[tid]; + const float C1 = ((global const float *)(C_seq + (ulong)t * C_nb2))[tid + 64]; + + state0 = state0 * dA + B0 * x_dt; + state1 = state1 * dA + B1 * x_dt; + const float partial = state0 * C0 + state1 * C1; + + const float sum = sub_group_reduce_add(partial); + if (tid == 0) { + y_seq[(ulong)t * y_dim_total + (ulong)head_id * head_dim + dim_id] = sum; + } + } + + s_warp[tid] = state0; + s_warp[tid + 64] = state1; +} + +// d_state = 256 (Falcon-H1). WG = 64 threads, each holds 4 state elements. +REQD_SUBGROUP_SIZE_64 +kernel void kernel_ssm_scan_f32_mamba2_d256( + global const char * src0_base, ulong src0_off, + global const char * src1_base, ulong src1_off, + global const char * src2_base, ulong src2_off, + global const char * src3_base, ulong src3_off, + global const char * src4_base, ulong src4_off, + global const char * src5_base, ulong src5_off, + global const char * src6_base, ulong src6_off, + global char * dst_base, ulong dst_off, + ulong s0_nb2, ulong s0_nb3, + ulong x_nb2, ulong x_nb3, + ulong dt_nb1, ulong dt_nb2, + ulong A_nb1, + ulong B_nb2, ulong B_nb3, + ulong C_nb2, ulong C_nb3, + ulong s_off_bytes, + int head_dim, int n_head, int n_group, int n_tokens +) { + const int d_state = 256; + + const int tid = (int) get_local_id(0); + const int wg_x = (int) get_group_id(0); + const int seq_id = (int) get_group_id(1); + + const int head_id = wg_x / head_dim; + const int dim_id = wg_x - head_id * head_dim; + const int g = head_id / (n_head / n_group); + + src0_base += src0_off; + src1_base += src1_off; + src2_base += src2_off; + src3_base += src3_off; + src4_base += src4_off; + src5_base += src5_off; + src6_base += src6_off; + dst_base += dst_off; + + const int seq_slot = ((global const int *) src6_base)[seq_id]; + + const ulong state_base_off = (ulong)seq_slot * s0_nb3 + (ulong)head_id * s0_nb2 + + (ulong)dim_id * d_state * sizeof(float); + global const float * s0_warp = (global const float *)(src0_base + state_base_off); + const ulong state_out_off = (ulong)seq_id * s0_nb3 + (ulong)head_id * s0_nb2 + + (ulong)dim_id * d_state * sizeof(float); + global float * s_warp = (global float *)(dst_base + s_off_bytes + state_out_off); + + global const char * x_seq = src1_base + (ulong)seq_id * x_nb3; + global const char * dt_seq = src2_base + (ulong)seq_id * dt_nb2; + global const char * B_seq = src4_base + (ulong)seq_id * B_nb3 + (ulong)g * d_state * sizeof(float); + global const char * C_seq = src5_base + (ulong)seq_id * C_nb3 + (ulong)g * d_state * sizeof(float); + + const ulong y_dim_total = (ulong)n_head * head_dim; + global float * y_seq = (global float *)dst_base + + (ulong)seq_id * (ulong)n_tokens * y_dim_total; + + const float A_val = ((global const float *)src3_base)[(ulong)head_id * A_nb1 / sizeof(float)]; + + // c_factor = 4: each thread owns 4 state elements. + float state0 = s0_warp[tid]; + float state1 = s0_warp[tid + 64]; + float state2 = s0_warp[tid + 128]; + float state3 = s0_warp[tid + 192]; + + for (int t = 0; t < n_tokens; ++t) { + const float dt_h = ((global const float *)(dt_seq + (ulong)t * dt_nb1))[head_id]; + const float dt_softplus = softplus_f32(dt_h); + const float dA = exp(dt_softplus * A_val); + const float x_val = ((global const float *)(x_seq + (ulong)t * x_nb2))[(ulong)head_id * head_dim + dim_id]; + const float x_dt = x_val * dt_softplus; + + global const float * B_t = (global const float *)(B_seq + (ulong)t * B_nb2); + global const float * C_t = (global const float *)(C_seq + (ulong)t * C_nb2); + + const float B0 = B_t[tid]; + const float B1 = B_t[tid + 64]; + const float B2 = B_t[tid + 128]; + const float B3 = B_t[tid + 192]; + const float C0 = C_t[tid]; + const float C1 = C_t[tid + 64]; + const float C2 = C_t[tid + 128]; + const float C3 = C_t[tid + 192]; + + state0 = state0 * dA + B0 * x_dt; + state1 = state1 * dA + B1 * x_dt; + state2 = state2 * dA + B2 * x_dt; + state3 = state3 * dA + B3 * x_dt; + const float partial = state0 * C0 + state1 * C1 + state2 * C2 + state3 * C3; + + const float sum = sub_group_reduce_add(partial); + if (tid == 0) { + y_seq[(ulong)t * y_dim_total + (ulong)head_id * head_dim + dim_id] = sum; + } + } + + s_warp[tid] = state0; + s_warp[tid + 64] = state1; + s_warp[tid + 128] = state2; + s_warp[tid + 192] = state3; +} + +kernel void kernel_ssm_scan_f32( + global const char * s_buf, + ulong s_off, + global const char * x_buf, + ulong x_off, + global const char * dt_buf, + ulong dt_off, + global const char * A_buf, + ulong A_off, + global const char * B_buf, + ulong B_off, + global const char * C_buf, + ulong C_off, + global const char * ids_buf, + ulong ids_off, + global char * dst_buf, + ulong dst_off, + ulong s_nb2, + ulong s_nb3, + ulong x_nb2, + ulong x_nb3, + ulong dt_nb1, + ulong dt_nb2, + ulong A_nb1, + ulong B_nb2, + ulong B_nb3, + ulong C_nb2, + ulong C_nb3, + ulong state_off, + int head_dim, + int n_head, + int n_group, + int n_tokens, + ulong s_nb1, + ulong x_nb1, + ulong B_nb1, + ulong C_nb1, + uint A_ne0, + uint d_state, + uint n_seqs, + uint K, + local float * reduce +) { + global const char * s_data = s_buf + s_off; + global const char * x_data = x_buf + x_off; + global const char * dt_data = dt_buf + dt_off; + global const char * A_data = A_buf + A_off; + global const char * B_data = B_buf + B_off; + global const char * C_data = C_buf + C_off; + global const int * ids_data = (global const int *) (ids_buf + ids_off); + global float * dst = (global float *) (dst_buf + dst_off); + const uint y_elems = state_off / sizeof(float); + + const uint tid = get_local_id(0); + const uint inner_idx = get_group_id(0); + const uint seq_idx = get_group_id(1); + const uint head_idx = inner_idx / head_dim; + const uint dim_idx = inner_idx - head_idx * head_dim; + const uint group_idx = head_idx / (n_head / n_group); + const uint state_slot = (uint) ids_data[seq_idx]; + + const ulong s_idx = (ulong) state_slot * s_nb3 + + (ulong) head_idx * s_nb2 + + (ulong) dim_idx * s_nb1 + + (ulong) tid * sizeof(float); + float state = *((global const float *) (s_data + s_idx)); + + const ulong A_idx = (ulong) head_idx * A_nb1 + + (ulong) (tid % A_ne0) * sizeof(float); + const float A_value = *((global const float *) (A_data + A_idx)); + + for (int token_idx = 0; token_idx < n_tokens; ++token_idx) { + const ulong x_idx = (ulong) head_idx * x_nb1 + + (ulong) token_idx * x_nb2 + + (ulong) seq_idx * x_nb3 + + (ulong) dim_idx * sizeof(float); + const ulong dt_idx = (ulong) token_idx * dt_nb1 + + (ulong) seq_idx * dt_nb2 + + (ulong) head_idx * sizeof(float); + const ulong B_idx = (ulong) group_idx * B_nb1 + + (ulong) token_idx * B_nb2 + + (ulong) seq_idx * B_nb3 + + (ulong) tid * sizeof(float); + const ulong C_idx = (ulong) group_idx * C_nb1 + + (ulong) token_idx * C_nb2 + + (ulong) seq_idx * C_nb3 + + (ulong) tid * sizeof(float); + + const float x_value = *((global const float *) (x_data + x_idx)); + const float dt_value = *((global const float *) (dt_data + dt_idx)); + const float B_value = *((global const float *) (B_data + B_idx)); + const float C_value = *((global const float *) (C_data + C_idx)); + const float dt_soft_plus = dt_value > 20.0f ? dt_value : log(1.0f + exp(dt_value)); + const float dA = exp(dt_soft_plus * A_value); + const float x_dt = x_value * dt_soft_plus; + + state = mad(state, dA, B_value * x_dt); + reduce[tid] = state * C_value; + barrier(CLK_LOCAL_MEM_FENCE); + + for (uint stride = d_state / 2; stride > 0; stride >>= 1) { + if (tid < stride) { + reduce[tid] += reduce[tid + stride]; + } + barrier(CLK_LOCAL_MEM_FENCE); + } + + if (tid == 0) { + const uint y_idx = dim_idx + head_idx * head_dim + + token_idx * n_head * head_dim + + seq_idx * n_tokens * n_head * head_dim; + dst[y_idx] = reduce[0]; + } + + const uint snapshot_slot = n_tokens - 1 - token_idx; + if (snapshot_slot > 0 && snapshot_slot < K) { + const uint snapshot_idx = y_elems + tid + dim_idx * d_state + + head_idx * d_state * head_dim + + (snapshot_slot * n_seqs + seq_idx) * d_state * head_dim * n_head; + dst[snapshot_idx] = state; + } + barrier(CLK_LOCAL_MEM_FENCE); + } + + const uint state_idx = y_elems + tid + dim_idx * d_state + + head_idx * d_state * head_dim + + seq_idx * d_state * head_dim * n_head; + dst[state_idx] = state; +} diff --git a/ggml/src/ggml-opencl/kernels/unary_ext.cl b/ggml/src/ggml-opencl/kernels/unary_ext.cl new file mode 100644 index 00000000..e86eadfa --- /dev/null +++ b/ggml/src/ggml-opencl/kernels/unary_ext.cl @@ -0,0 +1,85 @@ +#pragma OPENCL EXTENSION cl_khr_fp16 : enable + +//------------------------------------------------------------------------------ +// Extended elementwise unary ops, same variant shape as abs.cl: +// f32, f32_4 (vec4), f16, f16_4 (vec4), f32_nc, f16_nc (stride-addressed). +// +// sgn, step, elu, hardswish, hardsigmoid, floor, ceil, round, trunc. +// +// Semantics match the ggml CPU reference (ggml.c). Values are computed in float +// (the f16 variants read/write half and convert), so the conditional ops match +// the CPU bit-for-bit within tolerance. SEXPR is the scalar form, VEXPR the +// float4 form (vector ternaries need select()). +//------------------------------------------------------------------------------ + +#define UNARY_EXT(NAME, SEXPR, VEXPR) \ +kernel void kernel_##NAME##_f32( \ + global const float * src0, ulong offset0, \ + global float * dst, ulong offsetd) { \ + src0 = (global float*)((global char*)src0 + offset0); \ + dst = (global float*)((global char*)dst + offsetd); \ + float x = src0[get_global_id(0)]; \ + dst[get_global_id(0)] = (SEXPR); \ +} \ +kernel void kernel_##NAME##_f32_4( \ + global const float4 * src0, ulong offset0, \ + global float4 * dst, ulong offsetd) { \ + src0 = (global float4*)((global char*)src0 + offset0); \ + dst = (global float4*)((global char*)dst + offsetd); \ + float4 x = src0[get_global_id(0)]; \ + dst[get_global_id(0)] = (VEXPR); \ +} \ +kernel void kernel_##NAME##_f16( \ + global const half * src0, ulong offset0, \ + global half * dst, ulong offsetd) { \ + src0 = (global half*)((global char*)src0 + offset0); \ + dst = (global half*)((global char*)dst + offsetd); \ + float x = src0[get_global_id(0)]; \ + dst[get_global_id(0)] = (SEXPR); \ +} \ +kernel void kernel_##NAME##_f16_4( \ + global const half4 * src0, ulong offset0, \ + global half4 * dst, ulong offsetd) { \ + src0 = (global half4*)((global char*)src0 + offset0); \ + dst = (global half4*)((global char*)dst + offsetd); \ + float4 x = convert_float4(src0[get_global_id(0)]); \ + dst[get_global_id(0)] = convert_half4(VEXPR); \ +} \ +kernel void kernel_##NAME##_f32_nc( \ + global const char * src0, ulong offset0, \ + global char * dst, ulong offsetd, \ + int ne00, ulong nb00, ulong nb01, ulong nb02, ulong nb03, \ + ulong nb0, ulong nb1, ulong nb2, ulong nb3) { \ + src0 = src0 + offset0; dst = dst + offsetd; \ + const int i3 = get_group_id(2); \ + const int i2 = get_group_id(1); \ + const int i1 = get_group_id(0); \ + for (int i0 = get_local_id(0); i0 < ne00; i0 += get_local_size(0)) { \ + float x = *(global const float *)(src0 + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00); \ + *(global float *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0) = (SEXPR); \ + } \ +} \ +kernel void kernel_##NAME##_f16_nc( \ + global const char * src0, ulong offset0, \ + global char * dst, ulong offsetd, \ + int ne00, ulong nb00, ulong nb01, ulong nb02, ulong nb03, \ + ulong nb0, ulong nb1, ulong nb2, ulong nb3) { \ + src0 = src0 + offset0; dst = dst + offsetd; \ + const int i3 = get_group_id(2); \ + const int i2 = get_group_id(1); \ + const int i1 = get_group_id(0); \ + for (int i0 = get_local_id(0); i0 < ne00; i0 += get_local_size(0)) {\ + float x = *(global const half *)(src0 + i3*nb03 + i2*nb02 + i1*nb01 + i0*nb00); \ + *(global half *)(dst + i3*nb3 + i2*nb2 + i1*nb1 + i0*nb0) = (SEXPR); \ + } \ +} + +UNARY_EXT(sgn, sign(x), sign(x)) +UNARY_EXT(step, x > 0.0f ? 1.0f : 0.0f, select((float4)0.0f, (float4)1.0f, x > 0.0f)) +UNARY_EXT(elu, x > 0.0f ? x : expm1(x), select(expm1(x), x, x > 0.0f)) +UNARY_EXT(hardswish, x * fmin(1.0f, fmax(0.0f, (x + 3.0f) / 6.0f)), x * fmin((float4)1.0f, fmax((float4)0.0f, (x + 3.0f) / 6.0f))) +UNARY_EXT(hardsigmoid, fmin(1.0f, fmax(0.0f, (x + 3.0f) / 6.0f)), fmin((float4)1.0f, fmax((float4)0.0f, (x + 3.0f) / 6.0f))) +UNARY_EXT(floor, floor(x), floor(x)) +UNARY_EXT(ceil, ceil(x), ceil(x)) +UNARY_EXT(round, round(x), round(x)) +UNARY_EXT(trunc, trunc(x), trunc(x)) diff --git a/ggml/src/ggml-openvino/CMakeLists.txt b/ggml/src/ggml-openvino/CMakeLists.txt index cc089b72..af3e0758 100644 --- a/ggml/src/ggml-openvino/CMakeLists.txt +++ b/ggml/src/ggml-openvino/CMakeLists.txt @@ -1,6 +1,8 @@ find_package(OpenVINO REQUIRED COMPONENTS Runtime Threading) find_package(OpenCL REQUIRED) +message(STATUS "Found OpenVINO: ${OpenVINO_DIR} (found version \"${OpenVINO_VERSION}\")") + file(GLOB_RECURSE GGML_HEADERS_OPENVINO "*.h" "*.hpp") file(GLOB_RECURSE GGML_SOURCES_OPENVINO "*.cpp") diff --git a/ggml/src/ggml-openvino/ggml-decoder.cpp b/ggml/src/ggml-openvino/ggml-decoder.cpp index 599f41ae..cd06b22e 100644 --- a/ggml/src/ggml-openvino/ggml-decoder.cpp +++ b/ggml/src/ggml-openvino/ggml-decoder.cpp @@ -117,16 +117,7 @@ bool is_same_shape(const ggml_tensor * a, const ggml_tensor * b) { bool is_conv_states_all_tensor(const ggml_tensor * tensor) { return tensor != nullptr && strncmp(tensor->name, "conv_states_all", strlen("conv_states_all")) == 0; } - -// CPY writing the tail of conv_input (the concat of the previous conv state and the new tokens) -// back into a slot block of the recurrent state cache. Detected structurally because the rollback -// variant (cparams.n_rs_seq > 0) emits one such CPY per snapshot slot without naming them. -bool is_conv_state_writeback(const ggml_tensor * node) { - return node->op == GGML_OP_CPY && node->view_src != nullptr && GgmlOvDecoder::is_kvcache(node->view_src, nullptr) && - node->src[0] != nullptr && node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr && - node->src[0]->src[0]->op == GGML_OP_CONCAT && node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && - node->src[1]->view_src == node->view_src; -} +} // namespace // MoE expert aggregation (build_moe_ffn in llama-graph.cpp): each expert plane is // `ggml_view_2d(experts, n_embd, n_tokens, experts->nb[2], i*experts->nb[1])` and the planes @@ -174,20 +165,31 @@ bool is_moe_expert_sum_add(const ggml_tensor * node) { return base != nullptr && base->ne[1] > 1 && plane_indices.size() == static_cast(base->ne[1]); } -} // namespace -static std::string get_tensor_ov_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor) { +std::string GgmlOvDecoder::get_tensor_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor) { if (tensor == nullptr) { return ""; } - const size_t hash_pos = ggml_hash_find(&cgraph->visited_hash_set, tensor); - if (((tensor->flags & GGML_TENSOR_FLAG_COMPUTE) || GgmlOvDecoder::is_kvcache(tensor, nullptr)) && - hash_pos != GGML_HASHSET_FULL && ggml_bitset_get(cgraph->visited_hash_set.used, hash_pos)) { - return std::string(tensor->name) + "#" + std::to_string(hash_pos); + if ((tensor->flags & GGML_TENSOR_FLAG_COMPUTE) || is_kvcache(tensor, nullptr)) { + // Hash-table slots depend on tensor addresses and differ between contexts. + // Graph ordinals disambiguate duplicate names while keeping compiled-model + // ports identical for equivalent graphs in different contexts. + const auto * node = std::find(cgraph->nodes, cgraph->nodes + cgraph->n_nodes, tensor); + if (node != cgraph->nodes + cgraph->n_nodes) { + return std::string(tensor->name) + "#n" + std::to_string(node - cgraph->nodes); + } + const auto * leaf = std::find(cgraph->leafs, cgraph->leafs + cgraph->n_leafs, tensor); + if (leaf != cgraph->leafs + cgraph->n_leafs) { + return std::string(tensor->name) + "#l" + std::to_string(leaf - cgraph->leafs); + } } return tensor->name; } +static std::string get_tensor_ov_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor) { + return GgmlOvDecoder::get_tensor_name(cgraph, tensor); +} + static std::string get_tensor_graph_input_ov_name(const GgmlOvDecoder * decoder, const ggml_cgraph * cgraph, const ggml_tensor * tensor, @@ -198,8 +200,20 @@ static std::string get_tensor_graph_input_ov_name(const GgmlOvDecoder * decoder, if (GgmlOvDecoder::is_inp_emb(tensor, op)) { return "embd"; } - if (decoder->is_stateful() && GgmlOvDecoder::is_inp_mask(tensor, op)) { - return std::string(tensor->name).find("swa") == std::string::npos ? "self_kq_mask" : "self_kq_mask_swa"; + if (GgmlOvDecoder::is_inp_mask(tensor, op)) { + // Give the two attention masks distinct OV parameter names. build_attn_inp_kq_mask() + // names the full-attention mask and the sliding-window mask identically, so keying a + // parameter off the name alone makes the second mask overwrite the first and both + // attention types read one parameter. Tell them apart by tensor identity, using the + // SWA classification computed in compute_llm_params(). An empty swa_layers set means + // there is only one mask in play and the plain name is correct. + const bool is_swa = decoder->is_swa_mask(tensor); + if (decoder->is_stateful()) { + return is_swa ? "self_kq_mask_swa" : "self_kq_mask"; + } + if (is_swa) { + return get_tensor_ov_name(cgraph, tensor) + "_swa"; + } } return get_tensor_ov_name(cgraph, tensor); } @@ -231,7 +245,7 @@ void GgmlOvDecoder::set_input_output() { if (src->op == GGML_OP_VIEW) { // Traverse upward through nested VIEW operations std::remove_reference_t view_chain; - auto current = src; + auto * current = src; while (current != nullptr) { auto current_name = get_tensor_ov_name(m_cgraph, current); @@ -318,9 +332,7 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const { break; } case GGML_OP_MUL_MAT: { - if (node->src[0]->op == GGML_OP_VIEW && node->src[1]->op == GGML_OP_VIEW) { - op_case = 3; - } else if (node->src[1]->op == GGML_OP_SOFT_MAX) { + if (node->src[1]->op == GGML_OP_SOFT_MAX) { // In the case of `-fa off`, softmax is used, v_trans=true, the dynamic dim is ne[0] for cache_v op_case = 2; } @@ -357,6 +369,18 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const { break; } case GGML_OP_VIEW: { + if (m_is_static && node->src[0] != nullptr && + (node->src[0]->op == GGML_OP_GATED_DELTA_NET || node->src[0]->op == GGML_OP_CONCAT)) { + // VIEW slicing a GATED_DELTA_NET combined [attn|state] output, or the conv_input + // CONCAT. The consuming CPY/RMS_NORM op recovers the true window at runtime via + // ssm_state_size / the fixed conv kernel width, so this VIEW must stay an identity + // pass-through of the full source here too (it already is on the dynamic path); + // otherwise the generic static-mode Slice below would bake in the *captured* + // cgraph's token count, which is wrong once the compiled static model runs with a + // different token count (prefill chunk size or 1). + op_case = 1; + break; + } if (node->src[0]->op == GGML_OP_VIEW) { auto * src = node->src[0]; if (ggml_nelements(node) != ggml_nelements(src)) { @@ -408,11 +432,28 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const { } break; } + case GGML_OP_POOL_2D: { + const ggml_op_pool pool_mode = static_cast(node->op_params[0]); + switch (pool_mode) { + case GGML_OP_POOL_MAX: { + op_case = 1; + break; + } + case GGML_OP_POOL_AVG: { + op_case = 2; + break; + } + default: + op_case = 0; + break; + } + break; + } case GGML_OP_CPY: { if (node->src[0]->op == GGML_OP_VIEW) { if (node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) { op_case = 1; - } else if (is_conv_state_writeback(node)) { + } else if (GgmlOvDecoder::is_conv_state_writeback(node)) { op_case = 2; break; } else if (is_conv_states_all_tensor(node->view_src) && node->src[1] != nullptr && @@ -425,6 +466,31 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const { is_kvcache(node->src[1]->view_src, nullptr)) { // s_copy defrag remainder writeback: gathered extra state rows copied back into the cache op_case = 3; + } else if (node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr) { + // op_case 5: KV write for decoder self-attention (dynamic write offset) + // op_case 6: KV write for encoder self-attn or cross-attn (static offset) + const ggml_tensor * kv_buf = node->src[1]->view_src; + if (kv_buf->ne[1] == 1 && kv_buf->ne[2] == 1 && kv_buf->ne[3] == 1) { + op_case = 6; + // Forward-scan the graph for a FLASH_ATTN_EXT that reads from + // the same buffer. Having a mask (src[3] != nullptr) implies + // decoder self-attention and the write offset is dynamic. + for (int i = 0; i < m_cgraph->n_nodes; i++) { + const ggml_tensor * n = m_cgraph->nodes[i]; + if (n->op != GGML_OP_FLASH_ATTN_EXT) { + continue; + } + // K (src[1]) and V (src[2]) are 3-D views whose view_src is + // the flat KV buffer we are writing to. + if ((n->src[1] != nullptr && n->src[1]->view_src == kv_buf) || + (n->src[2] != nullptr && n->src[2]->view_src == kv_buf)) { + if (n->src[3] != nullptr) { + op_case = 5; // decoder self-attention: mask present + } + break; + } + } + } } break; } @@ -448,6 +514,15 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const { } break; } + case GGML_OP_FLASH_ATTN_EXT: { + if (node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr) { + const ggml_tensor * kv_buf = node->src[1]->view_src; + if (kv_buf->ne[1] == 1 && kv_buf->ne[2] == 1 && kv_buf->ne[3] == 1) { + op_case = (node->src[3] != nullptr) ? 1 : 2; + } + } + break; + } default: break; } @@ -469,6 +544,40 @@ std::optional extract_layer_from_name(const std::string & name) { return layer; } +// Recover the sliding window width from ggml's own SWA mask. llama.cpp never passes n_swa to a +// backend, but fill_mask() writes it into the mask: a query row keeps exactly the cells inside +// its window, so the widest row counts min(pos + 1, n_swa) unmasked cells. Counting rather than +// looking for a contiguous band is what makes this work on the KV-cache mask, where columns are +// physical cache cells in arbitrary order, not positions. +// Assumes LLAMA_SWA_TYPE_STANDARD, the only type the caller reconstructs. +static int get_swa_window_from_mask(const ggml_tensor * mask) { + if (mask->data == nullptr || !ggml_backend_buffer_is_host(mask->buffer)) { + return -1; + } + if (mask->type != GGML_TYPE_F16 && mask->type != GGML_TYPE_F32) { + return -1; + } + + const int64_t n_kv = mask->ne[0]; + const int64_t n_tokens = mask->ne[1]; + int64_t window = 0; + + for (int64_t r = 0; r < n_tokens; r++) { + int64_t kept = 0; + for (int64_t c = 0; c < n_kv; c++) { + const size_t i = (size_t) r * n_kv + c; + const float v = mask->type == GGML_TYPE_F16 ? ggml_fp16_to_fp32(((const ggml_fp16_t *) mask->data)[i]) : + ((const float *) mask->data)[i]; + if (v > -INFINITY) { + kept++; + } + } + window = std::max(window, kept); + } + + return window > 0 ? (int) window : -1; +} + std::pair GgmlOvDecoder::compute_llm_params(ggml_cgraph * cgraph, bool is_static) { ModelParams model_params; ComputeParams compute_params; @@ -479,23 +588,34 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr switch (node->op) { case GGML_OP_FLASH_ATTN_EXT: - if (node->src[0] == nullptr || node->src[1] == nullptr || node->src[3] == nullptr) { + if (node->src[0] == nullptr || node->src[1] == nullptr) { return -1; } switch (node->src[1]->op) { case GGML_OP_PERMUTE: - // case 0: node op is FLASH_ATTN_EXT, src 1 not null & op is PERMUTE & the permuted tensor src is the view of cache k - if (node->src[1]->src[0] != nullptr && node->src[1]->src[0]->op == GGML_OP_VIEW) { + // case 0: src[1] is PERMUTE of a cache VIEW, mask required + if (node->src[3] != nullptr && node->src[1]->src[0] != nullptr && + node->src[1]->src[0]->op == GGML_OP_VIEW) { return 0; } break; case GGML_OP_CPY: - // case 1: node op is FLASH_ATTN_EXT, src 1 not null & op is CPY & the copied tensor src is PERMUTE & the permuted tensor src is the view of cache k - if (node->src[1]->src[0] != nullptr && node->src[1]->src[0]->op == GGML_OP_PERMUTE && - node->src[1]->src[0]->src[0] != nullptr && node->src[1]->src[0]->src[0]->op == GGML_OP_VIEW) { + // case 1: src[1] is CPY of a PERMUTE(VIEW), mask required + if (node->src[3] != nullptr && node->src[1]->src[0] != nullptr && + node->src[1]->src[0]->op == GGML_OP_PERMUTE && node->src[1]->src[0]->src[0] != nullptr && + node->src[1]->src[0]->src[0]->op == GGML_OP_VIEW) { return 1; } break; + case GGML_OP_VIEW: + // cases 4/5/6: whisper - K is a direct non-contiguous VIEW_3D of a KV cache + if (node->src[1]->view_src != nullptr) { + if (node->src[3] != nullptr) { + return 4; // decoder self-attention + } + return 5; // cross-attention or encoder self-attention + } + break; default: break; } @@ -522,10 +642,100 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr return -1; }; + // Resolve the attention mask an attention node consumes, mirroring the src layout that + // get_attention_pattern_case() classifies. Used by the SWA pre-pass below. + auto get_attention_op_mask = [&get_attention_pattern_case](const ggml_tensor * node) -> const ggml_tensor * { + switch (get_attention_pattern_case(node)) { + case 0: + case 1: + return node->src[3]; + case 2: + case 3: + return node->src[1]; + default: + return nullptr; + } + }; + + // Pre-pass: classify sliding-window vs full-attention layers. + // + // An interleaved-SWA model keeps two KV caches and two attention masks, and hands each layer + // whichever pair matches its attention type. The mask tensor does not say which is which: both + // are named "attn_inp_kq_mask" by build_attn_inp_kq_mask(), and both carry the same n_kv because + // llama_kv_cache::get_n_kv() pads occupancy up to a common multiple. + // + // The KV cache does say. Each cache allocates cache_k_l once at load time with its own cell + // count: the windowed cache is sized from the window + // (PAD(min(size_base, n_swa*(unified ? n_seq_max : 1) + n_ubatch), 256), see + // llama_kv_cache_iswa), the full-attention one spans the whole context. Read the LEAF buffer + // behind the VIEW rather than the VIEW itself: the leaf extent is a constant per layer, known + // from the first graph onwards, while the view grows with context depth and would invert the + // comparison at shallow depth. + // + // Layers whose leaf is smaller than the largest leaf are the windowed ones. When every layer + // reports the same extent there is no distinction to draw -- either the model has no windowed + // layers, or the window is at least as large as the context so the two caches coincide, in + // which case a windowed layer and a full-attention one compute the same thing. + // + // Getting this wrong is silent and severe: with the windowed layers classified as + // full-attention, permute's KV slicing uses attention_size instead of attention_size_swa. The + // two agree while the context is shorter than the window, then diverge, and the mask add fails + // shape inference ("Failed to broadcast-merge input shapes") partway into a long prompt. + { + std::map layer_extent; // layer -> leaf cache_k cell count + std::map layer_mask; // layer -> mask it consumes + int64_t max_extent = 0; + + for (int i = 0; i < cgraph->n_nodes; i++) { + const ggml_tensor * mask = get_attention_op_mask(cgraph->nodes[i]); + if (mask == nullptr) { + continue; + } + const ggml_tensor * cache_k_permute = nullptr; + switch (get_attention_pattern_case(cgraph->nodes[i])) { + case 0: cache_k_permute = cgraph->nodes[i]->src[1]; break; + case 1: cache_k_permute = cgraph->nodes[i]->src[1]->src[0]; break; + case 2: cache_k_permute = cgraph->nodes[i]->src[0]->src[0]; break; + default: cache_k_permute = cgraph->nodes[i]->src[0]->src[0]->src[0]; break; + } + const ggml_tensor * cache_k_view = cache_k_permute->src[0]; + if (cache_k_view->op != GGML_OP_VIEW) { + continue; + } + const ggml_tensor * leaf = cache_k_view->src[0]; + auto layer = extract_layer_from_name(leaf->name); + if (!layer.has_value()) { + continue; + } + layer_extent[layer.value()] = leaf->ne[1]; + layer_mask[layer.value()] = mask; + max_extent = std::max(max_extent, leaf->ne[1]); + } + + for (const auto & [layer, extent] : layer_extent) { + if (extent < max_extent) { + model_params.swa_layers.push_back(layer); + if (model_params.swa_mask == nullptr) { + model_params.swa_mask = layer_mask[layer]; + } + } + } + std::sort(model_params.swa_layers.begin(), model_params.swa_layers.end()); + + if (ggml_openvino_getenv_int("GGML_OPENVINO_LOG_SWA_LAYERS")) { + std::string per_layer; + for (const auto & [layer, extent] : layer_extent) { + per_layer += " " + std::to_string(layer) + ":" + std::to_string(extent) + + (extent < max_extent ? "(swa)" : ""); + } + GGML_LOG_WARN("ov-swa: attn_layers=%zu max_extent=%ld swa_layers=%zu |%s\n", layer_extent.size(), + (long) max_extent, model_params.swa_layers.size(), per_layer.c_str()); + } + } + bool rope_seen = false; for (int i = 0; i < cgraph->n_nodes; i++) { - auto * node = cgraph->nodes[i]; - std::string name = std::string(node->name); + ggml_tensor * node = cgraph->nodes[i]; const int attention_pattern_case = get_attention_pattern_case(node); if (attention_pattern_case != -1) { ggml_tensor * cache_k_permute = nullptr; @@ -548,6 +758,18 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr cache_k_permute = node->src[0]->src[0]->src[0]; mask = node->src[1]; break; + case 4: + case 5: { + // whisper: K is a direct VIEW_3D of the KV buffer, no PERMUTE node + auto * cache_k_view = node->src[1]; // VIEW_3D of kv_self.k or kv_cross.k` + compute_params.token_len_per_seq = node->src[0]->ne[1]; + if (attention_pattern_case == 4) { + compute_params.attention_size = cache_k_view->ne[1]; + } else { + compute_params.attention_size_static = cache_k_view->ne[1]; + } + continue; + } default: break; } @@ -567,11 +789,14 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr ggml_tensor * cache_k = cache_k_view->src[0]; int layer = extract_layer_from_name(cache_k->name).value(); - std::string mask_name(mask->name); + // Classified by the pre-pass above, which groups layers by mask tensor identity. The + // mask NAME cannot be used: build_attn_inp_kq_mask() gives both masks the same name. + const bool layer_is_swa = std::find(model_params.swa_layers.begin(), model_params.swa_layers.end(), + layer) != model_params.swa_layers.end(); model_params.kv_buffer_ctx_id = ggml_backend_openvino_buffer_get_ctx_id(cache_k->buffer); - if (mask_name.find("swa") != std::string::npos) { - model_params.swa_layers.push_back(layer); + model_params.n_heads_kv_per_layer[layer] = cache_k_permute->ne[2]; + if (layer_is_swa) { model_params.ctx_per_seq_swa = cache_k->ne[1]; } else { model_params.ctx_per_seq = cache_k->ne[1]; @@ -584,8 +809,9 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr memcpy(&offset, cache_k_view->op_params, sizeof(size_t)); compute_params.seq_active_start = offset / seq_size; - if (mask_name.find("swa") != std::string::npos) { + if (layer_is_swa) { compute_params.attention_size_swa = mask->ne[0]; + compute_params.swa_window = get_swa_window_from_mask(mask); } else { compute_params.attention_size = mask->ne[0]; } @@ -621,11 +847,11 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr // mixed SWA/non-SWA layers with different n_dims or freq_base), we cannot // share a single precomputed rope_sin/rope_cos. Track divergence so the // translator falls back to per-op make_sin_cos in that case. - static_assert(sizeof(model_params.rope_params) == sizeof(int32_t) * 15, "rope_params size"); + static_assert(sizeof(model_params.rope_params) == sizeof(int32_t) * 16, "rope_params size"); if (!rope_seen) { - memcpy(model_params.rope_params, node->op_params, sizeof(int32_t) * 15); + memcpy(model_params.rope_params, node->op_params, sizeof(int32_t) * 16); rope_seen = true; - } else if (memcmp(model_params.rope_params, node->op_params, sizeof(int32_t) * 15) != 0) { + } else if (memcmp(model_params.rope_params, node->op_params, sizeof(int32_t) * 16) != 0) { model_params.mixed_rope_params = true; } } @@ -654,10 +880,8 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr ComputeParams::RsWriteback writeback; writeback.slot_begin = (int) (dest_view->view_offs / row_bytes); if (is_conv) { - // conv_input column the copied window starts at writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[0]); } else if (is_gdn) { - // first row of the state part of the gated-delta-net output writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[1]); } compute_params.rs_writebacks[get_tensor_ov_name(cgraph, node)] = writeback; @@ -667,8 +891,41 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr } } } + if (model_params.n_heads_kv == -1) { + for (int i = 0; i < cgraph->n_nodes; i++) { + const auto * node = cgraph->nodes[i]; + const ggml_tensor * mask = nullptr; + if (node->op == GGML_OP_SOFT_MAX) { + mask = node->src[1]; + } else if (node->op == GGML_OP_FLASH_ATTN_EXT) { + mask = node->src[3]; + } else { + continue; + } + if (mask == nullptr || mask->op != GGML_OP_NONE || !(mask->flags & GGML_TENSOR_FLAG_INPUT) || + node->src[0] == nullptr) { + continue; + } + model_params.is_cacheless_attn = true; + model_params.n_seq = 1; + model_params.ctx_per_seq = mask->ne[0]; + compute_params.input_len = node->src[0]->ne[1]; + compute_params.token_len_per_seq = compute_params.input_len; + break; + } + } + auto * output_tensor = cgraph->nodes[cgraph->n_nodes - 1]; compute_params.output_len = output_tensor->ne[1]; + if (model_params.is_cacheless_attn) { + for (int i = 0; i < cgraph->n_nodes; i++) { + const auto * node = cgraph->nodes[i]; + if (node->op == GGML_OP_GET_ROWS && is_output_idx(node->src[1], node)) { + compute_params.output_len = node->src[1]->ne[0]; + break; + } + } + } // for NPU, output_len is always 1 except for llama-perplexity if (is_static && compute_params.output_len == 0) { compute_params.output_len = 1; @@ -689,7 +946,6 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, if (m_naive) { return input != nullptr ? ov::PartialShape{get_shape(input)} : ov::PartialShape{get_shape(op)}; } - auto name = std::string(input->name); ov::PartialShape input_shape; if (is_inp_tok(input, op) || is_inp_pos(input, op)) { @@ -705,6 +961,10 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, // output index input_shape = ov::PartialShape{1, 1, 1, m_is_static ? m_compute_params.output_len : -1}; + } else if (is_inp_mean(input, op)) { + input_shape = m_is_static ? ov::PartialShape{1, 1, input->ne[1], m_prefill_chunk_size} : + ov::PartialShape{1, 1, -1, -1}; + } else if (is_inp_mask(input, op)) { // mask if (m_is_static) { @@ -718,18 +978,30 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, } else if (is_kvcache(input, op)) { // kvcache input_shape = ov::PartialShape{get_shape(input)}; - if (!m_is_static) { + // Whisper.cpp uses a fixed size 1D KV buffer [N, 1, 1, 1] (GGML) or [1, 1, 1, N] (OV). + // the token fill level is handled by token_len_per_seq + dynamic mask input. + // skip dynamic dim and stateful reshape for this layout. + const bool is_flat_kv = (input->ne[1] == 1 && input->ne[2] == 1 && input->ne[3] == 1); + if (!m_is_static && !is_flat_kv) { // do not fix ctx size to make llama-bench work across test params input_shape[2] = -1; } - if (is_stateful()) { + if (is_stateful() && !is_flat_kv) { // Convert stateless KV cache layout [1, 1, seq, n_heads_kv * head_size] // to stateful layout [1, seq, n_heads_kv, head_size]. + // NOTE: Gemma4 uses per-layer-type KV shapes, so no single scalar describes every + // layer. E2B varies only the head size (sliding 256, full 512); 12B also varies the + // head COUNT (sliding 8 x 256, full 1 x 512). Take the head count for this tensor's + // own layer type and derive the head size from its own combined dim, so both layer + // types get the correct split. Using the model-level count split 12B's sliding + // states as 1 x 2048 and decoded garbage. assert(input_shape.size() == 4 && input_shape[0] == 1 && input_shape[1] == 1 && - input_shape[2].is_dynamic() && - input_shape[3] == (m_model_params.n_heads_kv * m_model_params.head_size)); - input_shape = {input_shape[0], ov::Dimension::dynamic(), m_model_params.n_heads_kv, - m_model_params.head_size}; + input_shape[2].is_dynamic() && input_shape[3].is_static()); + const int n_heads_kv = get_n_heads_kv_for_tensor(input); + assert(n_heads_kv > 0 && input_shape[3].get_length() % n_heads_kv == 0); + const int64_t combined_dim = input_shape[3].get_length(); // n_heads_kv * head_size + const int64_t head_size = combined_dim / n_heads_kv; + input_shape = {input_shape[0], ov::Dimension::dynamic(), n_heads_kv, head_size}; } } else if (is_kv_idx(input, op)) { @@ -738,7 +1010,9 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, input_shape = ov::PartialShape{1, 1, 1, len}; } else if (is_inp_s_copy(input, op) || is_s_copy_leaf(input)) { - input_shape = ov::PartialShape{1, 1, 1, -1}; + // On NPU the total slot count (n_seq_max) is fixed at translation time, so the s_copy + // index list has a static length; on CPU/GPU it may change across compiles (defrag). + input_shape = m_is_static ? ov::PartialShape{get_shape(input)} : ov::PartialShape{1, 1, 1, -1}; } else { input_shape = ov::PartialShape{get_shape(input)}; @@ -749,8 +1023,14 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, if (op->op == GGML_OP_SOFT_MAX && op->src[1] != nullptr && op->src[1]->op == GGML_OP_NONE && op->src[1]->flags & GGML_TENSOR_FLAG_INPUT && op->src[1] == input) { // for softmax input mask, the shape is [1, 1, seq_active, seq_active], where seq_active is determined by the input active sequence length instead of the kv cache sequence length - input_shape[2] = -1; - input_shape[3] = -1; + if (m_is_static) { + const int64_t seq_active = m_is_prefill ? m_prefill_chunk_size : 1; + input_shape[2] = seq_active; + input_shape[3] = seq_active; + } else { + input_shape[2] = -1; + input_shape[3] = -1; + } } return input_shape; } @@ -790,16 +1070,23 @@ void GgmlOvDecoder::add_extra_inputs() { // see llama_kv_cache_unified::get_n_kv and llama_kv_cache_unified::get_padding. // 2. `n_seq_active` and `seq_active_start`, used in FLASH_ATTN_EXT to indicate the active sequences in the batch - auto create_1d_input = [this](const std::string & name, int64_t value) { - m_model_extra_inputs[name] = {ov::element::i64, ov::Shape{1}, value, !m_is_static}; + auto create_1d_input = [this](const std::string & name, int64_t value, bool force_parameter = false) { + m_model_extra_inputs[name] = {ov::element::i64, ov::Shape{1}, value, force_parameter || !m_is_static}; }; if (m_compute_params.attention_size != -1) { create_1d_input("attention_size", m_compute_params.attention_size); } + if (m_compute_params.attention_size_static != -1) { + create_1d_input("attention_size_static", m_compute_params.attention_size_static); + } if (m_compute_params.attention_size_swa != -1) { create_1d_input("attention_size_swa", m_compute_params.attention_size_swa); } + // only the stateful SWA mask consumes this + if (is_stateful() && m_compute_params.swa_window != -1) { + create_1d_input("swa_window", m_compute_params.swa_window); + } create_1d_input("n_seq_active", m_compute_params.n_seq_active); create_1d_input("seq_active_start", m_compute_params.seq_active_start); create_1d_input("seq_active_end", m_compute_params.seq_active_start + m_compute_params.n_seq_active); @@ -809,17 +1096,32 @@ void GgmlOvDecoder::add_extra_inputs() { // create_1d_input("token_len", m_compute_params.token_len_per_seq * m_compute_params.n_seq_active); if (m_compute_params.cache_rs_reset_idx != -1) { - create_1d_input("cache_rs_reset_idx", m_compute_params.cache_rs_reset_idx); - create_1d_input("cache_rs_reset_len", m_compute_params.cache_rs_reset_len); + // Whether/which cache slot to reset varies per compute call (e.g. a new sequence starting + // vs. continued decoding). can_reuse_statically() does not invalidate the cached static + // model on ComputeParams changes, so these must stay runtime Parameters even when static + // (scale.cpp op_case 1 only uses them in value comparisons, never as Slice bounds, so this + // does not reintroduce dynamic shapes). + create_1d_input("cache_rs_reset_idx", m_compute_params.cache_rs_reset_idx, /*force_parameter=*/true); + create_1d_input("cache_rs_reset_len", m_compute_params.cache_rs_reset_len, /*force_parameter=*/true); } if (m_compute_params.s_copy_active_slot_len != -1) { create_1d_input("s_copy_active_slot_len", m_compute_params.s_copy_active_slot_len); + if (m_is_static) { + // Number of real tokens in the current prefill chunk. The last chunk is padded with + // fabricated token ids; attention masks them out, but the recurrent (GDN/conv) path + // would otherwise fold them into cache_r/cache_s permanently. Varies per chunk, so it + // must stay a runtime Parameter; it is only compared against a Range or used as Gather + // indices, so it does not make any shape dynamic. + create_1d_input("chunk_valid_len", get_static_n_tokens(), /*force_parameter=*/true); + } } for (const auto & [node_name, writeback] : m_compute_params.rs_writebacks) { create_1d_input("rs_slot_begin_" + node_name, writeback.slot_begin); - create_1d_input("rs_src_begin_" + node_name, writeback.src_begin); + if (!m_is_static) { + create_1d_input("rs_src_begin_" + node_name, writeback.src_begin); + } } } @@ -1169,7 +1471,7 @@ std::shared_ptr GgmlOvDecoder::create_weight_node(ggml_tensor * tensor void GgmlOvDecoder::dump_cgraph(const ggml_cgraph * cgraph, std::string & filename) { std::ofstream file(filename); if (!file.is_open()) { - std::cerr << "Failed to open file" << std::endl; + std::cerr << "Failed to open file" << '\n'; return; } @@ -1275,11 +1577,11 @@ void print_tensor_address_map(const ggml_cgraph * cgraph) { } } for (const auto & pair : address_map) { - std::cout << "Address: " << pair.first << std::endl; + std::cout << "Address: " << pair.first << '\n'; for (const auto & name : pair.second) { std::cout << name << " ; "; } - std::cout << std::endl << std::endl; + std::cout << "\n\n"; } } @@ -1785,13 +2087,23 @@ void GgmlOvDecoder::compute_node_dynamic_dims() { auto dynamic_dim_stride = src_logical_nb[dynamic_dim_idx] / ggml_type_size(node->src[0]->type) * ggml_type_size(node->type); int matched_dim_count = 0; + int first_matched_dim = -1; for (int i = 0; i < GGML_MAX_DIMS; i++) { if (node->nb[i] == dynamic_dim_stride && node->ne[i] == node->src[0]->ne[dynamic_dim_idx]) { + if (first_matched_dim == -1) { + first_matched_dim = i; + } m_node_dynamic_dims[node] = i; matched_dim_count++; } } - if (matched_dim_count != 1) { + if (matched_dim_count > 1 && node->src[0]->ne[dynamic_dim_idx] == 1) { + // Single-token capture: every trailing dim is size 1 with the same stride, so + // the match is ambiguous. The lowest index is the real axis; the rest are + // ggml's size-1 padding. Bailing out here would bake the captured token count + // into the static prefill model, which then runs with a different one. + m_node_dynamic_dims[node] = first_matched_dim; + } else if (matched_dim_count != 1) { m_node_dynamic_dims[node] = -1; GGML_LOG_WARN("ggml-openvino: cannot determine dynamic dim for CONT node '%s', src[0]: '%s'\n", node->name, node->src[0]->name); @@ -1911,7 +2223,7 @@ void GgmlOvDecoder::compute_node_dynamic_dims() { std::cout << ", "; } } - std::cout << "]" << std::endl; + std::cout << "]" << '\n'; // print the src name & shape with the dynamic dim for debugging for (int j = 0; j < GGML_MAX_SRC; j++) { ggml_tensor * src = node->src[j]; @@ -1930,9 +2242,9 @@ void GgmlOvDecoder::compute_node_dynamic_dims() { std::cout << ", "; } } - std::cout << "]" << std::endl; + std::cout << "]" << '\n'; } - std::cout << std::endl; + std::cout << '\n'; } } } diff --git a/ggml/src/ggml-openvino/ggml-decoder.h b/ggml/src/ggml-openvino/ggml-decoder.h index 8e39a26c..056e39e8 100644 --- a/ggml/src/ggml-openvino/ggml-decoder.h +++ b/ggml/src/ggml-openvino/ggml-decoder.h @@ -21,18 +21,28 @@ struct ModelParams { int ctx_per_seq_swa = -1; int n_seq = 1; int n_heads_kv = -1; + // Per-layer KV head count. gemma-4 12B interleaves 8 x 256 sliding layers with 1 x 512 + // full-attention layers, so no single scalar describes every layer. Keyed by layer, not by + // layer TYPE, because the SWA classification depends on the context size (extents tie at a + // small -c) while the head count does not. + std::map n_heads_kv_per_layer; int head_size = -1; int state_size = -1; // for SSM molels, eg qwen35 - int32_t rope_params[15]; + int32_t rope_params[16]; bool mixed_rope_params = false; + bool is_cacheless_attn = false; std::vector swa_layers; + // The sliding-window mask tensor, identified in compute_llm_params() by grouping attention + // layers on the mask they consume. Only used to tell the two masks apart when naming OV + // parameters -- both carry the same tensor name. Null when the graph has a single mask. + const ggml_tensor * swa_mask = nullptr; std::vector kv_names; size_t kv_buffer_ctx_id = 0; bool same_rope_params(const ModelParams & other) const { return mixed_rope_params == other.mixed_rope_params && - memcmp(rope_params, other.rope_params, sizeof(int32_t) * 15) == 0; + memcmp(rope_params, other.rope_params, sizeof(int32_t) * 16) == 0; } bool can_reuse_dynamically(const ModelParams & other) const { return same_rope_params(other); } @@ -47,6 +57,12 @@ struct ComputeParams { int seq_active_start = 0; int attention_size = -1; int attention_size_swa = -1; + int attention_size_static = -1; // encoder/cross-attn KV fill level (whisper) + // Sliding window width, read back from the band of ggml's own SWA mask. ggml never passes + // n_swa down to a backend, but fill_mask() bakes it into the mask contents, so the widest + // unmasked row recovers it. Shorter than n_swa while the sequence is still short, which is + // harmless: every causal pair is inside the window then anyway. + int swa_window = -1; int input_len = -1; int token_len_per_seq = -1; int past_kv_len = -1; @@ -84,18 +100,26 @@ struct ComputeParams { struct RsWriteback { int slot_begin = 0; // first cache slot written by the CPY - int src_begin = 0; // where the copied data starts in the source tensor (in rows of it) + int src_begin = 0; // first source row or column copied by the CPY }; std::map rs_writebacks; - // Offsets of the state cache writeback CPY nodes, keyed by node name. They change with the - // batch (kv head, active sequence count, token count) and, with rollback enabled - // (cparams.n_rs_seq > 0), the conv state is written back once per snapshot slot, each snapshot - // taking a different conv_input window. Passed to the cached model as runtime inputs. + // Destination slot offset of each state cache writeback CPY node, keyed by node name. It + // changes with the batch (kv head, active sequence count) and, with rollback enabled + // (cparams.n_rs_seq > 0), the conv state is written back once per snapshot slot. Passed to the + // cached model as a runtime input. Dynamic models also receive the source-side offset; static + // models use a fixed end-anchored offset in the translator. }; +// defined below; declared here because GgmlOvDecoder uses it inline +std::optional extract_layer_from_name(const std::string & name); + +// detects the MoE expert-plane-sum ADD chain (see definition); used by supports_op too +bool is_moe_expert_sum_add(const ggml_tensor * node); + class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { public: + static std::string get_tensor_name(const ggml_cgraph * cgraph, const ggml_tensor * tensor); struct NodeInfo { ggml_tensor * node; std::string node_name; @@ -248,6 +272,21 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { m_model_params.swa_layers.end(); } + // KV head count for one layer. Sliding and full layers can differ (gemma-4 12B), so callers + // that reinterpret a KV buffer must use this and not the model-level n_heads_kv. + int get_n_heads_kv_for_layer(int layer) const { + auto it = m_model_params.n_heads_kv_per_layer.find(layer); + return it != m_model_params.n_heads_kv_per_layer.end() ? it->second : m_model_params.n_heads_kv; + } + + // Same, for a KV cache tensor: its layer comes from the leaf name (cache_k_l). + int get_n_heads_kv_for_tensor(const ggml_tensor * kv_tensor) const { + if (auto layer = extract_layer_from_name(std::string(kv_tensor->name)); layer.has_value()) { + return get_n_heads_kv_for_layer(layer.value()); + } + return m_model_params.n_heads_kv; + } + int get_past_kv_len() const { return m_compute_params.past_kv_len; } int get_input_len() const { return m_compute_params.input_len; } @@ -315,35 +354,41 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { void update_io(ggml_cgraph * cgraph); - inline static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op == GGML_OP_NONE; } - inline static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_ROPE && tensor == op->src[1]; } // IMROPE packs 4 stacked position planes (t/h/w/e) into inp_pos, each of length // n_tokens; other modes carry a single position per token. - inline static int get_inp_pos_n_planes(const ggml_tensor * op) { + static int get_inp_pos_n_planes(const ggml_tensor * op) { return op->op_params[2] == GGML_ROPE_TYPE_IMROPE ? 4 : 1; } - inline static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) { return tensor->op == GGML_OP_GET_ROWS && op->op == GGML_OP_RMS_NORM; } - inline static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_CPY || (op->op == GGML_OP_FLASH_ATTN_EXT && tensor == op->src[3]) || (op->op == GGML_OP_SOFT_MAX && tensor == op->src[1]); } - inline static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_mean(const ggml_tensor * tensor, const ggml_tensor * op) { + return op->op == GGML_OP_MUL_MAT && tensor == op->src[1] && tensor->op == GGML_OP_NONE && + (tensor->flags & GGML_TENSOR_FLAG_INPUT) && tensor->type == GGML_TYPE_F32 && + op->src[0] != nullptr && op->src[0]->op != GGML_OP_NONE; + } + + static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_ROPE && tensor == op->src[2]; } // also returns true for cache_s and cache_r in SSM/DeltaNet models - inline static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) { if (tensor == nullptr) { return false; } @@ -351,17 +396,28 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { (op != nullptr && op->op == GGML_OP_SET_ROWS && op->src[2] == tensor); } - inline static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_conv_state_writeback(const ggml_tensor * node) { + return node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) && + node->src[0] != nullptr && node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr && + node->src[0]->src[0]->op == GGML_OP_CONCAT && node->src[1] != nullptr && + node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src; + } + + static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_SET_ROWS && op->src[1] == tensor; } - inline static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) { + bool is_swa_mask(const ggml_tensor * tensor) const { + return m_model_params.swa_mask != nullptr && tensor == m_model_params.swa_mask; + } + + static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op != GGML_OP_NONE && op->src[1]->op == GGML_OP_NONE; } // the state permutation index input used in SSM/DeltaNet models (inp->s_copy in llama-graph.cpp) - inline static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY; } @@ -373,8 +429,22 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { if (is_inp_emb(tensor, op)) { return "embd"; } - if (is_stateful() && is_inp_mask(tensor, op)) { - return std::string(tensor->name).find("swa") == std::string::npos ? "self_kq_mask" : "self_kq_mask_swa"; + if (is_inp_mask(tensor, op)) { + // Give the two attention masks distinct OV parameter names. + // + // An interleaved-SWA model builds one full-attention mask and one sliding-window mask, + // but build_attn_inp_kq_mask() names them identically, so keying a parameter off + // tensor->name alone makes the second mask OVERWRITE the first in m_model_inputs: both + // attention types then read a single parameter, and the windowed layers silently run + // against an unbanded mask. Disambiguate using the SWA layer set computed in + // compute_llm_params(), which classifies by mask tensor identity rather than by name. + // + // When no SWA layer was found there is only one mask in play, so the plain name is + // correct and no _swa parameter is created. + if (m_model_params.swa_layers.empty()) { + return "self_kq_mask"; + } + return is_swa_mask(tensor) ? "self_kq_mask_swa" : "self_kq_mask"; } return tensor->name; } @@ -411,5 +481,3 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder { }; void print_tensor_address_map(const ggml_cgraph * cgraph); - -std::optional extract_layer_from_name(const std::string & name); diff --git a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp index 36c74924..216e3b8a 100644 --- a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp @@ -31,7 +31,10 @@ void ggml_openvino_device_config::init() { // String values (use ggml_openvino_getenv_str) "GGML_OPENVINO_DEVICE", "GGML_OPENVINO_CACHE_DIR", + "GGML_OPENVINO_SPILL_DIR", "GGML_OPENVINO_DEBUG_NODE", + "GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR", + "GGML_OPENVINO_NPU_COMPILE_CONFIG", // Integer values (use ggml_openvino_getenv_int) "GGML_OPENVINO_PREFILL_CHUNK_SIZE", // Boolean toggles (treated as int flags via ggml_openvino_getenv_int) @@ -41,6 +44,9 @@ void ggml_openvino_device_config::init() { "GGML_OPENVINO_DUMP_IR", "GGML_OPENVINO_DEBUG_INPUT", "GGML_OPENVINO_DEBUG_OUTPUT", + // Force the static (NPU-shape) compute path on any device, e.g. GGML_OPENVINO_DEVICE=CPU, + // to test the static-shape translation without NPUW/real NPU hardware in the loop. + "GGML_OPENVINO_FORCE_STATIC", "GGML_OPENVINO_PRINT_CGRAPH_TENSOR_ADDRESS", "GGML_OPENVINO_ENABLE_CACHE", "GGML_OPENVINO_DISABLE_CACHE", @@ -50,7 +56,12 @@ void ggml_openvino_device_config::init() { "GGML_OPENVINO_MEMORY_OPTIMIZE", "GGML_OPENVINO_RELEASE_WEIGHTS", "GGML_OPENVINO_REDUCE_COMPILE_MEM", - "GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR", + "GGML_OPENVINO_LOG_UNSUPPORTED_OPS", + "GGML_OPENVINO_LOG_SWA_LAYERS", + "GGML_OPENVINO_NATIVE_SOFTPLUS", + "GGML_OPENVINO_DISABLE_REMOTE_OUTPUTS", + "GGML_OPENVINO_REQUANT_KQUANT", + "GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT", }; for (const char * const & env_var : env_var_names) { @@ -85,6 +96,11 @@ void ggml_openvino_device_config::init() { compile_config["NPUW_CACHE_DIR"] = cache_dir; compile_config.insert(ov::cache_mode(ov::CacheMode::OPTIMIZE_SIZE)); } + const char * compilation_mode_params = + ggml_openvino_getenv_str("GGML_OPENVINO_NPU_COMPILE_CONFIG"); + if (compilation_mode_params && strlen(compilation_mode_params) > 0) { + compile_config["NPU_COMPILATION_MODE_PARAMS"] = compilation_mode_params; + } } else if (cache_dir && strlen(cache_dir) > 0) { compile_config.insert(ov::cache_dir(cache_dir)); compile_config.insert(ov::cache_mode(ov::CacheMode::OPTIMIZE_SIZE)); @@ -253,9 +269,81 @@ std::optional ggml_openvino_get_requant_type(const ggml_tensor * if (ggml_openvino_is_npu()) { return ExtraQuantType::Q4_0_128; } + // By default Q6_K/Q5_K are requantized to Q8_0_C, which *inflates* 6- and 5-bit weights to 8 + // while the rest of the model stays at 4 bits, and Q4_K keeps its native group-32 layout + // (an f16 scale plus an f16 zero point per 32 weights = 0.125 B/weight of metadata). + // Decode of a large model is bandwidth-bound, so both cost throughput. + // + // GGML_OPENVINO_REQUANT_KQUANT selects a 4-bit target instead. Names are + // q4_[_all]: says whether a per-group zero point is kept, + // is the group size, and the _all suffix sends Q4_K down the same path (without it only + // Q6_K/Q5_K are touched): + // q4_sym128 Q6_K/Q5_K -> Q4_0_128 (u4, group 128, symmetric) + // q4_sym128_all and Q4_K too -- drops Q4_K's per-32 zero point, which costs some accuracy + // q4_asym64_all Q6_K/Q5_K and Q4_K -> Q4_1_64 (u4, group 64, asymmetric) -- most of the + // metadata saving while keeping a real zero point + // native no requantization at all (keep Q6_K/Q5_K as they are) + // + // The asymmetric target is only offered in its _all form: leaving Q4_K at its native group 32 + // while Q6_K/Q5_K move to group 64 gives the Q/K/V projections different group counts, and the + // GPU plugin's FullyConnectedHorizontalFusion concatenates their scale constants, which then + // fails shape inference. Requantizing all three keeps the group size uniform. + const char * rq = ggml_openvino_getenv_str("GGML_OPENVINO_REQUANT_KQUANT"); + auto is_opt = [rq](const char * name) { + return rq && strcmp(rq, name) == 0; + }; + const bool sym128 = is_opt("q4_sym128"); + const bool sym128_all = is_opt("q4_sym128_all"); + const bool asym64_all = is_opt("q4_asym64_all"); + + if (tensor->type == GGML_TYPE_Q4_K) { + if (sym128_all) { + return ExtraQuantType::Q4_0_128; + } + if (asym64_all) { + return ExtraQuantType::Q4_1_64; + } + } + // MoE expert weights (3D, ne[2] = n_expert) stored as Q5_1/Q8_0 are the expert-side + // equivalent of Q6_K/Q5_K: kept at 8 bits by default while the rest of the model is at 4 + // (gemma-4 26B-A4B keeps its down projection there). Send them to 4 bits under the same + // option, at group 64 rather than 128: the down expert has k=704, which 64 divides + // (704/64 = 11) and 128 does not. + if (tensor->ne[2] > 1 && (tensor->type == GGML_TYPE_Q5_1 || tensor->type == GGML_TYPE_Q8_0)) { + if (sym128 || sym128_all) { + return ExtraQuantType::Q4_0_64; + } + if (asym64_all) { + return ExtraQuantType::Q4_1_64; + } + // TODO: temporary workaround for a known OpenVINO GPU-plugin bug -- remove once the + // plugin computes grouped 8-bit GatherMatmulCompressed correctly. This costs accuracy + // (5/8-bit -> 4-bit) on any model it applies to, so it must not outlive the bug. + // + // On GPU these would otherwise stay in their native *grouped 8-bit* layout, which the GPU + // plugin's GatherMatmulCompressed computes incorrectly -- gemma-4 26B-A4B (whose down + // projection is Q5_1) produces garbage, while the same graph is correct on CPU. It is + // specific to grouped 8 bit: the gate/up experts are grouped u4 *with* a zero point and + // are fine, and Qwen3.5 / granite are fine because their Q5_K/Q6_K down projections + // already requantize to per-channel Q8_0_C (grouped=0). Sending these to grouped 4 bit + // avoids the broken layout and restores correct output. + // Opt out with GGML_OPENVINO_REQUANT_KQUANT=native. + if (ggml_openvino_get_device_name() == "GPU" && !is_opt("native")) { + return ExtraQuantType::Q4_0_64; + } + } switch (tensor->type) { case GGML_TYPE_Q6_K: case GGML_TYPE_Q5_K: + if (sym128 || sym128_all) { + return ExtraQuantType::Q4_0_128; + } + if (asym64_all) { + return ExtraQuantType::Q4_1_64; + } + if (is_opt("native")) { + return std::nullopt; + } return ExtraQuantType::Q8_0_C; default: return std::nullopt; @@ -321,6 +409,16 @@ ggml_openvino_extracted_layout ggml_openvino_get_extracted_layout(const ggml_ten layout.weights_per_block = 128; layout.is_symmetric = true; break; + case ExtraQuantType::Q4_1_64: + layout.is_u4 = true; + layout.weights_per_block = 64; + layout.is_symmetric = false; + break; + case ExtraQuantType::Q4_0_64: + layout.is_u4 = true; + layout.weights_per_block = 64; + layout.is_symmetric = true; + break; case ExtraQuantType::Q4_0_C: layout.is_u4 = true; layout.weights_per_block = tensor->ne[0]; @@ -374,10 +472,6 @@ ggml_openvino_extracted_layout ggml_openvino_get_extracted_layout(const ggml_ten switch (tensor->type) { case GGML_TYPE_MXFP4: - layout.is_u4 = true; - layout.is_symmetric = true; - break; - case GGML_TYPE_Q4_0: layout.is_u4 = true; layout.is_symmetric = true; diff --git a/ggml/src/ggml-openvino/ggml-openvino-extra.h b/ggml/src/ggml-openvino/ggml-openvino-extra.h index 0916b416..9d827d96 100644 --- a/ggml/src/ggml-openvino/ggml-openvino-extra.h +++ b/ggml/src/ggml-openvino/ggml-openvino-extra.h @@ -15,7 +15,10 @@ #include // ExtraQuantType enum - defines requantization target formats -enum class ExtraQuantType { F16, Q4_0_C, Q8_1_C, Q4_0_128, Q8_0_C, Q8_0_32 }; +// Q4_1_64: u4, group 64, *true* asymmetric (per-group scale and zero point). Note that +// Q4_0_128/Q4_0_C are symmetric despite taking the unsigned branch of quantize_q4_0 -- that branch +// pins zp to 8 with d = max/-8, which is algebraically symmetric. +enum class ExtraQuantType { F16, Q4_0_C, Q8_1_C, Q4_0_128, Q4_0_64, Q8_0_C, Q8_0_32, Q4_1_64 }; ov::Core & ov_singleton_core(); diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp index cac83a1b..02c59622 100644 --- a/ggml/src/ggml-openvino/ggml-openvino.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino.cpp @@ -10,7 +10,10 @@ #include "ggml.h" #include +#include +#include #include +#include #include #include #include @@ -25,7 +28,7 @@ #include #include -#if defined(_WIN32) +#ifdef _WIN32 # define WIN32_LEAN_AND_MEAN # ifndef NOMINMAX # define NOMINMAX @@ -53,6 +56,7 @@ // - CPU repack buffer: tensor->extra stores tensor_traits with repacked data // ===================================================== +namespace { // Buffer context that manages per-tensor allocations (no contiguous buffer for weights) struct ggml_backend_openvino_buffer_context { int device; @@ -64,6 +68,11 @@ struct ggml_backend_openvino_buffer_context { size_t size; bool is_remote; + // Set when the buffer is a file-backed spill mapping (GGML_OPENVINO_SPILL_DIR); it must be + // munmap'd rather than freed. + void * spill_mapping = nullptr; + size_t spill_size = 0; + // Wrapping of the buffer std::shared_ptr ov_buffer; @@ -98,10 +107,56 @@ struct ggml_backend_openvino_buffer_context { data = usm_tensor.get(); ov_buffer = std::make_shared(std::move(usm_tensor)); } else { - data = ggml_aligned_malloc(size); - GGML_ASSERT(data); - memset(data, 0, size); - ov_buffer = std::make_shared(ov::element::u8, ov::Shape{size}, data); +#ifndef _WIN32 + if (const char * spill_dir = ggml_openvino_getenv_str("GGML_OPENVINO_SPILL_DIR")) { + // Disk-backed weight buffer: back the repacked weights with a temp file via MAP_SHARED + // instead of anonymous memory. Anonymous pages can only be evicted to swap, so the + // repacked buffer stays pinned alongside the mmap'd source and both are resident at once + // -- that double residency is the load-time peak. File-backed pages are reclaimable: the + // kernel can write them back and drop them under pressure, then re-read on demand, so RSS + // becomes a working set rather than the whole buffer. The file is unlinked immediately, + // so it disappears when the process exits. + // + // The directory must be real storage. Pointing this at a tmpfs mount (/tmp on many + // systems) backs the "spill" with RAM and makes matters worse. + char path[PATH_MAX]; + snprintf(path, sizeof(path), "%s/ggml-ov-weights-%d-XXXXXX", spill_dir, (int) getpid()); + int fd = mkstemp(path); + if (fd < 0) { + GGML_LOG_ERROR("%s: mkstemp(%s) failed: %s\n", __func__, path, strerror(errno)); + return; + } + unlink(path); // anonymous-but-file-backed: freed on process exit + if (ftruncate(fd, (off_t) size) != 0) { + GGML_LOG_ERROR("%s: ftruncate(%zu) failed: %s\n", __func__, size, strerror(errno)); + close(fd); + return; + } + void * m = mmap(nullptr, size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0); + close(fd); // the mapping keeps the file alive + if (m == MAP_FAILED) { + GGML_LOG_ERROR("%s: mmap(%zu) failed: %s\n", __func__, size, strerror(errno)); + return; + } + data = m; + spill_mapping = m; + spill_size = size; + GGML_LOG_INFO("%s: weight buffer spilled to %s (%zu MB, file-backed)\n", __func__, spill_dir, + size / 1024 / 1024); + ov_buffer = std::make_shared(ov::element::u8, ov::Shape{size}, data); + } else +#endif + { +#ifdef _WIN32 + if (ggml_openvino_getenv_str("GGML_OPENVINO_SPILL_DIR")) { + GGML_LOG_WARN("%s: GGML_OPENVINO_SPILL_DIR is not supported on Windows, ignoring\n", __func__); + } +#endif + data = ggml_aligned_malloc(size); + GGML_ASSERT(data); + memset(data, 0, size); + ov_buffer = std::make_shared(ov::element::u8, ov::Shape{size}, data); + } } if (data == nullptr) { @@ -124,6 +179,11 @@ struct ggml_backend_openvino_buffer_context { delete pair.second; } tensor_extras.clear(); +#ifndef _WIN32 + if (spill_mapping != nullptr) { + munmap(spill_mapping, spill_size); + } else +#endif if (!is_remote && data != nullptr) { ggml_aligned_free(data, size); } @@ -135,6 +195,7 @@ struct ggml_backend_openvino_buffer_type_context { int device; std::string name; }; +} // namespace // ===================================================== // Host weight-buffer release (GGML_OPENVINO_RELEASE_WEIGHTS) @@ -194,14 +255,16 @@ void ggml_openvino_release_weight_buffers() { for (const auto & b : reg.buffers) { // Align down/up to page boundaries so madvise only drops whole pages // fully owned by this buffer. - const long page = sysconf(_SC_PAGESIZE); - uintptr_t start = reinterpret_cast(b.first); - uintptr_t end = start + b.second; - uintptr_t astart = (start + page - 1) & ~(uintptr_t) (page - 1); - uintptr_t aend = end & ~(uintptr_t) (page - 1); - if (aend > astart) { - if (madvise(reinterpret_cast(astart), aend - astart, MADV_DONTNEED) == 0) { - total += aend - astart; + const size_t page = (size_t) sysconf(_SC_PAGESIZE); + const uintptr_t ustart = reinterpret_cast(b.first); + const size_t offset_to_page = (page - (ustart & (page - 1))) & (page - 1); + if (b.second > offset_to_page) { + const size_t aligned_len = (b.second - offset_to_page) & ~(page - 1); + if (aligned_len > 0) { + char * astart = static_cast(b.first) + offset_to_page; + if (madvise(astart, aligned_len, MADV_DONTNEED) == 0) { + total += aligned_len; + } } } } @@ -611,9 +674,7 @@ GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_openvino_buffer_type(in static const char * ggml_backend_openvino_host_buffer_type_get_name(ggml_backend_buffer_type_t buft) { ggml_backend_openvino_buffer_type_context * ctx = (ggml_backend_openvino_buffer_type_context *) buft->context; - static std::string name; - name = ctx->name + "_HOST"; - return name.c_str(); + return ctx->name.c_str(); } static bool ggml_backend_openvino_host_buffer_type_is_host(ggml_backend_buffer_type_t buft) { @@ -646,7 +707,7 @@ GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_openvino_host_buffer_ty for (int i = 0; i < device_count; i++) { buffer_type_contexts[i].device = i; - buffer_type_contexts[i].name = std::string(GGML_OPENVINO_NAME) + std::to_string(i); + buffer_type_contexts[i].name = std::string(GGML_OPENVINO_NAME) + std::to_string(i) + "_HOST"; buffer_types[i] = ggml_backend_buffer_type{ /* .iface = */ ggml_backend_openvino_host_buffer_type_interface, @@ -711,13 +772,16 @@ static void ggml_backend_openvino_free(ggml_backend_t backend) { if (ctx->runtime_context) { auto r_ctx = std::static_pointer_cast(ctx->runtime_context); - if (--r_ctx->backend_count == 0) { + auto cache = r_ctx->compiled_cache; + r_ctx->clear_caches(); + std::lock_guard cache_lock(cache->mutex); + if (--cache->backend_count == 0) { // If host weight buffers were released (GGML_OPENVINO_RELEASE_WEIGHTS), the // dropped pages can never be repopulated, so a recompile is impossible. Keep // the compiled-model cache alive across backend teardown so the next context // reuses it instead of recompiling against zeroed weights. if (!ggml_openvino_weight_buffers_released()) { - r_ctx->clear_caches(); + cache->graphs.clear(); } } } @@ -766,12 +830,14 @@ static ggml_guid_t ggml_backend_openvino_guid(void) { } static std::shared_ptr get_ov_runtime_context_ptr() { - static std::shared_ptr r_ctx = [] { - auto ctx = std::make_shared(); - ctx->device = ggml_openvino_get_device_name(); - ctx->stateful = is_stateful_enabled() && !ggml_openvino_is_npu(); - return ctx; - }(); + // Share compiled models, but give every backend its own requests and KV state. + static auto cache = std::make_shared(); + auto r_ctx = std::make_shared(); + r_ctx->device = ggml_openvino_get_device_name(); + r_ctx->stateful = is_stateful_enabled() && !ggml_openvino_is_npu(); + r_ctx->compiled_cache = cache; + std::lock_guard cache_lock(cache->mutex); + ++cache->backend_count; return r_ctx; } @@ -795,9 +861,6 @@ GGML_BACKEND_API ggml_backend_t ggml_backend_openvino_init(int device) { return nullptr; } - std::shared_ptr r_ctx = std::static_pointer_cast(ctx->runtime_context); - r_ctx->backend_count++; - ggml_backend_t openvino_backend = new ggml_backend{ /* .guid = */ ggml_backend_openvino_guid(), /* .interface = */ ggml_backend_openvino_interface, @@ -812,11 +875,13 @@ GGML_BACKEND_API bool ggml_backend_is_openvino(ggml_backend_t backend) { return backend != NULL && ggml_guid_matches(backend->guid, ggml_backend_openvino_guid()); } +namespace { struct ggml_backend_openvino_device_context { int device; std::string name; std::string description; }; +} static const char * ggml_backend_openvino_device_get_name(ggml_backend_dev_t dev) { ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; @@ -908,11 +973,31 @@ static bool has_non_contiguous_view_input(const ggml_tensor * op) { } static bool is_supported_flash_attn_pattern(const ggml_tensor * op) { - // pattern of q,k,v should be q->op==PERMUTE, q->src[0]->op==VIEW, q->src[0]->src[0]->view_src==nullptr + // Each Q/K/V input must follow one of: + // PERMUTE -> VIEW -> base (view_src==nullptr) (llama KV-cache path) + // PERMUTE -> RESHAPE -> base (view_src==nullptr) (whisper Q) + // VIEW -> base (view_src==nullptr) (whisper K/V from kv_pad) for (int i = 0; i < 3; i++) { const ggml_tensor * src = op->src[i]; - if (src->op != GGML_OP_PERMUTE || src->src[0] == nullptr || src->src[0]->op != GGML_OP_VIEW || - src->src[0]->src[0] == nullptr || src->src[0]->src[0]->view_src != nullptr) { + if (src->op == GGML_OP_PERMUTE) { + if (src->src[0] == nullptr) { + return false; + } + if (src->src[0]->op != GGML_OP_VIEW && src->src[0]->op != GGML_OP_RESHAPE) { + return false; + } + if (src->src[0]->src[0] == nullptr || src->src[0]->src[0]->view_src != nullptr) { + return false; + } + } else if (src->op == GGML_OP_VIEW) { + if (src->src[0] == nullptr || src->src[0]->view_src != nullptr) { + return false; + } + } else if (src->op == GGML_OP_CPY) { + if (src->src[0] == nullptr || src->src[0]->op != GGML_OP_PERMUTE || src->src[0]->src[0] == nullptr) { + return false; + } + } else { return false; } } @@ -979,7 +1064,7 @@ static bool cpy_output_view_is_supported(const ggml_tensor * op) { return false; } - return ggml_nbytes(op) == 0 || ggml_is_contiguous(op); + return ggml_nbytes(op) == 0 || ggml_is_contiguous(op) || GgmlOvDecoder::is_conv_state_writeback(op); } static bool mul_mat_id_requires_large_tmp(const ggml_tensor * op) { @@ -1030,18 +1115,29 @@ static bool is_msa_block_mask_expansion(const ggml_tensor * op) { return tensor_name_starts_with(src, "msa_block_mask"); } -static bool is_op_unsupported_case(const ggml_tensor * op) { +namespace { +struct ggml_openvino_op_support { + bool is_supported = true; + std::string reason; + + operator bool() const { + return is_supported; + } +}; +} // namespace + +static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) { if (is_msa_block_mask_expansion(op)) { - return true; + return {false, "MSA block mask expansion is not supported"}; } switch (op->op) { case GGML_OP_CONCAT: { if (op->type == GGML_TYPE_I64) { - return true; + return {false, "CONCAT with I64 type is not supported"}; } if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16 && has_view_op_input(op)) { - return true; + return {false, "CONCAT with BF16 type and VIEW input is not supported on GPU"}; } break; } @@ -1052,24 +1148,25 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // OpenVINO SET translation currently supports dst layouts that match src0 strides. if (op->src[0] == nullptr || nb1 != op->src[0]->nb[1] || nb2 != op->src[0]->nb[2] || nb3 != op->src[0]->nb[3]) { - // std::cout << "Unsupported SET op with dst nb1=" << nb1 << ", nb2=" << nb2 << ", nb3=" << nb3 - // << " that does not match src0 strides nb[1]=" - // << (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[1]) : "null") - // << ", nb[2]=" << (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[2]) : "null") - // << ", nb[3]=" << (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[3]) : "null") - // << std::endl; - return true; + return {false, "SET op with dst nb1=" + std::to_string(nb1) + ", nb2=" + std::to_string(nb2) + ", nb3=" + std::to_string(nb3) + + " that does not match src0 strides nb[1]=" + (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[1]) : "null") + + ", nb[2]=" + (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[2]) : "null") + + ", nb[3]=" + (op->src[0] != nullptr ? std::to_string(op->src[0]->nb[3]) : "null")}; } break; } case GGML_OP_GET_ROWS: case GGML_OP_SET_ROWS: { if (op->ne[3] != 1) { - return true; + return {false, "GET_ROWS/SET_ROWS with ne[3] != 1 (ne[3]=" + std::to_string(op->ne[3]) + ") is not supported"}; + } + if (op->op == GGML_OP_GET_ROWS && ggml_is_quantized(op->src[0]->type) && + op->src[0]->view_src != nullptr && op->src[0]->view_offs != 0) { + return {false, "GET_ROWS with a nonzero quantized src0 view offset is not supported"}; } if (op->op == GGML_OP_GET_ROWS && ggml_openvino_get_device_name() == "GPU" && op->src[0]->type == GGML_TYPE_BF16) { - return true; + return {false, "GET_ROWS with BF16 src0 is not supported on GPU"}; } if (op->ne[0] == 256 && (op->src[0]->type == GGML_TYPE_Q4_K || op->src[0]->type == GGML_TYPE_Q5_K || op->src[0]->type == GGML_TYPE_Q4_1 || op->src[0]->type == GGML_TYPE_Q5_1)) { @@ -1078,14 +1175,14 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // make_int8_weights/make_int4_weights: dequant is done in f16, not f32, to keep the // Convert/Subtract/Multiply chain fusable into GatherMatmulCompressed/FullyConnectedCompressed // for the shared non-test code paths). - return true; + return {false, "GET_ROWS/SET_ROWS with ne[0] == 256 and type " + std::string(ggml_type_name(op->src[0]->type)) + + " rejected due to f16-arithmetic dequant rounding errors that intermittently exceed 1e-7 NMSE threshold"}; } - break; } case GGML_OP_RESHAPE: { if (strncmp(op->name, "ffn_norm_exps", sizeof("ffn_norm_exps") - 1) == 0) { - return true; + return {false, "RESHAPE for ffn_norm_exps is not supported"}; } break; } @@ -1093,11 +1190,17 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { case GGML_OP_MUL: case GGML_OP_SUB: { if (op->src[1]->op == GGML_OP_PERMUTE) { - return true; + return {false, "ADD/MUL/SUB with PERMUTE src1 is not supported"}; + } + // >8-expert MoE ReduceSum drifts past the 1e-7 tolerance (f32 order vs CPU); intermittent. + if (op->op == GGML_OP_ADD && is_moe_expert_sum_add(op) && op->src[1]->src[0]->ne[1] > 8) { + return {false, "MoE expert-plane sum with more than 8 experts is not supported"}; } for (int i = 0; i < 4; i++) { if (op->src[0]->ne[i] != op->src[1]->ne[i] && (op->src[0]->ne[i] != 1 && op->src[1]->ne[i] != 1)) { - return true; + return {false, "ADD/MUL/SUB with incompatible broadcast shapes: src0->ne[" + std::to_string(i) + "]=" + + std::to_string(op->src[0]->ne[i]) + ", src1->ne[" + std::to_string(i) + "]=" + + std::to_string(op->src[1]->ne[i])}; } } break; @@ -1106,7 +1209,7 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // Keep support aligned with the CPU backend implementation, which only handles f32 inputs/output and i32 ids. if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32 || op->src[2]->type != GGML_TYPE_I32) { - return true; + return {false, "ADD_ID only supports F32 inputs/output and I32 ids"}; } break; } @@ -1116,14 +1219,27 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // until the fused GPU kernel is reliable. (falied case llama-arch-test mpt) if (ggml_openvino_get_device_name() == "GPU" && op->src[1]->ne[0] == op->ne[0] && op->src[1]->ne[1] == 1 && op->src[1]->ne[2] == 1 && op->src[1]->ne[3] == 1) { - return true; + return {false, "DIV per-channel scale broadcast is not supported on GPU"}; + } + break; + } + case GGML_OP_POOL_2D: { + const auto& name = ggml_openvino_get_device_name(); + if (name == "GPU") { + const int32_t * params = op->op_params; + const int k0 = params[1]; + const int k1 = params[2]; + const int p0 = params[5]; + const int p1 = params[6]; + if ((p0 > 0 || p1 > 0) && (k0 < 3 || k1 < 3)) { + return {false, "POOL_2D with padding and kernel size < 3 is not supported on " + name}; + } } break; } case GGML_OP_SUM_ROWS: { - // if the input is PERMUTE skip if (op->src[0]->op == GGML_OP_PERMUTE) { - return true; + return {false, "SUM_ROWS with PERMUTE input is not supported"}; } break; } @@ -1140,54 +1256,54 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // accuracy drift in the OpenVINO path. Restrict by scale=1.0 to avoid // affecting non-gemma3n models such as Llama-3.2. if (fabsf(scale - 1.0f) < 1e-6f && is_gemma3n_flash_attn_pattern(op)) { - return true; + return {false, "FLASH_ATTN_EXT gemma3n pattern on GPU is not supported"}; } if (op->src[4] != nullptr) { - // GGML_LOG_WARN("OpenVINO backend does not support FLASH_ATTN_EXT with sinks\n"); - return true; + return {false, "FLASH_ATTN_EXT with sinks is not supported"}; } if (!is_supported_flash_attn_pattern(op)) { - return true; + return {false, "FLASH_ATTN_EXT unsupported attention pattern"}; } if (max_bias > 0) { - // GGML_LOG_WARN("OpenVINO backend does not support FLASH_ATTN_EXT with max_bias > 0\n"); - return true; + return {false, "FLASH_ATTN_EXT with max_bias > 0 (max_bias=" + std::to_string(max_bias) + ") is not supported"}; } if (logit_softcap != 0) { - // GGML_LOG_WARN("OpenVINO backend does not support FLASH_ATTN_EXT with logit_softcap != 0\n"); - return true; + return {false, "FLASH_ATTN_EXT with logit_softcap != 0 (logit_softcap=" + std::to_string(logit_softcap) + ") is not supported"}; } break; } case GGML_OP_PERMUTE: { - if (op->type == GGML_TYPE_BF16) { - // err msg: [GPU] Could not find a suitable kernel for transpose - // GGML_LOG_WARN("OpenVINO backend does not support PERMUTE with BF16 type\n"); - return true; + if (op->type == GGML_TYPE_BF16 && ggml_openvino_get_device_name() == "GPU") { + return {false, "PERMUTE with BF16 type is not supported on GPU"}; } break; } case GGML_OP_CPY: { - if (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16) { - // GGML_LOG_WARN("OpenVINO backend does not support CPY with non-contiguous data or bf16 types\n"); - return true; + if (op->src[0]->type != GGML_TYPE_BF16 && op->src[1]->type == GGML_TYPE_BF16) { + return {false, "CPY with BF16 src[1] type is not supported"}; + } + if (ggml_openvino_get_device_name() == "NPU" && (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) { + return {false, "CPY with BF16 is not supported is not supported on NPU"}; } // CPY to a quantized destination (e.g. f32 -> q4_0) is numerically unstable with OpenVINO backend. if (ggml_is_quantized(op->type)) { - return true; + return {false, "CPY to quantized destination (e.g. f32 -> q4_0) is numerically unstable"}; } if (ggml_nelements(op->src[0]) != ggml_nelements(op->src[1])) { - return true; + return {false, "CPY with mismatched element counts is not supported: src0=" + std::to_string(ggml_nelements(op->src[0])) + + " != src1=" + std::to_string(ggml_nelements(op->src[1]))}; } // op test case with non-contiguous src or dst if ((op->ne[0] == 3 && op->ne[1] == 4 && op->ne[2] == 3 && op->ne[3] == 2) || (op->ne[0] == 1 && op->ne[1] == 4 && op->ne[2] == 3 && op->ne[3] == 2) || (op->ne[0] == 2 && op->ne[1] == 4 && op->ne[2] == 3 && op->ne[3] == 2)) { - return true; + return {false, "CPY with non-contiguous shape [" + std::to_string(op->ne[0]) + ", " + + std::to_string(op->ne[1]) + ", " + std::to_string(op->ne[2]) + ", " + + std::to_string(op->ne[3]) + "] is not supported"}; } if (!cpy_output_view_is_supported(op)) { - return true; + return {false, "CPY with non-contiguous output view is not supported"}; } break; } @@ -1196,13 +1312,18 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { ggml_is_quantized(op->src[0]->type) && strcmp(op->src[0]->name, "a") == 0 && strcmp(op->src[1]->name, "b") == 0 && op->src[0]->ne[1] == 1 && op->src[1]->ne[1] == 64 && op->src[0]->ne[0] == 256 && op->src[1]->ne[0] == 256) { - return true; + return {false, "MUL_MAT quantized benchmark test case on GPU is not supported"}; + } + if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_F32 && op->ne[0] == 1 && op->ne[1] == 1 && + (op->src[0]->buffer == nullptr || op->src[0]->buffer->usage != GGML_BACKEND_BUFFER_USAGE_WEIGHTS)) { + return {false, "MUL_MAT scalar dot product with non-weight src[0] on GPU is not supported"}; } if (op->src[0]->ne[3] != op->src[1]->ne[3] && op->src[0]->ne[3] != 1 && op->src[1]->ne[3] != 1) { - return true; + return {false, "MUL_MAT with incompatible broadcast on ne[3]: src0->ne[3]=" + std::to_string(op->src[0]->ne[3]) + + ", src1->ne[3]=" + std::to_string(op->src[1]->ne[3])}; } if (op->src[0]->op == GGML_OP_VIEW && op->src[1]->op == GGML_OP_VIEW) { - return true; + return {false, "MUL_MAT with both inputs as VIEW is not supported"}; } break; } @@ -1210,16 +1331,26 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // Single-expert (or empty) MUL_MAT_ID is a degenerate shape that stresses GatherMatmul edge // cases and never occurs in real MoE; let it fall back to CPU. if (op->src[0] != nullptr && op->src[0]->ne[2] <= 1) { - return true; - } - if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_BF16) { - return true; - } - // GPU MUL_MAT_ID uses a Gather+MatMul fallback because the GPU plugin rejects internal - // GatherMatmul for these test shapes. Skip cases that would materialize a large selected - // expert-weight temporary. - if (ggml_openvino_get_device_name() == "GPU" && mul_mat_id_requires_large_tmp(op)) { - return true; + return {false, "MUL_MAT_ID with single-expert or empty ne[2] <= 1 (ne[2]=" + + std::to_string(op->src[0]->ne[2]) + ") is not supported"}; + } + if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && !ggml_is_quantized(op->src[0]->type)) { + return {false, "MUL_MAT_ID with non-quantized weights on GPU is not supported"}; + } + // The GPU plugin's GatherMatmul returns wrong values for the layouts test-backend-ops + // produces: it builds a rank-4 input layout ([n_used, n_tokens, k, 1]) instead of rank 3 + // and the kernel misreads it, silently returning garbage (NMSE ~86) rather than asserting. + // The same graph is correct on the CPU plugin, and correct on GPU for every real model, + // which always feeds experts from a bound tensor buffer. Standalone op-test tensors have + // no buffer at all, so use that to exclude them and let the scheduler run them on CPU. + if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->buffer == nullptr) { + return {false, "MUL_MAT_ID with unbound expert tensors on GPU is not supported"}; + } + // Only MXFP4 still needs the large-temporary guard; every other quantized type goes + // through GatherMatmul, which never materializes the selected expert weights. + if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_MXFP4 && + mul_mat_id_requires_large_tmp(op)) { + return {false, "MUL_MAT_ID with MXFP4 weights requires large temporary on GPU"}; } break; } @@ -1227,49 +1358,51 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { const int32_t * op_params = op->op_params; const int n_dims = op_params[1]; const int mode = op_params[2]; + const int64_t n_offs = op_params[15]; if (mode != GGML_ROPE_TYPE_NORMAL && mode != GGML_ROPE_TYPE_NEOX && mode != GGML_ROPE_TYPE_IMROPE) { - // GGML_LOG_WARN("OpenVINO backend does not support ROPE with mode %d\n", mode); - return true; + return {false, "ROPE with mode " + std::to_string(mode) + " is not supported"}; + } + if (n_offs < 0 || (n_offs % 2) != 0) { + return {false, "ROPE with invalid n_offs=" + std::to_string(n_offs)}; } const int64_t head_dim = op->src[0]->ne[0]; const int64_t rope_dims = n_dims == 0 ? head_dim : n_dims; - if (rope_dims <= 0 || rope_dims > head_dim || (rope_dims % 2) != 0) { - // GGML_LOG_WARN("OpenVINO backend does not support ROPE with n_dims %d and src[0]->ne[0] %ld\n", n_dims, - // op->src[0]->ne[0]); - return true; + if (rope_dims <= 0 || rope_dims + n_offs > head_dim || (rope_dims % 2) != 0) { + return {false, "ROPE with n_dims=" + std::to_string(n_dims) + ", n_offs=" + std::to_string(n_offs) + + ", head_dim=" + std::to_string(head_dim) + " is not supported"}; } if (op->type != GGML_TYPE_F32 && op->type != GGML_TYPE_F16) { - // GGML_LOG_WARN("OpenVINO backend does not support ROPE with type %s\n", ggml_type_name(op->type)); - return true; - } - if (op->src[0]->op == GGML_OP_VIEW) { - if (op->src[0]->view_src->ne[1] != op->src[0]->ne[2]) { - // GGML_LOG_WARN( - // "OpenVINO backend does not support ROPE with src[0]->view_src->ne[1] %ld != src[0]->ne[2] " - // "%ld\n", - // op->src[0]->view_src->ne[1], op->src[0]->ne[2]); - return true; - } - } + return {false, "ROPE with type " + std::string(ggml_type_name(op->type)) + " is not supported"}; + } + if (op->view_src != nullptr && !ggml_is_contiguous(op->src[0])) { + return {false, "ROPE on VIEW / non-contiguous input is not supported"}; + } + if (op->src[0]->ne[3] > 1) { + // translate_rope's cos/sin tables cover one sequence only; ne[3] > 1 fails to broadcast. + return {false, "ROPE with multiple sequences (ne[3]=" + std::to_string(op->src[0]->ne[3]) + + ") is not supported"}; + } + float freq_scale; + float ext_factor; + float attn_factor; + memcpy(&freq_scale, op_params + 6, sizeof(float)); + memcpy(&ext_factor, op_params + 7, sizeof(float)); + memcpy(&attn_factor, op_params + 8, sizeof(float)); if (mode == GGML_ROPE_TYPE_IMROPE && - (op->src[2] != 0 || ((const float *) op_params)[6] != 1 || ((const float *) op_params)[7] != 0 || - ((const float *) op_params)[8] != 1)) { - // GGML_LOG_WARN("OpenVINO backend does not support IMROPE with freq_factors, freq_scale, ext_factor, and attn_factor\n"); - return true; + (op->src[2] != nullptr || freq_scale != 1.0f || ext_factor != 0.0f || attn_factor != 1.0f)) { + return {false, "IMROPE with freq_factors, freq_scale, ext_factor, or attn_factor is not supported"}; } break; } case GGML_OP_TRANSPOSE: { - // if the type is bf16, will return true if (op->type == GGML_TYPE_BF16) { - // GGML_LOG_WARN("OpenVINO backend does not support CONT with BF16 type\n"); - return true; + return {false, "TRANSPOSE with BF16 type is not supported"}; } break; } case GGML_OP_REPEAT: { if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16) { - return true; + return {false, "REPEAT with BF16 type is not supported on GPU"}; } break; } @@ -1281,15 +1414,15 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // return true; // } if (op->src[2]->op == GGML_OP_PERMUTE) { - return true; + return {false, "GATED_DELTA_NET with PERMUTE src2 is not supported"}; } // kda (per-key-dimension gating) not supported by fused GatedDeltaNet op if (op->src[3]->ne[0] != 1) { - return true; + return {false, "GATED_DELTA_NET with kda (per-key-dimension gating) is not supported"}; } // K > 1 (multiple state snapshots) not supported by fused op if (((const int32_t *) op->op_params)[0] > 1) { - return true; + return {false, "GATED_DELTA_NET with K > 1 (multiple state snapshots) is not supported"}; } break; } @@ -1303,17 +1436,17 @@ static bool is_op_unsupported_case(const ggml_tensor * op) { // Skip TOPK_MOE fused tests until it is fully supported. // The argsort_top_k VIEW wrapping ARGSORT is named "selected_experts" in test_topk_moe. if (strcmp(op->name, "selected_experts") == 0) { - return true; + return {false, "VIEW for selected_experts (argsort_top_k) is not supported"}; } break; } default: break; } - return false; + return {true, ""}; } -static bool ggml_backend_openvino_device_supports_op(ggml_backend_dev_t dev, const ggml_tensor * op) { +static ggml_openvino_op_support ggml_backend_openvino_device_supports_op_impl(ggml_backend_dev_t dev, const ggml_tensor * op) { GGML_ASSERT(dev->reg != nullptr); static std::unordered_set supported_types{ @@ -1363,48 +1496,41 @@ static bool ggml_backend_openvino_device_supports_op(ggml_backend_dev_t dev, con case GGML_OP_UNARY: { auto supported = supported_unary_ops.find(ggml_get_unary_op(op)) != supported_unary_ops.end(); if (!supported) { - // GGML_LOG_WARN("OpenVINO backend does not support unary op %s\n", ggml_unary_op_name(ggml_get_unary_op(op))); - return false; + return {false, "unary op " + std::string(ggml_unary_op_name(ggml_get_unary_op(op))) + " has no op translator"}; } if (ggml_get_unary_op(op) == GGML_UNARY_OP_EXP && op->type == GGML_TYPE_F32) { - return false; + return {false, "UNARY_EXP with F32 type is not supported"}; } break; } case GGML_OP_GLU: { auto supported = supported_glu_ops.find(ggml_get_glu_op(op)) != supported_glu_ops.end(); if (!supported) { - // GGML_LOG_WARN("OpenVINO backend does not support GLU op %s\n", ggml_glu_op_name(ggml_get_glu_op(op))); - return false; + return {false, "GLU op " + std::string(ggml_glu_op_name(ggml_get_glu_op(op))) + " has no op translator"}; } // if (has_view_op_input(op)) { - // // GGML_LOG_WARN("OpenVINO backend does not support unary op %s with view input\n", - // // ggml_glu_op_name(ggml_get_glu_op(op))); - // return false; + // return {false, "GLU op " + std::string(ggml_glu_op_name(ggml_get_glu_op(op))) + " with view input is not supported"}; // } if (op->src[1] == nullptr && op->src[0]->ne[0] % 2 != 0) { // triggers bug in ov gpu - return false; + return {false, "GLU op with odd src0 ne[0] and null src1 is not supported"}; } break; } default: { auto supported = supported_ops.find(op->op) != supported_ops.end(); if (!supported) { - // GGML_LOG_WARN("OpenVINO backend does not support op %s\n", ggml_op_name(op->op)); - return false; + return {false, "op " + std::string(ggml_op_name(op->op)) + " has no op translator"}; } static std::set ops_not_support_view_input{}; if (ops_not_support_view_input.find(op->op) != ops_not_support_view_input.end() && has_view_op_input(op)) { - // GGML_LOG_WARN("OpenVINO backend does not support op %s with view input\n", ggml_op_name(op->op)); - return false; + return {false, "op " + std::string(ggml_op_name(op->op)) + " with VIEW input is not supported"}; } } } if (supported_types.find(op->type) == supported_types.end()) { - // GGML_LOG_WARN("OpenVINO backend does not support tensor type %s\n", ggml_type_name(op->type)); - return false; + return {false, "tensor type " + std::string(ggml_type_name(op->type)) + " is not supported"}; } for (int i = 0; i < GGML_MAX_SRC; i++) { auto * src = op->src[i]; @@ -1412,21 +1538,32 @@ static bool ggml_backend_openvino_device_supports_op(ggml_backend_dev_t dev, con break; } if (supported_types.find(src->type) == supported_types.end()) { - // GGML_LOG_WARN("OpenVINO backend does not support tensor type %s\n", ggml_type_name(src->type)); - return false; + return {false, "src[" + std::to_string(i) + "] type " + std::string(ggml_type_name(src->type)) + " is not supported"}; } const bool is_supported_3d_moe_expert = op->op == GGML_OP_MUL_MAT_ID && i == 0 && (src->type == GGML_TYPE_MXFP4 || src->ne[3] == 1); if (ggml_is_quantized(src->type) && src->ne[2] != 1 && !is_supported_3d_moe_expert) { - // GGML_LOG_WARN("OpenVINO backend does not support 3D quantized tensors\n"); - return false; + return {false, "3D quantized tensor for src[" + std::to_string(i) + "] is not supported"}; } } - if (is_op_unsupported_case(op)) { - return false; + auto op_support_case = is_op_supported_case(op); + if (!op_support_case.is_supported) { + return op_support_case; } - return true; + return {true, ""}; +} + +static bool ggml_backend_openvino_device_supports_op(ggml_backend_dev_t dev, const ggml_tensor * op) { + auto res = ggml_backend_openvino_device_supports_op_impl(dev, op); + if (!res.is_supported) { + static const bool log_unsupported = ggml_openvino_getenv_int("GGML_OPENVINO_LOG_UNSUPPORTED_OPS") != 0; + if (log_unsupported) { + GGML_LOG_WARN("OpenVINO op unsupported: op '%s' (%s), type %s: %s\n", + op->name, ggml_op_name(op->op), ggml_type_name(op->type), res.reason.c_str()); + } + } + return res.is_supported; } static bool ggml_backend_openvino_device_supports_buft(ggml_backend_dev_t dev, ggml_backend_buffer_type_t buft) { @@ -1452,9 +1589,11 @@ static const struct ggml_backend_device_i ggml_backend_openvino_device_interface /* .event_synchronize = */ NULL, }; +namespace { struct ggml_backend_openvino_reg_context { std::vector devices; }; +} static const char * ggml_backend_openvino_reg_get_name(ggml_backend_reg_t reg) { return GGML_OPENVINO_NAME; diff --git a/ggml/src/ggml-openvino/ggml-quants.cpp b/ggml/src/ggml-openvino/ggml-quants.cpp index 120db01e..824d2447 100644 --- a/ggml/src/ggml-openvino/ggml-quants.cpp +++ b/ggml/src/ggml-openvino/ggml-quants.cpp @@ -34,6 +34,15 @@ #include #include +// From /src/common/transformations/include/transformations/utils/utils.hpp +namespace ov::op::util { +// From /src/common/transformations/include/transformations/utils/utils.hpp +bool get_single_value(const std::shared_ptr & const_node, + float & value, + bool check_value_range = true); +} // namespace ov::op::util + +namespace { void unpack_32_4(const uint8_t * data, uint8_t * dst) { std::fill_n(dst, 16, 0); for (int j = 0; j < 16; ++j) { @@ -48,11 +57,11 @@ void unpack_32_4(const uint8_t * data, uint8_t * dst) { } } -static constexpr size_t MXFP4_BLOCK_SIZE = 32; -static constexpr size_t MXFP4_BLOCK_QS_SIZE = MXFP4_BLOCK_SIZE / 2; -static constexpr size_t MXFP4_BLOCK_BYTES = sizeof(uint8_t) + MXFP4_BLOCK_QS_SIZE; +constexpr size_t MXFP4_BLOCK_SIZE = 32; +constexpr size_t MXFP4_BLOCK_QS_SIZE = MXFP4_BLOCK_SIZE / 2; +constexpr size_t MXFP4_BLOCK_BYTES = sizeof(uint8_t) + MXFP4_BLOCK_QS_SIZE; -static void pack_32_mxfp4_for_openvino(const uint8_t * data, uint8_t * dst) { +void pack_32_mxfp4_for_openvino(const uint8_t * data, uint8_t * dst) { for (int j = 0; j < static_cast(MXFP4_BLOCK_QS_SIZE); j += 2) { const uint8_t v0 = data[j] & 0x0F; const uint8_t v1 = (data[j + 1] & 0x0F) << 4; @@ -419,7 +428,7 @@ void extract_q6_k_data(const ggml_tensor * tensor, } } -static inline void get_scale_min_k4(int j, const uint8_t * q, uint8_t * d, uint8_t * m) { +inline void get_scale_min_k4(int j, const uint8_t * q, uint8_t * d, uint8_t * m) { if (j < 4) { *d = q[j] & 63; *m = q[j + 4] & 63; @@ -514,9 +523,9 @@ void extract_q5_k_data(const ggml_tensor * tensor, ov::Output make_int8_weights(ov::Tensor & weight, ov::Tensor & scales, ov::Tensor & zp, - size_t group_size, - bool use_bias, - bool for_gather_matmul) { + size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, + bool use_bias = false, + bool for_gather_matmul = false) { ov::Shape orig_shape = weight.get_shape(); bool is_signed = (weight.get_element_type() == ov::element::i8); // Symmetric: signed weights, no ZP @@ -611,13 +620,24 @@ ov::Output make_int8_weights(ov::Tensor & weight, return std::make_shared(result, ov::element::f32); } +// If for_gather_matmul is true, the weight tensor may be N-D (e.g. 3D MoE expert weights +// [n_expert, rows, cols]). The dequantization chain (Convert->[Subtract]->Multiply) is built as +// usual but left in f16 (no final Convert to f32) -- ov::pass::MarkDequantization (registered in +// translate_session.cpp) marks the chain so it survives model-build-time ConstantFolding -- see +// make_int8_weights.cpp/make_int4_weights.cpp. mul_mat_id.cpp constructs ov::op::internal::GatherMatmul +// directly from the resulting f16 dequant chain. +// +// When use_bias is true (explicitly, or implicitly because for_gather_matmul is true), the zp +// tensor is expected to hold an exact f16 bias value (rather than a rounded integer zero point); +// it is converted in place into an exact zero_point = -bias/scale and consumed via Subtract, not +// Add, so the chain still matches OpenVINO's Convert->Subtract->Multiply decompression pattern. // See make_int8_weights for the meaning of for_gather_matmul. ov::Output make_int4_weights(ov::Tensor & weight, ov::Tensor & scales, ov::Tensor & zp, - size_t group_size, - bool use_bias, - bool for_gather_matmul) { + size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, + bool use_bias = false, + bool for_gather_matmul = false) { ov::Shape orig_weight_shape = weight.get_shape(); bool is_signed = (weight.get_element_type() == ov::element::i4); // Symmetric: signed weights, no ZP @@ -746,13 +766,262 @@ ov::Output make_mxfp4_moe_packed_weights(ov::Tensor & weight) { return weights_node; } +void quantize_q4_0(const float * x, + ov::Tensor & weights_arr, + ov::Tensor & scales_arr, + ov::Tensor & zp_arr, + int64_t k, + int64_t qk) { + assert(k % qk == 0); + const int nb = k / qk; + + auto * weights = static_cast(weights_arr.data()); + auto * scales = scales_arr.data::value_type>(); + bool is_symmetric = (weights_arr.get_element_type() == ov::element::i4); // Signed i4 path + + if (!is_symmetric) { + auto * zp = static_cast(zp_arr.data()); + for (int i = 0; i < nb; i++) { + float amax = 0.0f; + float max = 0.0f; + for (int j = 0; j < qk; j++) { + const float v = x[i * qk + j]; + if (amax < fabsf(v)) { + amax = fabsf(v); + max = v; + } + } + const float d = max / -8; + if (d == 0) { + scales[i] = ov::float16(1.0f); + if (i % 2 == 0) { + zp[i / 2] = 8; + } else { + zp[i / 2] |= (8 << 4); + } + memset(weights + i * qk / 2, 8 | (8 << 4), qk / 2); + continue; + } + const float id = 1.0f / d; + scales[i] = ov::float16(d); + if (i % 2 == 0) { + zp[i / 2] = 8; + } else { + zp[i / 2] |= (8 << 4); + } + for (int j = 0; j < qk / 2; ++j) { + const float x0 = x[i * qk + 2 * j] * id; + const float x1 = x[i * qk + 2 * j + 1] * id; + const uint8_t xi0 = MIN(15, (int8_t) (x0 + 8.5f)); + const uint8_t xi1 = MIN(15, (int8_t) (x1 + 8.5f)); + weights[i * qk / 2 + j] = xi0 | (xi1 << 4); + } + } + } else { + // Symmetric: produce signed i4 values in [-8, 7] + for (int i = 0; i < nb; i++) { + float amax = 0.0f; + float max = 0.0f; + for (int j = 0; j < qk; j++) { + const float v = x[i * qk + j]; + if (amax < fabsf(v)) { + amax = fabsf(v); + max = v; + } + } + const float d = max / -8; + if (d == 0) { + scales[i] = ov::float16(1.0f); + // i4 value 0 packed: 0x00 + memset(weights + i * qk / 2, 0, qk / 2); + continue; + } + const float id = 1.0f / d; + scales[i] = ov::float16(d); + for (int j = 0; j < qk / 2; ++j) { + const float x0 = x[i * qk + 2 * j] * id; + const float x1 = x[i * qk + 2 * j + 1] * id; + // Signed i4: range [-8, 7]. Quantize as round(x*id), then pack as 4-bit two's complement. + int8_t si0 = (int8_t) std::max(-8, std::min(7, (int) roundf(x0))); + int8_t si1 = (int8_t) std::max(-8, std::min(7, (int) roundf(x1))); + weights[i * qk / 2 + j] = (si0 & 0x0F) | ((si1 & 0x0F) << 4); + } + } + } +} + +// Asymmetric u4 quantization with a per-group scale and zero point. +// +// Unlike quantize_q4_0's unsigned branch, which pins the zero point to 8 and is therefore +// symmetric, this keeps a real per-group zero point, so a group whose values are not centred on +// zero does not waste half its range. +void quantize_q4_1_asym(const float * x, + ov::Tensor & weights_arr, + ov::Tensor & scales_arr, + ov::Tensor & zp_arr, + int64_t k, + int64_t qk) { + assert(k % qk == 0); + const int nb = k / qk; + + auto * weights = static_cast(weights_arr.data()); + auto * scales = scales_arr.data::value_type>(); + auto * zp = static_cast(zp_arr.data()); + + // u4 zero points are packed two per byte, low nibble first, indexed by group -- the same + // convention as the unsigned branch of quantize_q4_0. + auto store_zp = [zp](int i, uint8_t v) { + if (i % 2 == 0) { + zp[i / 2] = v & 0x0F; + } else { + zp[i / 2] |= (uint8_t) ((v & 0x0F) << 4); + } + }; + + for (int i = 0; i < nb; i++) { + float vmin = x[i * qk]; + float vmax = x[i * qk]; + for (int j = 1; j < qk; j++) { + const float v = x[i * qk + j]; + vmin = std::min(vmin, v); + vmax = std::max(vmax, v); + } + // Include 0 in the range so an all-positive or all-negative group still represents zero + // exactly -- these are weights, so an exact zero matters. + vmin = std::min(vmin, 0.0f); + vmax = std::max(vmax, 0.0f); + + const float d = (vmax - vmin) / 15.0f; + if (d == 0.0f) { + scales[i] = ov::float16(1.0f); + store_zp(i, 0); + memset(weights + i * qk / 2, 0, qk / 2); + continue; + } + const float id = 1.0f / d; + + // The zero point is itself a 4-bit integer, so round it and dequantize as (q - zq) * d. + const int zq = std::max(0, std::min(15, (int) lroundf(-vmin * id))); + scales[i] = ov::float16(d); + store_zp(i, (uint8_t) zq); + + for (int j = 0; j < qk / 2; ++j) { + const float x0 = x[i * qk + 2 * j] * id; + const float x1 = x[i * qk + 2 * j + 1] * id; + const uint8_t q0 = (uint8_t) std::max(0, std::min(15, (int) lroundf(x0) + zq)); + const uint8_t q1 = (uint8_t) std::max(0, std::min(15, (int) lroundf(x1) + zq)); + weights[i * qk / 2 + j] = (uint8_t) (q0 | (q1 << 4)); + } + } +} + +void quantize_q8_0(const float * x, + ov::Tensor & weights_arr, + ov::Tensor & scales_arr, + ov::Tensor & zp_arr, + int64_t k, + int64_t qk, + int64_t block_offset = 0) { + assert(k % qk == 0); + const int nb = k / qk; + + // block_offset lets a caller quantize a chunk of blocks into the right place in the + // output buffers (used for streaming requant). x points at this chunk's first block; + // outputs are advanced by block_offset blocks. Q8 has one scale/zp per block (no + // nibble packing), so any block boundary is safe. + auto * weights = static_cast(weights_arr.data()) + block_offset * qk; + auto * scales = scales_arr.data::value_type>() + block_offset; + bool is_symmetric = (weights_arr.get_element_type() == ov::element::i8); // Signed i8 path + + if (!is_symmetric) { + auto * zp = static_cast(zp_arr.data()) + block_offset; + for (int i = 0; i < nb; i++) { + float amax = 0.0f; + for (int j = 0; j < qk; j++) { + const float v = x[i * qk + j]; + amax = std::max(amax, fabsf(v)); + } + const float d = amax / 127.0f; + const float id = d ? 1.0f / d : 0.0f; + scales[i] = ov::float16(d); + zp[i] = 128; + for (int j = 0; j < qk; ++j) { + const float x0 = x[i * qk + j] * id; + const int8_t xi0 = roundf(x0); + weights[i * qk + j] = (uint8_t) (xi0 + 128); + } + } + } else { + // Symmetric: store signed int8 values directly + auto * signed_weights = reinterpret_cast(weights); + for (int i = 0; i < nb; i++) { + float amax = 0.0f; + for (int j = 0; j < qk; j++) { + const float v = x[i * qk + j]; + amax = std::max(amax, fabsf(v)); + } + const float d = amax / 127.0f; + const float id = d ? 1.0f / d : 0.0f; + scales[i] = ov::float16(d); + for (int j = 0; j < qk; ++j) { + const float x0 = x[i * qk + j] * id; + signed_weights[i * qk + j] = (int8_t) roundf(x0); + } + } + } +} + +void quantize_q8_1(const float * x, + ov::Tensor & weights_arr, + ov::Tensor & scales_arr, + ov::Tensor & zp_arr, + int64_t k, + int64_t qk, + int64_t block_offset = 0) { + assert(k % qk == 0); + const int nb = k / qk; + + // See quantize_q8_0: block_offset places this chunk's output at the right block. + auto * weights = static_cast(weights_arr.data()) + block_offset * qk; + auto * scales = scales_arr.data::value_type>() + block_offset; + auto * zp = static_cast(zp_arr.data()) + block_offset; + for (int i = 0; i < nb; i++) { + float min = std::numeric_limits::max(); + float max = std::numeric_limits::lowest(); + + for (int j = 0; j < qk; j++) { + const float v = x[i * qk + j]; + min = std::min(v, min); + max = std::max(v, max); + } + + const float d = (max - min) / ((1 << 8) - 1); + const float id = d ? 1.0f / d : 0.0f; + scales[i] = ov::float16(d); + // zp = -min / scale (Q8_1 is asymmetric) + zp[i] = (d != 0.0f) ? (uint8_t) std::round(-min / d) : 0; + + for (int j = 0; j < qk; ++j) { + const float x0 = (x[i * qk + j] - min) * id; + const uint8_t xi0 = roundf(x0); + weights[i * qk + j] = xi0; + } + } +} + // Extract quantized weights from tensor and create weight subgraph +// If weights/scales/zp are provided (non-empty), uses them as output buffers +// Otherwise allocates new ov::Tensors internally +// Returns the weight node (make_int4_weights or make_int8_weights result) std::shared_ptr extract_quantized_weights(const ggml_tensor * tensor, - const void * data, + const void * data, // Source data pointer (may differ from tensor->data) ov::Tensor & weights, ov::Tensor & scales, ov::Tensor & zp, - bool use_bias) { + // Use an exact f16 zero point (vs. a rounded integer one); always + // used for for_gather_matmul (3D MoE expert) weights regardless of + // this flag, and also settable explicitly for test-backend-ops. + bool use_bias = false) { // Create a temporary tensor for extraction functions that read from tensor->data ggml_tensor temp_tensor = *tensor; temp_tensor.data = const_cast(data); @@ -837,9 +1106,11 @@ std::shared_ptr extract_quantized_weights(const ggml_tensor * tensor, return result; } -// Requantize weights to target format, writing to provided buffers +// Requantize weights from tensor to target format, writing to provided buffers +// For F16 target, only weights buffer is used (scales/zp ignored) +// Returns the weight node std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, - const void * data, + const void * data, // Source data pointer ExtraQuantType requant_type, int64_t block_size, ov::Tensor & weights, @@ -851,7 +1122,8 @@ std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, const auto * type_traits = ggml_get_type_traits(tensor->type); const size_t src_row_bytes = ggml_row_size(tensor->type, ne0); - bool is_u4 = (requant_type == ExtraQuantType::Q4_0_C || requant_type == ExtraQuantType::Q4_0_128); + bool is_u4 = (requant_type == ExtraQuantType::Q4_0_C || requant_type == ExtraQuantType::Q4_0_128 || + requant_type == ExtraQuantType::Q4_0_64 || requant_type == ExtraQuantType::Q4_1_64); // Streaming dequant (opt-in via GGML_OPENVINO_REDUCE_COMPILE_MEM or // GGML_OPENVINO_MEMORY_OPTIMIZE): instead of @@ -879,7 +1151,9 @@ std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, result->set_friendly_name(tensor->name); return result; } - if (is_u4) { + if (requant_type == ExtraQuantType::Q4_1_64) { + quantize_q4_1_asym(weights_f32.data(), weights, scales, zp, n_elements, block_size); + } else if (is_u4) { quantize_q4_0(weights_f32.data(), weights, scales, zp, n_elements, block_size); } else if (requant_type == ExtraQuantType::Q8_1_C) { quantize_q8_1(weights_f32.data(), weights, scales, zp, n_elements, block_size); @@ -930,6 +1204,7 @@ std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, result->set_friendly_name(tensor->name); return result; } +} // namespace OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, void * output_base_ptr, bool use_bias) { GGML_ASSERT(tensor != nullptr); @@ -1027,7 +1302,9 @@ OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, vo } else { result.weights = ov::Tensor(ov::element::f16, node_shape); } - ov::Tensor dummy_scales, dummy_zp; // Not used for F16 + // Not used for F16: + ov::Tensor dummy_scales; + ov::Tensor dummy_zp; result.weight_node = requantize_to_buffers(tensor, data, ExtraQuantType::F16, 0, result.weights, dummy_scales, dummy_zp); return result; @@ -1036,10 +1313,14 @@ OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, vo // Quantized path (normal extraction or quantized requant) // Create weight/scale/zp tensors - shared between both paths // For symmetric quantization, use signed types (i4/i8) and no ZP tensor - ov::element::Type weight_type = tensor->type == GGML_TYPE_MXFP4 ? - ov::element::f4e2m1 : - (layout.is_symmetric ? (layout.is_u4 ? ov::element::i4 : ov::element::i8) : - (layout.is_u4 ? ov::element::u4 : ov::element::u8)); + ov::element::Type weight_type; + if (tensor->type == GGML_TYPE_MXFP4) { + weight_type = ov::element::f4e2m1; + } else if (layout.is_symmetric) { + weight_type = layout.is_u4 ? ov::element::i4 : ov::element::i8; + } else { + weight_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; + } ov::Shape scale_shape = node_shape; scale_shape.back() /= layout.weights_per_block; @@ -1057,28 +1338,25 @@ OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, vo scale_shape.back() /= layout.weights_per_block; } + const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; + ov::element::Type zp_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; + if (zp_is_f16) { + zp_type = ov::element::f16; + } + if (output_base_ptr) { uint8_t * buf_base = static_cast(output_base_ptr); result.weights = ov::Tensor(weight_type, node_shape, buf_base + layout.weights_offset); - const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; result.scales = ov::Tensor(scale_type, scale_shape, buf_base + layout.scales_offset); if (!layout.is_symmetric) { - ov::element::Type zp_type = - zp_is_f16 ? ov::element::f16 : (layout.is_u4 ? ov::element::u4 : ov::element::u8); result.zp = ov::Tensor(zp_type, scale_shape, buf_base + layout.zp_offset); } // else: result.zp remains default-constructed (empty) for symmetric } else { result.weights = ov::Tensor(weight_type, node_shape); - const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; result.scales = ov::Tensor(scale_type, scale_shape); if (!layout.is_symmetric) { - if (zp_is_f16) { - result.zp = ov::Tensor(ov::element::f16, scale_shape); - } else { - ov::element::Type zp_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; - result.zp = ov::Tensor(zp_type, scale_shape); - } + result.zp = ov::Tensor(zp_type, scale_shape); } // else: result.zp remains default-constructed (empty) for symmetric } @@ -1093,181 +1371,3 @@ OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, vo return result; } - -void quantize_q4_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk) { - assert(k % qk == 0); - const int nb = k / qk; - - auto * weights = static_cast(weights_arr.data()); - auto * scales = scales_arr.data::value_type>(); - bool is_symmetric = (weights_arr.get_element_type() == ov::element::i4); // Signed i4 path - - if (!is_symmetric) { - auto * zp = static_cast(zp_arr.data()); - for (int i = 0; i < nb; i++) { - float amax = 0.0f; - float max = 0.0f; - for (int j = 0; j < qk; j++) { - const float v = x[i * qk + j]; - if (amax < fabsf(v)) { - amax = fabsf(v); - max = v; - } - } - const float d = max / -8; - if (d == 0) { - scales[i] = ov::float16(1.0f); - if (i % 2 == 0) { - zp[i / 2] = 8; - } else { - zp[i / 2] |= (8 << 4); - } - memset(weights + i * qk / 2, 8 | (8 << 4), qk / 2); - continue; - } - const float id = 1.0f / d; - scales[i] = ov::float16(d); - if (i % 2 == 0) { - zp[i / 2] = 8; - } else { - zp[i / 2] |= (8 << 4); - } - for (int j = 0; j < qk / 2; ++j) { - const float x0 = x[i * qk + 2 * j] * id; - const float x1 = x[i * qk + 2 * j + 1] * id; - const uint8_t xi0 = MIN(15, (int8_t) (x0 + 8.5f)); - const uint8_t xi1 = MIN(15, (int8_t) (x1 + 8.5f)); - weights[i * qk / 2 + j] = xi0 | (xi1 << 4); - } - } - } else { - // Symmetric: produce signed i4 values in [-8, 7] - for (int i = 0; i < nb; i++) { - float amax = 0.0f; - float max = 0.0f; - for (int j = 0; j < qk; j++) { - const float v = x[i * qk + j]; - if (amax < fabsf(v)) { - amax = fabsf(v); - max = v; - } - } - const float d = max / -8; - if (d == 0) { - scales[i] = ov::float16(1.0f); - // i4 value 0 packed: 0x00 - memset(weights + i * qk / 2, 0, qk / 2); - continue; - } - const float id = 1.0f / d; - scales[i] = ov::float16(d); - for (int j = 0; j < qk / 2; ++j) { - const float x0 = x[i * qk + 2 * j] * id; - const float x1 = x[i * qk + 2 * j + 1] * id; - // Signed i4: range [-8, 7]. Quantize as round(x*id), then pack as 4-bit two's complement. - int8_t si0 = (int8_t) std::max(-8, std::min(7, (int) roundf(x0))); - int8_t si1 = (int8_t) std::max(-8, std::min(7, (int) roundf(x1))); - weights[i * qk / 2 + j] = (si0 & 0x0F) | ((si1 & 0x0F) << 4); - } - } - } -} - -void quantize_q8_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset) { - assert(k % qk == 0); - const int nb = k / qk; - - // block_offset lets a caller quantize a chunk of blocks into the right place in the - // output buffers (used for streaming requant). x points at this chunk's first block; - // outputs are advanced by block_offset blocks. Q8 has one scale/zp per block (no - // nibble packing), so any block boundary is safe. - auto * weights = static_cast(weights_arr.data()) + block_offset * qk; - auto * scales = scales_arr.data::value_type>() + block_offset; - bool is_symmetric = (weights_arr.get_element_type() == ov::element::i8); // Signed i8 path - - if (!is_symmetric) { - auto * zp = static_cast(zp_arr.data()) + block_offset; - for (int i = 0; i < nb; i++) { - float amax = 0.0f; - for (int j = 0; j < qk; j++) { - const float v = x[i * qk + j]; - amax = std::max(amax, fabsf(v)); - } - const float d = amax / 127.0f; - const float id = d ? 1.0f / d : 0.0f; - scales[i] = ov::float16(d); - zp[i] = 128; - for (int j = 0; j < qk; ++j) { - const float x0 = x[i * qk + j] * id; - const int8_t xi0 = roundf(x0); - weights[i * qk + j] = (uint8_t) (xi0 + 128); - } - } - } else { - // Symmetric: store signed int8 values directly - auto * signed_weights = reinterpret_cast(weights); - for (int i = 0; i < nb; i++) { - float amax = 0.0f; - for (int j = 0; j < qk; j++) { - const float v = x[i * qk + j]; - amax = std::max(amax, fabsf(v)); - } - const float d = amax / 127.0f; - const float id = d ? 1.0f / d : 0.0f; - scales[i] = ov::float16(d); - for (int j = 0; j < qk; ++j) { - const float x0 = x[i * qk + j] * id; - signed_weights[i * qk + j] = (int8_t) roundf(x0); - } - } - } -} - -void quantize_q8_1(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset) { - assert(k % qk == 0); - const int nb = k / qk; - - // See quantize_q8_0: block_offset places this chunk's output at the right block. - auto * weights = static_cast(weights_arr.data()) + block_offset * qk; - auto * scales = scales_arr.data::value_type>() + block_offset; - auto * zp = static_cast(zp_arr.data()) + block_offset; - for (int i = 0; i < nb; i++) { - float min = std::numeric_limits::max(); - float max = std::numeric_limits::lowest(); - - for (int j = 0; j < qk; j++) { - const float v = x[i * qk + j]; - min = std::min(v, min); - max = std::max(v, max); - } - - const float d = (max - min) / ((1 << 8) - 1); - const float id = d ? 1.0f / d : 0.0f; - scales[i] = ov::float16(d); - // zp = -min / scale (Q8_1 is asymmetric) - zp[i] = (d != 0.0f) ? (uint8_t) std::round(-min / d) : 0; - - for (int j = 0; j < qk; ++j) { - const float x0 = (x[i * qk + j] - min) * id; - const uint8_t xi0 = roundf(x0); - weights[i * qk + j] = xi0; - } - } -} diff --git a/ggml/src/ggml-openvino/ggml-quants.h b/ggml/src/ggml-openvino/ggml-quants.h index e247255a..04fe0218 100644 --- a/ggml/src/ggml-openvino/ggml-quants.h +++ b/ggml/src/ggml-openvino/ggml-quants.h @@ -2,112 +2,12 @@ #include "ggml-openvino-extra.h" // For ExtraQuantType #include "ggml.h" -#include -#include #include +#include #include -void unpack_32_4(const uint8_t * data, uint8_t * dst); - -void extract_q4_0_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void extract_q4_1_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q5_1_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q8_0_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void unpack_256_4(const uint8_t * data, uint8_t * dst); - -void extract_q4_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q5_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q6_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void extract_mxfp4_data(const ggml_tensor * tensor, ov::Tensor & weights_arr, ov::Tensor & scales_arr); - static constexpr size_t GGML_QUANTIZATION_GROUP_SIZE = 32; -// If for_gather_matmul is true, the weight tensor may be N-D (e.g. 3D MoE expert weights -// [n_expert, rows, cols]). The dequantization chain (Convert->[Subtract]->Multiply) is built as -// usual but left in f16 (no final Convert to f32) -- ov::pass::MarkDequantization (registered in -// translate_session.cpp) marks the chain so it survives model-build-time ConstantFolding -- see -// make_int8_weights.cpp/make_int4_weights.cpp. mul_mat_id.cpp constructs ov::op::internal::GatherMatmul -// directly from the resulting f16 dequant chain. -// -// When use_bias is true (explicitly, or implicitly because for_gather_matmul is true), the zp -// tensor is expected to hold an exact f16 bias value (rather than a rounded integer zero point); -// it is converted in place into an exact zero_point = -bias/scale and consumed via Subtract, not -// Add, so the chain still matches OpenVINO's Convert->Subtract->Multiply decompression pattern. -ov::Output make_int8_weights(ov::Tensor & weight, - ov::Tensor & scales, - ov::Tensor & zp, - size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, - bool use_bias = false, - bool for_gather_matmul = false); - -ov::Output make_int4_weights(ov::Tensor & weight, - ov::Tensor & scales, - ov::Tensor & zp, - size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, - bool use_bias = false, - bool for_gather_matmul = false); - -ov::Output make_mxfp4_weights(ov::Tensor & weight, ov::Tensor & scales); - -ov::Output make_mxfp4_moe_packed_weights(ov::Tensor & weight); - -// Extract quantized weights from tensor and create weight subgraph -// If weights/scales/zp are provided (non-empty), uses them as output buffers -// Otherwise allocates new ov::Tensors internally -// Returns the weight node (make_int4_weights or make_int8_weights result) -std::shared_ptr extract_quantized_weights( - const ggml_tensor * tensor, - const void * data, // Source data pointer (may differ from tensor->data) - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp, - bool use_bias = false); // Use an exact f16 zero point (vs. a rounded integer one); always - // used for for_gather_matmul (3D MoE expert) weights regardless of - // this flag, and also settable explicitly for test-backend-ops. - -// Requantize weights from tensor to target format, writing to provided buffers -// For F16 target, only weights buffer is used (scales/zp ignored) -// Returns the weight node -std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, - const void * data, // Source data pointer - ExtraQuantType requant_type, - int64_t block_size, - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp); - inline const char * extra_quant_type_name(ExtraQuantType t) { switch (t) { case ExtraQuantType::F16: @@ -122,6 +22,10 @@ inline const char * extra_quant_type_name(ExtraQuantType t) { return "Q8_0_32"; case ExtraQuantType::Q8_1_C: return "Q8_1_C"; + case ExtraQuantType::Q4_0_64: + return "Q4_0_64"; + case ExtraQuantType::Q4_1_64: + return "Q4_1_64"; default: return "unknown"; } @@ -152,35 +56,3 @@ OvWeight process_weight_tensor( // always used for for_gather_matmul (3D MoE expert) weights // regardless of this flag, and also settable explicitly for // test-backend-ops. - -void quantize_q4_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk); -void quantize_q8_1(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset = 0); -void quantize_q8_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset = 0); - -namespace ov { -namespace op { -namespace util { -// From /src/common/transformations/include/transformations/utils/utils.hpp -bool get_single_value(const std::shared_ptr & const_node, - float & value, - bool check_value_range = true); -} // namespace util -} // namespace op -} // namespace ov diff --git a/ggml/src/ggml-openvino/model-cache.cpp b/ggml/src/ggml-openvino/model-cache.cpp index 3fc7028d..3725fbd2 100644 --- a/ggml/src/ggml-openvino/model-cache.cpp +++ b/ggml/src/ggml-openvino/model-cache.cpp @@ -237,7 +237,8 @@ bool ggml_openvino_model_cache_verify_manifest(const std::string & path, if (!f.is_open()) { return false; } - std::string tag, val; + std::string tag; + std::string val; // header: fingerprint if (!(f >> tag >> val) || tag != "fingerprint" || val != hex64(fingerprint)) { return false; diff --git a/ggml/src/ggml-openvino/openvino/frontend.cpp b/ggml/src/ggml-openvino/openvino/frontend.cpp index c2ba14e6..88de86fe 100644 --- a/ggml/src/ggml-openvino/openvino/frontend.cpp +++ b/ggml/src/ggml-openvino/openvino/frontend.cpp @@ -3,6 +3,7 @@ #include "input_model.h" #include "op_table.h" #include "translate_session.h" +#include namespace ov { namespace frontend { @@ -11,7 +12,7 @@ namespace ggml { FrontEnd::FrontEnd() {} std::shared_ptr FrontEnd::convert(const InputModel::Ptr & model, bool naive) { - auto ggml_model = std::dynamic_pointer_cast(model); + auto ggml_model = ov::as_type_ptr(model); FRONT_END_GENERAL_CHECK(ggml_model, "Invalid input model"); std::shared_ptr converted_model; const auto & supported_ops = get_supported_ops(); diff --git a/ggml/src/ggml-openvino/openvino/frontend.h b/ggml/src/ggml-openvino/openvino/frontend.h index 72134a3e..4e301d32 100644 --- a/ggml/src/ggml-openvino/openvino/frontend.h +++ b/ggml/src/ggml-openvino/openvino/frontend.h @@ -12,7 +12,6 @@ namespace ggml { class FrontEnd { public: - using Ptr = std::shared_ptr; FrontEnd(); static std::shared_ptr convert(const InputModel::Ptr & model, bool naive = false); diff --git a/ggml/src/ggml-openvino/openvino/node_context.h b/ggml/src/ggml-openvino/openvino/node_context.h index 2e275603..f1ea0e4f 100644 --- a/ggml/src/ggml-openvino/openvino/node_context.h +++ b/ggml/src/ggml-openvino/openvino/node_context.h @@ -143,6 +143,10 @@ class NodeContext : public frontend::NodeContext { bool has_input(const std::string & name) const { return m_tensor_map->find(name) != m_tensor_map->end(); } + void put_shared(const std::string & name, const Output & value) const { + m_tensor_map->insert({name, value}); + } + const std::string & get_name() const override { return m_decoder->get_op_name(m_node_idx); } ov::Any get_attribute_as_any(const std::string & name) const override { return m_decoder->get_attribute(name); } diff --git a/ggml/src/ggml-openvino/openvino/op/add.cpp b/ggml/src/ggml-openvino/openvino/op/add.cpp index c43eb67f..a45520d9 100644 --- a/ggml/src/ggml-openvino/openvino/op/add.cpp +++ b/ggml/src/ggml-openvino/openvino/op/add.cpp @@ -5,6 +5,7 @@ #include #include #include +#include #include #include @@ -35,7 +36,20 @@ OutputVector translate_add(const NodeContext & context) { auto input_0 = process_view_input_new(context, 0); auto input_1 = process_view_input_new(context, 1); - auto res = std::make_shared(input_0, input_1); + // opset1::Add needs matching types (e.g. fused ADD_ADD mixes f16/f32); add in f32, cast once. + auto output_type = context.get_output_type(); + if (input_0.get_element_type() != input_1.get_element_type()) { + if (input_0.get_element_type() != ov::element::f32) { + input_0 = std::make_shared(input_0, ov::element::f32); + } + if (input_1.get_element_type() != ov::element::f32) { + input_1 = std::make_shared(input_1, ov::element::f32); + } + } + ov::Output res = std::make_shared(input_0, input_1); + if (res.get_element_type() != output_type) { + res = std::make_shared(res, output_type); + } return rename_outputs_with_suffix({res}, context.get_name()); } diff --git a/ggml/src/ggml-openvino/openvino/op/add_id.cpp b/ggml/src/ggml-openvino/openvino/op/add_id.cpp index e54d700d..79bdbe87 100644 --- a/ggml/src/ggml-openvino/openvino/op/add_id.cpp +++ b/ggml/src/ggml-openvino/openvino/op/add_id.cpp @@ -20,7 +20,7 @@ namespace op { static ov::Output reshape_add_id_input_to_2d(const ov::Output & input, const ov::PartialShape & input_shape, const std::vector & dims) { - const auto actual_shape = input.get_partial_shape(); + const auto & actual_shape = input.get_partial_shape(); if (actual_shape.rank().is_static() && actual_shape.rank().get_length() == 2) { return input; } diff --git a/ggml/src/ggml-openvino/openvino/op/cont.cpp b/ggml/src/ggml-openvino/openvino/op/cont.cpp index 1d6cc672..9888f6b9 100644 --- a/ggml/src/ggml-openvino/openvino/op/cont.cpp +++ b/ggml/src/ggml-openvino/openvino/op/cont.cpp @@ -3,12 +3,9 @@ #include "../op_table.h" #include "../utils.h" -#include -#include #include #include #include -#include namespace ov { namespace frontend { diff --git a/ggml/src/ggml-openvino/openvino/op/cpy.cpp b/ggml/src/ggml-openvino/openvino/op/cpy.cpp index 5b387fc5..6f1e3477 100644 --- a/ggml/src/ggml-openvino/openvino/op/cpy.cpp +++ b/ggml/src/ggml-openvino/openvino/op/cpy.cpp @@ -3,8 +3,11 @@ #include "../utils.h" #include +#include +#include #include -#include +#include +#include #include #include #include @@ -12,9 +15,14 @@ #include #include #include +#include #include +#include #include #include +#include +#include +#include namespace ov { namespace frontend { @@ -61,10 +69,27 @@ OutputVector translate_cpy(const NodeContext & context) { return rename_outputs_with_suffix({res}, context.get_name()); } - // Recurrent state cache writeback into a slot block of the cache. Where the block starts and - // where the copied data starts in the source are runtime inputs, so the cached model works for - // any kv head, active sequence count and token count. The result is the full updated cache. + // Recurrent state cache writeback into a slot block of the cache. Where the block starts is a + // runtime input, so the cached model works for any kv head and active sequence count. The + // result is the full updated cache. // op_case 1: gated-delta-net state, op_case 2: conv state, op_case 3: defrag remainder. + if (op_case == 3) { + // With -np 1 (and generally whenever there is no defrag remainder) this GET_ROWS gathers + // zero rows: nothing to write back, and the cache is unchanged. NPU rejects zero-size + // tensors, so short-circuit instead of building a degenerate Slice/Concat chain. + bool is_empty = false; + if (input_shape.rank().is_static()) { + for (const auto & d : input_shape) { + if (d.is_static() && d.get_length() == 0) { + is_empty = true; + break; + } + } + } + if (is_empty) { + return {context.get_input(1)}; + } + } const std::string slot_begin_name = "rs_slot_begin_" + context.get_name(); const bool slice_assign = context.has_input(slot_begin_name) && !context.is_stateful() && (op_case >= 1 && op_case <= 3); @@ -81,19 +106,49 @@ OutputVector translate_cpy(const NodeContext & context) { ov::Output begin = context.get_input(slot_begin_name); auto base = context.get_input(1); if (op_case == 1) { - // GDN packs [attn | state snapshots]; the state part runs from src_begin to the end. - auto src_begin = context.get_input("rs_src_begin_" + context.get_name()); - auto state_part = std::make_shared(context.get_input(0), src_begin, int_max, one, axis); + ov::Output state_begin; + const std::string src_begin_name = "rs_src_begin_" + context.get_name(); + if (context.has_input(src_begin_name)) { + state_begin = context.get_input(src_begin_name); + } else { + auto ssm_state_size = context.get_ssm_state_size(); + if (context.has_input("s_copy_active_slot_len")) { + auto len = context.get_input("s_copy_active_slot_len"); + auto state_rows = std::make_shared( + ov::op::v0::Constant::create(ov::element::i64, {1}, {ssm_state_size}), len); + state_begin = std::make_shared(state_rows); + } else { + state_begin = ov::op::v0::Constant::create(ov::element::i64, {1}, {-ssm_state_size}); + } + } + auto state_part = + std::make_shared(context.get_input(0), state_begin, int_max, one, axis); src = std::make_shared(state_part, feature, false); } else if (op_case == 2) { - // conv_input is [previous conv state | new tokens]; copy the conv_kernel_size - 1 wide - // window starting at src_begin, which is the snapshot this writeback corresponds to. + // conv_input is [previous conv state | new tokens]; the snapshot is the conv_kernel_size - 1 + // columns ending at the last *valid* token. Gather (rather than Slice) keeps the output + // shape static even though the window start is a runtime value. auto window_size = (int64_t) input_shape[3].get_length(); - auto src_begin = context.get_input("rs_src_begin_" + context.get_name()); - auto src_end = std::make_shared( - src_begin, ov::op::v0::Constant::create(ov::element::i64, {1}, {window_size})); - auto window = std::make_shared(context.get_input(0), src_begin, src_end, one, - ov::op::v0::Constant::create(ov::element::i64, {1}, {3})); + ov::Output window; + auto col_axis = ov::op::v0::Constant::create(ov::element::i64, {1}, {3}); + const std::string src_begin_name = "rs_src_begin_" + context.get_name(); + if (context.has_input(src_begin_name)) { + auto src_begin = context.get_input(src_begin_name); + auto src_end = std::make_shared( + src_begin, ov::op::v0::Constant::create(ov::element::i64, {1}, {window_size})); + window = std::make_shared(context.get_input(0), src_begin, src_end, one, col_axis); + } else if (context.has_input("chunk_valid_len")) { + std::vector offsets(window_size); + std::iota(offsets.begin(), offsets.end(), 0); + auto indices = std::make_shared( + ov::op::v0::Constant::create(ov::element::i64, {(size_t) window_size}, offsets), + context.get_input("chunk_valid_len")); + window = std::make_shared(context.get_input(0), indices, col_axis); + } else { + auto window_begin = ov::op::v0::Constant::create(ov::element::i64, {1}, {-window_size}); + window = + std::make_shared(context.get_input(0), window_begin, int_max, one, col_axis); + } const auto base_shape = base.get_partial_shape(); FRONT_END_OP_CONVERSION_CHECK(base_shape.rank().is_static() && base_shape.rank().get_length() == 4, "CPY conv state cache update requires rank-4 base cache"); @@ -157,6 +212,63 @@ OutputVector translate_cpy(const NodeContext & context) { auto input = process_view_input_new(context, 0); + if (op_case == 5 || op_case == 6) { + auto input_shape = context.get_input_shape(0); + auto output_shape = context.get_output_shape(); + auto dst_ggml_shape = context.get_view_input_ggml_shape(1, 0); + auto dst_stride = context.get_view_input_stride(1, 0); + size_t offset_bytes = context.get_view_input_offset(1, 0); + auto n_state = (int64_t) context.get_input_shape(0)[3].get_length(); + auto n_state_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_state}); + auto kv_buf = context.get_input(1); // shape {1,1,1,N} + + Output token_len_per_seq; + Output n_write_dyn; + if (context.has_input("token_len_per_seq")) { + token_len_per_seq = context.get_input("token_len_per_seq"); + n_write_dyn = std::make_shared(token_len_per_seq, n_state_c); + } else { + n_write_dyn = ov::op::v0::Constant::create(ov::element::i64, {1}, {(int64_t) dst_ggml_shape[3]}); + } + size_t elem_size = dst_stride[3]; + FRONT_END_OP_CONVERSION_CHECK(elem_size > 0, "CPY KV cache view update has invalid element size"); + int64_t start_elem = (int64_t) (offset_bytes / elem_size); + // op_case 5: decoder self-attention – write offset advances each step. + // op_case 6: encoder self-attn or cross-attn – offset fixed at compile time. + const bool is_decoder_self_attn = (op_case == 5); + auto ones_c = ov::op::v0::Constant::create(ov::element::i64, {3}, std::vector{1, 1, 1}); + auto new_shape = std::make_shared(ov::OutputVector{ones_c, n_write_dyn}, 0); + + auto reshaped = std::make_shared(input, new_shape, false); + auto data = std::make_shared(reshaped, context.get_output_type()); + // Indices [start_elem .. start_elem + n_write) on axis 3 of {1,1,1,N} + // For decoder self-attention the write offset advances each step, so compute it + // dynamically from the model inputs: start = (attention_size - token_len_per_seq) * n_state. + // For encoder self-attn and cross-attn the offset is fixed at graph-compile time. + ov::Output start; + if (is_decoder_self_attn && context.has_input("attention_size") && context.has_input("token_len_per_seq")) { + auto attention_size_in = context.get_input("attention_size"); + auto token_len_in = context.get_input("token_len_per_seq"); + auto past_tokens = std::make_shared(attention_size_in, token_len_in); + auto new_start = std::make_shared(past_tokens, n_state_c); + start = std::make_shared( + new_start, ov::op::v0::Constant::create(ov::element::i64, {1}, {start_elem})); + } else { + start = ov::op::v0::Constant::create(ov::element::i64, {1}, {start_elem}); + } + auto start_squeezed = std::make_shared(start); + auto end = std::make_shared(start_squeezed, n_write_dyn); + auto end_squeezed = std::make_shared(end); + auto step = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); + auto step_squeezed = std::make_shared(step); + auto indices = + std::make_shared(start_squeezed, end_squeezed, step_squeezed, ov::element::i64); + auto axis = ov::op::v0::Constant::create(ov::element::i64, {1}, {3}); + + auto kv_updated = std::make_shared(kv_buf, indices, data, axis); + return rename_outputs_with_suffix({kv_updated}, context.get_name()); + } + if (input_shape != output_shape) { auto new_shape = ov::op::v0::Constant::create( ov::element::i64, {static_cast(output_shape.rank().get_length())}, output_shape.to_shape()); diff --git a/ggml/src/ggml-openvino/openvino/op/diag.cpp b/ggml/src/ggml-openvino/openvino/op/diag.cpp index dacea2f0..05e06489 100644 --- a/ggml/src/ggml-openvino/openvino/op/diag.cpp +++ b/ggml/src/ggml-openvino/openvino/op/diag.cpp @@ -3,11 +3,8 @@ #include "../utils.h" #include -#include +#include #include -#include -#include -#include namespace ov { namespace frontend { @@ -23,31 +20,13 @@ namespace op { OutputVector translate_diag(const NodeContext & context) { num_inputs_check(context, 1, 1); - auto x = context.get_input(0); // OV shape: [ne3, ne2, 1, ne0] + auto x = process_view_input_new(context, 0); // OV shape: [ne3, ne2, 1, ne0] - auto out_shape = context.get_output_shape().to_shape(); - int64_t n = static_cast(out_shape[3]); // ne0 + auto n = get_dimensions(x.get_node_shared_ptr(), {3}); + auto zero_diag = ov::op::v0::Constant::create(ov::element::i64, {}, {0}); - // Build index range [0, 1, ..., n-1] - auto start = ov::op::v0::Constant::create(ov::element::i64, {}, {int64_t(0)}); - auto stop = ov::op::v0::Constant::create(ov::element::i64, {}, {n}); - auto step = ov::op::v0::Constant::create(ov::element::i64, {}, {int64_t(1)}); - auto range = std::make_shared(start, stop, step, ov::element::i64); - - // col_idx shape [1, 1, 1, n] - auto col_shape = ov::op::v0::Constant::create(ov::element::i64, {4}, std::vector{1, 1, 1, n}); - auto col_idx = std::make_shared(range, col_shape, false); - - // row_idx shape [1, 1, n, 1] - auto row_shape = ov::op::v0::Constant::create(ov::element::i64, {4}, std::vector{1, 1, n, 1}); - auto row_idx = std::make_shared(range, row_shape, false); - - // mask: true where col == row (diagonal) - auto mask = std::make_shared(col_idx, row_idx); - - // Broadcast input from [ne3, ne2, 1, ne0] to [ne3, ne2, ne0, ne0] via select - auto zero = ov::op::v0::Constant::create(ov::element::f32, {}, {0.0f}); - auto res = std::make_shared(mask, x, zero); + auto eye = std::make_shared(n, n, zero_diag, x.get_element_type()); + auto res = std::make_shared(x, eye); return rename_outputs_with_suffix({res}, context.get_name()); } diff --git a/ggml/src/ggml-openvino/openvino/op/div.cpp b/ggml/src/ggml-openvino/openvino/op/div.cpp index 11dd9dec..2089ffd4 100644 --- a/ggml/src/ggml-openvino/openvino/op/div.cpp +++ b/ggml/src/ggml-openvino/openvino/op/div.cpp @@ -4,12 +4,14 @@ #include "ggml.h" #include +#include #include #include #include #include #include #include +#include #include #include #include @@ -33,22 +35,12 @@ bool is_silu_div_pattern(const ov::Output & numerator, return false; } - auto mul = std::dynamic_pointer_cast(numerator.get_node_shared_ptr()); - if (!mul) { - return false; - } - const auto denom_node = denominator.get_node_shared_ptr(); - const auto mul_input_0 = mul->input_value(0).get_node_shared_ptr(); - const auto mul_input_1 = mul->input_value(1).get_node_shared_ptr(); - auto sigmoid = std::dynamic_pointer_cast(mul_input_1); - if (mul_input_0 == denom_node && sigmoid && sigmoid->input_value(0).get_node_shared_ptr() == denom_node) { - return true; + if (auto swish = ov::as_type_ptr(numerator.get_node_shared_ptr())) { + return swish->input_value(0).get_node_shared_ptr() == denom_node; } - - sigmoid = std::dynamic_pointer_cast(mul_input_0); - return mul_input_1 == denom_node && sigmoid && sigmoid->input_value(0).get_node_shared_ptr() == denom_node; + return false; } ov::Output repeat_input_to_match(const NodeContext & context, diff --git a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp index 582df013..b06d01dc 100644 --- a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp +++ b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp @@ -3,8 +3,8 @@ #include "../utils.h" #include "ggml-openvino/ggml-openvino-extra.h" +#include #include -#include #include #include #include @@ -15,6 +15,7 @@ #include #include #include +#include #include #include #include @@ -24,13 +25,62 @@ namespace ov { namespace frontend { namespace ggml { namespace op { +static ov::Output reshape_flat_kv(const ov::Output & kv_flat, + size_t view_offset_bytes, + size_t nb1_bytes, + int64_t n_head, + int64_t head_size, + const ov::Output & attention_size) { + int64_t n_state = n_head * head_size; + int64_t layer_start_elem = (int64_t) (view_offset_bytes / (nb1_bytes / n_state)); + // Dynamic slice: [layer_start_elem, layer_start_elem + n_kv * n_state) + auto start_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {layer_start_elem}); + auto n_state_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_state}); + // end = start + attention_size * n_state (both static + dynamic) + auto kv_len_elems = std::make_shared(attention_size, n_state_c); + auto end_c = std::make_shared(start_c, kv_len_elems); + auto step_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); + auto axis_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {3}); + auto sliced = std::make_shared(kv_flat, start_c, end_c, step_c, axis_c); + + // KV cache is laid out as {n_kv, n_head, head_size} in memory + // Reshape to {1, n_kv, n_head, head_size}, then transpose to {1, n_head, n_kv, head_size} + // as required by SDPA. + auto one_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); + auto n_head_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_head}); + auto head_size_c = ov::op::v0::Constant::create(ov::element::i64, {1}, {head_size}); + // reshape: {n_kv*n_state} -> {1, n_kv, n_head, head_size} + auto new_shape = + std::make_shared(ov::OutputVector{one_c, attention_size, n_head_c, head_size_c}, 0); + auto reshaped = std::make_shared(sliced, new_shape, false); + // transpose: {1, n_kv, n_head, head_size} -> {1, n_head, n_kv, head_size} + auto perm = ov::op::v0::Constant::create(ov::element::i64, {4}, {0, 2, 1, 3}); + auto ret = std::make_shared(reshaped, perm); + return ret; +} OutputVector translate_flash_attn_ext(const NodeContext & context) { - num_inputs_check(context, 4, 4); + num_inputs_check(context, 3, 4); + const bool has_mask = context.get_input_size() == 4; auto q_f32 = context.get_input(0); auto k = context.get_input(1); auto v = context.get_input(2); - auto mask = context.get_input(3); + const int op_case = context.get_op_case(); + + if (op_case == 1 || op_case == 2) { + int64_t n_state_head = (int64_t) context.get_view_input_ggml_shape(1, 0)[3]; + int64_t n_head = (int64_t) context.get_view_input_ggml_shape(1, 0)[1]; + size_t nb1 = context.get_view_input_stride(1, 0)[2]; + size_t offset = context.get_view_input_offset(1, 0); + ov::Output attention_size; + if (op_case == 1) { + attention_size = context.get_input("attention_size"); + } else { + attention_size = context.get_input("attention_size_static"); + } + k = reshape_flat_kv(k, offset, nb1, n_head, n_state_head, attention_size); + v = reshape_flat_kv(v, offset, nb1, n_head, n_state_head, attention_size); + } float * params = reinterpret_cast(context.get_output_op_params()); float scale = params[0]; @@ -43,16 +93,19 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { ov::Output res; // For stateful - std::string mask_name = "KQ_mask_sliced"; - if (context.get_input_names()[3].find("swa") != std::string::npos) { - mask_name = "KQ_mask_swa_sliced"; - } - if (context.has_input(mask_name)) { - mask = context.get_input(mask_name); - } - - if (mask.get_element_type() != ov::element::f16) { - mask = std::make_shared(mask, ov::element::f16); + ov::Output mask; + if (has_mask) { + mask = context.get_input(3); + std::string mask_name = "KQ_mask_sliced"; + if (context.get_input_names()[3].find("swa") != std::string::npos) { + mask_name = "KQ_mask_swa_sliced"; + } + if (context.has_input(mask_name)) { + mask = context.get_input(mask_name); + } + if (mask.get_element_type() != ov::element::f16) { + mask = std::make_shared(mask, ov::element::f16); + } } //auto tile_kv = [&](int64_t num_heads, int64_t num_heads_kv, int64_t head_size, ov::Output kv) { @@ -108,10 +161,14 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { // get [B, 1, 1, S_q, S_k], which NUMPY-broadcasts cleanly against the // [B, num_heads_kv, factor, S_q, S_k] scores: B==B, then 1→num_heads_kv and // 1→factor on the head dims. - auto mask_unsq1 = - std::make_shared(mask, ov::op::v0::Constant::create(ov::element::i64, {1}, {2})); - // mask_unsq1: [B, 1, 1, S_q, S_k] (rank 5) - ov::Output qk_masked = std::make_shared(qk_scaled, mask_unsq1); + ov::Output qk_masked; + if (has_mask) { + auto mask_unsq1 = + std::make_shared(mask, ov::op::v0::Constant::create(ov::element::i64, {1}, {2})); + qk_masked = std::make_shared(qk_scaled, mask_unsq1); + } else { + qk_masked = qk_scaled; + } auto softmax = std::make_shared(qk_masked, /*axis=*/-1); @@ -138,7 +195,9 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { auto tile_kv = [&](int64_t n_heads, int64_t n_heads_kv, int64_t hs, ov::Output kv) { int64_t f = n_heads / n_heads_kv; if (f > 1 && n_heads_kv > 1) { - ov::Output kv_broadcast_shape, kv_unsqueezed, new_kv_shape; + ov::Output kv_broadcast_shape; + ov::Output kv_unsqueezed; + ov::Output new_kv_shape; auto unsqueeze_axes = ov::op::v0::Constant::create(ov::element::i64, Shape{}, {2}); kv_unsqueezed = std::make_shared(kv, unsqueeze_axes); @@ -164,9 +223,16 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { k = tile_kv(num_heads, num_heads_kv, head_size, k); v = tile_kv(num_heads, num_heads_kv, head_size, v); - auto sdpa = std::make_shared(q, k, v, mask, scale_node, false); - res = std::make_shared(sdpa, - ov::op::v0::Constant::create(ov::element::i64, {4}, {0, 2, 1, 3})); + constexpr auto causal = false; + if (has_mask) { + auto sdpa = std::make_shared(q, k, v, mask, scale_node, causal); + res = std::make_shared( + sdpa, ov::op::v0::Constant::create(ov::element::i64, {4}, {0, 2, 1, 3})); + } else { + auto sdpa = std::make_shared(q, k, v, scale_node, causal); + res = std::make_shared( + sdpa, ov::op::v0::Constant::create(ov::element::i64, {4}, {0, 2, 1, 3})); + } res = std::make_shared(res, ov::element::f32); return rename_outputs_with_suffix({res}, context.get_name()); } diff --git a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp index 66c74828..8d07c90b 100644 --- a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp +++ b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp @@ -7,12 +7,15 @@ #include #include #include +#include #include #include #include #include +#include #include #include +#include #include #include #include @@ -80,6 +83,28 @@ OutputVector translate_gated_delta_net(const NodeContext & context) { g = std::make_shared(g, ov::op::v0::Constant::create(ov::element::i64, {1}, {3})); beta = std::make_shared(beta, ov::op::v0::Constant::create(ov::element::i64, {1}, {3})); + if (context.has_input("chunk_valid_len")) { + // The last prefill chunk is padded with fabricated tokens. The recurrence is + // S_t = S_{t-1} * exp(g_t) + k_t (x) ((v_t - S_{t-1}^T k_t) * beta_t) + // so forcing g = 0 and beta = 0 makes a padded step an exact identity and keeps the final + // state equal to the state after the last real token. Attention output at those positions + // is garbage but never read. + const auto & g_ps = g.get_partial_shape(); + FRONT_END_OP_CONVERSION_CHECK(g_ps.rank().is_static() && g_ps.rank().get_length() == 3 && g_ps[1].is_static(), + "GATED_DELTA_NET pad masking requires a static token dimension"); + const int64_t n_tokens = g_ps[1].get_length(); + std::vector positions(n_tokens); + std::iota(positions.begin(), positions.end(), 0); + auto valid = std::make_shared( + ov::op::v0::Constant::create(ov::element::i64, {(size_t) n_tokens}, positions), + context.get_input("chunk_valid_len")); + auto mask = std::make_shared( + std::make_shared(valid, g.get_element_type()), + ov::op::v0::Constant::create(ov::element::i64, {2}, std::vector{0, 2})); + g = std::make_shared(g, mask); + beta = std::make_shared(beta, mask); + } + // std::cout << "GatedDeltaNet input shapes: q=" << q.get_partial_shape() << ", k=" << k.get_partial_shape() // << ", v=" << v.get_partial_shape() << ", g=" << g.get_partial_shape() // << ", beta=" << beta.get_partial_shape() << ", state=" << state.get_partial_shape() << std::endl; @@ -171,7 +196,7 @@ static OutputVector translate_gated_delta_net_ref(const NodeContext & context) { } // Merge batch and head dims: [B*H_v, T, S_v] - auto merge_bh = [&](ov::Output x, int64_t last_dim) { + auto merge_bh = [&](const ov::Output & x, int64_t last_dim) { auto shape = ov::op::v0::Constant::create(ov::element::i64, {3}, std::vector{B * H_v, T, last_dim}); return std::make_shared(x, shape, false); }; diff --git a/ggml/src/ggml-openvino/openvino/op/glu_geglu_quick.cpp b/ggml/src/ggml-openvino/openvino/op/glu_geglu_quick.cpp new file mode 100644 index 00000000..385d75f5 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/op/glu_geglu_quick.cpp @@ -0,0 +1,62 @@ +#include "../node_context.h" +#include "../op_table.h" +#include "../utils.h" + +#include +#include +#include +#include +#include +#include + +namespace ov { +namespace frontend { +namespace ggml { +namespace op { + +OutputVector translate_glu_geglu_quick(const NodeContext & context) { + num_inputs_check(context, 1, 2); + + ov::Output src0; + ov::Output src1; + if (context.get_input_size() == 2) { + src0 = process_view_input_new(context, 0); + src1 = process_view_input_new(context, 1); + } else { + // split along last axis, nc = ne[0] / 2 + auto combined = process_view_input_new(context, 0); + auto combined_shape = combined.get_partial_shape(); + int64_t last_dim_val = combined_shape[combined_shape.rank().get_length() - 1].get_length(); + int64_t nc = last_dim_val / 2; + + auto axis = ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}); + auto step = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); + auto start0 = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); + auto stop0 = ov::op::v0::Constant::create(ov::element::i64, {1}, {nc}); + auto start1 = ov::op::v0::Constant::create(ov::element::i64, {1}, {nc}); + auto stop1 = ov::op::v0::Constant::create(ov::element::i64, {1}, {2 * nc}); + + src0 = std::make_shared(combined, start0, stop0, step, axis); + src1 = std::make_shared(combined, start1, stop1, step, axis); + } + + int32_t * params = context.get_output_op_params(); + const int32_t swapped = params[1]; + if (swapped) { + std::swap(src0, src1); + } + + // GELU_QUICK(x) = x * sigmoid(1.702 * x) + // Create the constant in the same type as src0 to avoid f16/f32 mismatch. + auto input_type = src0.get_element_type(); + auto coef = ov::op::v0::Constant::create(input_type, ov::Shape{}, {1.702f}); + auto gated = std::make_shared(src0, coef); + auto res = std::make_shared(gated, src1); + + return rename_outputs_with_suffix({res}, context.get_name()); +} + +} // namespace op +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp b/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp index d220f2f5..7eea81d9 100644 --- a/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp +++ b/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp @@ -9,9 +9,10 @@ #include #include #include +#include #include -#include #include +#include namespace ov { namespace frontend { @@ -61,8 +62,7 @@ static std::pair, ov::Output> get_glu_inputs(cons OutputVector translate_glu_swiglu(const NodeContext & context) { auto [src0, src1] = get_glu_inputs(context); - auto sigmoid = std::make_shared(src0); - auto silu = std::make_shared(src0, sigmoid); + auto silu = std::make_shared(src0); auto res = std::make_shared(silu, src1); return rename_outputs_with_suffix({res}, context.get_name()); @@ -77,9 +77,7 @@ OutputVector translate_glu_swiglu_oai(const NodeContext & context) { auto gate = std::make_shared(src0, -std::numeric_limits::infinity(), limit); auto alpha_const = ov::op::v0::Constant::create(ov::element::f32, {}, {alpha}); - auto scaled_gate = std::make_shared(gate, alpha_const); - auto sigmoid = std::make_shared(scaled_gate); - auto out_glu = std::make_shared(gate, sigmoid); + auto out_glu = std::make_shared(gate, alpha_const); auto up = std::make_shared(src1, -limit, limit); auto one = ov::op::v0::Constant::create(ov::element::f32, {}, {1.0f}); @@ -89,6 +87,32 @@ OutputVector translate_glu_swiglu_oai(const NodeContext & context) { return rename_outputs_with_suffix({res}, context.get_name()); } +OutputVector translate_glu_swiglu_clamp(const NodeContext & context) { + auto [src0, src1] = get_glu_inputs(context); + + const int32_t * params = context.get_output_op_params(); + const float limit = reinterpret_cast(params)[3]; + + // Compute in f32: f16 Swish/Clamp rounding drifts past the 1e-7 test tolerance. + auto output_type = context.get_output_type(); + if (src0.get_element_type() != ov::element::f32) { + src0 = std::make_shared(src0, ov::element::f32); + } + if (src1.get_element_type() != ov::element::f32) { + src1 = std::make_shared(src1, ov::element::f32); + } + + auto gate = std::make_shared(src0, -std::numeric_limits::infinity(), limit); + auto silu = std::make_shared(gate); + auto up = std::make_shared(src1, -limit, limit); + ov::Output res = std::make_shared(silu, up); + if (res.get_element_type() != output_type) { + res = std::make_shared(res, output_type); + } + + return rename_outputs_with_suffix({res}, context.get_name()); +} + } // namespace op } // namespace ggml } // namespace frontend diff --git a/ggml/src/ggml-openvino/openvino/op/im2col.cpp b/ggml/src/ggml-openvino/openvino/op/im2col.cpp index 856e97f7..08b53f26 100644 --- a/ggml/src/ggml-openvino/openvino/op/im2col.cpp +++ b/ggml/src/ggml-openvino/openvino/op/im2col.cpp @@ -1,7 +1,6 @@ #include "../node_context.h" #include "../op_table.h" #include "../utils.h" -#include "ggml-impl.h" #include #include diff --git a/ggml/src/ggml-openvino/openvino/op/moe_compressed.hpp b/ggml/src/ggml-openvino/openvino/op/moe_compressed.hpp new file mode 100644 index 00000000..07e94c69 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/op/moe_compressed.hpp @@ -0,0 +1,90 @@ +// Copyright (C) 2018-2026 Intel Corporation +// SPDX-License-Identifier: Apache-2.0 +// +// Local mirror of OpenVINO's internal ov::op::internal::MOE and MOECompressed ops. +// +// The class bodies are provided by the linked libopenvino.so; only the declarations are +// needed here so the backend can construct the node directly (same approach as +// GatherMatmul and GatedDeltaNet). The class layout must stay in sync with +// openvino/src/core/dev_api/openvino/op/moe.hpp +// openvino/src/common/transformations/include/ov_ops/moe_compressed.hpp +// +// \note MOE op classes are under development and subject to change. + +#pragma once + +#include + +#include "openvino/core/type/element_type.hpp" +#include "openvino/op/op.hpp" + +namespace ov::op::internal { + +class OPENVINO_API MOE : public ov::op::Op { +public: + OPENVINO_OP("MOE") + + MOE() = default; + + MOE(const OutputVector & args) : Op(args) {} + + enum class Expert_type { GEMM2_BIAS_SWIGLU_CLAMP, GEMM3_SWIGLU }; + + enum class Activation_type { SWIGLU, GEGLU_TANH, GEGLU_ERF }; + + struct Config { + Expert_type expert_type{ Expert_type::GEMM2_BIAS_SWIGLU_CLAMP }; + float expert_alpha{ 0.0f }; + float expert_beta{ 1.0f }; + size_t gate_idx{ 0 }; + Activation_type activation_type{ Activation_type::SWIGLU }; + }; + + MOE(const OutputVector & args, const Config & config); + + const Config & get_config() const; + void set_config(const Config & config); + + bool visit_attributes(AttributeVisitor & visitor) override; + void validate_and_infer_types() override; + std::shared_ptr clone_with_new_inputs(const OutputVector & new_args) const override; + +private: + Config m_config; +}; + +class OPENVINO_API MOECompressed : public MOE { +public: + OPENVINO_OP("MOECompressed", "", ov::op::internal::MOE) + + MOECompressed() = default; + + struct Config : public MOE::Config { + size_t hidden_size = 0; + size_t inter_size = 0; + size_t num_expert = 0; + size_t num_shared_expert = 0; + size_t top_k = 0; + // numeric_limits::max() means per_channel compression (single group) + size_t group_size = 0; + bool has_batch_dim = false; + bool has_zp = false; + ov::element::Type out_type = ov::element::dynamic; + std::optional scale_factor; + }; + + MOECompressed(const OutputVector & args, const Config & config); + + const Config & get_config() const { return m_config; } + + void set_scale_factor(float scale_factor) { m_config.scale_factor = scale_factor; } + + bool visit_attributes(AttributeVisitor & visitor) override; + void validate_and_infer_types() override; + std::shared_ptr clone_with_new_inputs(const OutputVector & new_args) const override; + +protected: + Config m_config; +}; + +} // namespace ov::op::internal diff --git a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp index f1b28c85..a336924e 100644 --- a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp +++ b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp @@ -42,7 +42,7 @@ ov::Output slice_axis(const ov::Output & input, int64_t axis ov::Output static_shape_dims_or_shapeof(const ov::Output & input, const std::vector & dims) { - const auto partial_shape = input.get_partial_shape(); + const auto & partial_shape = input.get_partial_shape(); if (partial_shape.is_static()) { std::vector values; values.reserve(dims.size()); @@ -56,54 +56,6 @@ ov::Output static_shape_dims_or_shapeof(const ov::Output & i return get_dimensions(shape, dims); } -ov::Output translate_mul_mat_id_gather_matmul_fallback(const NodeContext & context, - ov::Output expert_weights, - ov::Output activations, - ov::Output ids) { - auto gather_axis = ov::op::v0::Constant::create(ov::element::i32, ov::Shape{}, {0}); - ov::Output selected_weights = std::make_shared(expert_weights, ids, gather_axis); - - const auto output_type = context.get_output_type(); - if (selected_weights.get_element_type() != ov::element::f32) { - selected_weights = std::make_shared(selected_weights, ov::element::f32); - } - if (activations.get_element_type() != ov::element::f32) { - activations = std::make_shared(activations, ov::element::f32); - } - - auto activations_shape = std::make_shared(activations, ov::element::i64); - auto ids_shape = std::make_shared(ids, ov::element::i64); - ov::Output acts_target_dims = std::make_shared( - ov::OutputVector{ - get_dimensions(activations_shape, {0}), - get_dimensions(ids_shape, {1}), - get_dimensions(activations_shape, {2}), - }, - 0); - ov::Output acts_broadcasted = - std::make_shared(activations, acts_target_dims, ov::op::BroadcastType::BIDIRECTIONAL); - - auto activations_expanded = std::make_shared(acts_broadcasted, const_i64({2})); - ov::Output result = - std::make_shared(activations_expanded, selected_weights, false, true); - - auto output_shape = context.get_output_shape(); - FRONT_END_OP_CONVERSION_CHECK(output_shape.rank().is_static() && output_shape.rank().get_length() == 4, - "Unexpected MUL_MAT_ID output rank"); - FRONT_END_OP_CONVERSION_CHECK(output_shape[3].is_static(), "Expected static row dimension for MUL_MAT_ID output"); - - auto batch_dim = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); - auto row_dim = ov::op::v0::Constant::create(ov::element::i64, {1}, {output_shape[3].get_length()}); - auto result_target_dims = std::make_shared( - ov::OutputVector{batch_dim, get_dimensions(ids_shape, {0, 1}), row_dim}, 0); - result = std::make_shared(result, result_target_dims, false); - - if (result.get_element_type() != output_type) { - result = std::make_shared(result, output_type); - } - return result; -} - ov::Output translate_mul_mat_id_mxfp4_packed(const NodeContext & context, ov::Output expert_weights, ov::Output activations, @@ -229,7 +181,6 @@ OutputVector translate_mul_mat_id(const NodeContext & context) { auto expert_weights_rank = expert_weights.get_partial_shape().rank(); FRONT_END_OP_CONVERSION_CHECK(expert_weights_rank.is_static(), "Expected static rank for MUL_MAT_ID expert weights"); - const bool use_gpu_fallback = ggml_openvino_get_device_name() == "GPU"; if (expert_weights_rank.get_length() == 4) { auto expert_weights_shape_3d = static_shape_dims_or_shapeof(expert_weights, {1, 2, 3}); expert_weights = std::make_shared(expert_weights, expert_weights_shape_3d, false); @@ -246,14 +197,9 @@ OutputVector translate_mul_mat_id(const NodeContext & context) { } const auto output_type = context.get_output_type(); - if (activations.get_element_type() != ov::element::f32) { - activations = std::make_shared(activations, ov::element::f32); - } - - if (use_gpu_fallback || !expert_weights.get_partial_shape().is_static() || !activations.get_partial_shape().is_static() || - !ids.get_partial_shape().is_static()) { - return rename_outputs_with_suffix({translate_mul_mat_id_gather_matmul_fallback(context, expert_weights, activations, ids)}, - context.get_name()); + const auto activations_type = ggml_openvino_get_device_name() == "GPU" ? ov::element::f16 : ov::element::f32; + if (activations.get_element_type() != activations_type) { + activations = std::make_shared(activations, activations_type); } // GatherMatmul's A input is [n_used_or_1, n_tokens, k]; activations_3d is diff --git a/ggml/src/ggml-openvino/openvino/op/mulmat.cpp b/ggml/src/ggml-openvino/openvino/op/mulmat.cpp index 41d7c54a..9d4315aa 100644 --- a/ggml/src/ggml-openvino/openvino/op/mulmat.cpp +++ b/ggml/src/ggml-openvino/openvino/op/mulmat.cpp @@ -29,19 +29,11 @@ OutputVector translate_mulmat(const NodeContext & context) { int op_case = context.get_op_case(); - ov::Output res; - ov::Output B; - ov::Output A; - if (op_case == 3) { - B = process_view_input(context, 0); - A = process_view_input(context, 1); - } else { - B = process_view_input_new(context, 0); - A = process_view_input_new(context, 1); - } + ov::Output B = process_view_input_new(context, 0); + ov::Output A = process_view_input_new(context, 1); if (A.get_element_type() != B.get_element_type()) { - B = std::make_shared(context.get_input(0), context.get_input_type(1)); + B = std::make_shared(B, context.get_input_type(1)); } auto B_shape = context.get_input_shape(0).to_shape(); @@ -84,7 +76,7 @@ OutputVector translate_mulmat(const NodeContext & context) { } bool transpose_b = true; - res = std::make_shared(A, B, false, transpose_b); + ov::Output res = std::make_shared(A, B, false, transpose_b); const auto output_type = context.get_output_type(); if (res.get_element_type() != output_type) { diff --git a/ggml/src/ggml-openvino/openvino/op/norm.cpp b/ggml/src/ggml-openvino/openvino/op/norm.cpp index c8bedb6d..8660c652 100644 --- a/ggml/src/ggml-openvino/openvino/op/norm.cpp +++ b/ggml/src/ggml-openvino/openvino/op/norm.cpp @@ -2,15 +2,10 @@ #include "../op_table.h" #include "../utils.h" +#include #include -#include #include -#include -#include -#include -#include -#include -#include +#include namespace ov { namespace frontend { @@ -21,33 +16,11 @@ OutputVector translate_norm(const NodeContext & context) { num_inputs_check(context, 1, 1); auto input_node = process_view_input_new(context, 0); - - // Step 1: Calculate mean along the last dimension - // mean = reduce_mean(input, axis=-1, keepdims=true) - auto mean = std::make_shared( - input_node, ov::op::v0::Constant::create(ov::element::i64, ov::Shape{1}, {-1}), true); - - // Step 2: Calculate (input - mean) - auto centered = std::make_shared(input_node, mean); - - // Step 3: Calculate squared differences (input - mean)^2 - auto squared = std::make_shared( - centered, ov::op::v0::Constant::create(ov::element::f32, ov::Shape{1}, {2.0f})); - - // Step 4: Calculate variance = mean((input - mean)^2) - auto variance = std::make_shared( - squared, ov::op::v0::Constant::create(ov::element::i64, ov::Shape{1}, {-1}), true); - - // Step 5: Get epsilon from op_params float eps; memcpy(&eps, context.get_output_op_params(), sizeof(float)); - // Step 6: Calculate std = sqrt(variance + eps) - auto std_dev = std::make_shared(std::make_shared( - variance, ov::op::v0::Constant::create(ov::element::f32, ov::Shape{1}, {eps}))); - - // Step 7: Normalize: output = (input - mean) / std - auto res = std::make_shared(centered, std_dev); + auto axes = ov::op::v0::Constant::create(ov::element::i64, ov::Shape{1}, {-1}); + auto res = std::make_shared(input_node, axes, true, eps, ov::op::MVNEpsMode::INSIDE_SQRT); return rename_outputs_with_suffix({res}, context.get_name()); } diff --git a/ggml/src/ggml-openvino/openvino/op/pad.cpp b/ggml/src/ggml-openvino/openvino/op/pad.cpp index 492033d1..ae3d7be1 100644 --- a/ggml/src/ggml-openvino/openvino/op/pad.cpp +++ b/ggml/src/ggml-openvino/openvino/op/pad.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include namespace ov { @@ -20,7 +21,7 @@ namespace { ov::Output translate_circular_pad(ov::Output input, const std::array & pads, const ov::Shape & input_shape) { - ov::Output result = input; + ov::Output result = std::move(input); const std::array pads_begin = {pads[6], pads[4], pads[2], pads[0]}; const std::array pads_end = {pads[7], pads[5], pads[3], pads[1]}; @@ -60,9 +61,7 @@ OutputVector translate_pad(const NodeContext & context) { auto input = process_view_input_new(context, 0); if (context.get_input_shape(0) == context.get_output_shape()) { - auto input_shape = std::make_shared(input); - auto res = std::make_shared(input, input_shape, false); - return rename_outputs_with_suffix({res}, context.get_name()); + return rename_outputs_with_suffix({input}, context.get_name()); } const int32_t * op_params = context.get_output_op_params(); diff --git a/ggml/src/ggml-openvino/openvino/op/permute.cpp b/ggml/src/ggml-openvino/openvino/op/permute.cpp index 85550bff..df4f0389 100644 --- a/ggml/src/ggml-openvino/openvino/op/permute.cpp +++ b/ggml/src/ggml-openvino/openvino/op/permute.cpp @@ -45,11 +45,22 @@ OutputVector translate_permute(const NodeContext & context) { static_cast(perm_values.size() - 1 - input_axis); } } - auto perm = ov::op::v0::Constant::create(ov::element::i64, {4}, perm_values); - if (op_case == 1 || context.is_stateful()) { + // The stateful path carries hidden-state tensors in a rank-3 layout (the + // leading batch dim is dropped, e.g. Gemma4's per-layer-embedding path). The + // perm above is rank-4; when the actual input is rank-3, drop the batch axis + // (perm[0], which is always the identity 0 here) and shift the rest down by 1 + // so the transpose order matches the input rank. + std::vector perm_used = perm_values; + const auto & src_ps = src.get_partial_shape(); + if (src_ps.rank().is_static() && src_ps.rank().get_length() == 3 && perm_values.size() == 4 && + perm_values[0] == 0) { + perm_used = {perm_values[1] - 1, perm_values[2] - 1, perm_values[3] - 1}; + } + auto perm = ov::op::v0::Constant::create(ov::element::i64, ov::Shape{perm_used.size()}, perm_used); res = std::make_shared(src, perm); } else if (op_case == 2) { + auto perm = ov::op::v0::Constant::create(ov::element::i64, {4}, perm_values); auto output_shape = context.get_output_shape().to_shape(); auto n_heads = ov::op::v0::Constant::create(ov::element::i64, {1}, {output_shape[1]}); auto head_size = ov::op::v0::Constant::create(ov::element::i64, {1}, {output_shape[3]}); @@ -68,6 +79,7 @@ OutputVector translate_permute(const NodeContext & context) { auto reshaped = std::make_shared(src, new_shape, true); res = std::make_shared(reshaped, perm); } else { + auto perm = ov::op::v0::Constant::create(ov::element::i64, {4}, perm_values); auto cache_shape = src.get_partial_shape(); auto output_shape = context.get_output_shape().to_shape(); int64_t head_size = output_shape[3]; diff --git a/ggml/src/ggml-openvino/openvino/op/pool_2d.cpp b/ggml/src/ggml-openvino/openvino/op/pool_2d.cpp new file mode 100644 index 00000000..fb633317 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/op/pool_2d.cpp @@ -0,0 +1,53 @@ +#include "../node_context.h" +#include "../op_table.h" +#include "../utils.h" + +#include +#include +#include + +namespace ov { +namespace frontend { +namespace ggml { +namespace op { + +OutputVector translate_pool_2d(const NodeContext & context) { + num_inputs_check(context, 1, 1); + const int32_t * params = context.get_output_op_params(); + + const int k0 = params[1]; + const int k1 = params[2]; + const int s0 = params[3]; + const int s1 = params[4]; + const int p0 = params[5]; + const int p1 = params[6]; + + const int op_case = context.get_op_case(); + ov::Output input = context.get_input(0); + ov::Strides strides{static_cast(s1), static_cast(s0)}; + ov::Shape pads_begin{static_cast(p1), static_cast(p0)}; + ov::Shape pads_end{static_cast(p1), static_cast(p0)}; + ov::Shape kernel{static_cast(k1), static_cast(k0)}; + ov::Output res; + + switch (op_case) { + case 1: // GGML_OP_POOL_MAX + { + res = std::make_shared(input, strides, pads_begin, pads_end, kernel); + break; + } + case 2: // GGML_OP_POOL_AVG + { + res = std::make_shared(input, strides, pads_begin, pads_end, kernel, false); + break; + } + default: + break; + } + return rename_outputs_with_suffix({res}, context.get_name()); +} + +} // namespace op +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/op/repeat.cpp b/ggml/src/ggml-openvino/openvino/op/repeat.cpp index d58b59e4..b7aeaa24 100644 --- a/ggml/src/ggml-openvino/openvino/op/repeat.cpp +++ b/ggml/src/ggml-openvino/openvino/op/repeat.cpp @@ -1,7 +1,6 @@ #include "../node_context.h" #include "../op_table.h" #include "../utils.h" -#include "ggml.h" #include #include diff --git a/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp b/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp index 9cbce7db..25c95354 100644 --- a/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp +++ b/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp @@ -25,9 +25,7 @@ OutputVector translate_rms_norm(const NodeContext & context) { auto op_case = context.get_op_case(); ov::Output input_node; - if (op_case == 1) { - input_node = process_view_input_new(context, 0); - } else if (op_case == 2) { + if (op_case == 2) { auto ssm_state_size = context.get_ssm_state_size(); // The GDN op packs [attn | new_state] along the row axis; the state occupies the last // ssm_state_size * n_seqs rows. Slice it off (scaling by the active sequence count) to keep diff --git a/ggml/src/ggml-openvino/openvino/op/roll.cpp b/ggml/src/ggml-openvino/openvino/op/roll.cpp new file mode 100644 index 00000000..e8d1b8e5 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/op/roll.cpp @@ -0,0 +1,36 @@ +#include "../node_context.h" +#include "../op_table.h" +#include "../utils.h" + +#include +#include + +namespace ov { +namespace frontend { +namespace ggml { +namespace op { + +OutputVector translate_roll(const NodeContext & context) { + num_inputs_check(context, 1, 1); + const int32_t * params = context.get_output_op_params(); + + int64_t s0 = params[0]; + int64_t s1 = params[1]; + int64_t s2 = params[2]; + int64_t s3 = params[3]; + + auto input = context.get_input(0); + + auto shift = ov::op::v0::Constant::create( + ov::element::i64, ov::Shape{4}, std::vector{s3, s2, s1, s0}); + auto axes = ov::op::v0::Constant::create( + ov::element::i64, ov::Shape{4}, std::vector{0, 1, 2, 3}); + + auto roll = std::make_shared(input, shift, axes); + return rename_outputs_with_suffix({roll}, context.get_name()); +} + +} // namespace op +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/op/rope.cpp b/ggml/src/ggml-openvino/openvino/op/rope.cpp index 8f20a0d1..a3da7d1f 100644 --- a/ggml/src/ggml-openvino/openvino/op/rope.cpp +++ b/ggml/src/ggml-openvino/openvino/op/rope.cpp @@ -11,16 +11,11 @@ #include #include #include -#include -#include #include #include -#include -#include #include #include #include -#include #include #include #include @@ -37,13 +32,14 @@ OutputVector translate_rope(const NodeContext & context) { ov::Output res; - auto data_node = context.get_input(0).get_node_shared_ptr(); + auto data_node = process_view_input_new(context, 0).get_node_shared_ptr(); auto output_shape = context.get_output_shape().to_shape(); int32_t * op_params = context.get_output_op_params(); const int mode = op_case; const int64_t head_dim = static_cast(output_shape[3]); const int64_t configured_n_dims = static_cast(op_params[1]); const int64_t n_dims = configured_n_dims == 0 ? head_dim : configured_n_dims; + const int64_t n_offs = static_cast(op_params[15]); constexpr int TYPE_NORMAL = 0; constexpr int TYPE_NEOX = 1; @@ -55,27 +51,27 @@ OutputVector translate_rope(const NodeContext & context) { cos_theta_node = context.get_input("rope_cos"); sin_theta_node = context.get_input("rope_sin"); } else { - auto inp_pos = context.get_input(1).get_node_shared_ptr(); - std::shared_ptr rope_freqs_weight; + std::string cache_key = "rope_sin_cos"; + for (int i = 0; i < 15; i++) { + cache_key += "_" + std::to_string(op_params[i]); + } if (context.get_input_size() == 3) { - rope_freqs_weight = context.get_input(2).get_node_shared_ptr(); + cache_key += "_ff_" + context.get_input_names()[2]; } - auto sin_cos = make_sin_cos(op_params, inp_pos, rope_freqs_weight, mode == TYPE_IMROPE, false); - sin_theta_node = sin_cos.first; - cos_theta_node = sin_cos.second; - } - - if (context.get_view_input_size(0) > 0) { - data_node = process_view_input_new(context, 0).get_node_shared_ptr(); - if (context.is_stateful()) { - auto data_shape = ov::op::v0::Constant::create( - ov::element::i64, {3}, std::vector{-1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); - data_node = std::make_shared(data_node, data_shape, false); + if (context.has_input(cache_key + "_cos")) { + cos_theta_node = context.get_input(cache_key + "_cos"); + sin_theta_node = context.get_input(cache_key + "_sin"); } else { - auto data_shape = ov::op::v0::Constant::create( - ov::element::i64, {4}, - std::vector{1, -1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); - data_node = std::make_shared(data_node, data_shape, false); + auto inp_pos = context.get_input(1).get_node_shared_ptr(); + std::shared_ptr rope_freqs_weight; + if (context.get_input_size() == 3) { + rope_freqs_weight = context.get_input(2).get_node_shared_ptr(); + } + auto sin_cos = make_sin_cos(op_params, inp_pos, rope_freqs_weight, mode == TYPE_IMROPE, false); + sin_theta_node = sin_cos.first; + cos_theta_node = sin_cos.second; + context.put_shared(cache_key + "_cos", cos_theta_node); + context.put_shared(cache_key + "_sin", sin_theta_node); } } @@ -84,52 +80,34 @@ OutputVector translate_rope(const NodeContext & context) { data_node = std::make_shared(data_node, ov::element::f32); } - FRONT_END_OP_CONVERSION_CHECK(n_dims > 0 && n_dims <= head_dim && (n_dims % 2 == 0), - "ROPE expects even n_dims in [1, head_dim]"); - - // TODO(openvino-gpu-rope-fusion): TEMPORARY WORKAROUND - do NOT revert until the - // OpenVINO GPU plugin is updated. - // + FRONT_END_OP_CONVERSION_CHECK(n_offs >= 0 && (n_offs % 2 == 0), + "ROPE expects non-negative even n_offs"); + FRONT_END_OP_CONVERSION_CHECK(n_dims > 0 && n_dims + n_offs <= head_dim && (n_dims % 2 == 0), + "ROPE expects even n_dims in [1, head_dim - n_offs]"); + + // RoPEFusionFlux requires rank_equals(4) on x, t_cos and t_sin. The cos/sin + // tables are already built rank-4 ([1, S, 1, head_size/2]) for both modes. In + // stateful mode the data arrives rank-3 ([S, n_heads, head_size]), so lift it + // to rank-4 ([1, S, n_heads, head_size]) here. Stateful RoPE already produced + // rank-4 output, so downstream attention is unaffected. + if (context.is_stateful()) { + auto r4_shape = ov::op::v0::Constant::create( + ov::element::i64, {4}, + std::vector{1, -1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); + data_node = std::make_shared(data_node, r4_shape, false); + } // For TYPE_NORMAL rope (both stateful and stateless) we emit the Flux-style // interleaved pattern below so the GPU plugin's RoPEFusionFlux matcher folds it - // into ov::op::internal::RoPE. The matcher requires rank-4 inputs, which is why - // the original even/odd Slice translation (kept in the `else if (mode == - // TYPE_NORMAL)` branch below for reference) does not get fused. - // - // Once the GPU plugin's RoPE fusion is extended to also recognize the original - // even/odd Slice form, this Flux rewrite should be removed and both modes should - // be restored to the captured even/odd translation. Until then, keep both paths: - // the active Flux rewrite here and the previous translation preserved below. + // into ov::op::internal::RoPE. if (mode == TYPE_NORMAL) { auto axis_last = ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}); - auto zero = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); auto step_one = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); - // Emit the Flux-style interleaved-RoPE pattern so the GPU plugin's - // RoPEFusionFlux matcher folds this subgraph into ov::op::internal::RoPE: - // x_paired = Reshape(x_rot, [1, S, n_heads, n_dims/2, 2]) - // x0, x1 = Split(x_paired, axis=-1, num_splits=2) - // x1_neg = x1 * -1 - // x_rotated = Reshape(Concat([x1_neg, x0], axis=-1), [1, S, n_heads, n_dims]) - // y_rot = x_rot * t_cos + x_rotated * t_sin - // y = Concat([y_rot, x_tail], axis=-1) if n_dims < head_dim - // Mathematically equivalent to the even/odd Slice form below. - // - // RoPEFusionFlux requires rank_equals(4) on x, t_cos and t_sin. The cos/sin - // tables are already built rank-4 ([1, S, 1, head_size/2]) for both modes. In - // stateful mode the data arrives rank-3 ([S, n_heads, head_size]), so lift it - // to rank-4 ([1, S, n_heads, head_size]) here. Stateful RoPE already produced - // rank-4 output, so downstream attention is unaffected. - if (context.is_stateful()) { - auto r4_shape = ov::op::v0::Constant::create( - ov::element::i64, {4}, - std::vector{1, -1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); - data_node = std::make_shared(data_node, r4_shape, false); - } const int64_t n_heads = static_cast(output_shape[2]); const int64_t half = n_dims / 2; - auto rot_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_dims}); - auto rot_data = std::make_shared(data_node, zero, rot_end, step_one, axis_last); + auto rot_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs}); + auto rot_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs + n_dims}); + auto rot_data = std::make_shared(data_node, rot_start, rot_end, step_one, axis_last); auto neg_one_f = ov::op::v0::Constant::create(data_node->get_element_type(), ov::Shape{}, {-1.0f}); @@ -153,7 +131,7 @@ OutputVector translate_rope(const NodeContext & context) { // Expand cos/sin from [..., n_dims/2] to [..., n_dims] by repeating each // entry twice. Use special_zero on the final Reshape so the seq dim passes // through dynamically. Final rank is 4 to satisfy the matcher's predicate. - auto expand_cos_sin = [&](Output cs) { + auto expand_cos_sin = [&](const Output& cs) { auto cs_unsq = std::make_shared( cs, ov::op::v0::Constant::create(ov::element::i64, {1}, {-1})); auto bcast_target = ov::op::v0::Constant::create( @@ -170,123 +148,80 @@ OutputVector translate_rope(const NodeContext & context) { auto y2 = std::make_shared(x_rotated, sin_full); auto rotated = std::make_shared(y1, y2); - if (n_dims < head_dim) { - auto tail_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_dims}); + ov::OutputVector concat_parts; + if (n_offs > 0) { + auto head_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); + auto head_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs}); + auto head = std::make_shared(data_node, head_start, head_end, step_one, axis_last); + concat_parts.push_back(head); + } + concat_parts.push_back(rotated); + if (n_offs + n_dims < head_dim) { + auto tail_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs + n_dims}); auto tail_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {head_dim}); auto tail = std::make_shared(data_node, tail_start, tail_end, step_one, axis_last); - res = std::make_shared(ov::OutputVector{rotated, tail}, -1); - } else { - res = rotated; + concat_parts.push_back(tail); } - } - // PRESERVED PREVIOUS TRANSLATION - Re-enable this branch (and remove the Flux branch above) once - // the GPU plugin's RoPE fusion is updated to recognize the even/odd Slice form; - // see the TODO(openvino-gpu-rope-fusion) note above. Do not delete. - // - // Original even/odd Slice form. In stateless mode it ran on rank-4 data - // ([1, S, n_heads, head_size]); in stateful mode on rank-3 data - // ([S, n_heads, head_size]). Either way it does not match RoPEFusionFlux - // (which needs rank-4 x in the interleaved layout), so the RoPE stays as - // discrete elementwise ops. - // - // } else if (mode == TYPE_NORMAL) { - // auto neg_one = ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}); - // auto zero = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); - // auto one = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); - // auto two = ov::op::v0::Constant::create(ov::element::i64, {1}, {2}); - // auto end = ov::op::v0::Constant::create(ov::element::i64, {1}, {output_shape[3]}); - // Output even_slice; - // Output odd_slice; - // // stateful data is rank 3 (unsqueeze at axis 3), stateless is rank 4 (axis 4) - // int32_t unsqueeze_dim = context.is_stateful() ? 3 : 4; - // even_slice = std::make_shared(data_node, zero, end, two, neg_one); - // odd_slice = std::make_shared(data_node, one, end, two, neg_one); - // - // Output first_half = - // std::make_shared(std::make_shared(even_slice, cos_theta_node), - // std::make_shared(odd_slice, sin_theta_node)); - // Output second_half = - // std::make_shared(std::make_shared(even_slice, sin_theta_node), - // std::make_shared(odd_slice, cos_theta_node)); - // - // first_half = std::make_shared(first_half, - // ov::op::v0::Constant::create(ov::element::i64, {1}, {unsqueeze_dim})); - // second_half = std::make_shared(second_half, - // ov::op::v0::Constant::create(ov::element::i64, {1}, {unsqueeze_dim})); - // auto stack = std::make_shared(OutputVector{first_half, second_half}, unsqueeze_dim); - // - // auto data_shape = ov::op::v0::Constant::create( - // ov::element::i64, {4}, std::vector{1, -1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); - // res = std::make_shared(stack, data_shape, false); - else if (mode == TYPE_NEOX) { - // In stateful mode the data arrives rank-3 ([S, n_heads, head_size]) while the - // cos/sin tables are rank-4 ([1, S, 1, n_dims/2]). The resulting mixed-rank - // broadcast in the Multiply below is miscomputed by the OpenVINO GPU plugin, - // corrupting the rotated Q/K. Lift the data to rank-4 ([1, S, n_heads, head_size]) - // first so the RoPE Multiplies are equal-rank, matching the TYPE_NORMAL branch. - // Stateful RoPE already produced rank-4 output, so downstream attention is unaffected. - if (context.is_stateful()) { - auto r4_shape = ov::op::v0::Constant::create( - ov::element::i64, {4}, - std::vector{1, -1, (int64_t) output_shape[2], (int64_t) output_shape[3]}); - data_node = std::make_shared(data_node, r4_shape, false); + if (concat_parts.size() == 1) { + res = rotated; + } else { + res = std::make_shared(concat_parts, -1); } - auto axis_last = ov::op::v0::Constant::create(ov::element::i64, ov::Shape{}, {-1}); - std::vector split_lengths = {n_dims / 2, n_dims / 2}; - if (n_dims < head_dim) { - split_lengths.push_back(head_dim - n_dims); + } else if (mode == TYPE_NEOX || mode == TYPE_IMROPE) { + if (mode == TYPE_IMROPE) { + auto cos_sin_shape = std::make_shared(ov::element::i64, ov::Shape{4}, + std::vector{1, -1, 1, (n_dims >> 1)}); + cos_theta_node = std::make_shared(cos_theta_node, cos_sin_shape, true); + sin_theta_node = std::make_shared(sin_theta_node, cos_sin_shape, true); } - auto data_split = std::make_shared( - data_node, axis_last, - ov::op::v0::Constant::create(ov::element::i64, {split_lengths.size()}, split_lengths)); - Output slice_data_node_0 = data_split->outputs()[0]; - Output slice_data_node_1 = data_split->outputs()[1]; - - auto first_half_node = std::make_shared( - std::make_shared(slice_data_node_0, cos_theta_node), - std::make_shared(slice_data_node_1, sin_theta_node)); - - auto second_half_node = std::make_shared( - std::make_shared(slice_data_node_0, sin_theta_node), - std::make_shared(slice_data_node_1, cos_theta_node)); + auto axis_last = ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}); + auto step_one = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); - if (n_dims < head_dim) { - Output tail = data_split->outputs()[2]; - res = std::make_shared(ov::OutputVector{first_half_node, second_half_node, tail}, -1); - } else { - res = std::make_shared(ov::OutputVector{first_half_node, second_half_node}, -1); + Output rot_data = data_node; + if (n_offs > 0 || n_offs + n_dims < head_dim) { + auto rot_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs}); + auto rot_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs + n_dims}); + rot_data = std::make_shared(data_node, rot_start, rot_end, step_one, axis_last); } - } else if (mode == TYPE_IMROPE) { - auto cos_sin_shape = std::make_shared(ov::element::i64, ov::Shape{4}, - std::vector{1, -1, 1, (n_dims >> 1)}); - auto cos_reshaped = std::make_shared(cos_theta_node, cos_sin_shape, true); - auto sin_reshaped = std::make_shared(sin_theta_node, cos_sin_shape, true); + + const int64_t half = n_dims / 2; + auto neg_one_f = ov::op::v0::Constant::create(data_node->get_element_type(), ov::Shape{}, {-1.0f}); auto split_axis = ov::op::v0::Constant::create(ov::element::i64, ov::Shape{}, {3}); - std::vector split_lengths = {n_dims / 2, n_dims / 2}; - if (n_dims < head_dim) { - split_lengths.push_back(head_dim - n_dims); - } + auto split_lengths = ov::op::v0::Constant::create(ov::element::i64, {2}, {half, half}); + auto data_split = std::make_shared(rot_data, split_axis, split_lengths); + Output x1 = data_split->outputs()[0]; + Output x2 = data_split->outputs()[1]; + + auto x2_neg = std::make_shared(x2, neg_one_f); + auto x_rotate_half = std::make_shared(ov::OutputVector{x2_neg, x1}, -1); - auto split_a = std::make_shared( - data_node, split_axis, - ov::op::v0::Constant::create(ov::element::i64, {split_lengths.size()}, split_lengths)); - auto x0 = split_a->output(0); - auto x1 = split_a->output(1); - auto mul_a = std::make_shared(x0, cos_reshaped); - auto mul_b = std::make_shared(x1, sin_reshaped); - auto sub = std::make_shared(mul_a, mul_b); + auto cos_full = std::make_shared(ov::OutputVector{cos_theta_node, cos_theta_node}, -1); + auto sin_full = std::make_shared(ov::OutputVector{sin_theta_node, sin_theta_node}, -1); - auto mul_c = std::make_shared(x0, sin_reshaped); - auto mul_d = std::make_shared(x1, cos_reshaped); - auto add = std::make_shared(mul_c, mul_d); + auto y1 = std::make_shared(rot_data, cos_full); + auto y2 = std::make_shared(x_rotate_half, sin_full); + auto rotated = std::make_shared(y1, y2); - if (n_dims < head_dim) { - auto tail = split_a->output(2); - res = std::make_shared(ov::OutputVector{sub, add, tail}, 3); + ov::OutputVector concat_parts; + if (n_offs > 0) { + auto head_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); + auto head_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs}); + auto head = std::make_shared(data_node, head_start, head_end, step_one, axis_last); + concat_parts.push_back(head); + } + concat_parts.push_back(rotated); + if (n_offs + n_dims < head_dim) { + auto tail_start = ov::op::v0::Constant::create(ov::element::i64, {1}, {n_offs + n_dims}); + auto tail_end = ov::op::v0::Constant::create(ov::element::i64, {1}, {head_dim}); + auto tail = std::make_shared(data_node, tail_start, tail_end, step_one, axis_last); + concat_parts.push_back(tail); + } + if (concat_parts.size() == 1) { + res = rotated; } else { - res = std::make_shared(ov::OutputVector{sub, add}, 3); + res = std::make_shared(concat_parts, -1); } } diff --git a/ggml/src/ggml-openvino/openvino/op/set_rows.cpp b/ggml/src/ggml-openvino/openvino/op/set_rows.cpp index 0fe8e0a8..3b606c82 100644 --- a/ggml/src/ggml-openvino/openvino/op/set_rows.cpp +++ b/ggml/src/ggml-openvino/openvino/op/set_rows.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -75,7 +76,7 @@ OutputVector translate_set_rows(const NodeContext & context) { res = std::make_shared(dst, ind_squeezed, data_reshaped, axes); } - auto dst_reshape = std::dynamic_pointer_cast(dst.get_node_shared_ptr()); + auto dst_reshape = ov::as_type_ptr(dst.get_node_shared_ptr()); if (!multidim_indices && dst_reshape) { // Fix the case of multiple sequences, reshape back to original shape [1, n_seq, ctx_per_seq, emb] // ctx_per_seq is not fixed due to llama-bench compatibility diff --git a/ggml/src/ggml-openvino/openvino/op/transpose.cpp b/ggml/src/ggml-openvino/openvino/op/transpose.cpp index 8d89ca55..0651a410 100644 --- a/ggml/src/ggml-openvino/openvino/op/transpose.cpp +++ b/ggml/src/ggml-openvino/openvino/op/transpose.cpp @@ -14,9 +14,7 @@ OutputVector translate_transpose(const NodeContext & context) { // Compute permute order from input/output shape and stride information // so it adapts to different input and output layouts. - auto input_shape = context.get_input_shape(0).to_shape(); auto input_stride = context.get_input_stride(0); - auto output_shape = context.get_output_shape().to_shape(); auto output_stride = context.get_output_stride(); // Compute permute order by matching output and input stride rankings. diff --git a/ggml/src/ggml-openvino/openvino/op/unary_silu.cpp b/ggml/src/ggml-openvino/openvino/op/unary_silu.cpp deleted file mode 100644 index 48ee0431..00000000 --- a/ggml/src/ggml-openvino/openvino/op/unary_silu.cpp +++ /dev/null @@ -1,27 +0,0 @@ -#include "../node_context.h" -#include "../op_table.h" -#include "../utils.h" - -#include -#include -#include - -namespace ov { -namespace frontend { -namespace ggml { -namespace op { - -OutputVector translate_unary_silu(const NodeContext & context) { - num_inputs_check(context, 1, 1); - - auto input = process_view_input_new(context, 0); - auto sigmoid = std::make_shared(input); - auto res = std::make_shared(input, sigmoid); - - return rename_outputs_with_suffix({res}, context.get_name()); -} - -} // namespace op -} // namespace ggml -} // namespace frontend -} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/op/unary_softplus.cpp b/ggml/src/ggml-openvino/openvino/op/unary_softplus.cpp index 756d9c33..a9e495c3 100644 --- a/ggml/src/ggml-openvino/openvino/op/unary_softplus.cpp +++ b/ggml/src/ggml-openvino/openvino/op/unary_softplus.cpp @@ -1,6 +1,7 @@ #include "../node_context.h" #include "../op_table.h" #include "../utils.h" +#include "ggml-openvino/ggml-openvino-extra.h" #include #include @@ -9,6 +10,7 @@ #include #include #include +#include namespace ov { namespace frontend { @@ -18,6 +20,10 @@ namespace op { OutputVector translate_unary_softplus(const NodeContext & context) { num_inputs_check(context, 1, 1); + if (ggml_openvino_getenv_int("GGML_OPENVINO_NATIVE_SOFTPLUS") != 0) { + return translate_1to1_match_1_input(context); + } + auto input = process_view_input_new(context, 0); const auto element_type = input.get_element_type(); auto one = ov::op::v0::Constant::create(element_type, ov::Shape{}, {1.0f}); diff --git a/ggml/src/ggml-openvino/openvino/op/view.cpp b/ggml/src/ggml-openvino/openvino/op/view.cpp index 138526cb..ca2d2dc0 100644 --- a/ggml/src/ggml-openvino/openvino/op/view.cpp +++ b/ggml/src/ggml-openvino/openvino/op/view.cpp @@ -7,7 +7,6 @@ #include #include #include -#include namespace ov { namespace frontend { @@ -17,6 +16,13 @@ namespace op { OutputVector translate_view(const NodeContext & context) { num_inputs_check(context, 1, 1); + if (context.get_op_case() == 1) { + // Static-mode identity pass-through for VIEWs over a GATED_DELTA_NET combined output or + // the conv_input CONCAT; the consuming op (CPY/RMS_NORM) does its own runtime-correct + // slicing on the full tensor (see ggml-decoder.cpp compute_op_case, GGML_OP_VIEW). + return {context.get_input(0)}; + } + if (!context.is_static()) { // On the stateless/non-static path VIEW is normally a no-op (consumers re-slice). // EXCEPTION: the MoE expert aggregation slices each expert plane out of @@ -146,7 +152,8 @@ OutputVector translate_view(const NodeContext & context) { return {input}; } - int64_t src_elems = 1, dst_elems = 1; + int64_t src_elems = 1; + int64_t dst_elems = 1; for (int64_t i = 0; i < src_shape.rank().get_length(); ++i) { if (src_shape[i].is_dynamic()) { return {input}; diff --git a/ggml/src/ggml-openvino/openvino/op_table.cpp b/ggml/src/ggml-openvino/openvino/op_table.cpp index 3c26fe83..f249a06b 100644 --- a/ggml/src/ggml-openvino/openvino/op_table.cpp +++ b/ggml/src/ggml-openvino/openvino/op_table.cpp @@ -10,8 +10,10 @@ #include #include #include +#include #include #include +#include #include namespace ov { @@ -49,16 +51,18 @@ std::unordered_map get_supported_ops() { {"GGML_OP_TRANSPOSE", op::translate_transpose }, {"GGML_UNARY_OP_GELU", op::translate_1to1_match_1_input }, {"GGML_UNARY_OP_SIGMOID", op::translate_1to1_match_1_input }, - {"GGML_UNARY_OP_SILU", op::translate_unary_silu }, + {"GGML_UNARY_OP_SILU", op::translate_1to1_match_1_input }, {"GGML_UNARY_OP_SOFTPLUS", op::translate_unary_softplus }, {"GGML_UNARY_OP_TANH", op::translate_1to1_match_1_input }, - {"GGML_UNARY_OP_SIGMOID", op::translate_1to1_match_1_input }, {"GGML_UNARY_OP_EXP", op::translate_1to1_match_1_input }, {"GGML_UNARY_OP_NEG", op::translate_1to1_match_1_input }, + {"GGML_UNARY_OP_RELU", op::translate_1to1_match_1_input }, {"GGML_OP_VIEW", op::translate_view }, {"GGML_GLU_OP_SWIGLU", op::translate_glu_swiglu }, {"GGML_GLU_OP_SWIGLU_OAI", op::translate_glu_swiglu_oai }, + {"GGML_GLU_OP_SWIGLU_CLAMP", op::translate_glu_swiglu_clamp }, {"GGML_GLU_OP_GEGLU", op::translate_glu_geglu }, + {"GGML_GLU_OP_GEGLU_QUICK", op::translate_glu_geglu_quick }, {"GGML_OP_SET_ROWS", op::translate_set_rows }, {"GGML_OP_CPY", op::translate_cpy }, {"GGML_OP_FLASH_ATTN_EXT", op::translate_flash_attn_ext }, @@ -72,6 +76,8 @@ std::unordered_map get_supported_ops() { {"GGML_OP_DIAG", op::translate_diag }, {"GGML_OP_TRI", op::translate_tri }, {"GGML_OP_SET", op::translate_set }, + {"GGML_OP_POOL_2D", op::translate_pool_2d }, + {"GGML_OP_ROLL", op::translate_roll }, // solve_tri has accuracy issues on GPU // {"GGML_OP_SOLVE_TRI", op::translate_solve_tri }, }; diff --git a/ggml/src/ggml-openvino/openvino/op_table.h b/ggml/src/ggml-openvino/openvino/op_table.h index d4b9292d..3dc98bd9 100644 --- a/ggml/src/ggml-openvino/openvino/op_table.h +++ b/ggml/src/ggml-openvino/openvino/op_table.h @@ -30,14 +30,15 @@ GGML_OP_CONVERTER(translate_sqr); GGML_OP_CONVERTER(translate_rope); GGML_OP_CONVERTER(translate_scale); GGML_OP_CONVERTER(translate_sqrt); -GGML_OP_CONVERTER(translate_unary_silu); GGML_OP_CONVERTER(translate_unary_softplus); GGML_OP_CONVERTER(translate_soft_max); GGML_OP_CONVERTER(translate_transpose); GGML_OP_CONVERTER(translate_view); GGML_OP_CONVERTER(translate_glu_swiglu); GGML_OP_CONVERTER(translate_glu_swiglu_oai); +GGML_OP_CONVERTER(translate_glu_swiglu_clamp); GGML_OP_CONVERTER(translate_glu_geglu); +GGML_OP_CONVERTER(translate_glu_geglu_quick); GGML_OP_CONVERTER(translate_set_rows); GGML_OP_CONVERTER(translate_cpy); GGML_OP_CONVERTER(translate_argsort); @@ -53,6 +54,8 @@ GGML_OP_CONVERTER(translate_set); GGML_OP_CONVERTER(translate_diag); GGML_OP_CONVERTER(translate_tri); GGML_OP_CONVERTER(translate_solve_tri); +GGML_OP_CONVERTER(translate_pool_2d); +GGML_OP_CONVERTER(translate_roll); } // namespace op diff --git a/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.cpp b/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.cpp new file mode 100644 index 00000000..c4872ac2 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.cpp @@ -0,0 +1,273 @@ +#include "fuse_moe_compressed.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "../op/gather_matmul.hpp" +#include "../op/moe_compressed.hpp" + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +namespace { + +struct dequant_inputs { + ov::Output weight; + ov::Output scale; + ov::Output zp; + bool has_zp = false; + bool ok = false; +}; + +// Peel the chain built by make_int4_weights/make_int8_weights back to its Constant inputs. +// Grouped weights keep the pre-Reshape rank-4 form [n_expert, n, k/group, group] with scale +// and zp at [n_expert, n, k/group, 1], which is the layout MOECompressed expects. Channel-wise +// weights stay rank-3 with a rank-3 scale and carry no zp. +dequant_inputs unwrap_dequant(const ov::Output & b) { + dequant_inputs res; + + auto node = b.get_node_shared_ptr(); + while (ov::is_type(node) || ov::is_type(node)) { + node = node->get_input_node_shared_ptr(0); + } + + auto mul = ov::as_type_ptr(node); + if (!mul) { + return res; + } + res.scale = mul->input_value(1); + + auto lhs = mul->get_input_node_shared_ptr(0); + if (auto sub = ov::as_type_ptr(lhs)) { + // Take the zero point down to its Constant: an integer zp is wrapped in a Convert to f16, + // and the op wants the integer form. A natively quantized expert instead carries an exact + // f16 zp (-min/scale) with no integer behind it, which the MoE kernel does not accept. + auto zp_node = sub->get_input_node_shared_ptr(1); + while (ov::is_type(zp_node)) { + zp_node = zp_node->get_input_node_shared_ptr(0); + } + res.zp = zp_node->output(0); + res.has_zp = true; + lhs = sub->get_input_node_shared_ptr(0); + } + while (ov::is_type(lhs)) { + lhs = lhs->get_input_node_shared_ptr(0); + } + if (!ov::is_type(lhs)) { + return res; + } + + res.weight = lhs->output(0); + res.ok = res.scale.get_partial_shape().is_static() && res.weight.get_partial_shape().is_static(); + return res; +} + +size_t logical_k(const ov::Shape & shape) { + return shape.size() == 4 ? shape[2] * shape[3] : shape.back(); +} + +} // namespace + +FuseMoeCompressed::FuseMoeCompressed() { + using namespace ov::pass::pattern; + + // The gate and up projections each get their own Reshape/Transpose of the hidden state and + // their own Reshape of the routing ids, so every branch needs its own sub-pattern. On GPU + // mul_mat_id also converts the activations to f16 before the op and back to f32 after it, + // so those Converts are matched as optional. + auto hidden_gate_m = any_input(); + auto a_gate_reshape_m = wrap_type({ hidden_gate_m, any_input() }); + auto a_gate_m = + wrap_type({ optional({ a_gate_reshape_m }), any_input() }); + auto hidden_up_m = any_input(); + auto a_up_m = wrap_type( + { optional({ wrap_type({ hidden_up_m, any_input() }) }), + any_input() }); + + auto gate_w_m = any_input(); + auto up_w_m = any_input(); + auto down_w_m = any_input(); + auto ids_gate_m = any_input(); + auto ids_up_m = any_input(); + auto ids_down_m = any_input(); + + auto bgm_gate_m = wrap_type({ a_gate_m, gate_w_m, ids_gate_m, any_input() }); + auto gate_u_m = optional({ wrap_type( + { wrap_type({ bgm_gate_m, any_input() }), any_input() }) }); + + auto silu_m = wrap_type({ gate_u_m }); + + auto bgm_up_m = wrap_type({ a_up_m, up_w_m, ids_up_m, any_input() }); + auto up_u_m = optional({ wrap_type( + { wrap_type({ bgm_up_m, any_input() }), any_input() }) }); + auto swiglu_m = wrap_type({ silu_m, up_u_m }); + + auto d_t_m = wrap_type( + { optional({ wrap_type({ swiglu_m, any_input() }) }), + any_input() }); + auto bgm_down_m = wrap_type({ d_t_m, down_w_m, ids_down_m, any_input() }); + auto down_u_m = optional({ wrap_type( + { wrap_type({ bgm_down_m, any_input() }), any_input() }) }); + + auto routing_m = any_input(); + auto weighted_m = wrap_type({ down_u_m, routing_m }); + auto root_m = wrap_type({ weighted_m, any_input() }); + + const auto callback = [=](Matcher & m) { + auto & pm = m.get_pattern_value_map(); + + const auto gate = unwrap_dequant(pm.at(gate_w_m)); + const auto up = unwrap_dequant(pm.at(up_w_m)); + const auto down = unwrap_dequant(pm.at(down_w_m)); + if (!gate.ok || !up.ok || !down.ok) { + return false; + } + + const auto gate_shape = gate.weight.get_shape(); + const auto up_shape = up.weight.get_shape(); + const auto down_shape = down.weight.get_shape(); + if (gate_shape != up_shape || gate_shape.size() < 3 || down_shape.size() < 3) { + return false; + } + + // MOECompressed carries one group_size and one has_zp for all three projections, so a + // model whose down-proj is quantized differently from gate/up cannot be described. This + // happens when ggml requantizes Q5_K/Q6_K experts to channel-wise int8. + if (gate.has_zp != down.has_zp || gate_shape.size() != down_shape.size()) { + return false; + } + + // The kernel only takes an integer zero point (moe_3gemm_swiglu_opt validate_impl). + if (gate.has_zp) { + static const std::set int_zp_types = { ov::element::u4, ov::element::i4, + ov::element::u8, ov::element::i8 }; + if (int_zp_types.count(gate.zp.get_element_type()) == 0 || + int_zp_types.count(down.zp.get_element_type()) == 0) { + return false; + } + } + + // Config holds a single group_size for all three projections. + const auto group_of = [](const dequant_inputs & w) { + const auto s = w.weight.get_shape(); + return s.size() == 4 ? s[3] : logical_k(s); + }; + if (group_of(gate) != group_of(up) || group_of(gate) != group_of(down)) { + return false; + } + + // all three branches must route the same hidden state through the same experts + if (pm.at(hidden_gate_m) != pm.at(hidden_up_m)) { + return false; + } + + auto ids = pm.at(ids_down_m); + const auto ids_pshape = ids.get_partial_shape(); + if (ids_pshape.rank().is_dynamic() || ids_pshape[ids_pshape.rank().get_length() - 1].is_dynamic()) { + return false; + } + const size_t top_k = ids_pshape[ids_pshape.rank().get_length() - 1].get_length(); + + // routing weights arrive as [1, n_tokens, top_k, 1]; the op wants [..., top_k] + auto routing = pm.at(routing_m); + const auto routing_pshape = routing.get_partial_shape(); + if (routing_pshape.rank().is_dynamic() || routing_pshape.rank().get_length() != 4 || + routing_pshape[3] != 1) { + return false; + } + // MOE requires routing weights and ids to have the same shape. Drop the trailing 1 of the + // routing weights and give the ids the leading batch dim, so both become [1, n_tokens, top_k]. + routing = std::make_shared( + routing, ov::op::v0::Constant::create(ov::element::i64, ov::Shape{ 1 }, { 3 })); + if (ids_pshape.rank().get_length() == 2) { + ids = std::make_shared( + ids, ov::op::v0::Constant::create(ov::element::i64, ov::Shape{ 1 }, { 0 })); + } + if (routing.get_partial_shape() != ids.get_partial_shape()) { + return false; + } + + const size_t down_k = logical_k(down_shape); + const auto down_scale_shape = down.scale.get_shape(); + const size_t down_groups = down_scale_shape.size() >= 3 ? down_scale_shape[2] : 1; + + ov::op::internal::MOECompressed::Config config; + config.expert_type = ov::op::internal::MOE::Expert_type::GEMM3_SWIGLU; + config.activation_type = ov::op::internal::MOE::Activation_type::SWIGLU; + config.expert_alpha = 0.0f; + config.expert_beta = 1.0f; + config.gate_idx = 0; + config.hidden_size = logical_k(gate_shape); + config.inter_size = gate_shape[1]; + config.num_expert = gate_shape[0]; + config.num_shared_expert = 0; + config.top_k = top_k; + config.group_size = down_groups <= 1 ? std::numeric_limits::max() : down_k / down_groups; + config.has_batch_dim = true; + config.has_zp = gate.has_zp; + // dynamic makes the output follow the hidden state, so the plugin can lower this region + // to f16 together with the rest of the graph + config.out_type = ov::element::dynamic; + + auto absent_zp = [] { + auto zp = std::make_shared(ov::element::dynamic, ov::Shape{ 0 }); + ov::pass::disable_constant_folding(zp); + return zp->output(0); + }; + + // MOE takes its output type from the hidden state. Transpose the activations before the + // f16 Convert that mul_mat_id adds on GPU, so the op stays f32 like the block it replaces + // and the plugin can lower the whole region uniformly. + const auto a_transpose = pm.at(a_gate_m).get_node_shared_ptr(); + ov::Output hidden = + std::make_shared(pm.at(a_gate_reshape_m), a_transpose->input_value(1)); + + const ov::OutputVector args = { + hidden, routing, ids, + gate.weight, gate.scale, gate.has_zp ? gate.zp : absent_zp(), + up.weight, up.scale, up.has_zp ? up.zp : absent_zp(), + down.weight, down.scale, down.has_zp ? down.zp : absent_zp(), + }; + + auto moe = std::make_shared(args, config); + + // MOE takes its output type from the hidden state, which is f16 on GPU, while the rest of + // the ggml graph works in f32. + ov::Output result = moe->output(0); + const auto root_type = m.get_match_root()->get_output_element_type(0); + if (result.get_element_type() != root_type) { + result = std::make_shared(result, root_type); + } + + result.get_node_shared_ptr()->set_friendly_name(m.get_match_root()->get_friendly_name()); + ov::copy_runtime_info(m.get_matched_nodes(), result.get_node_shared_ptr()); + ov::replace_node(m.get_match_root(), result.get_node_shared_ptr()); + register_new_node(moe); + return true; + }; + + register_matcher(std::make_shared(root_m, "ov::frontend::ggml::pass::FuseMoeCompressed"), callback); +} + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.h b/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.h new file mode 100644 index 00000000..5500bed6 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/fuse_moe_compressed.h @@ -0,0 +1,19 @@ +#include "openvino/pass/matcher_pass.hpp" + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +// Folds the MoE expert block emitted for MUL_MAT_ID (3 GatherMatmul + SwiGLU + routing +// weighting + expert reduction) into a single ov::op::internal::MOECompressed. +class FuseMoeCompressed : public ov::pass::MatcherPass { +public: + OPENVINO_MATCHER_PASS_RTTI("ov::frontend::ggml::pass::FuseMoeCompressed") + FuseMoeCompressed(); +}; + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.cpp b/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.cpp new file mode 100644 index 00000000..21801c0f --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.cpp @@ -0,0 +1,212 @@ +#include "fuse_to_conv.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace opp = ov::pass::pattern; + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +// This pass fuses an IM2COL + MatMul convolution into OpenVINO's Convolution op for performance gains. +// Reference the im2col.cpp translator for reference on the pattern being matched. + +FuseToConv::FuseToConv() { + const auto m_wei = opp::any_input(); + const auto m_act = opp::any_input(); + const auto m_matmul = opp::wrap_type({m_wei, m_act}); + + const auto callback = [=](ov::pass::pattern::Matcher & m) { + const auto & pm = m.get_pattern_value_map(); + + auto matmul_node = ov::as_type_ptr(pm.at(m_matmul).get_node_shared_ptr()); + if (!matmul_node || matmul_node->get_transpose_a() || !matmul_node->get_transpose_b()) { + return false; + } + + auto trace = matmul_node->input_value(1); + + // Optional Convert + if (auto n = ov::as_type_ptr(trace.get_node_shared_ptr())) { + trace = n->input_value(0); + } + + for (int i = 0; i < 2; ++i) { + auto n = ov::as_type_ptr(trace.get_node_shared_ptr()); + if (!n) { + return false; + } + trace = n->input_value(0); + } + + if (auto n = ov::as_type_ptr(trace.get_node_shared_ptr())) { + trace = n->input_value(0); + } else { + return false; + } + + if (auto n = ov::as_type_ptr(trace.get_node_shared_ptr())) { + trace = n->input_value(0); + } else { + return false; + } + + if (auto n = ov::as_type_ptr(trace.get_node_shared_ptr())) { + trace = n->input_value(0); + } else { + return false; + } + + auto eip = ov::as_type_ptr(trace.get_node_shared_ptr()); + if (!eip) { + return false; + } + const auto eip_strides = eip->get_strides(); // {stride_h, stride_w} + const auto eip_rates = eip->get_rates(); // {dil_h, dil_w} + + auto pad = ov::as_type_ptr(eip->input_value(0).get_node_shared_ptr()); + if (!pad) { + return false; + } + auto pads_begin_const = + ov::as_type_ptr(pad->input_value(1).get_node_shared_ptr()); + + const auto pads_begin_vals = pads_begin_const->cast_vector(); // {0, 0, pad_h, pad_w} + const std::ptrdiff_t pad_h = static_cast(pads_begin_vals[2]); + const std::ptrdiff_t pad_w = static_cast(pads_begin_vals[3]); + + auto image_input = pad->input_value(0); // [N, IC, 1, IW] NCHW + + auto w_trace = matmul_node->input_value(0); + if (auto n = ov::as_type_ptr(w_trace.get_node_shared_ptr())) { + w_trace = n->input_value(0); + } + for (int i = 0; i < 2; ++i) { + auto n = ov::as_type_ptr(w_trace.get_node_shared_ptr()); + if (!n) { + break; + } + w_trace = n->input_value(0); + } + + auto weight_const = ov::as_type_ptr(w_trace.get_node_shared_ptr()); + if (!weight_const) { + return false; + } + + // Reshape weight to [OC, IC, 1, KW] (OIHW). + const auto w_shape = weight_const->get_shape(); + ov::Shape conv_w_shape; + if (w_shape.size() == 3) { + conv_w_shape = {w_shape[0], w_shape[1], 1, w_shape[2]}; + } else if (w_shape.size() == 4) { + conv_w_shape = {w_shape[1], w_shape[2], 1, w_shape[3]}; + } else { + return false; + } + + auto weight_reshaped = register_new_node(weight_const->get_element_type(), conv_w_shape, + weight_const->get_data_ptr()); + + ov::Output weight_input = weight_reshaped; + if (weight_reshaped->get_element_type() != image_input.get_element_type()) { + weight_input = register_new_node(weight_reshaped, image_input.get_element_type()); + } + + auto conv = register_new_node( + image_input, weight_input, + ov::Strides{static_cast(eip_strides[0]), static_cast(eip_strides[1])}, + ov::CoordinateDiff{pad_h, pad_w}, ov::CoordinateDiff{pad_h, pad_w}, + ov::Strides{static_cast(eip_rates[0]), static_cast(eip_rates[1])}, + ov::op::PadType::EXPLICIT); + + constexpr auto target_type = ov::element::f32; + ov::Output conv_out = conv; + if (conv_out.get_element_type() != target_type) { + conv_out = register_new_node(conv_out, target_type); + } + + std::shared_ptr add_node; + ov::Output bias_input; + for (const auto & consumer_in : matmul_node->output(0).get_target_inputs()) { + auto cast = ov::as_type_ptr(consumer_in.get_node()->shared_from_this()); + if (!cast) { + continue; + } + for (const auto & add_in : cast->output(0).get_target_inputs()) { + auto add = ov::as_type_ptr(add_in.get_node()->shared_from_this()); + if (!add) { + continue; + } + for (size_t i = 0; i < 2; ++i) { + if (ov::as_type_ptr(add->input_value(i).get_node_shared_ptr())) { + bias_input = add->input_value(i); + add_node = add; + break; + } + } + if (add_node) { + break; + } + } + if (add_node) { + break; + } + } + + ov::Output final_out; + std::shared_ptr target_node; + + if (add_node) { + // Reshape bias [OC, 1] → [1, OC, 1, 1] for NCHW broadcasting. + ov::Output bias = bias_input; + if (bias.get_element_type() != target_type) { + bias = register_new_node(bias, target_type); + } + const auto oc = static_cast(conv_w_shape[0]); + auto bias_shape = register_new_node(ov::element::i64, ov::Shape{4}, + std::vector{1, oc, 1, 1}); + bias = register_new_node(bias, bias_shape, false); + final_out = register_new_node(conv_out, bias); + target_node = add_node; + } else { + final_out = conv_out; + target_node = matmul_node; + } + + // Reshape final output back to the target node's original shape if needed. + auto orig_shape = target_node->get_output_partial_shape(0); + if (orig_shape.is_static() && final_out.get_partial_shape() != orig_shape) { + auto shape_const = register_new_node(ov::element::i64, ov::Shape{orig_shape.size()}, + orig_shape.to_shape()); + final_out = register_new_node(final_out, shape_const, false); + } + + final_out.get_node_shared_ptr()->set_friendly_name(target_node->get_friendly_name()); + ov::copy_runtime_info(m.get_matched_nodes(), final_out.get_node_shared_ptr()); + ov::replace_node(target_node, final_out.get_node_shared_ptr()); + + return true; + }; + + register_matcher(std::make_shared(m_matmul, "ov::frontend::ggml::pass::FuseToConv"), callback); +} + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.h b/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.h new file mode 100644 index 00000000..feac14b1 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/fuse_to_conv.h @@ -0,0 +1,17 @@ +#include "openvino/pass/matcher_pass.hpp" + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +class FuseToConv : public ov::pass::MatcherPass { +public: + OPENVINO_MATCHER_PASS_RTTI("ov::frontend::ggml::pass::FuseToConv") + FuseToConv(); +}; + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp new file mode 100644 index 00000000..04de2d00 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp @@ -0,0 +1,114 @@ +#include "kv_state_seq_axis.h" + +#include +#include +#include +#include +#include +#include +#include +#include + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +namespace { + +const std::vector & seq_axis_perm() { + // [1, seq, n_heads_kv, head_size] <-> [1, n_heads_kv, seq, head_size] + static const std::vector perm{0, 2, 1, 3}; + return perm; +} + +// True when the state still has the frontend's stateful KV layout, so the sequence axis +// can be moved: rank 4, batch and both head dims static, and seq the only dynamic dim, +// at dim 1. Any KV head count is fine. With a single head the rewrite is pure metadata +// ([1, seq, 1, head] and [1, 1, seq, head] are the same memory); with several heads it +// also drops the reader-side transpose of the whole accumulated state, which is where +// most of the gain comes from at depth. +bool can_move_seq_axis(const ov::PartialShape & shape) { + return shape.rank().is_static() && shape.rank().get_length() == 4 && shape[0].is_static() && + shape[1].is_dynamic() && shape[2].is_static() && shape[3].is_static(); +} + +std::shared_ptr match_kv_append(const std::shared_ptr & assign) { + auto concat = ov::as_type_ptr(assign->get_input_node_shared_ptr(0)); + if (!concat || concat->get_input_size() != 2 || concat->get_axis() != 1) { + return nullptr; + } + auto read_value = ov::as_type_ptr(concat->get_input_node_shared_ptr(0)); + if (!read_value || read_value->get_variable() != assign->get_variable()) { + return nullptr; + } + if (!can_move_seq_axis(read_value->get_output_partial_shape(0))) { + return nullptr; + } + return concat; +} + +} // namespace + +bool KVStateSeqAxis::run_on_model(const std::shared_ptr & model) { + std::vector> assigns; + for (const auto & op : model->get_ops()) { + if (auto assign = ov::as_type_ptr(op)) { + assigns.push_back(assign); + } + } + + bool changed = false; + for (const auto & assign : assigns) { + auto concat = match_kv_append(assign); + if (!concat) { + continue; + } + auto read_value = ov::as_type_ptr(concat->get_input_node_shared_ptr(0)); + + auto variable = read_value->get_variable(); + auto info = variable->get_info(); + const auto & shape = info.data_shape; + info.data_shape = ov::PartialShape{shape[0], shape[2], shape[1], shape[3]}; + variable->update(info); + read_value->validate_and_infer_types(); + + auto readers = concat->output(0).get_target_inputs(); + + auto new_rows = concat->input_value(1); + auto perm_in = ov::op::v0::Constant::create(ov::element::i64, {4}, seq_axis_perm()); + concat->set_argument(1, std::make_shared(new_rows, perm_in)); + concat->set_axis(2); + concat->validate_and_infer_types(); + + // Readers still expect seq at dim 1. A reader that is itself the inverse + // Transpose wanted seq at dim 2 all along, so drop it; give anything else the + // inverse Transpose so its input is unchanged. + for (const auto & reader : readers) { + auto * node = reader.get_node(); + if (ov::is_type(node)) { + continue; + } + bool dropped = false; + if (auto * transpose = ov::as_type(node)) { + auto order = ov::as_type_ptr(transpose->get_input_node_shared_ptr(1)); + if (order && order->cast_vector() == seq_axis_perm()) { + ov::replace_output_update_name(transpose->output(0), concat->output(0)); + dropped = true; + } + } + if (!dropped) { + auto perm_out = ov::op::v0::Constant::create(ov::element::i64, {4}, seq_axis_perm()); + reader.replace_source_output(std::make_shared(concat->output(0), perm_out)); + } + } + changed = true; + } + + return changed; +} + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.h b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.h new file mode 100644 index 00000000..579022c4 --- /dev/null +++ b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.h @@ -0,0 +1,24 @@ +#include "openvino/pass/pass.hpp" + +namespace ov { +namespace frontend { +namespace ggml { +namespace pass { + +// Moves the sequence axis of the stateful KV cache from dim 1 to dim 2, i.e. from +// [1, seq, n_heads_kv, head_size] to [1, n_heads_kv, seq, head_size], and updates the +// Concat that appends to it. Two wins: the GPU plugin only appends new tokens in place +// when the growing axis is a spatial axis, and the reader no longer has to transpose the +// whole accumulated state every token (that cost grows with context length, so it is the +// larger win at depth for a model with several KV heads). Only rewrites states that still +// match the frontend layout, so it no-ops if that layout ever changes. +class KVStateSeqAxis : public ov::pass::ModelPass { +public: + OPENVINO_MODEL_PASS_RTTI("ov::frontend::ggml::pass::KVStateSeqAxis") + bool run_on_model(const std::shared_ptr & model) override; +}; + +} // namespace pass +} // namespace ggml +} // namespace frontend +} // namespace ov diff --git a/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.cpp b/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.cpp index 20a3a374..09c213f3 100644 --- a/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.cpp +++ b/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.cpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -26,7 +27,7 @@ SqueezeMatmul::SqueezeMatmul() { const auto callback = [=](ov::pass::pattern::Matcher & m) { const auto & pattern_map = m.get_pattern_value_map(); auto matmul_node = - std::dynamic_pointer_cast(pattern_map.at(m_matmul).get_node_shared_ptr()); + ov::as_type_ptr(pattern_map.at(m_matmul).get_node_shared_ptr()); auto act = pattern_map.at(m_act); auto wei = pattern_map.at(m_wei); auto act_shape = act.get_partial_shape(); diff --git a/ggml/src/ggml-openvino/openvino/translate_session.cpp b/ggml/src/ggml-openvino/openvino/translate_session.cpp index 35598aba..e56a4e41 100644 --- a/ggml/src/ggml-openvino/openvino/translate_session.cpp +++ b/ggml/src/ggml-openvino/openvino/translate_session.cpp @@ -5,6 +5,9 @@ #include "ggml-openvino/openvino/node_context.h" #include "ggml-openvino/openvino/utils.h" #include "input_model.h" +#include "pass/fuse_moe_compressed.h" +#include "pass/fuse_to_conv.h" +#include "pass/kv_state_seq_axis.h" #include "pass/mark_decompression_convert_constant_folding.h" #include "pass/mark_dequantization_subgraph.h" #include "pass/squeeze_matmul.h" @@ -18,28 +21,36 @@ #include #include #include +#include #include #include #include #include +#include #include #include #include #include #include +#include +#include +#include #include #include #include #include #include +#include #include #include #include #include +#include #include #include #include #include +#include #include namespace ov { @@ -109,7 +120,8 @@ ov::pass::MakeStateful::ParamResPairs get_kv_param_res_pairs( void add_sliced_mask_stateful(TensorMap & tensor_map) { auto create_sliced_mask = [&](const std::string & mask_name, const std::string & sliced_name) { if ((tensor_map.find(mask_name) != tensor_map.end()) && - (tensor_map.find("token_len_per_seq") != tensor_map.end())) { + (tensor_map.find("token_len_per_seq") != tensor_map.end()) && + (tensor_map.find("inp_pos") != tensor_map.end())) { auto token_len_per_seq = tensor_map.at("token_len_per_seq").get_node_shared_ptr(); auto mask = tensor_map.at(mask_name).get_node_shared_ptr(); std::shared_ptr mask_sliced = mask; @@ -137,9 +149,68 @@ void add_sliced_mask_stateful(TensorMap & tensor_map) { }; create_sliced_mask("self_kq_mask", "KQ_mask_sliced"); + create_sliced_mask("KQ_mask", "KQ_mask_sliced"); create_sliced_mask("self_kq_mask_swa", "KQ_mask_swa_sliced"); } +// Rebuild the sliding-window mask from absolute positions. +// ggml caps self_kq_mask_swa at the size of its own SWA cache, but the stateful KV state is +// Concat-appended and grows without bound, so past that cap the two disagree on length and the +// mask add fails. A pure-Concat state is ordered by position, so positions can rebuild the mask. +// swa_window holds the real n_swa, read back from the ggml mask in ggml-decoder.cpp. +// No-op when the graph has no SWA mask, or when the window could not be read back. +void add_position_mask_stateful_swa(TensorMap & tensor_map) { + if (tensor_map.find("self_kq_mask_swa") == tensor_map.end() || tensor_map.find("inp_pos") == tensor_map.end() || + tensor_map.find("swa_window") == tensor_map.end()) { + return; + } + + auto inp_pos = tensor_map.at("inp_pos").get_node_shared_ptr(); + + auto zero_i64 = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); + auto one_i64 = ov::op::v0::Constant::create(ov::element::i64, {1}, {1}); + auto three = ov::op::v0::Constant::create(ov::element::i64, {1}, {3}); + auto neg_one = ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}); + + auto query_pos = std::make_shared(inp_pos, ov::element::i64); + auto query_pos_1d = std::make_shared( + query_pos, ov::op::v0::Constant::create(ov::element::i64, {1}, {-1}), false); + + auto last_pos = std::make_shared(inp_pos, neg_one, three); + auto last_pos_1d = std::make_shared(last_pos, one_i64, false); + auto last_pos_cvt = std::make_shared(last_pos_1d, ov::element::i64); + auto total_len = std::make_shared(last_pos_cvt, one_i64); + auto total_len_scalar = std::make_shared(total_len); + + auto cached_pos = std::make_shared( + ov::op::v0::Constant::create(ov::element::i64, {}, {0}), total_len_scalar, + ov::op::v0::Constant::create(ov::element::i64, {}, {1}), ov::element::i64); + + auto query_col = std::make_shared( + query_pos_1d, ov::op::v0::Constant::create(ov::element::i64, {2}, {-1, 1}), false); + auto cached_row = std::make_shared( + cached_pos, ov::op::v0::Constant::create(ov::element::i64, {2}, {1, -1}), false); + auto diff = std::make_shared(query_col, cached_row); + + auto swa_window = tensor_map.at("swa_window").get_node_shared_ptr(); + auto window = std::make_shared(swa_window, ov::element::i64); + auto causal_ok = std::make_shared(diff, zero_i64); + auto window_ok = std::make_shared(diff, window); + auto keep = std::make_shared(causal_ok, window_ok); + + auto zero_f = ov::op::v0::Constant::create(ov::element::f32, {}, {0.0f}); + auto neg_inf_f = ov::op::v0::Constant::create(ov::element::f32, {}, {-std::numeric_limits::infinity()}); + std::shared_ptr mask = std::make_shared(keep, zero_f, neg_inf_f); + + auto batch_axis = ov::op::v0::Constant::create(ov::element::i64, {1}, {0}); + mask = std::make_shared(mask, batch_axis); + mask = std::make_shared(mask, batch_axis); + mask = std::make_shared(mask, ov::element::f16); + mask->set_friendly_name("KQ_mask_swa_sliced"); + + tensor_map["KQ_mask_swa_sliced"] = mask->output(0); +} + void add_rope_sin_cos(TensorMap & tensor_map, GgmlDecoder & ggml_model_decoder) { // When ROPE ops in the graph have divergent op_params (e.g. gemma4's mixed // SWA/non-SWA layers with different n_dims or freq_base), a shared sin/cos @@ -172,6 +243,7 @@ void add_rope_sin_cos(TensorMap & tensor_map, GgmlDecoder & ggml_model_decoder) void preprocess(TensorMap & tensor_map, GgmlDecoder & ggml_model_decoder) { if (ggml_model_decoder.is_stateful()) { add_sliced_mask_stateful(tensor_map); + add_position_mask_stateful_swa(tensor_map); } // This optimization is error-prone // add_rope_sin_cos(tensor_map, ggml_model_decoder); @@ -201,7 +273,7 @@ std::shared_ptr TranslateSession::translate_graph(const frontend::InputMo auto tensor_map = std::make_shared(); std::shared_ptr resulting_model; - const auto & ggml_model = std::dynamic_pointer_cast(input_model); + const auto & ggml_model = ov::as_type_ptr(input_model); std::shared_ptr ggml_model_decoder = ggml_model->get_model_decoder(); for (const auto & it : ggml_model_decoder->get_model_inputs()) { @@ -213,7 +285,7 @@ std::shared_ptr TranslateSession::translate_graph(const frontend::InputMo for (const auto & it : ggml_model_decoder->get_model_extra_inputs()) { auto input_node = create_extra_input(it.first, it.second); if (it.second.is_parameter) { - params.push_back(std::dynamic_pointer_cast(input_node)); + params.push_back(ov::as_type_ptr(input_node)); } (*tensor_map)[it.first] = input_node; } @@ -272,7 +344,7 @@ std::shared_ptr TranslateSession::translate_graph(const frontend::InputMo } }; - auto node_visitor = [&](std::shared_ptr decoder, int node_idx) { + auto node_visitor = [&](const std::shared_ptr & decoder, int node_idx) { auto converted_outputs = translate_node(decoder, node_idx); if (converted_outputs.empty()) { return; @@ -384,7 +456,7 @@ std::shared_ptr TranslateSession::translate_graph(const frontend::InputMo } std::shared_ptr TranslateSession::apply_transformations(std::shared_ptr model) { - auto ggml_model_decoder = std::dynamic_pointer_cast(m_input_model)->get_model_decoder(); + auto ggml_model_decoder = ov::as_type_ptr(m_input_model)->get_model_decoder(); { ov::pass::Manager manager; manager.set_per_pass_validation(true); @@ -395,11 +467,22 @@ std::shared_ptr TranslateSession::apply_transformations(std::shared_ptr( std::vector{ov::element::u8, ov::element::i8, ov::element::u4, ov::element::i4}); + manager.register_pass(); + + // MOECompressed has no CPU plugin implementation, so keep the GatherMatmul path + // everywhere else. Opt-in while the fused path is being brought up. + if (ggml_openvino_get_device_name() == "GPU" && getenv("GGML_OPENVINO_MOE_OP")) { + manager.register_pass(); + } if (ggml_model_decoder->is_stateful()) { const auto kv_param_res_names = ggml_model_decoder->get_kv_param_res_names(); const auto kv_param_res_pairs = get_kv_param_res_pairs(model, kv_param_res_names); manager.register_pass(kv_param_res_pairs); + // Must run after MakeStateful, which is what creates the ReadValue/Assign pairs. + if (!ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT")) { + manager.register_pass(); + } } if (ggml_model_decoder->is_static()) { diff --git a/ggml/src/ggml-openvino/openvino/utils.cpp b/ggml/src/ggml-openvino/openvino/utils.cpp index 504d74b7..98a85e63 100644 --- a/ggml/src/ggml-openvino/openvino/utils.cpp +++ b/ggml/src/ggml-openvino/openvino/utils.cpp @@ -1,7 +1,5 @@ #include "utils.h" -#include "ggml-impl.h" - #include #include #include @@ -28,13 +26,6 @@ namespace ov { namespace frontend { namespace ggml { -std::string getCurrentTime() { - std::time_t now = std::time(nullptr); - char buf[100]; - std::strftime(buf, sizeof(buf), "%Y-%m-%d %H:%M:%S", std::localtime(&now)); - return buf; -} - void num_inputs_check(const NodeContext & context, size_t min_inputs, size_t max_inputs) { auto input_size = context.get_input_size(); FRONT_END_OP_CONVERSION_CHECK(input_size >= min_inputs, "Got less inputs than expected"); @@ -72,6 +63,7 @@ OutputVector rename_outputs_with_suffix(const OutputVector & outputs, const std: name += "_"; name += suffix; node->set_friendly_name(name); + // Uncomment to dump every node's inferred shape (used to hunt down dynamic dims on NPU). // std::cout << name << " " << output.get_partial_shape() << std::endl; } return outputs; @@ -81,7 +73,7 @@ namespace { ov::Output rope_yarn_ramp_mix(int n_dims, const float corr_dims[2], float ext_factor) { int half_n_dims = n_dims / 2; std::vector dim_ids_vec(half_n_dims); - std::iota(dim_ids_vec.begin(), dim_ids_vec.end(), 0); + std::iota(dim_ids_vec.begin(), dim_ids_vec.end(), 0.0f); auto dim_ids = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, (size_t) half_n_dims}, dim_ids_vec); auto corr_low = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, 1}, {corr_dims[0]}); auto corr_high = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, 1}, {corr_dims[1]}); @@ -550,6 +542,7 @@ ov::Output process_view_input_new(const NodeContext & context, int inp if (tail_begin >= 0 && tail_end <= tail_src_elems) { std::vector flat_shape; + flat_shape.reserve(slice_dim); for (int i = 0; i < slice_dim; ++i) { flat_shape.push_back(static_cast(view_src_ggml_shape[i])); } diff --git a/ggml/src/ggml-openvino/openvino/utils.h b/ggml/src/ggml-openvino/openvino/utils.h index 5d4c3538..d9858f92 100644 --- a/ggml/src/ggml-openvino/openvino/utils.h +++ b/ggml/src/ggml-openvino/openvino/utils.h @@ -14,8 +14,6 @@ namespace ggml { std::string getCurrentTime(); -void dump_ov_model(std::shared_ptr model); - void num_inputs_check(const NodeContext & context, size_t min_inputs, size_t max_inputs); int non_cont_dim(std::vector ne, std::vector nb); diff --git a/ggml/src/ggml-openvino/utils.cpp b/ggml/src/ggml-openvino/utils.cpp index 4df8381d..b1ee792f 100644 --- a/ggml/src/ggml-openvino/utils.cpp +++ b/ggml/src/ggml-openvino/utils.cpp @@ -2,6 +2,7 @@ #include "ggml-impl.h" #include "ggml-openvino-extra.h" +#include "ggml-openvino.h" #include "ggml-openvino/ggml-decoder.h" #include "ggml.h" #include "model-cache.h" @@ -16,6 +17,7 @@ #include #include #include +#include #include #include #include @@ -36,36 +38,7 @@ #include #include -// Suppress deprecation warning for ov::Tensor::data() -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wdeprecated-declarations" - -enum ggml_status ov_graph_compute(ggml_cgraph * cgraph, ggml_backend_t backend) { - ggml_backend_openvino_context * ctx = (ggml_backend_openvino_context *) backend->context; - try { - if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_CGRAPH")) { - std::string filename = "cgraph_ov.txt"; - GgmlOvDecoder::dump_cgraph(cgraph, filename); - } - - const auto is_static = ggml_openvino_is_npu(); - - GGML_ASSERT(ctx->runtime_context != nullptr); - std::shared_ptr r_ctx = std::static_pointer_cast(ctx->runtime_context); - - return is_static ? ov_graph_compute_static(cgraph, r_ctx) : ov_graph_compute_dynamic(cgraph, r_ctx); - } catch (const ov::Exception & e) { - GGML_LOG_ERROR("GGML OpenVINO backend ov::Exception: %s\n", e.what()); - return GGML_STATUS_FAILED; - } catch (const std::exception & e) { - GGML_LOG_ERROR("GGML OpenVINO backend std::exception: %s\n", e.what()); - return GGML_STATUS_FAILED; - } catch (...) { - GGML_LOG_ERROR("GGML OpenVINO backend unknown exception\n"); - return GGML_STATUS_FAILED; - } -} - +namespace { // For a KV cache input, return an ov::Tensor sized to n_kv (== attention_size // for that layer) instead of the fully-allocated ctx_per_seq. Pre-conditions: // * non-static (CPU/GPU) backend, single sequence, seq_active_start == 0 @@ -76,9 +49,9 @@ enum ggml_status ov_graph_compute(ggml_cgraph * cgraph, ggml_backend_t backend) // n_kv rows no longer contain the live prefix // On any unmet pre-condition returns std::nullopt; the caller falls back to // the full-size tensor. -static std::optional try_make_kv_sliced_tensor(std::shared_ptr ggml_decoder, - const std::string & name, - const ggml_tensor * ggml_tensor) { +std::optional try_make_kv_sliced_tensor(const std::shared_ptr & ggml_decoder, + const std::string & name, + const ggml_tensor * ggml_tensor) { static const bool kv_slice_disabled = ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_SLICE"); if (kv_slice_disabled) { return std::nullopt; @@ -116,7 +89,7 @@ static std::optional try_make_kv_sliced_tensor(std::shared_ptrget_shape(ggml_tensor); + ov::Shape full_shape = GgmlOvDecoder::get_shape(ggml_tensor); if (full_shape.size() != 4 || full_shape[0] != 1 || full_shape[1] != 1 || static_cast(full_shape[2]) != ctx_per_seq) { return std::nullopt; @@ -132,16 +105,16 @@ static std::optional try_make_kv_sliced_tensor(std::shared_ptrget_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); // } - return ov::Tensor(ggml_decoder->get_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); + return ov::Tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); } -static uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, bool stateful) { +uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, bool stateful) { const char * manual_gqa_env = ggml_openvino_getenv_str("GGML_OPENVINO_MANUAL_GQA_ATTN"); const bool manual_gqa_enabled = manual_gqa_env != nullptr ? ggml_openvino_getenv_int("GGML_OPENVINO_MANUAL_GQA_ATTN") > 0 : device == "GPU"; - uint64_t extra_cfg = 0; + uint64_t extra_cfg = 1; // Graph-ordinal port names (invalidate older disk-cache blobs). extra_cfg = extra_cfg * 131 + (stateful ? 1u : 0u); extra_cfg = extra_cfg * 131 + (ggml_openvino_reduce_compile_mem_enabled() ? 1u : 0u); extra_cfg = extra_cfg * 131 + (ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_SLICE") ? 1u : 0u); @@ -149,8 +122,95 @@ static uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, return extra_cfg; } -ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, - std::shared_ptr infer_request, +std::map> get_weight_names(ggml_cgraph * cgraph) { + std::map> names; + for (const auto & name : GgmlOvDecoder::collect_weight_names(cgraph)) { + names[name] = nullptr; + } + return names; +} + +// A conservative, exact in-process key, evaluated only on a context-local cache +// miss. Include topology, layouts, op parameters, constant extra inputs and weight +// allocation identities. Never use a sampled weight hash or a graph name alone: +// different models can have identical topology. OV buffer IDs survive address reuse. +std::string compiled_graph_key(const ggml_cgraph * graph, + const GgmlOvDecoder & decoder, + const std::string & device, + int prefill_chunk_size = 0) { + std::string key; + auto append = [&key](const auto & value) { + key.append(reinterpret_cast(&value), sizeof(value)); + }; + auto append_string = [&](const std::string & value) { + append(value.size()); + key.append(value); + }; + append_string(device); + append(decoder.is_static()); + append(decoder.is_stateful()); + append(prefill_chunk_size); + bool has_weight_buffer_id = false; + std::unordered_map ids; + std::function visit = [&](const ggml_tensor * tensor) { + if (!tensor) { + append(size_t(0)); + return; + } + auto inserted = ids.emplace(tensor, ids.size() + 1); + append(inserted.first->second); + if (!inserted.second) { + return; + } + append_string(tensor->name); + append(tensor->type); + append(tensor->op); + append(tensor->flags); + append(tensor->ne); + append(tensor->nb); + append(tensor->op_params); + append(tensor->view_offs); + const auto * base = tensor->view_src ? tensor->view_src : tensor; + const bool weight = base->buffer && base->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS; + append(weight); + if (weight) { + const size_t buffer_id = ggml_backend_openvino_buffer_get_ctx_id(base->buffer); + has_weight_buffer_id |= buffer_id != 0; + append(buffer_id); + append(tensor->data); + } + visit(tensor->view_src); + for (const auto * src : tensor->src) { + visit(src); + } + }; + append(graph->n_nodes); + for (int i = 0; i < graph->n_nodes; ++i) { + visit(graph->nodes[i]); + } + append(graph->n_leafs); + for (int i = 0; i < graph->n_leafs; ++i) { + visit(graph->leafs[i]); + } + for (const auto & input : decoder.get_model_extra_inputs()) { + append_string(input.first); + append_string(input.second.type.get_type_name()); + append(input.second.shape.size()); + for (auto dim : input.second.shape) { + append(dim); + } + append(input.second.is_parameter); + if (!input.second.is_parameter) { + append(input.second.value); + } + } + // Without an allocation generation, pointer reuse could select stale weights. + // Such graphs still get private requests; they simply do not share compilation. + return has_weight_buffer_id ? key : std::string{}; +} + +ov::Tensor create_ov_output_tensor(const std::shared_ptr & ggml_decoder, + const std::shared_ptr & infer_request, int output_index, const ggml_tensor * ggml_tensor) { if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, std::string(ggml_tensor->name), ggml_tensor)) { @@ -166,19 +226,409 @@ ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, // } // } - auto output_type = ggml_decoder->get_ov_type(ggml_tensor); + auto output_type = GgmlOvDecoder::get_ov_type(ggml_tensor); ov::Shape output_shape; + void * output_data = ggml_tensor->data; if (ggml_decoder->is_static()) { output_shape = infer_request->get_output_tensor(output_index).get_shape(); } else { - output_shape = ggml_decoder->get_shape(ggml_tensor); + // For a CPY into a padded view_src (e.g. a padded KV cache buffer), the + // OV ScatterUpdate node outputs the full view_src shape, not the CPY node's + // own (smaller) shape. Using the CPY shape here causes set_output_tensor to + // fail with a shape-incompatibility error. Use view_src's shape and data + // pointer instead so the OV tensor matches the model output exactly. + if (ggml_tensor->op == GGML_OP_CPY && ggml_tensor->view_src != nullptr && + ggml_nbytes(ggml_tensor) != ggml_nbytes(ggml_tensor->view_src)) { + output_shape = GgmlOvDecoder::get_shape(ggml_tensor->view_src); + output_data = ggml_tensor->view_src->data; + } else { + output_shape = GgmlOvDecoder::get_shape(ggml_tensor); + } } - - ov::Tensor output_tensor(output_type, output_shape, ggml_tensor->data); + ov::Tensor output_tensor(output_type, output_shape, output_data); return output_tensor; } -enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr r_ctx) { +// Rewrite ggml's KV rows into a relayout state that keeps the sequence on dim 2. +// ggml stores [seq][n_heads_kv * head_size]; the state wants [1, n_heads_kv, seq, head_size], +// a different element order, so the rows are copied instead of reinterpreted. +ov::Tensor kv_rows_to_seq_axis_2(const ov::Tensor & kv_tensor, size_t n_heads_kv) { + const size_t rows = kv_tensor.get_shape()[2]; + const size_t head_size = kv_tensor.get_shape()[3] / n_heads_kv; + const size_t elem = kv_tensor.get_element_type().size(); + const size_t head_bytes = head_size * elem; + + ov::Tensor out(kv_tensor.get_element_type(), ov::Shape{1, n_heads_kv, rows, head_size}); + const auto * src = static_cast(kv_tensor.data()); + auto * dst = static_cast(out.data()); + for (size_t s = 0; s < rows; s++) { + for (size_t h = 0; h < n_heads_kv; h++) { + memcpy(dst + (h * rows + s) * head_bytes, src + (s * n_heads_kv + h) * head_bytes, head_bytes); + } + } + return out; +} + +template void set_zero_diagonal(std::vector & matrix, size_t rows, size_t cols, T zero_value = T{}) { + for (size_t i = 0; i < rows; ++i) { + size_t diag_col = std::min(i, cols - 1); + matrix[i * cols + diag_col] = zero_value; + } +} + +ov::Tensor make_contiguous_split_input_tensor(const struct ggml_tensor * ggml_tensor, const ov::Shape & input_shape) { + const size_t element_size = ggml_type_size(ggml_tensor->type); + const size_t block_size = ggml_blck_size(ggml_tensor->type); + + GGML_ASSERT(block_size == 1 && "non-contiguous split inputs must be plain element types"); + + const struct ggml_tensor * source_tensor = ggml_tensor->view_src != nullptr ? ggml_tensor->view_src : ggml_tensor; + const size_t source_offset = ggml_tensor->view_src != nullptr ? ggml_tensor->view_offs : 0; + + std::vector source_data(ggml_nbytes(source_tensor)); + ggml_backend_tensor_get(source_tensor, source_data.data(), 0, source_data.size()); + + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + auto * dst = static_cast(input_tensor.data()); + size_t dst_offset = 0; + + for (size_t i3 = 0; i3 < static_cast(ggml_tensor->ne[3]); ++i3) { + for (size_t i2 = 0; i2 < static_cast(ggml_tensor->ne[2]); ++i2) { + for (size_t i1 = 0; i1 < static_cast(ggml_tensor->ne[1]); ++i1) { + for (size_t i0 = 0; i0 < static_cast(ggml_tensor->ne[0]); ++i0) { + const size_t src_offset = source_offset + i3 * ggml_tensor->nb[3] + i2 * ggml_tensor->nb[2] + + i1 * ggml_tensor->nb[1] + i0 * ggml_tensor->nb[0]; + std::memcpy(dst + dst_offset, source_data.data() + src_offset, element_size); + dst_offset += element_size; + } + } + } + } + + return input_tensor; +} + +ov::Tensor convert_ggml_input_to_ov(const std::shared_ptr & ggml_decoder, const std::string & name) { + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(name); + + if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, name, ggml_tensor)) { + return *sliced; + } + + if (ggml_tensor->extra != nullptr && !ggml_decoder->is_splited_model()) { + auto * extra_base = static_cast(ggml_tensor->extra); + if (extra_base->type == ggml_openvino_extra_base::Type::TENSOR) { + // GGML_LOG_DEBUG("Using ggml_tensor->extra as ov::Tensor for input: %s\n", name.c_str()); + auto * tensor_extra = static_cast(extra_base); + return *tensor_extra->tensor; + } + } + + // GGML_LOG_DEBUG("Converting ggml tensor to ov::Tensor for input: %s\n", name.c_str()); + auto * input_data = ggml_tensor->data; + ov::Shape input_shape; + if (ggml_tensor->op == GGML_OP_VIEW && !ggml_decoder->is_splited_model()) { + // This case is added to make test-backend-ops work + input_shape = GgmlOvDecoder::get_shape(ggml_tensor->view_src); + } else { + input_shape = GgmlOvDecoder::get_shape(ggml_tensor); + } + + if (ggml_decoder->is_splited_model() && !ggml_is_contiguous(ggml_tensor)) { + return make_contiguous_split_input_tensor(ggml_tensor, input_shape); + } + + auto input_tensor = ov::Tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape, input_data); + return input_tensor; +} + +ov::Tensor get_ov_input_tensor(const std::shared_ptr & ggml_decoder, const std::string & param_name) { + ov::Tensor input_tensor; + auto extra_input = ggml_decoder->get_model_extra_inputs().find(param_name); + if (extra_input != ggml_decoder->get_model_extra_inputs().end()) { + input_tensor = ov::Tensor(extra_input->second.type, extra_input->second.shape); + *input_tensor.data() = extra_input->second.value; + } else { + input_tensor = convert_ggml_input_to_ov(ggml_decoder, param_name); + } + return input_tensor; +} + +ov::Tensor get_ov_input_tensor_static_decode(const std::shared_ptr & ggml_decoder, + const std::string & param_name) { + // NPU decoding stage + if (ggml_decoder->get_model_extra_inputs().count(param_name)) { + return get_ov_input_tensor(ggml_decoder, param_name); + } + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); + const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); + + if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || + GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { + // IMROPE's inp_pos holds one value per t/h/w/e plane instead of a single position; + // with a single decode token the planes are still contiguous, so a flat copy works. + const int n_planes = GgmlOvDecoder::is_inp_pos(ggml_tensor, op) ? GgmlOvDecoder::get_inp_pos_n_planes(op) : 1; + assert(ggml_tensor->ne[0] == n_planes); + ov::Shape input_shape = {1, 1, 1, (size_t) n_planes}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + std::memcpy(input_tensor.data(), ggml_tensor->data, n_planes * ggml_type_size(ggml_tensor->type)); + return input_tensor; + } + + if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { + ov::Shape input_shape = {1, 1, 1, 1}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + int32_t inp_out_id = *((int32_t *) ggml_tensor->data); + assert(ggml_tensor->ne[0] == 1); + assert(inp_out_id == 0); + *input_tensor.data() = inp_out_id; + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { + size_t context_size = ggml_decoder->get_ctx_size(); + if (ggml_tensor->type == GGML_TYPE_F16) { + std::vector padded_data = + pad_input(ggml_tensor, 1, context_size, GGML_FP32_TO_FP16(-INFINITY)); + ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, 1, context_size}); + std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); + return input_tensor; + } + + std::vector padded_data = pad_input(ggml_tensor, 1, context_size, -INFINITY); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, 1, context_size}); + auto * data_ptr = input_tensor.data(); + std::copy(padded_data.begin(), padded_data.begin() + context_size, data_ptr); + return input_tensor; + } + + return get_ov_input_tensor(ggml_decoder, param_name); +} + +ov::Tensor get_ov_input_tensor_static_prefill(const std::shared_ptr & ggml_decoder, + const std::string & param_name, + int chunk_index) { + // NPU prompt processing stage + const size_t input_len = ggml_decoder->get_input_len(); + const size_t chunk_size = ggml_decoder->m_prefill_chunk_size; + const size_t chunk_valid_size = std::min(chunk_size, input_len - chunk_index * chunk_size); + const size_t chunk_pad_size = chunk_size - chunk_valid_size; + + if (param_name == "chunk_valid_len") { + ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); + *input_tensor.data() = (int64_t) chunk_valid_size; + return input_tensor; + } + if (chunk_index > 0 && param_name == "cache_rs_reset_len") { + // The recurrent-state clear belongs to the start of the sequence. Re-applying it on every + // chunk would wipe the state accumulated by the preceding chunks, so disable it (a zero + // length makes scale.cpp's keep-mask select every slot) after the first chunk. + ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); + *input_tensor.data() = 0; + return input_tensor; + } + if (ggml_decoder->get_model_extra_inputs().count(param_name)) { + return get_ov_input_tensor(ggml_decoder, param_name); + } + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); + const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); + + if (GgmlOvDecoder::is_inp_pos(ggml_tensor, op) && GgmlOvDecoder::get_inp_pos_n_planes(op) > 1) { + // IMROPE: inp_pos stacks n_planes (t/h/w/e) position planes, each of length + // input_len; pad every plane independently so they stay aligned to chunk_size. + const int n_planes = GgmlOvDecoder::get_inp_pos_n_planes(op); + const size_t element_size = ggml_type_size(ggml_tensor->type); + ov::Shape input_shape = {1, 1, 1, (size_t) n_planes * chunk_size}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + for (int p = 0; p < n_planes; p++) { + const char * src = + (const char *) ggml_tensor->data + (p * input_len + chunk_index * chunk_size) * element_size; + char * dst = (char *) input_tensor.data() + p * chunk_size * element_size; + std::memcpy(dst, src, chunk_valid_size * element_size); + if (chunk_pad_size > 0) { + if (ggml_tensor->type == GGML_TYPE_I32) { + int32_t last_value = *((const int32_t *) src + chunk_valid_size - 1); + int32_t * out = (int32_t *) dst; + std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); + } else if (ggml_tensor->type == GGML_TYPE_I64) { + int64_t last_value = *((const int64_t *) src + chunk_valid_size - 1); + int64_t * out = (int64_t *) dst; + std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); + } else { + throw std::runtime_error("Unexpected tensor type for " + param_name); + } + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || + GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { + ov::Shape input_shape = {1, 1, 1, chunk_size}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + // copy the chunk_index-th chunk from ggml_tensor + size_t element_size = ggml_type_size(ggml_tensor->type); + void * input_data = (char *) ggml_tensor->data + chunk_index * chunk_size * element_size; + std::memcpy(input_tensor.data(), input_data, chunk_valid_size * element_size); + // pad the rest with last_value + 1, so that kv's of padded positions are inserted + // to the next row after the valids row in the kvcache + if (chunk_pad_size > 0) { + if (ggml_tensor->type == GGML_TYPE_I32) { + int32_t last_value = + *((int32_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); + int32_t * output_data = input_tensor.data(); + std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); + } else if (ggml_tensor->type == GGML_TYPE_I64) { + int64_t last_value = + *((int64_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); + int64_t * output_data = input_tensor.data(); + std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); + } else { + throw std::runtime_error("Unexpected tensor type for " + param_name); + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { + size_t output_len = ggml_decoder->get_compute_params().output_len; + ov::Shape input_shape = {1, 1, 1, output_len}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + if (ggml_tensor->ne[0] == 0) { + *input_tensor.data() = 0; + } else { + auto * data_addr = input_tensor.data(); + for (size_t i = 0; i < output_len; i++) { + data_addr[i] = ((int32_t *) ggml_tensor->data)[i] % chunk_size; + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mean(ggml_tensor, op)) { + const size_t n_seqs = ggml_tensor->ne[1]; + const size_t src_stride = ggml_tensor->ne[0]; + const size_t copy_len = std::min(chunk_valid_size, src_stride - chunk_index * chunk_size); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, n_seqs, chunk_size}); + auto * dst = input_tensor.data(); + std::fill(dst, dst + n_seqs * chunk_size, 0.0f); + const auto * src = static_cast(ggml_tensor->data) + chunk_index * chunk_size; + for (size_t s = 0; s < n_seqs; s++) { + std::memcpy(dst + s * chunk_size, src + s * src_stride, copy_len * sizeof(float)); + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { + size_t cols = ggml_tensor->ne[0]; + size_t rows = ggml_tensor->ne[1]; + size_t chunk_valid_rows = std::min(chunk_size, rows - chunk_index * chunk_size); + size_t context_size = ggml_decoder->get_ctx_size(); + if (ggml_tensor->type == GGML_TYPE_F16) { + const auto * ggml_data = + static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; + std::vector padded_data = pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, + context_size, GGML_FP32_TO_FP16(-INFINITY)); + set_zero_diagonal(padded_data, chunk_size, context_size, GGML_FP32_TO_FP16(0.0f)); + ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, chunk_size, context_size}); + std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); + return input_tensor; + } + + const auto * ggml_data = static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; + std::vector padded_data = + pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, context_size, -INFINITY); + set_zero_diagonal(padded_data, chunk_size, context_size); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, chunk_size, context_size}); + auto * data_ptr = input_tensor.data(); + std::copy(padded_data.begin(), padded_data.begin() + chunk_size * context_size, data_ptr); + return input_tensor; + } + + return get_ov_input_tensor(ggml_decoder, param_name); +} + +enum ggml_status naive_compute(ggml_cgraph * cgraph, + ov::Core & core, + const std::string & device, + const ov::AnyMap & config, + ov_compiled_model_cache & cache) { + if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { + return GGML_STATUS_SUCCESS; + } + + std::unique_lock compile_lock(cache.mutex); + bool naive = true; + auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); + auto decoder = std::make_shared(cgraph, model_weights); + auto input_model = std::make_shared(decoder); + auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); + if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { + ov::serialize(model, "IR_naive.xml"); + } + + std::shared_ptr infer_request; + auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; + if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { + // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; + } else { + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; + } + if (remote_context.has_value()) { + infer_request = std::make_shared( + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); + } else { + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); + } + std::vector input_names; + std::vector output_names; + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); + + for (size_t i = 0; i < input_names.size(); i++) { + const auto & param_name = input_names[i]; + auto input_tensor = get_ov_input_tensor(decoder, param_name); + infer_request->set_input_tensor(i, input_tensor); + } + + // Use get_output_tensor + memcpy instead of set_output_tensor to avoid memory overwritten + // when i/o buffer overlaps, e.g. the cgraph is a single PERMUTE + + infer_request->infer(); + + for (size_t i = 0; i < output_names.size(); i++) { + auto output_tensor = infer_request->get_output_tensor(i); + const auto & model_outputs = decoder->get_model_outputs(); + auto model_output_it = model_outputs.find(output_names[i]); + if (model_output_it == model_outputs.end()) { + // Debug-only output added via GGML_OPENVINO_DEBUG_NODE; nothing to copy into. + if (ggml_openvino_getenv_int("GGML_OPENVINO_DEBUG_OUTPUT") || + ggml_openvino_getenv_str("GGML_OPENVINO_DEBUG_NODE")) { + print_output_tensor_info(output_names[i], output_tensor, output_tensor.data()); + } + continue; + } + auto * ggml_tensor = model_output_it->second; + std::memcpy(ggml_tensor->data, output_tensor.data(), output_tensor.get_byte_size()); + } + return GGML_STATUS_SUCCESS; +} + +enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, const std::shared_ptr & r_ctx) { auto & core = ov_singleton_core(); const auto & config = ggml_openvino_get_compile_config(); const auto & device = r_ctx->device; @@ -203,7 +653,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< if (is_naive(cgraph)) { if (!model_is_splitted) { - return naive_compute(cgraph, core, device, config); + return naive_compute(cgraph, core, device, config, *r_ctx->compiled_cache); } } @@ -247,6 +697,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } std::lock_guard lock(*(entry->mutex)); + cache_hit = cache_hit && entry->ptr && r_ctx->infer_request_cache.count(key) != 0; if (cache_hit) { ggml_decoder = entry->ptr; @@ -277,39 +728,96 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< if (stateful) { const auto * inp_pos = get_inp_pos_tensor(cgraph); int32_t * pos_data = (int32_t *) inp_pos->data; - auto pos_shape = ggml_decoder->get_shape(inp_pos); + auto pos_shape = GgmlOvDecoder::get_shape(inp_pos); if (pos_data[0] == 0) { infer_request->reset_state(); r_ctx->stateful_kv_size = pos_shape[3]; } else if (r_ctx->stateful_kv_size == static_cast(pos_data[0])) { r_ctx->stateful_kv_size += pos_shape[3]; } else { + const size_t pos_begin = static_cast(pos_data[0]); + const bool refill = pos_begin > r_ctx->stateful_kv_size; + + // A refill seeds the state from ggml's KV cache, so it needs that cache to be a + // plain prefix: cell i must hold position i. An SWA layer keeps only the last + // n_swa positions, so once a position leaves the window ggml drops it and the + // remaining cells shift - cell i stops holding position i. While every position + // is still inside the window nothing has been dropped and the refill is sound. + if (refill && !ggml_decoder->get_model_params().swa_layers.empty()) { + const int n_swa = ggml_decoder->get_compute_params().swa_window; + if (n_swa < 0 || static_cast(n_swa) < pos_begin) { + GGML_LOG_ERROR( + "GGML OpenVINO backend stateful inference failed: cannot resume at position %zu from a " + "state that holds %zu tokens, because the sliding-window layers keep only the last %d " + "positions. Run without GGML_OPENVINO_STATEFUL_EXECUTION.\n", + pos_begin, r_ctx->stateful_kv_size, n_swa); + return GGML_STATUS_FAILED; + } + } + + const bool relayout_enabled = !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT"); + auto states = infer_request->query_state(); for (auto state : states) { auto state_tensor = state.get_state(); auto state_tensor_shape = state_tensor.get_shape(); - if (static_cast(pos_data[0]) > r_ctx->stateful_kv_size) { - std::string state_name; - try { - state_name = r_ctx->kv_state_input_name_map.at(state.get_name()); - } catch (...) { + + std::string state_name; + if (auto it = r_ctx->kv_state_input_name_map.find(state.get_name()); + it != r_ctx->kv_state_input_name_map.end()) { + state_name = it->second; + } + + // Which axis holds the sequence: pass::KVStateSeqAxis moves it from dim 1 + // to dim 2. The head count is still needed below, because only a 1-head + // state stays byte-compatible with ggml's cache buffer. gemma-4 12B mixes + // 1-head full layers with 8-head sliding layers, so it is per state. + int n_heads_kv = ggml_decoder->get_model_params().n_heads_kv; + if (auto layer = extract_layer_from_name(state_name); layer.has_value()) { + n_heads_kv = ggml_decoder->get_n_heads_kv_for_layer(layer.value()); + } + const bool relayout_this_state = relayout_enabled; + const size_t seq_axis = relayout_this_state ? 2 : 1; + const size_t head_axis = seq_axis == 2 ? 1 : 2; + + if (refill) { + if (state_name.empty()) { GGML_LOG_ERROR( "GGML OpenVINO backend stateful inference failed: no input found for the state\n"); return GGML_STATUS_FAILED; } auto kv_tensor = get_ov_input_tensor(ggml_decoder, state_name); - kv_tensor.set_shape({state_tensor_shape[0], kv_tensor.get_shape()[2], state_tensor_shape[2], - state_tensor_shape[3]}); - state_tensor = kv_tensor; + if (relayout_this_state && n_heads_kv != 1) { + // several heads with seq on dim 2: not the same bytes as ggml's + // buffer, so the rows have to be copied into the new order + state_tensor = kv_rows_to_seq_axis_2(kv_tensor, (size_t) n_heads_kv); + } else { + ov::Shape refill_shape(4); + refill_shape[0] = state_tensor_shape[0]; + refill_shape[seq_axis] = kv_tensor.get_shape()[2]; + refill_shape[head_axis] = state_tensor_shape[head_axis]; + refill_shape[3] = state_tensor_shape[3]; + kv_tensor.set_shape(refill_shape); + state_tensor = kv_tensor; + } state_tensor_shape = state_tensor.get_shape(); } + // Only ever shrink to a prefix the source really has. Slicing past it used to + // surface as a bare ov::Exception from the ROI constructor. + if (state_tensor_shape[seq_axis] < pos_begin) { + GGML_LOG_ERROR( + "GGML OpenVINO backend stateful inference failed: state '%s' holds %zu tokens on axis " + "%zu, cannot resume at position %zu\n", + state.get_name().c_str(), state_tensor_shape[seq_axis], seq_axis, pos_begin); + return GGML_STATUS_FAILED; + } ov::Coordinate begin = {0, 0, 0, 0}; - ov::Coordinate end = {state_tensor_shape[0], static_cast(pos_data[0]), - state_tensor_shape[2], state_tensor_shape[3]}; + ov::Coordinate end(state_tensor_shape.begin(), state_tensor_shape.end()); + end[seq_axis] = pos_begin; ov::Tensor new_state_tensor(state_tensor, begin, end); state.set_state(new_state_tensor); } - r_ctx->stateful_kv_size = pos_data[0] + pos_shape[3]; + r_ctx->stateful_kv_size = pos_begin + pos_shape[3]; } } @@ -317,11 +825,30 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< conversion_end_time = decoder_end_time; compile_end_time = decoder_end_time; } else { + // Compilation can mutate shared weight nodes, so serialize cold paths. + // The lock is released before binding tensors or running inference. + auto shared_cache = r_ctx->compiled_cache; + std::unique_lock compile_lock(shared_cache->mutex); + auto weight_names = get_weight_names(cgraph); + ggml_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, is_static, + stateful, model_is_splitted); + const std::string shared_key = cache_enabled ? compiled_graph_key(cgraph, *ggml_decoder, device) : ""; + ov::CompiledModel shared_model; + bool imported = false; + auto shared_it = shared_cache->graphs.find(shared_key); + if (!shared_key.empty() && shared_it != shared_cache->graphs.end()) { + shared_model = shared_it->second.decode; + infer_request = std::make_shared(shared_model.create_infer_request()); + ov_input_names = shared_it->second.input_names; + ov_output_names = shared_it->second.output_names; + imported = true; + GGML_LOG_DEBUG("ggml-openvino: shared compiled model HIT (dynamic)\n"); + } // Fail fast: a cache-miss recompile feeds weight data to compile_model, but // GGML_OPENVINO_RELEASE_WEIGHTS (or GGML_OPENVINO_MEMORY_OPTIMIZE on GPU) // may have already dropped the host weight pages // (they would read as zeros). That mode requires stable graph shapes. - if (ggml_openvino_weight_buffers_released()) { + if (!imported && ggml_openvino_weight_buffers_released()) { GGML_ABORT( "ggml-openvino: a new graph needs to be compiled but host weight buffers were already " "released via GGML_OPENVINO_RELEASE_WEIGHTS/GGML_OPENVINO_MEMORY_OPTIMIZE. This mode requires " @@ -340,8 +867,8 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< // the weights are baked into the imported CompiledModel. const std::string model_cache_dir = ggml_openvino_model_cache_dir(); uint64_t model_fp = 0; - std::string blob_path, manifest_path; - bool imported = false; + std::string blob_path; + std::string manifest_path; // When the frontend model cache is active it supersedes the plugin-level // ov::cache_dir: a blob exported from a model compiled WITH cache_dir cannot // be re-imported (import returns an uninitialized model). Strip cache_dir / @@ -351,16 +878,17 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< mc_config.erase("CACHE_DIR"); mc_config.erase("CACHE_MODE"); } - if (!model_cache_dir.empty() && !model_is_splitted) { + if (!imported && !model_cache_dir.empty() && !model_is_splitted) { const uint64_t extra_cfg = ggml_openvino_model_cache_extra_cfg(device, stateful); - model_fp = ggml_openvino_model_fingerprint(cgraph, device, /*fa=*/true, m_params.rope_params, - 15, extra_cfg); + model_fp = + ggml_openvino_model_fingerprint(cgraph, device, /*fa=*/true, m_params.rope_params, 16, extra_cfg); blob_path = ggml_openvino_model_cache_blob_path(model_cache_dir, model_fp); manifest_path = ggml_openvino_model_cache_manifest_path(model_cache_dir, model_fp); std::ifstream blob_in(blob_path, std::ios::binary); bool blob_ok = blob_in.is_open(); - bool manifest_ok = blob_ok && ggml_openvino_model_cache_verify_manifest(manifest_path, cgraph, model_fp); + bool manifest_ok = + blob_ok && ggml_openvino_model_cache_verify_manifest(manifest_path, cgraph, model_fp); if (blob_ok && manifest_ok) { int64_t import_start = ggml_time_us(); try { @@ -380,6 +908,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< ggml_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, is_static, stateful, model_is_splitted); infer_request = std::make_shared(cm.create_infer_request()); + shared_model = cm; entry->ptr = ggml_decoder; // Names must match the decoder's ggml-tensor keys. The non-cached // path keys off Parameter/Result *friendly names* (set by the @@ -473,6 +1002,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } infer_request = std::make_shared(compiled_model.create_infer_request()); + shared_model = compiled_model; entry->ptr = ggml_decoder; for (const auto & ov_param : model->get_parameters()) { @@ -483,6 +1013,11 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } } // end non-imported (compile) path + entry->ptr = ggml_decoder; + if (!shared_key.empty() && shared_it == shared_cache->graphs.end()) { + shared_cache->graphs.emplace(shared_key, + ov_compiled_graph{shared_model, {}, ov_input_names, ov_output_names}); + } if (cache_enabled) { std::lock_guard map_lock(r_ctx->ctx_mutex); r_ctx->infer_request_cache[key] = infer_request; @@ -492,7 +1027,19 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< if (stateful && cache_enabled) { const auto * inp_pos = get_inp_pos_tensor(cgraph); - auto pos_shape = ggml_decoder->get_shape(inp_pos); + auto pos_shape = GgmlOvDecoder::get_shape(inp_pos); + // A freshly compiled model starts with an empty state, so it can only serve a + // sequence from its beginning. A non-zero start position means the KV history was + // built elsewhere (a restored ggml cache), which the state cannot adopt. + const int32_t pos_begin = ((int32_t *) inp_pos->data)[0]; + if (pos_begin != 0) { + GGML_LOG_ERROR( + "GGML OpenVINO backend stateful inference failed: a new model was compiled for a sequence that " + "starts at position %d, but its state is empty. Run without " + "GGML_OPENVINO_STATEFUL_EXECUTION.\n", + pos_begin); + return GGML_STATUS_FAILED; + } r_ctx->stateful_kv_size = pos_shape[3]; const auto kv_param_res_names = ggml_decoder->get_kv_param_res_names(); for (const auto & pair : kv_param_res_names) { @@ -502,7 +1049,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } for (size_t i = 0; i < ov_input_names.size(); i++) { - auto param_name = ov_input_names[i]; + const auto & param_name = ov_input_names[i]; auto input_tensor = get_ov_input_tensor(ggml_decoder, param_name); infer_request->set_input_tensor(i, input_tensor); @@ -557,22 +1104,32 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< // GGML_OPENVINO_RELEASE_WEIGHTS (or GGML_OPENVINO_MEMORY_OPTIMIZE on GPU): the plugin holds its own device copy of // every weight after compile, so the host weight buffers can be dropped to reclaim - // RSS. The GPU backend uses a single dynamic-shape model for both prefill and decode, - // so once a graph is compiled it is reused for the whole session — the only thing - // that forces a recompile is clear_caches() on backend teardown. We therefore release - // on the first cache-hit (model compiled, plugin has its copy) and, crucially, pin the - // compiled-model cache so it survives backend teardown (see ggml_backend_openvino_free). - // Without the pin, a later test/context would recompile against the now-dropped pages. - // A genuinely new graph still fails fast at the cache-miss compile branch. - if (cache_hit && ggml_openvino_release_weights_enabled(device) && - !ggml_openvino_weight_buffers_released()) { - ggml_openvino_release_weight_buffers(); + // RSS. Release only while holding the compilation mutex so another context cannot + // be reading host weights during conversion/compilation. Pin the shared compiled + // models across backend teardown; a later context can create its own request without + // reading the dropped pages. A new, uncached graph still fails fast above. + if (cache_hit && ggml_openvino_release_weights_enabled(device)) { + std::lock_guard compile_lock(r_ctx->compiled_cache->mutex); + if (!ggml_openvino_weight_buffers_released()) { + ggml_openvino_release_weight_buffers(); + } } return GGML_STATUS_SUCCESS; } -enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptr r_ctx) { +ov::AnyMap without_npuw(const ov::AnyMap & config) { + ov::AnyMap out; + for (const auto & kv : config) { + if (kv.first.rfind("NPUW", 0) == 0 || kv.first == "NPU_USE_NPUW") { + continue; + } + out.insert(kv); + } + return out; +} + +enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, const std::shared_ptr & r_ctx) { auto & core = ov_singleton_core(); auto get_prefill_chunk_size = [] { @@ -583,7 +1140,9 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrcompiled_cache); } auto start_time = ggml_time_us(); @@ -603,7 +1162,12 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrne[0]; + } graph_key key(cgraph); static const bool cache_enabled = !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_CACHE"); bool cache_hit = false; @@ -637,6 +1201,8 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptr lock(*(entry->mutex)); + cache_hit = cache_hit && entry->ptr && r_ctx->infer_request_cache.count(key) != 0 && + r_ctx->infer_request_cache_prefill.count(key) != 0; if (cache_hit) { ggml_decoder = entry->ptr; @@ -674,78 +1240,127 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrinfer_request_cache_prefill.erase(key); } - std::shared_ptr model; - auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph); + // Static execution shares a compiled prefill/decode pair. Each backend + // creates and retains its own requests for both phases. + auto shared_cache = r_ctx->compiled_cache; + std::unique_lock compile_lock(shared_cache->mutex); + auto weight_names = get_weight_names(cgraph); + auto local_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, is_static, + stateful, false, is_prefill, prefill_chunk_size); + const std::string shared_key = + cache_enabled ? compiled_graph_key(cgraph, *local_decoder, device, prefill_chunk_size) : ""; + auto shared_it = shared_cache->graphs.find(shared_key); + if (!shared_key.empty() && shared_it != shared_cache->graphs.end()) { + auto & compiled = shared_it->second; + auto prefill_request = std::make_shared(compiled.prefill.create_infer_request()); + auto decode_request = no_kv_cache ? + prefill_request : + std::make_shared(compiled.decode.create_infer_request()); + ggml_decoder = local_decoder; + entry->ptr = ggml_decoder; + infer_request = is_prefill ? prefill_request : decode_request; + ov_input_names_local = compiled.input_names; + ov_output_names_local = compiled.output_names; + r_ctx->infer_request_cache_prefill[key] = prefill_request; + r_ctx->infer_request_cache[key] = decode_request; + r_ctx->ov_input_names_cache[key] = ov_input_names_local; + r_ctx->ov_output_names_cache[key] = ov_output_names_local; + decoder_end_time = conversion_end_time = compile_end_time = ggml_time_us(); + GGML_LOG_DEBUG("ggml-openvino: shared compiled model HIT (static)\n"); + } else { + std::shared_ptr model; + auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph); + + auto ggml_decoder_prefill = std::make_shared( + cgraph, m_params, c_params, model_weights, is_static, stateful, false, true, prefill_chunk_size); + auto ggml_decoder_decode = + no_kv_cache ? ggml_decoder_prefill : + std::make_shared(cgraph, m_params, c_params, model_weights, is_static, + stateful, false, false, prefill_chunk_size); + decoder_end_time = ggml_time_us(); - if (m_params.n_heads_kv == -1) { - // graph is not a LLM, e.g. context-shift graph - prefill_chunk_size = inp_pos->ne[0]; - } - auto ggml_decoder_prefill = std::make_shared( - cgraph, m_params, c_params, model_weights, is_static, stateful, false, true, prefill_chunk_size); - auto ggml_decoder_decode = std::make_shared(cgraph, m_params, c_params, model_weights, is_static, - stateful, false, false, prefill_chunk_size); - decoder_end_time = ggml_time_us(); + const bool dump_ir = ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR"); + const auto dump_ir_timestamp = static_cast(ggml_time_us()); - auto input_model_prefill = std::make_shared(ggml_decoder_prefill); - auto input_model_decode = std::make_shared(ggml_decoder_decode); - - auto model_prefill = ov::frontend::ggml::FrontEnd::convert(input_model_prefill); - ggml_decoder_prefill->clear_model_weights(); - auto model_decode = ov::frontend::ggml::FrontEnd::convert(input_model_decode); - ggml_decoder_decode->clear_model_weights(); - conversion_end_time = ggml_time_us(); - - if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { - char timestamped_filename[64]; - auto timestamp = (long long) ggml_time_us(); - snprintf(timestamped_filename, sizeof(timestamped_filename), "model_prefill_%lld.xml", timestamp); - ov::serialize(model_prefill, timestamped_filename); - snprintf(timestamped_filename, sizeof(timestamped_filename), "model_decode_%lld.xml", timestamp); - ov::serialize(model_decode, timestamped_filename); - } + auto build_static_model = [&core, &compile_config, dump_ir, dump_ir_timestamp]( + const std::shared_ptr & decoder, const char * tag, + std::shared_ptr & model, ov::CompiledModel & compiled_model, + std::shared_ptr & infer_request, + int64_t & local_conversion_end_time, int64_t & local_compile_end_time) { + auto input_model = std::make_shared(decoder); + model = ov::frontend::ggml::FrontEnd::convert(input_model); + decoder->clear_model_weights(); + local_conversion_end_time = ggml_time_us(); - ov::CompiledModel compiled_model_prefill; - ov::CompiledModel compiled_model_decode; - auto remote_context = ggml_openvino_get_remote_context(); - if (remote_context.has_value()) { - compiled_model_prefill = core.compile_model(model_prefill, remote_context.value(), config); - compiled_model_decode = core.compile_model(model_decode, remote_context.value(), config); - } else { - compiled_model_prefill = core.compile_model(model_prefill, device, config); - compiled_model_decode = core.compile_model(model_decode, device, config); - } + if (dump_ir) { + char timestamped_filename[64]; + snprintf(timestamped_filename, sizeof(timestamped_filename), "model_%s_%lld.xml", tag, + dump_ir_timestamp); + ov::serialize(model, timestamped_filename); + } - auto infer_request_prefill = std::make_shared(compiled_model_prefill.create_infer_request()); - auto infer_request_decode = std::make_shared(compiled_model_decode.create_infer_request()); - compile_end_time = ggml_time_us(); + compiled_model = core.compile_model(model, device, compile_config); + infer_request = std::make_shared(compiled_model.create_infer_request()); + local_compile_end_time = ggml_time_us(); + }; + std::shared_ptr model_prefill; + std::shared_ptr model_decode; + ov::CompiledModel compiled_model_prefill; + ov::CompiledModel compiled_model_decode; + std::shared_ptr infer_request_prefill; + std::shared_ptr infer_request_decode; + int64_t prefill_conversion_end_time; + int64_t decode_conversion_end_time; + int64_t prefill_compile_end_time; + int64_t decode_compile_end_time; + build_static_model(ggml_decoder_prefill, "prefill", model_prefill, compiled_model_prefill, + infer_request_prefill, prefill_conversion_end_time, prefill_compile_end_time); + if (no_kv_cache) { + model_decode = model_prefill; + compiled_model_decode = compiled_model_prefill; + infer_request_decode = infer_request_prefill; + decode_conversion_end_time = prefill_conversion_end_time; + decode_compile_end_time = prefill_compile_end_time; + } else { + build_static_model(ggml_decoder_decode, "decode", model_decode, compiled_model_decode, + infer_request_decode, decode_conversion_end_time, decode_compile_end_time); + } + conversion_end_time = std::max(prefill_conversion_end_time, decode_conversion_end_time); + compile_end_time = std::max(prefill_compile_end_time, decode_compile_end_time); - model = is_prefill ? model_prefill : model_decode; - ggml_decoder = is_prefill ? ggml_decoder_prefill : ggml_decoder_decode; - infer_request = is_prefill ? infer_request_prefill : infer_request_decode; - entry->ptr = ggml_decoder; + model = is_prefill ? model_prefill : model_decode; + ggml_decoder = is_prefill ? ggml_decoder_prefill : ggml_decoder_decode; + infer_request = is_prefill ? infer_request_prefill : infer_request_decode; + entry->ptr = ggml_decoder; - for (const auto & ov_param : model->get_parameters()) { - ov_input_names_local.push_back(ov_param->get_friendly_name()); - } - for (const auto & ov_output : model->get_results()) { - ov_output_names_local.push_back(ov_output->get_friendly_name()); - } + for (const auto & ov_param : model->get_parameters()) { + ov_input_names_local.push_back(ov_param->get_friendly_name()); + } + for (const auto & ov_output : model->get_results()) { + ov_output_names_local.push_back(ov_output->get_friendly_name()); + } - if (cache_enabled) { - std::lock_guard map_lock(r_ctx->ctx_mutex); - r_ctx->infer_request_cache_prefill[key] = infer_request_prefill; - r_ctx->infer_request_cache[key] = infer_request_decode; - r_ctx->ov_input_names_cache[key] = ov_input_names_local; - r_ctx->ov_output_names_cache[key] = ov_output_names_local; + if (!shared_key.empty()) { + shared_cache->graphs.emplace( + shared_key, ov_compiled_graph{compiled_model_decode, compiled_model_prefill, ov_input_names_local, + ov_output_names_local}); + } + + if (cache_enabled) { + std::lock_guard map_lock(r_ctx->ctx_mutex); + r_ctx->infer_request_cache_prefill[key] = infer_request_prefill; + r_ctx->infer_request_cache[key] = infer_request_decode; + r_ctx->ov_input_names_cache[key] = ov_input_names_local; + r_ctx->ov_output_names_cache[key] = ov_output_names_local; + } } } if (is_prefill) { - auto inp_len = inp_pos->ne[0]; + auto inp_len = get_inp_pos_n_tokens(cgraph, inp_pos); for (int chunk_index = 0; chunk_index * prefill_chunk_size < inp_len; chunk_index++) { for (size_t i = 0; i < ov_input_names_local.size(); i++) { - auto param_name = ov_input_names_local[i]; + const auto & param_name = ov_input_names_local[i]; auto input_tensor = get_ov_input_tensor_static_prefill(ggml_decoder, param_name, chunk_index); infer_request->set_input_tensor(i, input_tensor); @@ -762,6 +1377,11 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrsecond; + if (ggml_nbytes(ggml_tensor) == 0) { + // Zero-row in-place writeback (e.g. the empty s_copy defrag remainder). The OV + // Result is the full cache, so binding it over this 0-byte buffer overflows it. + continue; + } auto output_tensor = create_ov_output_tensor(ggml_decoder, infer_request, i, ggml_tensor); infer_request->set_output_tensor(i, output_tensor); } @@ -781,7 +1401,7 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrset_input_tensor(i, input_tensor); @@ -798,6 +1418,9 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrsecond; + if (ggml_nbytes(ggml_tensor) == 0) { + continue; + } auto output_tensor = create_ov_output_tensor(ggml_decoder, infer_request, i, ggml_tensor); infer_request->set_output_tensor(i, output_tensor); } @@ -816,420 +1439,139 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrsrc. -// Step 2 verifies that node inputs come from model nodes/weights/leafs; external sources imply split. -bool is_model_splitted(ggml_cgraph * cgraph) { - static const bool fallback_enabled = ggml_openvino_getenv_int("GGML_OPENVINO_ENABLE_FALLBACK") != 0; - if (!fallback_enabled) { - return false; - } - - // Backend op tests execute each node through ggml_graph_view(), which preserves the original - // graph use_counts while exposing only one node. Treat those single-node views as regular - // naive graphs so intermediate ops do not look like split-model fragments. - if (cgraph->n_nodes <= 1 && cgraph->n_leafs == 0) { - return false; - } - - // check the nodes of the model are used by the following nodes, through compare the node's use count and the count of nodes that use it as input. If does not match, return true, else return false. - for (int i = 0; i < cgraph->n_nodes; i++) { - ggml_tensor * node = cgraph->nodes[i]; - int use_count = cgraph->use_counts[ggml_hash_find(&cgraph->visited_hash_set, node)]; - // TODO: this is a workround for the tests case from llama.cpp, fix should from the root cause in the future. - if ((cgraph->n_nodes <= 1 && use_count == 0) || - (cgraph->n_nodes <= 1 && node->op == GGML_OP_VIEW && use_count == 1 && node->src[0] != nullptr && - node->src[0]->op == GGML_OP_NONE)) { - return false; - } - if (cgraph->n_nodes == 1 && - (cgraph->nodes[0]->op == GGML_OP_TRANSPOSE || cgraph->nodes[0]->op == GGML_OP_PERMUTE)) { - return false; - } - int input_use_count = 0; - for (int j = 0; j < cgraph->n_nodes; j++) { - ggml_tensor * other_node = cgraph->nodes[j]; - for (int k = 0; k < GGML_MAX_SRC; k++) { - if (other_node->src[k] == node) { - input_use_count++; - } - } - } - if (use_count != input_use_count && node->op != GGML_OP_NONE) { - return true; - } - } - // if all nodes's src node's src is not come from the nodes in the model, we think the model is splitted. This is a complementary check for the above check, because for some special case like the output node is not used by any node, the use count and input use count are both 0, we can not determine whether the model is splitted or not just based on the first check. - // Only weight-name membership is needed below. With GGML_OPENVINO_REDUCE_COMPILE_MEM - // use the name-only collector (no weight extraction); otherwise keep the original - // behavior of building (naive) weight nodes and take their names. - std::set model_weights; - if (ggml_openvino_reduce_compile_mem_enabled()) { - model_weights = GgmlOvDecoder::collect_weight_names(cgraph); - } else { - for (const auto & kv : GgmlOvDecoder::create_weight_nodes(cgraph, true)) { - model_weights.insert(kv.first); - } - } - std::set model_nodes(cgraph->nodes, cgraph->nodes + cgraph->n_nodes); - // leaf nodes - std::set model_leafs(cgraph->leafs, cgraph->leafs + cgraph->n_leafs); - for (int i = 0; i < cgraph->n_nodes; i++) { - ggml_tensor * node = cgraph->nodes[i]; - for (int j = 0; j < GGML_MAX_SRC; j++) { - ggml_tensor * src = node->src[j]; - // the src is also not the model weights, we think the model is splitted. - // the src is also not in model leafs, we think the model is splitted. - if (src != nullptr && model_nodes.find(src) == model_nodes.end() && - model_weights.find(std::string(src->name)) == model_weights.end() && !model_leafs.empty() == false && - model_leafs.find(src) == model_leafs.end()) { - if (GgmlOvDecoder::is_inp_tok(src, node)) { - return false; - } - return true; - } - } - } - return false; -} - -bool is_naive(ggml_cgraph * cgraph) { - constexpr int naive_graph_size_threshold = 20; - int count = 0; - for (int i = 0; i < cgraph->n_nodes; i++) { - if (cgraph->nodes[i]->op != GGML_OP_NONE) { - count++; - } - } - return count < naive_graph_size_threshold; -} - -enum ggml_status naive_compute(ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config) { - if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { - return GGML_STATUS_SUCCESS; - } - - bool naive = true; - auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); - auto decoder = std::make_shared(cgraph, model_weights); - auto input_model = std::make_shared(decoder); - auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); - if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { - ov::serialize(model, "IR_naive.xml"); - } - - std::shared_ptr infer_request; - auto remote_context = ggml_openvino_get_remote_context(); - if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { - // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::PERFORMANCE)); - } else { - core.set_property(device, ov::hint::execution_mode(ov::hint::ExecutionMode::ACCURACY)); - } - if (remote_context.has_value()) { - infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), config).create_infer_request()); - } else { - infer_request = - std::make_shared(core.compile_model(model, device, config).create_infer_request()); - } - - auto ov_params = model->get_parameters(); - for (size_t i = 0; i < ov_params.size(); i++) { - auto param_name = ov_params[i]->get_friendly_name(); - auto input_tensor = get_ov_input_tensor(decoder, param_name); - infer_request->set_input_tensor(i, input_tensor); - } - - // Use get_output_tensor + memcpy instead of set_output_tensor to avoid memory overwritten - // when i/o buffer overlaps, e.g. the cgraph is a single PERMUTE - - infer_request->infer(); - - auto ov_results = model->get_results(); - for (size_t i = 0; i < ov_results.size(); i++) { - auto output_tensor = infer_request->get_output_tensor(i); - const auto & model_outputs = decoder->get_model_outputs(); - auto model_output_it = model_outputs.find(ov_results[i]->get_friendly_name()); - if (model_output_it == model_outputs.end()) { - // Debug-only output added via GGML_OPENVINO_DEBUG_NODE; nothing to copy into. - if (ggml_openvino_getenv_int("GGML_OPENVINO_DEBUG_OUTPUT") || - ggml_openvino_getenv_str("GGML_OPENVINO_DEBUG_NODE")) { - print_output_tensor_info(ov_results[i]->get_friendly_name(), output_tensor, output_tensor.data()); - } - continue; - } - auto * ggml_tensor = model_output_it->second; - std::memcpy(ggml_tensor->data, output_tensor.data(), output_tensor.get_byte_size()); - } - return GGML_STATUS_SUCCESS; -} - -namespace { -template void set_zero_diagonal(std::vector & matrix, size_t rows, size_t cols, T zero_value = T{}) { - for (size_t i = 0; i < rows; ++i) { - size_t diag_col = std::min(i, cols - 1); - matrix[i * cols + diag_col] = zero_value; - } -} - -ov::Tensor make_contiguous_split_input_tensor(std::shared_ptr ggml_decoder, - const struct ggml_tensor * ggml_tensor, - const ov::Shape & input_shape) { - const size_t element_size = ggml_type_size(ggml_tensor->type); - const size_t block_size = ggml_blck_size(ggml_tensor->type); - - GGML_ASSERT(block_size == 1 && "non-contiguous split inputs must be plain element types"); - - const struct ggml_tensor * source_tensor = ggml_tensor->view_src != nullptr ? ggml_tensor->view_src : ggml_tensor; - const size_t source_offset = ggml_tensor->view_src != nullptr ? ggml_tensor->view_offs : 0; - - std::vector source_data(ggml_nbytes(source_tensor)); - ggml_backend_tensor_get(source_tensor, source_data.data(), 0, source_data.size()); - - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - auto * dst = static_cast(input_tensor.data()); - size_t dst_offset = 0; - - for (size_t i3 = 0; i3 < static_cast(ggml_tensor->ne[3]); ++i3) { - for (size_t i2 = 0; i2 < static_cast(ggml_tensor->ne[2]); ++i2) { - for (size_t i1 = 0; i1 < static_cast(ggml_tensor->ne[1]); ++i1) { - for (size_t i0 = 0; i0 < static_cast(ggml_tensor->ne[0]); ++i0) { - const size_t src_offset = source_offset + i3 * ggml_tensor->nb[3] + i2 * ggml_tensor->nb[2] + - i1 * ggml_tensor->nb[1] + i0 * ggml_tensor->nb[0]; - std::memcpy(dst + dst_offset, source_data.data() + src_offset, element_size); - dst_offset += element_size; - } - } - } - } - - return input_tensor; -} - -ov::Tensor convert_ggml_input_to_ov(std::shared_ptr ggml_decoder, const std::string & name) { - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(name); - - if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, name, ggml_tensor)) { - return *sliced; - } - - if (ggml_tensor->extra != nullptr && !ggml_decoder->is_splited_model()) { - auto * extra_base = static_cast(ggml_tensor->extra); - if (extra_base->type == ggml_openvino_extra_base::Type::TENSOR) { - // GGML_LOG_DEBUG("Using ggml_tensor->extra as ov::Tensor for input: %s\n", name.c_str()); - auto * tensor_extra = static_cast(extra_base); - return *tensor_extra->tensor; - } - } - - // GGML_LOG_DEBUG("Converting ggml tensor to ov::Tensor for input: %s\n", name.c_str()); - auto * input_data = ggml_tensor->data; - ov::Shape input_shape; - if (ggml_tensor->op == GGML_OP_VIEW && !ggml_decoder->is_splited_model()) { - // This case is added to make test-backend-ops work - input_shape = ggml_decoder->get_shape(ggml_tensor->view_src); - } else { - input_shape = ggml_decoder->get_shape(ggml_tensor); - } - - if (ggml_decoder->is_splited_model() && !ggml_is_contiguous(ggml_tensor)) { - return make_contiguous_split_input_tensor(ggml_decoder, ggml_tensor, input_shape); + if (ggml_openvino_getenv_int("GGML_OPENVINO_PROFILING")) { + GGML_LOG_INFO("\nGGML OpenVINO Backend: \n"); + GGML_LOG_INFO(" - Graph decoder time: %.3f ms \n", (decoder_end_time - start_time) / 1000.0); + if (!cache_hit) { + GGML_LOG_INFO(" - Graph conversion time: %.3f ms \n", (conversion_end_time - decoder_end_time) / 1000.0); + GGML_LOG_INFO(" - Graph compile time: %.3f ms \n", (compile_end_time - conversion_end_time) / 1000.0); + } + GGML_LOG_INFO(" - Graph inference time: %.3f ms \n", (infer_end_time - compile_end_time) / 1000.0); + GGML_LOG_INFO(" - OV raw infer time: %.3f ms \n", ov_raw_infer_total / 1000.0); } - auto input_tensor = ov::Tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape, input_data); - return input_tensor; + return GGML_STATUS_SUCCESS; } } // namespace -ov::Tensor get_ov_input_tensor(std::shared_ptr ggml_decoder, const std::string & param_name) { - ov::Tensor input_tensor; - auto extra_input = ggml_decoder->get_model_extra_inputs().find(param_name); - if (extra_input != ggml_decoder->get_model_extra_inputs().end()) { - input_tensor = ov::Tensor(extra_input->second.type, extra_input->second.shape); - *input_tensor.data() = extra_input->second.value; - } else { - input_tensor = convert_ggml_input_to_ov(ggml_decoder, param_name); - } - return input_tensor; -} - -ov::Tensor get_ov_input_tensor_static_decode(std::shared_ptr ggml_decoder, - const std::string & param_name) { - // NPU decoding stage - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); - const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); - - if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || - GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { - // IMROPE's inp_pos holds one value per t/h/w/e plane instead of a single position; - // with a single decode token the planes are still contiguous, so a flat copy works. - const int n_planes = GgmlOvDecoder::is_inp_pos(ggml_tensor, op) ? GgmlOvDecoder::get_inp_pos_n_planes(op) : 1; - assert(ggml_tensor->ne[0] == n_planes); - ov::Shape input_shape = {1, 1, 1, (size_t) n_planes}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - std::memcpy(input_tensor.data(), ggml_tensor->data, n_planes * ggml_type_size(ggml_tensor->type)); - return input_tensor; - } +// Both execution paths use two cache levels: +// 1. Reuse this backend's decoder/request via graph_key and compatibility checks. +// 2. On a local miss, look up compiled_graph_key in the shared compilation cache, +// compile if needed, then create a private request from the compiled model. +// The shared lock covers compilation and frontend cleanup, never inference. +enum ggml_status ov_graph_compute(ggml_cgraph * cgraph, ggml_backend_t backend) { + ggml_backend_openvino_context * ctx = (ggml_backend_openvino_context *) backend->context; + try { + if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_CGRAPH")) { + std::string filename = "cgraph_ov.txt"; + GgmlOvDecoder::dump_cgraph(cgraph, filename); + } - if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { - ov::Shape input_shape = {1, 1, 1, 1}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - int32_t inp_out_id = *((int32_t *) ggml_tensor->data); - assert(ggml_tensor->ne[0] == 1); - assert(inp_out_id == 0); - *input_tensor.data() = inp_out_id; - return input_tensor; - } + const auto is_static = ggml_openvino_is_npu() || ggml_openvino_getenv_int("GGML_OPENVINO_FORCE_STATIC"); - if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { - size_t context_size = ggml_decoder->get_ctx_size(); - if (ggml_tensor->type == GGML_TYPE_F16) { - std::vector padded_data = - pad_input(ggml_tensor, 1, context_size, GGML_FP32_TO_FP16(-INFINITY)); - ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, 1, context_size}); - std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); - return input_tensor; - } + GGML_ASSERT(ctx->runtime_context != nullptr); + std::shared_ptr r_ctx = std::static_pointer_cast(ctx->runtime_context); + std::lock_guard execution_lock(r_ctx->execution_mutex); - std::vector padded_data = pad_input(ggml_tensor, 1, context_size, -INFINITY); - ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, 1, context_size}); - auto * data_ptr = input_tensor.data(); - std::copy(padded_data.begin(), padded_data.begin() + context_size, data_ptr); - return input_tensor; + return is_static ? ov_graph_compute_static(cgraph, r_ctx) : ov_graph_compute_dynamic(cgraph, r_ctx); + } catch (const ov::Exception & e) { + GGML_LOG_ERROR("GGML OpenVINO backend ov::Exception: %s\n", e.what()); + return GGML_STATUS_FAILED; + } catch (const std::exception & e) { + GGML_LOG_ERROR("GGML OpenVINO backend std::exception: %s\n", e.what()); + return GGML_STATUS_FAILED; + } catch (...) { + GGML_LOG_ERROR("GGML OpenVINO backend unknown exception\n"); + return GGML_STATUS_FAILED; } - - return get_ov_input_tensor(ggml_decoder, param_name); } -ov::Tensor get_ov_input_tensor_static_prefill(std::shared_ptr ggml_decoder, - const std::string & param_name, - int chunk_index) { - // NPU prompt processing stage - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); - const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); +// Detect whether a cgraph is a split subgraph or not. +// Step 1 compares each node's recorded use_count with actual fan-out references in node->src. +// Step 2 verifies that node inputs come from model nodes/weights/leafs; external sources imply split. +bool is_model_splitted(ggml_cgraph * cgraph) { + static const bool fallback_enabled = ggml_openvino_getenv_int("GGML_OPENVINO_ENABLE_FALLBACK") != 0; + if (!fallback_enabled) { + return false; + } - const size_t input_len = ggml_decoder->get_input_len(); - const size_t chunk_size = ggml_decoder->m_prefill_chunk_size; - const size_t chunk_valid_size = std::min(chunk_size, input_len - chunk_index * chunk_size); - const size_t chunk_pad_size = chunk_size - chunk_valid_size; + // Backend op tests execute each node through ggml_graph_view(), which preserves the original + // graph use_counts while exposing only one node. Treat those single-node views as regular + // naive graphs so intermediate ops do not look like split-model fragments. + if (cgraph->n_nodes <= 1 && cgraph->n_leafs == 0) { + return false; + } - if (GgmlOvDecoder::is_inp_pos(ggml_tensor, op) && GgmlOvDecoder::get_inp_pos_n_planes(op) > 1) { - // IMROPE: inp_pos stacks n_planes (t/h/w/e) position planes, each of length - // input_len; pad every plane independently so they stay aligned to chunk_size. - const int n_planes = GgmlOvDecoder::get_inp_pos_n_planes(op); - const size_t element_size = ggml_type_size(ggml_tensor->type); - ov::Shape input_shape = {1, 1, 1, (size_t) n_planes * chunk_size}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - for (int p = 0; p < n_planes; p++) { - const char * src = - (const char *) ggml_tensor->data + (p * input_len + chunk_index * chunk_size) * element_size; - char * dst = (char *) input_tensor.data() + p * chunk_size * element_size; - std::memcpy(dst, src, chunk_valid_size * element_size); - if (chunk_pad_size > 0) { - if (ggml_tensor->type == GGML_TYPE_I32) { - int32_t last_value = *((const int32_t *) src + chunk_valid_size - 1); - int32_t * out = (int32_t *) dst; - std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); - } else if (ggml_tensor->type == GGML_TYPE_I64) { - int64_t last_value = *((const int64_t *) src + chunk_valid_size - 1); - int64_t * out = (int64_t *) dst; - std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); - } else { - throw std::runtime_error("Unexpected tensor type for " + param_name); + // check the nodes of the model are used by the following nodes, through compare the node's use count and the count of nodes that use it as input. If does not match, return true, else return false. + for (int i = 0; i < cgraph->n_nodes; i++) { + ggml_tensor * node = cgraph->nodes[i]; + int use_count = cgraph->use_counts[ggml_hash_find(&cgraph->visited_hash_set, node)]; + // TODO: this is a workround for the tests case from llama.cpp, fix should from the root cause in the future. + if ((cgraph->n_nodes <= 1 && use_count == 0) || + (cgraph->n_nodes <= 1 && node->op == GGML_OP_VIEW && use_count == 1 && node->src[0] != nullptr && + node->src[0]->op == GGML_OP_NONE)) { + return false; + } + if (cgraph->n_nodes == 1 && + (cgraph->nodes[0]->op == GGML_OP_TRANSPOSE || cgraph->nodes[0]->op == GGML_OP_PERMUTE)) { + return false; + } + int input_use_count = 0; + for (int j = 0; j < cgraph->n_nodes; j++) { + ggml_tensor * other_node = cgraph->nodes[j]; + for (int k = 0; k < GGML_MAX_SRC; k++) { + if (other_node->src[k] == node) { + input_use_count++; } } } - return input_tensor; + if (use_count != input_use_count && node->op != GGML_OP_NONE) { + return true; + } } - - if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || - GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { - ov::Shape input_shape = {1, 1, 1, chunk_size}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - // copy the chunk_index-th chunk from ggml_tensor - size_t element_size = ggml_type_size(ggml_tensor->type); - void * input_data = (char *) ggml_tensor->data + chunk_index * chunk_size * element_size; - std::memcpy(input_tensor.data(), input_data, chunk_valid_size * element_size); - // pad the rest with last_value + 1, so that kv's of padded positions are inserted - // to the next row after the valids row in the kvcache - if (chunk_pad_size > 0) { - if (ggml_tensor->type == GGML_TYPE_I32) { - int32_t last_value = - *((int32_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); - int32_t * output_data = input_tensor.data(); - std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); - } else if (ggml_tensor->type == GGML_TYPE_I64) { - int64_t last_value = - *((int64_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); - int64_t * output_data = input_tensor.data(); - std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); - } else { - throw std::runtime_error("Unexpected tensor type for " + param_name); - } + // if all nodes's src node's src is not come from the nodes in the model, we think the model is splitted. This is a complementary check for the above check, because for some special case like the output node is not used by any node, the use count and input use count are both 0, we can not determine whether the model is splitted or not just based on the first check. + // Only weight-name membership is needed below. With GGML_OPENVINO_REDUCE_COMPILE_MEM + // use the name-only collector (no weight extraction); otherwise keep the original + // behavior of building (naive) weight nodes and take their names. + std::set model_weights; + if (ggml_openvino_reduce_compile_mem_enabled()) { + model_weights = GgmlOvDecoder::collect_weight_names(cgraph); + } else { + for (const auto & kv : GgmlOvDecoder::create_weight_nodes(cgraph, true)) { + model_weights.insert(kv.first); } - return input_tensor; } - - if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { - size_t output_len = ggml_decoder->get_compute_params().output_len; - ov::Shape input_shape = {1, 1, 1, output_len}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - if (ggml_tensor->ne[0] == 0) { - *input_tensor.data() = 0; - } else { - auto * data_addr = input_tensor.data(); - for (size_t i = 0; i < output_len; i++) { - data_addr[i] = ((int32_t *) ggml_tensor->data)[i] % chunk_size; + std::set model_nodes(cgraph->nodes, cgraph->nodes + cgraph->n_nodes); + // leaf nodes + std::set model_leafs(cgraph->leafs, cgraph->leafs + cgraph->n_leafs); + for (int i = 0; i < cgraph->n_nodes; i++) { + ggml_tensor * node = cgraph->nodes[i]; + for (int j = 0; j < GGML_MAX_SRC; j++) { + ggml_tensor * src = node->src[j]; + // the src is also not the model weights, we think the model is splitted. + // the src is also not in model leafs, we think the model is splitted. + if (src != nullptr && model_nodes.find(src) == model_nodes.end() && + model_weights.find(std::string(src->name)) == model_weights.end() && !model_leafs.empty() == false && + model_leafs.find(src) == model_leafs.end()) { + if (GgmlOvDecoder::is_inp_tok(src, node)) { + return false; + } + return true; } } - return input_tensor; } + return false; +} - if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { - size_t cols = ggml_tensor->ne[0]; - size_t rows = ggml_tensor->ne[1]; - size_t chunk_valid_rows = std::min(chunk_size, rows - chunk_index * chunk_size); - size_t context_size = ggml_decoder->get_ctx_size(); - if (ggml_tensor->type == GGML_TYPE_F16) { - const auto * ggml_data = - static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; - std::vector padded_data = pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, - context_size, GGML_FP32_TO_FP16(-INFINITY)); - set_zero_diagonal(padded_data, chunk_size, context_size, GGML_FP32_TO_FP16(0.0f)); - ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, chunk_size, context_size}); - std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); - return input_tensor; +bool is_naive(ggml_cgraph * cgraph) { + constexpr int naive_graph_size_threshold = 20; + int count = 0; + for (int i = 0; i < cgraph->n_nodes; i++) { + if (cgraph->nodes[i]->op != GGML_OP_NONE) { + count++; } - - const auto * ggml_data = static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; - std::vector padded_data = - pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, context_size, -INFINITY); - set_zero_diagonal(padded_data, chunk_size, context_size); - ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, chunk_size, context_size}); - auto * data_ptr = input_tensor.data(); - std::copy(padded_data.begin(), padded_data.begin() + chunk_size * context_size, data_ptr); - return input_tensor; } - - return get_ov_input_tensor(ggml_decoder, param_name); + return count < naive_graph_size_threshold; } size_t checksum(const void * data, size_t size) { @@ -1303,15 +1645,15 @@ bool save_ggml_tensor_data_to_txt(const ggml_tensor * tensor, const std::string void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor) { std::cout << "Input name: " << name << ", Input shape: " << tensor.get_shape() << ", Address: " << tensor.data() - << std::endl; + << '\n'; switch (tensor.get_element_type()) { case ov::element::f32: { - if (name.find("self_kq_mask") == std::string::npos) { - std::cout << *(tensor.data()) << std::endl; + if (name.find("self_kq_mask") == std::string::npos && name.find("KQ_mask") == std::string::npos) { + std::cout << *(tensor.data()) << '\n'; } else { size_t rows = tensor.get_shape()[2]; size_t cols = tensor.get_shape()[3]; - auto * data = tensor.data(); + const float * data = tensor.data(); for (size_t i = 0; i < rows; ++i) { for (size_t j = 0; j < cols; ++j) { float val = data[i * cols + j]; @@ -1321,26 +1663,26 @@ void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor std::cout << std::setw(5) << val; } } - std::cout << std::endl; + std::cout << '\n'; } } break; } case ov::element::f16: - std::cout << *(tensor.data()) << std::endl; + std::cout << *(tensor.data()) << '\n'; break; case ov::element::i32: for (size_t i = 0; i < tensor.get_size(); ++i) { - std::cout << tensor.data()[i] << " "; + std::cout << tensor.data()[i] << ' '; } - std::cout << std::endl; + std::cout << '\n'; break; case ov::element::i64: for (size_t i = 0; i < tensor.get_size(); ++i) { - std::cout << tensor.data()[i] << " "; + std::cout << tensor.data()[i] << ' '; } - std::cout << std::endl; + std::cout << '\n'; break; default: break; @@ -1349,7 +1691,7 @@ void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor void print_output_tensor_info(const std::string & name, const ov::Tensor & tensor, const void * output_dst) { std::cout << "Output name: " << name << ", Output shape: " << tensor.get_shape() << ", Address: " << output_dst - << std::endl; + << '\n'; auto print_float_stats = [](const std::string & type_name, size_t size, auto get_value) { if (size == 0) { @@ -1363,20 +1705,16 @@ void print_output_tensor_info(const std::string & name, const ov::Tensor & tenso for (size_t i = 1; i < size; ++i) { float v = get_value(i); - if (v < min) { - min = v; - } - if (v > max) { - max = v; - } + min = std::min(v, min); + max = std::max(v, max); sum += v; } double mean = sum / size; std::cout << std::right << std::setw(6) << type_name << std::right << std::setw(12) << "First" << std::setw(12) - << "Min" << std::setw(12) << "Max" << std::setw(12) << "Mean" << std::endl; + << "Min" << std::setw(12) << "Max" << std::setw(12) << "Mean" << '\n'; std::cout << std::right << std::setw(6) << "" << std::right << std::setw(12) << first << std::setw(12) << min - << std::setw(12) << max << std::setw(12) << mean << std::endl; + << std::setw(12) << max << std::setw(12) << mean << '\n'; }; switch (tensor.get_element_type()) { @@ -1414,8 +1752,22 @@ const ggml_tensor * get_inp_pos_tensor(ggml_cgraph * cgraph) { throw std::runtime_error("get_inp_pos_tensor: inp_pos not found in cgraph"); } -bool get_is_prefill(const ggml_tensor * inp_pos) { - return inp_pos->ne[0] > 1; +int64_t get_inp_pos_n_tokens(ggml_cgraph * cgraph, const ggml_tensor * inp_pos) { + // IMROPE stacks n_planes (t/h/w/e) position planes into inp_pos, so ne[0] is + // n_planes * n_tokens. Callers that need a token count must divide the planes out. + int n_planes = 1; + for (int i = 0; i < cgraph->n_nodes; ++i) { + auto * op = cgraph->nodes[i]; + for (int j = 0; j < GGML_MAX_SRC; ++j) { + if (op->src[j] == inp_pos) { + n_planes = GgmlOvDecoder::get_inp_pos_n_planes(op); + break; + } + } + } + return inp_pos->ne[0] / n_planes; } -#pragma GCC diagnostic pop +bool get_is_prefill(ggml_cgraph * cgraph, const ggml_tensor * inp_pos) { + return get_inp_pos_n_tokens(cgraph, inp_pos) > 1; +} diff --git a/ggml/src/ggml-openvino/utils.h b/ggml/src/ggml-openvino/utils.h index 513fa83c..74c25f0a 100644 --- a/ggml/src/ggml-openvino/utils.h +++ b/ggml/src/ggml-openvino/utils.h @@ -2,7 +2,6 @@ #include "ggml-impl.h" #include -#include #include #include #include @@ -14,6 +13,8 @@ #include #include +// Local execution-cache key. A match still needs the ModelParams compatibility +// check; this key alone does not identify weights or a compiled model. struct graph_key { int n_nodes; std::string first_node_name; @@ -26,14 +27,13 @@ struct graph_key { last_node_name = cgraph->nodes[n_nodes - 1]->name; } - auto get_input_key_name = [](const ggml_cgraph * graph, const ggml_tensor * tensor) { - std::string name = tensor->name; - const size_t hash_pos = ggml_hash_find(&graph->visited_hash_set, tensor); - if (((tensor->flags & GGML_TENSOR_FLAG_COMPUTE) || GgmlOvDecoder::is_kvcache(tensor, nullptr)) && - hash_pos != GGML_HASHSET_FULL && ggml_bitset_get(graph->visited_hash_set.used, hash_pos)) { - name += "#" + std::to_string(hash_pos); + std::unordered_map names; + auto get_input_key_name = [&names](const ggml_cgraph * graph, const ggml_tensor * tensor) { + auto it = names.find(tensor); + if (it == names.end()) { + it = names.emplace(tensor, GgmlOvDecoder::get_tensor_name(graph, tensor)).first; } - return name; + return it->second; }; std::vector node_names; @@ -90,7 +90,27 @@ struct decoder_runtime_ctx { std::shared_ptr ptr; }; +struct ov_compiled_graph { + ov::CompiledModel decode; + ov::CompiledModel prefill; + std::vector input_names; + std::vector output_names; +}; + +// Only compilation and cache publication use this mutex. Requests, decoders and +// sequence state belong to individual backend contexts and never enter this cache. +struct ov_compiled_model_cache { + std::mutex mutex; + std::unordered_map graphs; + size_t backend_count = 0; +}; + +// Private to one backend instance. Only compiled_cache is shared with other +// instances; clearing these local caches cannot invalidate their requests. struct ov_runtime_context { + // Serializes calls on this backend only, not inference in other contexts. + std::mutex execution_mutex; + std::shared_ptr compiled_cache; mutable std::mutex ctx_mutex; std::string device; bool stateful; @@ -99,13 +119,10 @@ struct ov_runtime_context { std::unordered_map, graph_key_hash> infer_request_cache_prefill; std::unordered_map, graph_key_hash> ov_input_names_cache; std::unordered_map, graph_key_hash> ov_output_names_cache; - //TODO: Stateful is only supported for single request at a time. - // Simultanous stateful inference request support to be added. size_t stateful_kv_size; std::map kv_state_input_name_map; - std::atomic backend_count; - ov_runtime_context() : device("CPU"), stateful(false), stateful_kv_size(0), backend_count(0) {} + ov_runtime_context() : device("CPU"), stateful(false), stateful_kv_size(0) {} void clear_caches_locked() { decoder_cache.clear(); @@ -125,9 +142,6 @@ struct ov_runtime_context { enum ggml_status ov_graph_compute(struct ggml_cgraph * cgraph, ggml_backend_t backend); -enum ggml_status ov_graph_compute_dynamic(struct ggml_cgraph * cgraph, std::shared_ptr r_ctx); -enum ggml_status ov_graph_compute_static(struct ggml_cgraph * cgraph, std::shared_ptr r_ctx); - size_t checksum(const void * data, size_t size); bool save_ggml_tensor_data_to_txt(const ggml_tensor * tensor, const std::string & file_path); @@ -164,19 +178,9 @@ std::vector pad_input(const ggml_tensor * tensor, size_t padded_rows, size_t const ggml_tensor * get_inp_pos_tensor(struct ggml_cgraph * cgraph); -bool get_is_prefill(const ggml_tensor * inp_pos); - -ov::Tensor get_ov_input_tensor(std::shared_ptr ggml_decoder, const std::string & param_name); -ov::Tensor get_ov_input_tensor_static_decode(std::shared_ptr ggml_decoder, - const std::string & param_name); -ov::Tensor get_ov_input_tensor_static_prefill(std::shared_ptr ggml_decoder, - const std::string & param_name, - int chunk_index); +int64_t get_inp_pos_n_tokens(struct ggml_cgraph * cgraph, const ggml_tensor * inp_pos); -ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, - std::shared_ptr infer_request, - int output_index, - const ggml_tensor * ggml_tensor); +bool get_is_prefill(struct ggml_cgraph * cgraph, const ggml_tensor * inp_pos); bool is_naive(struct ggml_cgraph * cgraph); @@ -186,8 +190,3 @@ bool is_naive(struct ggml_cgraph * cgraph); * @return true if the graph is identified as split; otherwise false. */ bool is_model_splitted(struct ggml_cgraph * cgraph); - -enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config); diff --git a/ggml/src/ggml-quants.c b/ggml/src/ggml-quants.c index 1ebc50a7..55db802c 100644 --- a/ggml/src/ggml-quants.c +++ b/ggml/src/ggml-quants.c @@ -4771,80 +4771,51 @@ static void quantize_row_iq1_m_impl(const float * GGML_RESTRICT x, void * GGML_R // 1: +, - // 2: -, + // 3: -, - - for (int i1 = 0; i1 <= block_size; ++i1) { - for (int i2 = i1; i2 <= block_size; ++i2) { - memset(sumqx, 0, 4*sizeof(float)); - memset(sumq2, 0, 4*sizeof(float)); - for (int j = 0; j < i1; ++j) { - int i = idx[2*j]; - if (i < block_size/2) { - sumqx[0] += weight[i]*x_p[0]*xb[i]; - sumqx[1] += weight[i]*x_p[0]*xb[i]; - sumqx[2] += weight[i]*x_m[0]*xb[i]; - sumqx[3] += weight[i]*x_m[0]*xb[i]; - sumq2[0] += weight[i]*x_p[0]*x_p[0]; - sumq2[1] += weight[i]*x_p[0]*x_p[0]; - sumq2[2] += weight[i]*x_m[0]*x_m[0]; - sumq2[3] += weight[i]*x_m[0]*x_m[0]; - } else { - sumqx[0] += weight[i]*x_p[0]*xb[i]; - sumqx[2] += weight[i]*x_p[0]*xb[i]; - sumqx[1] += weight[i]*x_m[0]*xb[i]; - sumqx[3] += weight[i]*x_m[0]*xb[i]; - sumq2[0] += weight[i]*x_p[0]*x_p[0]; - sumq2[2] += weight[i]*x_p[0]*x_p[0]; - sumq2[1] += weight[i]*x_m[0]*x_m[0]; - sumq2[3] += weight[i]*x_m[0]*x_m[0]; - } - } - for (int j = i1; j < i2; ++j) { - int i = idx[2*j]; - if (i < block_size/2) { - sumqx[0] += weight[i]*x_p[1]*xb[i]; - sumqx[1] += weight[i]*x_p[1]*xb[i]; - sumqx[2] += weight[i]*x_m[1]*xb[i]; - sumqx[3] += weight[i]*x_m[1]*xb[i]; - sumq2[0] += weight[i]*x_p[1]*x_p[1]; - sumq2[1] += weight[i]*x_p[1]*x_p[1]; - sumq2[2] += weight[i]*x_m[1]*x_m[1]; - sumq2[3] += weight[i]*x_m[1]*x_m[1]; - } else { - sumqx[0] += weight[i]*x_p[1]*xb[i]; - sumqx[2] += weight[i]*x_p[1]*xb[i]; - sumqx[1] += weight[i]*x_m[1]*xb[i]; - sumqx[3] += weight[i]*x_m[1]*xb[i]; - sumq2[0] += weight[i]*x_p[1]*x_p[1]; - sumq2[2] += weight[i]*x_p[1]*x_p[1]; - sumq2[1] += weight[i]*x_m[1]*x_m[1]; - sumq2[3] += weight[i]*x_m[1]*x_m[1]; - } - } - for (int j = i2; j < block_size; ++j) { - int i = idx[2*j]; - if (i < block_size/2) { - sumqx[0] += weight[i]*x_p[2]*xb[i]; - sumqx[1] += weight[i]*x_p[2]*xb[i]; - sumqx[2] += weight[i]*x_m[2]*xb[i]; - sumqx[3] += weight[i]*x_m[2]*xb[i]; - sumq2[0] += weight[i]*x_p[2]*x_p[2]; - sumq2[1] += weight[i]*x_p[2]*x_p[2]; - sumq2[2] += weight[i]*x_m[2]*x_m[2]; - sumq2[3] += weight[i]*x_m[2]*x_m[2]; - } else { - sumqx[0] += weight[i]*x_p[2]*xb[i]; - sumqx[2] += weight[i]*x_p[2]*xb[i]; - sumqx[1] += weight[i]*x_m[2]*xb[i]; - sumqx[3] += weight[i]*x_m[2]*xb[i]; - sumq2[0] += weight[i]*x_p[2]*x_p[2]; - sumq2[2] += weight[i]*x_p[2]*x_p[2]; - sumq2[1] += weight[i]*x_m[2]*x_m[2]; - sumq2[3] += weight[i]*x_m[2]*x_m[2]; + // prefix sums are kept per half of the block because each half can use a different sign (x_p or x_m) + // since v[0]-v[1] = v[1]-v[2] = -1 for both x_p and x_m, the 3-group sum for a split collapses to T*v[2] - px[i1] - px[i2] + { + float px[2][IQ1M_BLOCK_SIZE+1]; + float pw[2][IQ1M_BLOCK_SIZE+1]; + px[0][0] = px[1][0] = 0; + pw[0][0] = pw[1][0] = 0; + for (int j = 0; j < block_size; ++j) { + const int i = idx[2*j]; + const int h = i < block_size/2 ? 0 : 1; + px[h][j+1] = px[h][j] + weight[i]*xb[i]; + px[1-h][j+1] = px[1-h][j]; + pw[h][j+1] = pw[h][j] + weight[i]; + pw[1-h][j+1] = pw[1-h][j]; + } + const float txs[2] = {px[0][block_size], px[1][block_size]}; // total weight*x per half + const float tws[2] = {pw[0][block_size], pw[1][block_size]}; // total weight per half + const float p2 = x_p[2], m2 = x_m[2]; + const float cp1 = x_p[0]*x_p[0] - x_p[1]*x_p[1]; + const float cp2 = x_p[1]*x_p[1] - x_p[2]*x_p[2]; + const float cm1 = x_m[0]*x_m[0] - x_m[1]*x_m[1]; + const float cm2 = x_m[1]*x_m[1] - x_m[2]*x_m[2]; + for (int i1 = 0; i1 <= block_size; ++i1) { + for (int i2 = i1; i2 <= block_size; ++i2) { + float qx_p[2], qx_m[2], q2_p[2], q2_m[2]; + for (int h = 0; h < 2; ++h) { + const float sx = px[h][i1] + px[h][i2]; + qx_p[h] = txs[h]*p2 - sx; + qx_m[h] = txs[h]*m2 - sx; + q2_p[h] = tws[h]*p2*p2 + pw[h][i1]*cp1 + pw[h][i2]*cp2; + q2_m[h] = tws[h]*m2*m2 + pw[h][i1]*cm1 + pw[h][i2]*cm2; } - } - for (int k = 0; k < 4; ++k) { - if (sumq2[k] > 0 && sumqx[k]*sumqx[k] > best_score*sumq2[k]) { - scale = sumqx[k]/sumq2[k]; best_score = scale*sumqx[k]; - besti1 = i1; besti2 = i2; best_k = k; + sumqx[0] = qx_p[0] + qx_p[1]; + sumqx[1] = qx_p[0] + qx_m[1]; + sumqx[2] = qx_m[0] + qx_p[1]; + sumqx[3] = qx_m[0] + qx_m[1]; + sumq2[0] = q2_p[0] + q2_p[1]; + sumq2[1] = q2_p[0] + q2_m[1]; + sumq2[2] = q2_m[0] + q2_p[1]; + sumq2[3] = q2_m[0] + q2_m[1]; + for (int k = 0; k < 4; ++k) { + if (sumq2[k] > 0 && sumqx[k]*sumqx[k] > best_score*sumq2[k]) { + scale = sumqx[k]/sumq2[k]; best_score = scale*sumqx[k]; + besti1 = i1; besti2 = i2; best_k = k; + } } } } diff --git a/ggml/src/ggml-rpc/CMakeLists.txt b/ggml/src/ggml-rpc/CMakeLists.txt index 40e11fea..e3d0c9b4 100644 --- a/ggml/src/ggml-rpc/CMakeLists.txt +++ b/ggml/src/ggml-rpc/CMakeLists.txt @@ -9,10 +9,18 @@ if (WIN32) target_link_libraries(ggml-rpc PRIVATE ws2_32) endif() -# RDMA auto-detection (Linux only, requires libibverbs) -if (NOT WIN32 AND NOT APPLE) - find_library(IBVERBS_LIB ibverbs) - if (IBVERBS_LIB) +# RDMA auto-detection: Linux RoCE/IB via libibverbs, Apple RDMA-over-Thunderbolt via librdma +if (APPLE) + set(RDMA_LIB_NAME rdma) + set(RDMA_DESC "Apple RDMA-over-Thunderbolt, UC") +elseif (NOT WIN32) + set(RDMA_LIB_NAME ibverbs) + set(RDMA_DESC "auto-detected") +endif() + +if (RDMA_LIB_NAME) + find_library(RDMA_LIB ${RDMA_LIB_NAME}) + if (RDMA_LIB) option(GGML_RPC_RDMA "ggml: enable RDMA transport for RPC" ON) else() option(GGML_RPC_RDMA "ggml: enable RDMA transport for RPC" OFF) @@ -22,12 +30,22 @@ else() endif() if (GGML_RPC_RDMA) - if (NOT IBVERBS_LIB) - find_library(IBVERBS_LIB ibverbs REQUIRED) + if (NOT RDMA_LIB) + find_library(RDMA_LIB ${RDMA_LIB_NAME} REQUIRED) endif() target_compile_definitions(ggml-rpc PRIVATE GGML_RPC_RDMA) - target_link_libraries(ggml-rpc PRIVATE ${IBVERBS_LIB}) - message(STATUS " RDMA transport enabled (auto-detected)") + if (APPLE) + # librdma.dylib only exists on macOS 26.2 and later. Link it weakly so a build made + # where it exists still loads where it does not; checked at runtime before use + # but with BUILD_SHARED_LIBS=OFF ggml-rpc is a static archive and never links + # so the librdma symbols used by transport-apple.cpp stay undefined. + target_link_options(ggml-rpc PUBLIC "LINKER:-weak_library,${RDMA_LIB}") + target_compile_definitions(ggml-rpc PRIVATE GGML_RPC_RDMA_APPLE) + target_sources(ggml-rpc PRIVATE transport-apple.cpp) + else() + target_link_libraries(ggml-rpc PRIVATE ${RDMA_LIB}) + endif() + message(STATUS " RDMA transport enabled (${RDMA_DESC})") else() message(STATUS " RDMA transport disabled") endif() diff --git a/ggml/src/ggml-rpc/ggml-rpc.cpp b/ggml/src/ggml-rpc/ggml-rpc.cpp index e9de0d0a..c24caad7 100644 --- a/ggml/src/ggml-rpc/ggml-rpc.cpp +++ b/ggml/src/ggml-rpc/ggml-rpc.cpp @@ -9,6 +9,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -17,6 +20,8 @@ #include #include #include +#include +#include static const char * RPC_DEBUG = std::getenv("GGML_RPC_DEBUG"); @@ -47,7 +52,7 @@ struct rpc_tensor { uint64_t data; char name[GGML_MAX_NAME]; - char padding[4]; + int32_t use_count; }; static_assert(sizeof(rpc_tensor) % 8 == 0, "rpc_tensor size must be multiple of 8"); @@ -72,6 +77,7 @@ enum rpc_cmd { RPC_CMD_DEVICE_COUNT, RPC_CMD_GRAPH_RECOMPUTE, RPC_CMD_MEMSET_TENSOR, + RPC_CMD_NONE, RPC_CMD_COUNT, }; @@ -223,24 +229,24 @@ struct ggml_backend_rpc_buffer_type_context { size_t max_size; }; +class rpc_dispatcher; struct ggml_backend_rpc_context { - std::string endpoint; - uint32_t device; - std::string name; + std::shared_ptr dispatcher; + uint32_t device; + std::string name; }; struct ggml_backend_rpc_buffer_context { - std::shared_ptr sock; - void * base_ptr; - uint64_t remote_ptr; + std::shared_ptr dispatcher; + void * base_ptr; + uint64_t remote_ptr; }; // RPC helper functions // Computes FNV-1a hash of the data -static uint64_t fnv_hash(const uint8_t * data, size_t len) { +static uint64_t fnv_hash(const uint8_t * data, size_t len, uint64_t hash = 0xcbf29ce484222325ULL) { const uint64_t fnv_prime = 0x100000001b3ULL; - uint64_t hash = 0xcbf29ce484222325ULL; for (size_t i = 0; i < len; ++i) { hash ^= data[i]; @@ -253,7 +259,10 @@ static bool send_msg(socket_ptr sock, const void * msg, size_t msg_size) { if (!sock->send_data(&msg_size, sizeof(msg_size))) { return false; } - return sock->send_data(msg, msg_size); + if (!sock->send_data(msg, msg_size)) { + return false; + } + return sock->flush(); } static bool recv_msg(socket_ptr sock, void * msg, size_t msg_size) { @@ -308,7 +317,7 @@ static bool send_rpc_cmd(socket_ptr sock, enum rpc_cmd cmd, const void * input, if (!sock->send_data(input, input_size)) { return false; } - return true; + return sock->flush(); } // RPC request : | rpc_cmd (1 byte) | request_size (8 bytes) | request_data (request_size bytes) | @@ -354,44 +363,248 @@ static bool negotiate_hello(const std::shared_ptr & sock) { return true; } -static std::shared_ptr get_socket(const std::string & endpoint) { - static std::mutex mutex; - std::lock_guard lock(mutex); - static std::unordered_map> sockets; +template +class message_queue { +public: + message_queue() {} + + bool push(const T &value) { + std::unique_lock lock(mutex); + if (interrupted) { + return false; + } + queue.push(value); + cvar.notify_all(); + return true; + } - auto it = sockets.find(endpoint); - if (it != sockets.end()) { - if (auto sock = it->second.lock()) { - return sock; + bool pop(T* out) { + std::unique_lock lock(mutex); + cvar.wait(lock, [this] { return !queue.empty() || interrupted; }); + if (interrupted) { + return false; } + *out = queue.front(); + queue.pop(); + return true; + } + + void interrupt() { + std::unique_lock lock(mutex); + interrupted = true; + lock.unlock(); + cvar.notify_all(); } + +private: + bool interrupted = false; + std::queue queue; + std::mutex mutex; + std::condition_variable cvar; +}; + +class rpc_dispatcher { +public: + rpc_dispatcher() { + } + + void send(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size); + void send(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size, void * output, size_t output_size); + void send_async(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size); + void send_async(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size, void * output, size_t output_size); + + ggml_backend_event_t event_new(ggml_backend_dev_t dev); + void event_free(ggml_backend_event_t event); + void event_synchronize(ggml_backend_event_t event); + void event_record(ggml_backend_event_t event); + void synchronize(); + + void start(const std::string & endpoint); + void work(); + + ~rpc_dispatcher(); + +private: + struct rpc_msg { + rpc_cmd cmd; + std::shared_ptr input; + size_t input_size; + void * output; + size_t output_size; + std::promise completion; + }; + using rpc_msg_ptr = std::shared_ptr; + using rpc_msg_queue = message_queue; + struct rpc_event { + rpc_msg_ptr msg; + std::shared_future sf; + }; + rpc_msg_queue queue; + socket_ptr sock; + std::atomic_bool running; + std::thread thread; +}; + +static void rpc_dispatcher_trampoline(rpc_dispatcher * dispatcher) +{ + dispatcher->work(); +} + +void rpc_dispatcher::send(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size) { + auto msg = std::make_shared(); + msg->cmd = cmd; + msg->input = input; + msg->input_size = input_size; + msg->output = nullptr; + msg->output_size = 0; + GGML_ASSERT(queue.push(msg)); + auto future = msg->completion.get_future(); + future.wait(); +} + +void rpc_dispatcher::send_async(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size) { + auto msg = std::make_shared(); + msg->cmd = cmd; + msg->input = input; + msg->input_size = input_size; + msg->output = nullptr; + msg->output_size = 0; + GGML_ASSERT(queue.push(msg)); +} + +void rpc_dispatcher::send(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size, void * output, size_t output_size) { + auto msg = std::make_shared(); + msg->cmd = cmd; + msg->input = input; + msg->input_size = input_size; + msg->output = output; + msg->output_size = output_size; + GGML_ASSERT(queue.push(msg)); + auto future = msg->completion.get_future(); + future.wait(); +} + +void rpc_dispatcher::send_async(enum rpc_cmd cmd, std::shared_ptr input, size_t input_size, void * output, size_t output_size) { + auto msg = std::make_shared(); + msg->cmd = cmd; + msg->input = input; + msg->input_size = input_size; + msg->output = output; + msg->output_size = output_size; + GGML_ASSERT(queue.push(msg)); +} + +ggml_backend_event_t rpc_dispatcher::event_new(ggml_backend_dev_t dev) { + rpc_event * ev = new rpc_event; + ev->msg = std::make_shared(); + ev->msg->cmd = RPC_CMD_NONE; + ev->sf = ev->msg->completion.get_future().share(); + GGML_ASSERT(queue.push(ev->msg)); + return new ggml_backend_event { + /* .device = */ dev, + /* .context = */ ev, + }; +} + +void rpc_dispatcher::event_free(ggml_backend_event_t event) { + rpc_event * ev = (rpc_event *)event->context; + delete ev; +} + +void rpc_dispatcher::event_synchronize(ggml_backend_event_t event) { + rpc_event * ev = (rpc_event *)event->context; + ev->sf.wait(); +} + +void rpc_dispatcher::event_record(ggml_backend_event_t event) { + rpc_event * ev = (rpc_event *)event->context; + ev->msg = std::make_shared(); + ev->msg->cmd = RPC_CMD_NONE; + ev->sf = ev->msg->completion.get_future().share(); + GGML_ASSERT(queue.push(ev->msg)); +} + +void rpc_dispatcher::synchronize() { + // to ensure all messages are processed, submit dummy message and wait for it to complete + auto msg = std::make_shared(); + msg->cmd = RPC_CMD_NONE; + GGML_ASSERT(queue.push(msg)); + msg->completion.get_future().wait(); +} + +void rpc_dispatcher::start(const std::string & endpoint) { std::string host; int port; if (!parse_endpoint(endpoint, host, port)) { - GGML_LOG_ERROR("Failed to parse endpoint: %s\n", endpoint.c_str()); - return nullptr; + GGML_ABORT("Failed to parse endpoint: %s\n", endpoint.c_str()); } - if (!rpc_transport_init()) { - return nullptr; + GGML_ABORT("RPC transport initialization failed\n"); } - auto sock = socket_t::connect(host.c_str(), port); + + sock = socket_t::connect(host.c_str(), port); if (sock == nullptr) { - return nullptr; + GGML_ABORT("Failed to connect to %s\n", endpoint.c_str()); } if (!negotiate_hello(sock)) { - return nullptr; + GGML_ABORT("RPC handshake failed for %s\n", endpoint.c_str()); } LOG_DBG("[%s] connected to %s\n", __func__, endpoint.c_str()); - sockets[endpoint] = sock; - return sock; + running = true; + thread = std::thread(rpc_dispatcher_trampoline, this); +} + +void rpc_dispatcher::work() { + while (running) { + rpc_msg_ptr msg_ptr; + if (!queue.pop(&msg_ptr)) { + break; + } + if (msg_ptr->cmd != RPC_CMD_NONE) { + if (msg_ptr->output) { + bool status = send_rpc_cmd(sock, msg_ptr->cmd, msg_ptr->input.get(), msg_ptr->input_size, msg_ptr->output, msg_ptr->output_size); + RPC_STATUS_ASSERT(status); + } else { + bool status = send_rpc_cmd(sock, msg_ptr->cmd, msg_ptr->input.get(), msg_ptr->input_size); + RPC_STATUS_ASSERT(status); + } + } + msg_ptr->completion.set_value(); + } +} + +rpc_dispatcher::~rpc_dispatcher() { + running = false; + queue.interrupt(); + sock = nullptr; + if (thread.joinable()) { + thread.join(); + } +} + +static std::shared_ptr get_dispatcher(const std::string & endpoint) { + static std::mutex mutex; + std::lock_guard lock(mutex); + static std::unordered_map> dispatchers; + + auto it = dispatchers.find(endpoint); + if (it != dispatchers.end()) { + if (auto dispatcher = it->second.lock()) { + return dispatcher; + } + } + + auto dispatcher = std::make_shared(); + dispatcher->start(endpoint); + dispatchers[endpoint] = dispatcher; + return dispatcher; } static void ggml_backend_rpc_buffer_free_buffer(ggml_backend_buffer_t buffer) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - rpc_msg_free_buffer_req request = {ctx->remote_ptr}; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_FREE_BUFFER, &request, sizeof(request), nullptr, 0); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->remote_ptr = ctx->remote_ptr; + ctx->dispatcher->send(RPC_CMD_FREE_BUFFER, request, sizeof(*request)); delete ctx; } @@ -400,10 +613,10 @@ static void * ggml_backend_rpc_buffer_get_base(ggml_backend_buffer_t buffer) { if (ctx->base_ptr != nullptr) { return ctx->base_ptr; } - rpc_msg_buffer_get_base_req request = {ctx->remote_ptr}; + auto request = std::make_shared(); + request->remote_ptr = ctx->remote_ptr; rpc_msg_buffer_get_base_rsp response; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_BUFFER_GET_BASE, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + ctx->dispatcher->send(RPC_CMD_BUFFER_GET_BASE, request, sizeof(*request), &response, sizeof(response)); ctx->base_ptr = reinterpret_cast(response.base_ptr); return ctx->base_ptr; } @@ -412,7 +625,7 @@ static bool ggml_backend_buffer_is_rpc(ggml_backend_buffer_t buffer) { return buffer->iface.free_buffer == ggml_backend_rpc_buffer_free_buffer; } -static rpc_tensor serialize_tensor(const ggml_tensor * tensor) { +static rpc_tensor serialize_tensor(const ggml_tensor * tensor, const std::shared_ptr & dispatcher = nullptr) { rpc_tensor result; if (!tensor) { memset(&result, 0, sizeof(result)); @@ -424,8 +637,14 @@ static rpc_tensor serialize_tensor(const ggml_tensor * tensor) { if (tensor->buffer && ggml_backend_buffer_is_rpc(tensor->buffer)) { ggml_backend_buffer_t buffer = tensor->buffer; ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - result.buffer = ctx != nullptr ? ctx->remote_ptr : 0; - result.data = reinterpret_cast(tensor->data); + // ref: https://github.com/ggml-org/llama.cpp/pull/26500 + if (ctx != nullptr && (dispatcher == nullptr || ctx->dispatcher == dispatcher)) { + result.buffer = ctx->remote_ptr; + result.data = reinterpret_cast(tensor->data); + } else { + result.buffer = 0; + result.data = 0; + } } else { result.buffer = 0; result.data = 0; @@ -447,7 +666,7 @@ static rpc_tensor serialize_tensor(const ggml_tensor * tensor) { // Avoid sending uninitialized data over the wire memset(result.name, 0, sizeof(result.name)); - memset(result.padding, 0, sizeof(result.padding)); + result.use_count = 0; snprintf(result.name, GGML_MAX_NAME, "%s", tensor->name); return result; @@ -460,12 +679,9 @@ static enum ggml_status ggml_backend_rpc_buffer_init_tensor(ggml_backend_buffer_ // Due to bandwidth constraints, we only call the server init tensor functions if necessary. // In particular, only quantized tensors need padding if (ggml_is_quantized(tensor->type) && (tensor->ne[0] % 512 != 0) && (tensor->view_src == nullptr)) { - rpc_msg_init_tensor_req request; - - request.tensor = serialize_tensor(tensor); - - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_INIT_TENSOR, &request, sizeof(request), nullptr, 0); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->tensor = serialize_tensor(tensor); + ctx->dispatcher->send(RPC_CMD_INIT_TENSOR, request, sizeof(*request)); } return GGML_STATUS_SUCCESS; } @@ -473,50 +689,64 @@ static enum ggml_status ggml_backend_rpc_buffer_init_tensor(ggml_backend_buffer_ static void ggml_backend_rpc_buffer_memset_tensor( ggml_backend_buffer_t buffer, ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - rpc_msg_memset_tensor_req request = { - /* .tensor = */ serialize_tensor(tensor), - /* .offset = */ offset, - /* .size = */ size, - /* .value = */ value, - }; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_MEMSET_TENSOR, &request, sizeof(request), nullptr, 0); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->tensor = serialize_tensor(tensor); + request->offset = offset; + request->size = size; + request->value = value; + ctx->dispatcher->send(RPC_CMD_MEMSET_TENSOR, request, sizeof(*request)); +} + +// input serialization format: | rpc_tensor | cache_flag (1 byte) | offset (8 bytes) | data (size bytes) +static std::shared_ptr serialize_set_tensor(const rpc_tensor & rpc_tensor, uint8_t cache_flag, uint64_t offset, const void * data, size_t size, size_t & input_size) { + input_size = sizeof(rpc_tensor) + sizeof(cache_flag) + sizeof(offset) + size; + uint8_t * input = new uint8_t[input_size](); + uint8_t * p = input; + memcpy(p, &rpc_tensor, sizeof(rpc_tensor)); p += sizeof(rpc_tensor); + memcpy(p, &cache_flag, sizeof(cache_flag)); p += sizeof(cache_flag); + memcpy(p, &offset, sizeof(offset)); p += sizeof(offset); + memcpy(p, data, size); + return std::shared_ptr(input, std::default_delete()); +} + +// the hash cache is meant for weights, so that a model reload can skip re-sending them. +// compute-buffer inputs (the activations ggml_backend_sched copies between backends) must not +// take this path, otherwise with `rpc-server -c` every ubatch above the threshold is written +// to the cache directory and later served from there. +static bool rpc_use_hash_cache(const ggml_tensor * tensor, size_t size) { + return size > HASH_THRESHOLD && tensor->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS; } static void ggml_backend_rpc_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; rpc_tensor rpc_tensor = serialize_tensor(tensor); - if (size > HASH_THRESHOLD) { - rpc_msg_set_tensor_hash_req request; - request.tensor = rpc_tensor; - request.offset = offset; - request.hash = fnv_hash((const uint8_t*)data, size); + uint8_t cache_flag = 0; + if (rpc_use_hash_cache(tensor, size)) { + auto request = std::make_shared(); + request->tensor = rpc_tensor; + request->offset = offset; + request->hash = fnv_hash((const uint8_t*)data, size); rpc_msg_set_tensor_hash_rsp response; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_SET_TENSOR_HASH, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + ctx->dispatcher->send(RPC_CMD_SET_TENSOR_HASH, request, sizeof(*request), &response, sizeof(response)); if (response.result) { // the server has the same data, no need to send it return; } + // the server has no cache entry for this tensor - ask it to save one + cache_flag = 1; } - // input serialization format: | rpc_tensor | offset (8 bytes) | data (size bytes) - size_t input_size = sizeof(rpc_tensor) + sizeof(uint64_t) + size; - std::vector input(input_size, 0); - memcpy(input.data(), &rpc_tensor, sizeof(rpc_tensor)); - memcpy(input.data() + sizeof(rpc_tensor), &offset, sizeof(offset)); - memcpy(input.data() + sizeof(rpc_tensor) + sizeof(offset), data, size); - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_SET_TENSOR, input.data(), input.size()); - RPC_STATUS_ASSERT(status); + size_t input_size; + auto input = serialize_set_tensor(rpc_tensor, cache_flag, offset, data, size, input_size); + ctx->dispatcher->send(RPC_CMD_SET_TENSOR, input, input_size); } static void ggml_backend_rpc_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - rpc_msg_get_tensor_req request; - request.tensor = serialize_tensor(tensor); - request.offset = offset; - request.size = size; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_GET_TENSOR, &request, sizeof(request), data, size); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->tensor = serialize_tensor(tensor); + request->offset = offset; + request->size = size; + ctx->dispatcher->send(RPC_CMD_GET_TENSOR, request, sizeof(*request), data, size); } static bool ggml_backend_rpc_buffer_cpy_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * src, ggml_tensor * dst) { @@ -526,16 +756,15 @@ static bool ggml_backend_rpc_buffer_cpy_tensor(ggml_backend_buffer_t buffer, con ggml_backend_rpc_buffer_context * src_ctx = (ggml_backend_rpc_buffer_context *)src_buffer->context; ggml_backend_buffer_t dst_buffer = dst->buffer; ggml_backend_rpc_buffer_context * dst_ctx = (ggml_backend_rpc_buffer_context *)dst_buffer->context; - if (src_ctx->sock != dst_ctx->sock) { + if (src_ctx->dispatcher != dst_ctx->dispatcher) { return false; } ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - rpc_msg_copy_tensor_req request; - request.src = serialize_tensor(src); - request.dst = serialize_tensor(dst); + auto request = std::make_shared(); + request->src = serialize_tensor(src); + request->dst = serialize_tensor(dst); rpc_msg_copy_tensor_rsp response; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_COPY_TENSOR, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + ctx->dispatcher->send(RPC_CMD_COPY_TENSOR, request, sizeof(*request), &response, sizeof(response)); return response.result; } return false; @@ -543,9 +772,10 @@ static bool ggml_backend_rpc_buffer_cpy_tensor(ggml_backend_buffer_t buffer, con static void ggml_backend_rpc_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) { ggml_backend_rpc_buffer_context * ctx = (ggml_backend_rpc_buffer_context *)buffer->context; - rpc_msg_buffer_clear_req request = {ctx->remote_ptr, value}; - bool status = send_rpc_cmd(ctx->sock, RPC_CMD_BUFFER_CLEAR, &request, sizeof(request), nullptr, 0); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->remote_ptr = ctx->remote_ptr; + request->value = value; + ctx->dispatcher->send(RPC_CMD_BUFFER_CLEAR, request, sizeof(*request)); } static ggml_backend_buffer_i ggml_backend_rpc_buffer_interface = { @@ -569,15 +799,17 @@ static const char * ggml_backend_rpc_buffer_type_name(ggml_backend_buffer_type_t static ggml_backend_buffer_t ggml_backend_rpc_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) { ggml_backend_rpc_buffer_type_context * buft_ctx = (ggml_backend_rpc_buffer_type_context *)buft->context; - rpc_msg_alloc_buffer_req request = {buft_ctx->device, size}; + auto request = std::make_shared(); + request->device = buft_ctx->device; + request->size = size; rpc_msg_alloc_buffer_rsp response; - auto sock = get_socket(buft_ctx->endpoint); - bool status = send_rpc_cmd(sock, RPC_CMD_ALLOC_BUFFER, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + + auto dispatcher = get_dispatcher(buft_ctx->endpoint); + dispatcher->send(RPC_CMD_ALLOC_BUFFER, request, sizeof(*request), &response, sizeof(response)); if (response.remote_ptr != 0) { ggml_backend_buffer_t buffer = ggml_backend_buffer_init(buft, ggml_backend_rpc_buffer_interface, - new ggml_backend_rpc_buffer_context{sock, nullptr, response.remote_ptr}, + new ggml_backend_rpc_buffer_context{dispatcher, nullptr, response.remote_ptr}, response.remote_size); return buffer; } else { @@ -585,11 +817,11 @@ static ggml_backend_buffer_t ggml_backend_rpc_buffer_type_alloc_buffer(ggml_back } } -static size_t get_alignment(const std::shared_ptr & sock, uint32_t device) { - rpc_msg_get_alignment_req request = {device}; +static size_t get_alignment(const std::shared_ptr & dispatcher, uint32_t device) { + auto request = std::make_shared(); + request->device = device; rpc_msg_get_alignment_rsp response; - bool status = send_rpc_cmd(sock, RPC_CMD_GET_ALIGNMENT, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + dispatcher->send(RPC_CMD_GET_ALIGNMENT, request, sizeof(*request), &response, sizeof(response)); return response.alignment; } @@ -598,11 +830,11 @@ static size_t ggml_backend_rpc_buffer_type_get_alignment(ggml_backend_buffer_typ return buft_ctx->alignment; } -static size_t get_max_size(const std::shared_ptr & sock, uint32_t device) { - rpc_msg_get_max_size_req request = {device}; +static size_t get_max_size(const std::shared_ptr & dispatcher, uint32_t device) { + auto request = std::make_shared(); + request->device = device; rpc_msg_get_max_size_rsp response; - bool status = send_rpc_cmd(sock, RPC_CMD_GET_MAX_SIZE, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + dispatcher->send(RPC_CMD_GET_MAX_SIZE, request, sizeof(*request), &response, sizeof(response)); return response.max_size; } @@ -618,30 +850,70 @@ static size_t ggml_backend_rpc_buffer_type_get_alloc_size(ggml_backend_buffer_ty // See comments in init_tensor. rpc_get |= ggml_is_quantized(tensor->type) && (tensor->ne[0] % 512 != 0) && (tensor->view_src == nullptr); - // ops that require additional memory for fleeting data on certain backends + // [TAG_ALLOC_SIZE_EXPAND] + // ops that may require additional memory for fleeting data on certain backends // ref: https://github.com/ggml-org/llama.cpp/pull/15966 - rpc_get |= tensor->op == GGML_OP_FLASH_ATTN_EXT; - rpc_get |= tensor->op == GGML_OP_MUL_MAT_ID; + rpc_get |= ggml_op_alloc_size_may_expand(tensor->op); if (rpc_get) { ggml_backend_rpc_buffer_type_context * buft_ctx = (ggml_backend_rpc_buffer_type_context *)buft->context; - auto sock = get_socket(buft_ctx->endpoint); - rpc_msg_get_alloc_size_req request = { - /*.device =*/ buft_ctx->device, - /*.tensor =*/ serialize_tensor(tensor), - /*.srcs =*/ {}, + // Cache key for calls to read the alloc_size. + // We deliberately exclude src tensor dimensions from the key because: + // 1. For CPU backends, alloc_size = ggml_nbytes(output) regardless of src shapes + // 2. For GPU backends, the reservation graph uses max dimensions, so the + // cached value from reservation is always >= any subsequent request + // 3. Including src dims causes cache misses per-ubatch (e.g. growing KV cache) + // which blocks the main thread behind in-flight GRAPH_COMPUTE commands + struct alloc_size_cache_key { + uint32_t device; + uint32_t type; + uint32_t op; + int32_t op_params[GGML_MAX_OP_PARAMS / sizeof(int32_t)]; + uint32_t ne[GGML_MAX_DIMS]; }; + alloc_size_cache_key key = {}; + key.device = buft_ctx->device; + key.type = tensor->type; + key.op = tensor->op; + memcpy(key.op_params, tensor->op_params, sizeof(key.op_params)); + for (int i = 0; i < GGML_MAX_DIMS; i++) { + key.ne[i] = (uint32_t)tensor->ne[i]; + } + + uint64_t cache_hash = fnv_hash((const uint8_t *)&key, sizeof(key)); + cache_hash = fnv_hash((const uint8_t *)buft_ctx->endpoint.data(), buft_ctx->endpoint.size(), cache_hash); + + // alloc sizes are immutable for a given tensor configuration + static std::mutex cache_mutex; + static std::unordered_map cache; + + { + std::lock_guard lock(cache_mutex); + auto it = cache.find(cache_hash); + if (it != cache.end()) { + return it->second; + } + } + + auto request = std::make_shared(); + request->device = buft_ctx->device; + request->tensor = serialize_tensor(tensor); + // .get_alloc_size could be a function of the tensor's srcs, so we must serialize them as well for (int i = 0; i < GGML_MAX_SRC; i++) { - request.srcs[i] = serialize_tensor(tensor->src[i]); + request->srcs[i] = serialize_tensor(tensor->src[i]); } - // TODO: cache the alloc responses to avoid extra RPC calls? rpc_msg_get_alloc_size_rsp response; - bool status = send_rpc_cmd(sock, RPC_CMD_GET_ALLOC_SIZE, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + auto dispatcher = get_dispatcher(buft_ctx->endpoint); + dispatcher->send(RPC_CMD_GET_ALLOC_SIZE, request, sizeof(*request), &response, sizeof(response)); + + { + std::lock_guard lock(cache_mutex); + cache[cache_hash] = response.alloc_size; + } return response.alloc_size; } @@ -670,12 +942,45 @@ static void ggml_backend_rpc_free(ggml_backend_t backend) { delete backend; } +static void ggml_backend_rpc_set_tensor_async(ggml_backend_t backend, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { + ggml_backend_rpc_context * ctx = (ggml_backend_rpc_context *)backend->context; + rpc_tensor rpc_tensor = serialize_tensor(tensor); + uint8_t cache_flag = 0; + if (rpc_use_hash_cache(tensor, size)) { + auto request = std::make_shared(); + request->tensor = rpc_tensor; + request->offset = offset; + request->hash = fnv_hash((const uint8_t*)data, size); + rpc_msg_set_tensor_hash_rsp response; + // TODO: make this async + ctx->dispatcher->send(RPC_CMD_SET_TENSOR_HASH, request, sizeof(*request), &response, sizeof(response)); + if (response.result) { + // the server has the same data, no need to send it + return; + } + // the server has no cache entry for this tensor - ask it to save one + cache_flag = 1; + } + size_t input_size; + auto input = serialize_set_tensor(rpc_tensor, cache_flag, offset, data, size, input_size); + ctx->dispatcher->send_async(RPC_CMD_SET_TENSOR, input, input_size); +} + +static void ggml_backend_rpc_get_tensor_async(ggml_backend_t backend, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { + ggml_backend_rpc_context * ctx = (ggml_backend_rpc_context *)backend->context; + auto request = std::make_shared(); + request->tensor = serialize_tensor(tensor); + request->offset = offset; + request->size = size; + ctx->dispatcher->send_async(RPC_CMD_GET_TENSOR, request, sizeof(*request), data, size); +} + static void ggml_backend_rpc_synchronize(ggml_backend_t backend) { - GGML_UNUSED(backend); - // this is no-op because we don't have any async operations + ggml_backend_rpc_context * rpc_ctx = (ggml_backend_rpc_context *)backend->context; + rpc_ctx->dispatcher->synchronize(); } -static void add_tensor(ggml_tensor * tensor, std::vector & tensors, std::unordered_set & visited) { +static void add_tensor(ggml_tensor * tensor, const ggml_cgraph * cgraph, const std::shared_ptr & dispatcher, std::vector & tensors, std::unordered_set & visited) { if (tensor == nullptr) { return; } @@ -684,25 +989,30 @@ static void add_tensor(ggml_tensor * tensor, std::vector & tensors, } visited.insert(tensor); for (int i = 0; i < GGML_MAX_SRC; i++) { - add_tensor(tensor->src[i], tensors, visited); + add_tensor(tensor->src[i], cgraph, dispatcher, tensors, visited); + } + add_tensor(tensor->view_src, cgraph, dispatcher, tensors, visited); + rpc_tensor result = serialize_tensor(tensor, dispatcher); + const size_t hash_pos = ggml_hash_find(&cgraph->visited_hash_set, tensor); + if (hash_pos != GGML_HASHSET_FULL && ggml_bitset_get(cgraph->visited_hash_set.used, hash_pos)) { + result.use_count = cgraph->use_counts[hash_pos]; } - add_tensor(tensor->view_src, tensors, visited); - tensors.push_back(serialize_tensor(tensor)); + tensors.push_back(result); } -static void serialize_graph(uint32_t device, const ggml_cgraph * cgraph, std::vector & output) { +static uint8_t * serialize_graph(uint32_t device, const ggml_cgraph * cgraph, const std::shared_ptr & dispatcher, size_t * output_size) { uint32_t n_nodes = cgraph->n_nodes; std::vector tensors; std::unordered_set visited; for (uint32_t i = 0; i < n_nodes; i++) { - add_tensor(cgraph->nodes[i], tensors, visited); + add_tensor(cgraph->nodes[i], cgraph, dispatcher, tensors, visited); } // serialization format: // | device (4 bytes) | n_nodes (4 bytes) | nodes (n_nodes * sizeof(uint64_t) | n_tensors (4 bytes) | tensors (n_tensors * sizeof(rpc_tensor)) | uint32_t n_tensors = tensors.size(); - int output_size = 2*sizeof(uint32_t) + n_nodes * sizeof(uint64_t) + sizeof(uint32_t) + n_tensors * sizeof(rpc_tensor); - output.resize(output_size, 0); - uint8_t * dest = output.data(); + *output_size = 2*sizeof(uint32_t) + n_nodes * sizeof(uint64_t) + sizeof(uint32_t) + n_tensors * sizeof(rpc_tensor); + uint8_t * output = new uint8_t[*output_size](); + uint8_t * dest = output; memcpy(dest, &device, sizeof(device)); dest += sizeof(device); memcpy(dest, &n_nodes, sizeof(n_nodes)); @@ -715,6 +1025,7 @@ static void serialize_graph(uint32_t device, const ggml_cgraph * cgraph, std::ve dest += sizeof(n_tensors); rpc_tensor * out_tensors = (rpc_tensor *)dest; memcpy(out_tensors, tensors.data(), n_tensors * sizeof(rpc_tensor)); + return output; } static enum ggml_status ggml_backend_rpc_graph_compute(ggml_backend_t backend, ggml_cgraph * cgraph) { @@ -725,27 +1036,35 @@ static enum ggml_status ggml_backend_rpc_graph_compute(ggml_backend_t backend, g GGML_ASSERT(cgraph->n_nodes > 0); bool reuse = cgraph->uid != 0 && rpc_dev_ctx->last_graph_uid == cgraph->uid; if (reuse) { - rpc_msg_graph_recompute_req request; - request.device = rpc_ctx->device; - auto sock = get_socket(rpc_ctx->endpoint); - bool status = send_rpc_cmd(sock, RPC_CMD_GRAPH_RECOMPUTE, &request, sizeof(request)); - RPC_STATUS_ASSERT(status); + auto request = std::make_shared(); + request->device = rpc_ctx->device; + rpc_ctx->dispatcher->send_async(RPC_CMD_GRAPH_RECOMPUTE, request, sizeof(*request)); } else { rpc_dev_ctx->last_graph_uid = cgraph->uid; - std::vector input; - serialize_graph(rpc_ctx->device, cgraph, input); - auto sock = get_socket(rpc_ctx->endpoint); - bool status = send_rpc_cmd(sock, RPC_CMD_GRAPH_COMPUTE, input.data(), input.size()); - RPC_STATUS_ASSERT(status); + size_t input_size = 0; + uint8_t * input = serialize_graph(rpc_ctx->device, cgraph, rpc_ctx->dispatcher, &input_size); + std::shared_ptr input_ptr(input, std::default_delete()); + rpc_ctx->dispatcher->send_async(RPC_CMD_GRAPH_COMPUTE, input_ptr, input_size); } return GGML_STATUS_SUCCESS; } +static void ggml_backend_rpc_event_record(ggml_backend_t backend, ggml_backend_event_t event) { + ggml_backend_rpc_context * rpc_ctx = (ggml_backend_rpc_context *)backend->context; + rpc_ctx->dispatcher->event_record(event); +} + +static void ggml_backend_rpc_event_wait(ggml_backend_t backend, ggml_backend_event_t event) { + // this is noop for RPC as we have a single stream + GGML_UNUSED(backend); + GGML_UNUSED(event); +} + static ggml_backend_i ggml_backend_rpc_interface = { /* .get_name = */ ggml_backend_rpc_name, /* .free = */ ggml_backend_rpc_free, - /* .set_tensor_async = */ NULL, - /* .get_tensor_async = */ NULL, + /* .set_tensor_async = */ ggml_backend_rpc_set_tensor_async, + /* .get_tensor_async = */ ggml_backend_rpc_get_tensor_async, /* .set_tensor_2d_async = */ NULL, /* .get_tensor_2d_async = */ NULL, /* .cpy_tensor_async = */ NULL, @@ -755,8 +1074,8 @@ static ggml_backend_i ggml_backend_rpc_interface = { /* .graph_plan_update = */ NULL, /* .graph_plan_compute = */ NULL, /* .graph_compute = */ ggml_backend_rpc_graph_compute, - /* .event_record = */ NULL, - /* .event_wait = */ NULL, + /* .event_record = */ ggml_backend_rpc_event_record, + /* .event_wait = */ ggml_backend_rpc_event_wait, /* .graph_optimize = */ NULL, }; @@ -770,13 +1089,9 @@ ggml_backend_buffer_type_t ggml_backend_rpc_buffer_type(const char * endpoint, u if (it != buft_map.end()) { return it->second; } - auto sock = get_socket(endpoint); - if (sock == nullptr) { - GGML_LOG_ERROR("Failed to connect to %s\n", endpoint); - return nullptr; - } - size_t alignment = get_alignment(sock, device); - size_t max_size = get_max_size(sock, device); + auto dispatcher = get_dispatcher(endpoint); + size_t alignment = get_alignment(dispatcher, device); + size_t max_size = get_max_size(dispatcher, device); ggml_backend_rpc_buffer_type_context * buft_ctx = new ggml_backend_rpc_buffer_type_context { /* .endpoint = */ endpoint, /* .device = */ device, @@ -796,10 +1111,11 @@ ggml_backend_buffer_type_t ggml_backend_rpc_buffer_type(const char * endpoint, u ggml_backend_t ggml_backend_rpc_init(const char * endpoint, uint32_t device) { std::string dev_name = "RPC" + std::to_string(device) + "[" + std::string(endpoint) + "]"; + auto dispatcher = get_dispatcher(endpoint); ggml_backend_rpc_context * ctx = new ggml_backend_rpc_context { - /* .endpoint = */ endpoint, - /* .device = */ device, - /* .name = */ dev_name, + /* .dispatcher = */ dispatcher, + /* .device = */ device, + /* .name = */ dev_name, }; auto reg = ggml_backend_rpc_add_server(endpoint); ggml_backend_t backend = new ggml_backend { @@ -815,26 +1131,16 @@ bool ggml_backend_is_rpc(ggml_backend_t backend) { return backend != NULL && ggml_guid_matches(backend->guid, ggml_backend_rpc_guid()); } -static void get_device_memory(const std::shared_ptr & sock, uint32_t device, size_t * free, size_t * total) { - rpc_msg_get_device_memory_req request; - request.device = device; +void ggml_backend_rpc_get_device_memory(const char * endpoint, uint32_t device, size_t * free, size_t * total) { + auto dispatcher = get_dispatcher(endpoint); + auto request = std::make_shared(); + request->device = device; rpc_msg_get_device_memory_rsp response; - bool status = send_rpc_cmd(sock, RPC_CMD_GET_DEVICE_MEMORY, &request, sizeof(request), &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + dispatcher->send(RPC_CMD_GET_DEVICE_MEMORY, request, sizeof(*request), &response, sizeof(response)); *free = response.free_mem; *total = response.total_mem; } -void ggml_backend_rpc_get_device_memory(const char * endpoint, uint32_t device, size_t * free, size_t * total) { - auto sock = get_socket(endpoint); - if (sock == nullptr) { - *free = 0; - *total = 0; - return; - } - get_device_memory(sock, device, free, total); -} - // RPC server-side implementation class rpc_server { @@ -995,6 +1301,11 @@ bool rpc_server::free_buffer(const rpc_msg_free_buffer_req & request) { GGML_LOG_ERROR("[%s] buffer not found\n", __func__); return false; } + // Discard all cached graphs to avoid use-after-free in graph_recompute, + // since their nodes may hold pointers to the buffer being freed. + for (auto & sg : stored_graphs) { + sg.graph = nullptr; + } ggml_backend_buffer_free(buffer); buffers.erase(buffer); return true; @@ -1108,14 +1419,17 @@ ggml_tensor * rpc_server::deserialize_tensor(struct ggml_context * ctx, const rp bool rpc_server::set_tensor(const std::vector & input) { - // serialization format: | rpc_tensor | offset (8 bytes) | data (size bytes) | - if (input.size() < sizeof(rpc_tensor) + sizeof(uint64_t)) { + // serialization format: | rpc_tensor | cache_flag (1 byte) | offset (8 bytes) | data (size bytes) | + uint8_t cache_flag; + uint64_t offset; + const size_t header_size = sizeof(rpc_tensor) + sizeof(cache_flag) + sizeof(offset); + if (input.size() < header_size) { return false; } const rpc_tensor * in_tensor = (const rpc_tensor *)input.data(); - uint64_t offset; - memcpy(&offset, input.data() + sizeof(rpc_tensor), sizeof(offset)); - const size_t size = input.size() - sizeof(rpc_tensor) - sizeof(offset); + memcpy(&cache_flag, input.data() + sizeof(rpc_tensor), sizeof(cache_flag)); + memcpy(&offset, input.data() + sizeof(rpc_tensor) + sizeof(cache_flag), sizeof(offset)); + const size_t size = input.size() - header_size; struct ggml_init_params params { /*.mem_size =*/ ggml_tensor_overhead(), @@ -1144,8 +1458,8 @@ bool rpc_server::set_tensor(const std::vector & input) { } } - const void * data = input.data() + sizeof(rpc_tensor) + sizeof(offset); - if (cache_dir && size > HASH_THRESHOLD) { + const void * data = input.data() + header_size; + if (cache_dir && cache_flag) { uint64_t hash = fnv_hash((const uint8_t*)data, size); char hash_str[17]; snprintf(hash_str, sizeof(hash_str), "%016" PRIx64, hash); @@ -1443,7 +1757,6 @@ bool rpc_server::graph_compute(const std::vector & input) { int64_t id; memcpy(&id, &nodes[i], sizeof(id)); graph->nodes[i] = create_node(id, ctx, tensor_ptrs, tensor_map); - // Check if create_node failed for a *non-zero* ID. // If id was 0, create_node returning nullptr is expected. // If id was non-zero and create_node returned nullptr, it indicates a deserialization error. @@ -1451,6 +1764,10 @@ bool rpc_server::graph_compute(const std::vector & input) { GGML_LOG_ERROR("[%s] failed to create graph node %d (id=%" PRId64 ")\n", __func__, i, id); return false; } + if (graph->nodes[i] != nullptr) { + const size_t hash_pos = ggml_hash_insert(&graph->visited_hash_set, graph->nodes[i]); + graph->use_counts[hash_pos] = tensor_ptrs.at(id)->use_count; + } } ggml_status status = ggml_backend_graph_compute(backends[device], graph); GGML_ASSERT(status == GGML_STATUS_SUCCESS && "Unsuccessful graph computations are not supported with RPC"); @@ -1635,9 +1952,6 @@ static void rpc_serve_client(const std::vector & backends, const if (!server.free_buffer(request)) { return; } - if (!send_msg(sock, nullptr, 0)) { - return; - } break; } case RPC_CMD_BUFFER_CLEAR: { @@ -1648,9 +1962,6 @@ static void rpc_serve_client(const std::vector & backends, const if (!server.buffer_clear(request)) { return; } - if (!send_msg(sock, nullptr, 0)) { - return; - } break; } case RPC_CMD_MEMSET_TENSOR: { @@ -1661,9 +1972,6 @@ static void rpc_serve_client(const std::vector & backends, const if (!server.memset_tensor(request)) { return; } - if (!send_msg(sock, nullptr, 0)) { - return; - } break; } case RPC_CMD_SET_TENSOR: { @@ -1698,9 +2006,6 @@ static void rpc_serve_client(const std::vector & backends, const if (!server.init_tensor(request)) { return; } - if (!send_msg(sock, nullptr, 0)) { - return; - } break; } case RPC_CMD_GET_TENSOR: { @@ -1877,10 +2182,10 @@ static void ggml_backend_rpc_device_get_props(ggml_backend_dev_t dev, struct ggm props->type = ggml_backend_rpc_device_get_type(dev); ggml_backend_rpc_device_get_memory(dev, &props->memory_free, &props->memory_total); props->caps = { - /* .async = */ false, + /* .async = */ true, /* .host_buffer = */ false, /* .buffer_from_host_ptr = */ false, - /* .events = */ false, + /* .events = */ true, /* .mmap_support = */ true, }; } @@ -1917,6 +2222,24 @@ static bool ggml_backend_rpc_device_supports_buft(ggml_backend_dev_t dev, ggml_b return buft_ctx->endpoint == dev_ctx->endpoint && buft_ctx->device == dev_ctx->device; } +static ggml_backend_event_t ggml_backend_rpc_device_event_new(ggml_backend_dev_t dev) { + ggml_backend_rpc_device_context * ctx = (ggml_backend_rpc_device_context *)dev->context; + auto dispatcher = get_dispatcher(ctx->endpoint); + return dispatcher->event_new(dev); +} + +static void ggml_backend_rpc_device_event_free(ggml_backend_dev_t dev, ggml_backend_event_t event) { + ggml_backend_rpc_device_context * ctx = (ggml_backend_rpc_device_context *)dev->context; + auto dispatcher = get_dispatcher(ctx->endpoint); + dispatcher->event_free(event); +} + +static void ggml_backend_rpc_device_event_synchronize(ggml_backend_dev_t dev, ggml_backend_event_t event) { + ggml_backend_rpc_device_context * ctx = (ggml_backend_rpc_device_context *)dev->context; + auto dispatcher = get_dispatcher(ctx->endpoint); + dispatcher->event_synchronize(event); +} + static const struct ggml_backend_device_i ggml_backend_rpc_device_i = { /* .get_name = */ ggml_backend_rpc_device_get_name, /* .get_description = */ ggml_backend_rpc_device_get_description, @@ -1930,9 +2253,9 @@ static const struct ggml_backend_device_i ggml_backend_rpc_device_i = { /* .supports_op = */ ggml_backend_rpc_device_supports_op, /* .supports_buft = */ ggml_backend_rpc_device_supports_buft, /* .offload_op = */ NULL, - /* .event_new = */ NULL, - /* .event_free = */ NULL, - /* .event_synchronize = */ NULL, + /* .event_new = */ ggml_backend_rpc_device_event_new, + /* .event_free = */ ggml_backend_rpc_device_event_free, + /* .event_synchronize = */ ggml_backend_rpc_device_event_synchronize, }; // backend reg interface @@ -1992,14 +2315,9 @@ ggml_backend_reg_t ggml_backend_rpc_reg(void) { } static uint32_t ggml_backend_rpc_get_device_count(const char * endpoint) { - auto sock = get_socket(endpoint); - if (sock == nullptr) { - GGML_LOG_ERROR("Failed to connect to %s\n", endpoint); - return 0; - } + auto dispatcher = get_dispatcher(endpoint); rpc_msg_device_count_rsp response; - bool status = send_rpc_cmd(sock, RPC_CMD_DEVICE_COUNT, nullptr, 0, &response, sizeof(response)); - RPC_STATUS_ASSERT(status); + dispatcher->send(RPC_CMD_DEVICE_COUNT, nullptr, 0, &response, sizeof(response)); return response.device_count; } diff --git a/ggml/src/ggml-rpc/transport-apple.cpp b/ggml/src/ggml-rpc/transport-apple.cpp new file mode 100644 index 00000000..b1934175 --- /dev/null +++ b/ggml/src/ggml-rpc/transport-apple.cpp @@ -0,0 +1,481 @@ +#include "transport-apple.h" +#include "transport.h" +#include "ggml-impl.h" + +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +// Apple RDMA-over-Thunderbolt (see Apple TN3205). +// +// Apple's RDMA is quite different from what's supported in Linux - deserving of its own transport implementation. +// see https://developer.apple.com/documentation/technotes/tn3205-low-latency-communication-with-rdma-over-thunderbolt for details +// at a high level the main differences are: +// UC(unreliable connection) on Apple vs RC(reliable connection) QP transport types on Linux (though in practice UC on Apple is still lossless) +// fixed 128KiB stride on Apple vs variable chunk size on Linux +// relying on Apple's hardware credit based flow control vs RNR NAKs + retries on Linux +// +// on Apple a SEND and its corresponding RECV must cover the same number of 4 KiB Thunderbolt frames, +// so every SEND posts a whole 128KiB stride over the wire, even when partially filled. +// (In testing 128KiB was the best performing among 32, 64, 128, 256) + +static constexpr uint32_t RDMA_SEG_MAGIC = 0x52534547u; // "RSEG" +static constexpr int RDMA_NBUF = 16; // ring depth (frames per direction) +static constexpr size_t RDMA_FRAME = 4096; // Thunderbolt frame (fixed on Apple) +static constexpr size_t RDMA_STRIDE = 128 * 1024; // 32 Thunderbolt frames; NBUF x this = 2 MiB pinned per direction +static constexpr uint32_t RDMA_PSN = 0; // any value works if both sides match: UC has no retransmit +static constexpr size_t RDMA_GID_SIZE = 16; + +static_assert(RDMA_STRIDE % RDMA_FRAME == 0, "RDMA_STRIDE must be a whole number of frames"); +// TN3205 counts queue depth in Thunderbolt frames, not work requests. +static constexpr uint32_t RDMA_QP_WR = (uint32_t)RDMA_NBUF * (RDMA_STRIDE / RDMA_FRAME); +static constexpr uint64_t RDMA_RECV_WR = 1ull << 20; // wr_id bit tagging recv completions +static constexpr uint64_t RDMA_WR_IDX_MASK = 0xffff; // buffer index in the low bits of wr_id +static constexpr uint8_t RDMA_SYNC_READY = 0x2A; // readiness-handshake byte (peer activated) + +struct rdma_seg_hdr { + uint32_t magic; // RDMA_SEG_MAGIC; a mismatch means the stream desynced + uint32_t len; // payload bytes in this frame; the rest of the stride is padding +}; +static constexpr size_t RDMA_PAYLOAD = RDMA_STRIDE - sizeof(rdma_seg_hdr); + +struct apple_rdma_caps { + uint32_t qpn; + uint16_t lid; + uint16_t reserved; + uint8_t gid[RDMA_GID_SIZE]; +}; + +static_assert(sizeof(apple_rdma_caps) == RPC_CONN_CAPS_SIZE, "apple_rdma_caps must match conn_caps size"); + +struct apple_rdma::impl { + int fd = -1; // bootstrap TCP socket, kept as the liveness anchor + + struct ibv_context * ctx = nullptr; + struct ibv_pd * pd = nullptr; + struct ibv_cq * cq = nullptr; // one CQ for both directions; RDMA_RECV_WR tags recv completions + struct ibv_qp * qp = nullptr; + + uint8_t * send_mem = nullptr; + struct ibv_mr * send_mr = nullptr; + uint8_t * recv_mem = nullptr; + struct ibv_mr * recv_mr = nullptr; + + int send_busy[RDMA_NBUF] = {}; // 1 while this buffer has a send in flight + // completed recv frames, oldest first: ring index, bytes already handed to + // the reader, and total payload length + struct { int buf; uint32_t off; uint32_t len; } inq[RDMA_NBUF] = {}; + int inq_head = 0; + int inq_count = 0; + int pend_buf = -1; + uint32_t pend_len = 0; + bool broken = false; + + uint32_t qpn = 0; + uint8_t port = 0; + int gid_idx = 0; + enum ibv_mtu path_mtu = IBV_MTU_1024; + + int progress(); + bool acquire_pending(); + bool post_pending(); + + bool post_recv(int i) { + struct ibv_sge sge = {}; + sge.addr = (uintptr_t)(recv_mem + (size_t)i * RDMA_STRIDE); + sge.length = (uint32_t)RDMA_STRIDE; + sge.lkey = recv_mr->lkey; + struct ibv_recv_wr wr = {}, * bad = nullptr; + wr.wr_id = RDMA_RECV_WR | (uint64_t)i; + wr.sg_list = &sge; + wr.num_sge = 1; + return ibv_post_recv(qp, &wr, &bad) == 0; + } + + bool post_send(int i, size_t len) { + struct ibv_sge sge = {}; + sge.addr = (uintptr_t)(send_mem + (size_t)i * RDMA_STRIDE); + sge.length = (uint32_t)len; + sge.lkey = send_mr->lkey; + struct ibv_send_wr wr = {}, * bad = nullptr; + wr.wr_id = (uint64_t)i; + wr.sg_list = &sge; + wr.num_sge = 1; + wr.opcode = IBV_WR_SEND; + wr.send_flags = IBV_SEND_SIGNALED; + return ibv_post_send(qp, &wr, &bad) == 0; + } + + ~impl() { + broken = true; + // destroy the QP first: it can still write to the rings until it is gone. + // no IBV_QPS_ERR before it - Apple's provider then fails every region unmap. + if (qp) ibv_destroy_qp(qp); + if (send_mr) ibv_dereg_mr(send_mr); + if (recv_mr) ibv_dereg_mr(recv_mr); + free(send_mem); + free(recv_mem); + if (cq) ibv_destroy_cq(cq); + if (pd) ibv_dealloc_pd(pd); + if (ctx) ibv_close_device(ctx); + } +}; + +apple_rdma::apple_rdma(std::unique_ptr p) : pimpl(std::move(p)) {} + +apple_rdma::~apple_rdma() = default; + +bool apple_rdma::broken() const { + return pimpl->broken; +} + +// The readiness handshake below still runs over the bootstrap socket, one byte +// each way, before the transport is declared live. +static bool tcp_send_byte(int fd, uint8_t b) { + ssize_t n; + do { n = ::send(fd, &b, sizeof(b), 0); } while (n < 0 && errno == EINTR); + return n == sizeof(b); +} + +static bool tcp_recv_byte(int fd, uint8_t * b) { + ssize_t n; + do { n = ::recv(fd, b, sizeof(*b), 0); } while (n < 0 && errno == EINTR); + return n == (ssize_t)sizeof(*b); +} + +// Index of the GID on this port equal to the target, or -1. Thunderbolt GIDs are +// RoCEv2 IPv4-mapped (::ffff:a.b.c.d), so this matches the local TCP address. +static int rdma_match_gid(struct ibv_context * ctx, uint8_t port, int gid_tbl_len, + const uint8_t * target, union ibv_gid * out) { + for (int i = 0; i < gid_tbl_len; i++) { + union ibv_gid g; + if (ibv_query_gid(ctx, port, i, &g) != 0) continue; + if (memcmp(g.raw, target, RDMA_GID_SIZE) != 0) continue; + if (out) *out = g; + return i; + } + return -1; +} + +// First ACTIVE port on the device. Only a cabled, up Thunderbolt link reports +// ACTIVE, and it is not always port 1, so the port cannot be hardcoded the way +// the Linux path does. Returns 0 if none. +static uint8_t rdma_first_active_port(struct ibv_context * ctx, struct ibv_port_attr * out) { + struct ibv_device_attr da; + if (ibv_query_device(ctx, &da) != 0) return 0; + for (uint8_t p = 1; p <= da.phys_port_cnt; p++) { + struct ibv_port_attr pa; + if (ibv_query_port(ctx, p, &pa) != 0) continue; + if (pa.state == IBV_PORT_ACTIVE) { if (out) *out = pa; return p; } + } + return 0; +} + +// librdma.dylib is weak-linked, so its symbols are null when it is absent. Nothing may +// call one before this has returned true. +static bool rdma_library_present() { + static const bool present = [] { + void * handle = dlopen("/usr/lib/librdma.dylib", RTLD_LAZY); + if (handle == nullptr) { + return false; + } + dlclose(handle); + return true; + }(); + return present; +} + +// Called before the endpoints are exchanged: pick the local device facing this +// peer, create a UC QP and register the frame rings. RDMA is point-to-point, so +// the device is the one whose GID equals the bootstrap connection's local +// address, i.e. the one cabled to the peer. +std::unique_ptr apple_rdma::probe(int fd, const uint8_t * target_gid, uint8_t * caps) { + if (!rdma_library_present()) { + return nullptr; + } + int ndev = 0; + ibv_device ** devs = ibv_get_device_list(&ndev); + if (!devs) return nullptr; + + ibv_context * ctx = nullptr; + uint8_t port = 0; + struct ibv_port_attr pa = {}; + union ibv_gid gid = {}; + int gid_idx = -1; + std::string matched; + for (int d = 0; d < ndev; d++) { + ibv_context * c = ibv_open_device(devs[d]); + if (!c) continue; + struct ibv_port_attr p = {}; + uint8_t pt = rdma_first_active_port(c, &p); + int gi = pt ? rdma_match_gid(c, pt, p.gid_tbl_len, target_gid, &gid) : -1; + if (gi < 0) { ibv_close_device(c); continue; } + ctx = c; port = pt; pa = p; gid_idx = gi; + const char * name = ibv_get_device_name(devs[d]); + matched = name ? name : ""; + break; + } + ibv_free_device_list(devs); + if (!ctx) return nullptr; + + std::unique_ptr c(new impl()); + c->fd = fd; + c->ctx = ctx; + c->port = port; + c->gid_idx = gid_idx; + c->path_mtu = pa.active_mtu; + + c->pd = ibv_alloc_pd(ctx); + if (!c->pd) return nullptr; + + c->cq = ibv_create_cq(ctx, 2 * RDMA_QP_WR + 1, nullptr, nullptr, 0); + if (!c->cq) return nullptr; + + ibv_qp_init_attr qia = {}; + qia.send_cq = c->cq; + qia.recv_cq = c->cq; + qia.qp_type = IBV_QPT_UC; + qia.cap.max_send_wr = RDMA_QP_WR; + qia.cap.max_recv_wr = RDMA_QP_WR; + qia.cap.max_send_sge = 1; + qia.cap.max_recv_sge = 1; + c->qp = ibv_create_qp(c->pd, &qia); + if (!c->qp) return nullptr; + + { + ibv_qp_attr a = {}; + a.qp_state = IBV_QPS_INIT; + a.pkey_index = 0; + a.port_num = port; + a.qp_access_flags = IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_REMOTE_READ | IBV_ACCESS_REMOTE_WRITE; + if (ibv_modify_qp(c->qp, &a, + IBV_QP_STATE | IBV_QP_PKEY_INDEX | IBV_QP_PORT | IBV_QP_ACCESS_FLAGS) != 0) { + return nullptr; + } + } + + long page = sysconf(_SC_PAGESIZE); + if (page <= 0) page = 4096; + const size_t ring_bytes = (size_t)RDMA_NBUF * RDMA_STRIDE; + if (posix_memalign((void **)&c->send_mem, (size_t)page, ring_bytes) != 0) c->send_mem = nullptr; + if (posix_memalign((void **)&c->recv_mem, (size_t)page, ring_bytes) != 0) c->recv_mem = nullptr; + if (!c->send_mem || !c->recv_mem) return nullptr; + + // Apple's provider rejects LOCAL_WRITE-only MRs even for two-sided SEND/RECV. + const int mr_flags = IBV_ACCESS_LOCAL_WRITE | IBV_ACCESS_REMOTE_READ | IBV_ACCESS_REMOTE_WRITE; + c->send_mr = ibv_reg_mr(c->pd, c->send_mem, ring_bytes, mr_flags); + c->recv_mr = ibv_reg_mr(c->pd, c->recv_mem, ring_bytes, mr_flags); + if (!c->send_mr || !c->recv_mr) return nullptr; + + // Recvs are posted in activate() after the RTS transition, not here: Apple's + // provider rejects ibv_post_recv on a QP that has not reached RTS. + + c->qpn = c->qp->qp_num; + + apple_rdma_caps rc = {}; + rc.qpn = c->qpn; + rc.lid = pa.lid; + memcpy(rc.gid, gid.raw, RDMA_GID_SIZE); + memcpy(caps, &rc, sizeof(rc)); + + GGML_LOG_INFO("RDMA(Apple/UC) probed: dev=%s port=%u gid=%d qpn=%u lid=%u mtu=%d ring=%d x %zu KiB\n", + matched.c_str(), port, gid_idx, c->qpn, (unsigned)pa.lid, 128 << c->path_mtu, + RDMA_NBUF, RDMA_STRIDE / 1024); + return std::unique_ptr(new apple_rdma(std::move(c))); +} + +// Called once the peer's endpoint has arrived: INIT -> RTR -> RTS (UC: GID/GRH +// addressing, no timeout/retry/rnr/rd_atomic), then the readiness handshake. +bool apple_rdma::activate(const uint8_t * caps) { + impl * c = pimpl.get(); + + apple_rdma_caps rc = {}; + memcpy(&rc, caps, sizeof(rc)); + + bool ok = true; + { + ibv_qp_attr a = {}; + a.qp_state = IBV_QPS_RTR; + a.path_mtu = c->path_mtu; + a.rq_psn = RDMA_PSN; + a.dest_qp_num = rc.qpn; + a.ah_attr.is_global = 1; + a.ah_attr.port_num = c->port; + a.ah_attr.sl = 0; + a.ah_attr.src_path_bits = 0; + a.ah_attr.dlid = rc.lid; + a.ah_attr.grh.hop_limit = 1; + a.ah_attr.grh.sgid_index = (uint8_t)c->gid_idx; + memcpy(&a.ah_attr.grh.dgid, rc.gid, RDMA_GID_SIZE); + if (ibv_modify_qp(c->qp, &a, + IBV_QP_STATE | IBV_QP_AV | IBV_QP_PATH_MTU | IBV_QP_DEST_QPN | IBV_QP_RQ_PSN) != 0) { + GGML_LOG_ERROR("RDMA(Apple/UC) RTR failed: %s\n", strerror(errno)); + ok = false; + } + } + if (ok) { + ibv_qp_attr a = {}; + a.qp_state = IBV_QPS_RTS; + a.sq_psn = RDMA_PSN; + if (ibv_modify_qp(c->qp, &a, IBV_QP_STATE | IBV_QP_SQ_PSN) != 0) { + GGML_LOG_ERROR("RDMA(Apple/UC) RTS failed: %s\n", strerror(errno)); + ok = false; + } + } + + // Recvs are posted only now: the controller starts processing them at RTR. + for (int i = 0; ok && i < RDMA_NBUF; i++) { + if (!c->post_recv(i)) { + GGML_LOG_ERROR("RDMA(Apple/UC) post_recv %d/%d failed\n", i, RDMA_NBUF); + ok = false; + } + } + + // A queue pair processes receives only after RTR and the transitions above can + // fail on one side alone, so neither peer sends a frame until both report their + // recvs posted. + uint8_t peer_ready = 0; + if (!tcp_send_byte(c->fd, ok ? RDMA_SYNC_READY : 0) || !tcp_recv_byte(c->fd, &peer_ready)) { + return false; + } + if (!ok || peer_ready != RDMA_SYNC_READY) { + return false; + } + + GGML_LOG_INFO("RDMA(Apple/UC) activated: qpn=%u->%u mtu=%d rx_depth=%d\n", + c->qpn, rc.qpn, 128 << c->path_mtu, RDMA_NBUF); + return true; +} + +// Drain the CQ: release completed send buffers, queue completed recv frames for +// the reader. Returns the number of completions reaped, or -1 on error. +int apple_rdma::impl::progress() { + struct ibv_wc wc[RDMA_NBUF * 2]; + int n = ibv_poll_cq(cq, RDMA_NBUF * 2, wc); + if (n < 0) { GGML_LOG_ERROR("RDMA(Apple/UC) poll_cq failed\n"); broken = true; return -1; } + for (int j = 0; j < n; j++) { + uint64_t id = wc[j].wr_id; + bool is_recv = (id & RDMA_RECV_WR) != 0; + if (wc[j].status != IBV_WC_SUCCESS) { + GGML_LOG_ERROR("RDMA(Apple/UC) %s wc error: status=%d\n", is_recv ? "recv" : "send", wc[j].status); + broken = true; + return -1; + } + if (is_recv) { + int b = (int)(id & RDMA_WR_IDX_MASK); + const rdma_seg_hdr * h = (const rdma_seg_hdr *)(recv_mem + (size_t)b * RDMA_STRIDE); + if (h->magic != RDMA_SEG_MAGIC) { GGML_LOG_ERROR("RDMA(Apple/UC) bad frame magic\n"); broken = true; return -1; } + if (h->len > RDMA_PAYLOAD) { GGML_LOG_ERROR("RDMA(Apple/UC) frame len %u exceeds payload\n", h->len); broken = true; return -1; } + int slot = (inq_head + inq_count) % RDMA_NBUF; + inq[slot].buf = b; + inq[slot].off = 0; + inq[slot].len = h->len; + inq_count++; + } else { + send_busy[(int)(id & RDMA_WR_IDX_MASK)] = 0; + } + } + return n; +} + +// Reserve a free send buffer to coalesce into, waiting on progress if none free. +bool apple_rdma::impl::acquire_pending() { + if (pend_buf >= 0) return true; + for (;;) { + if (broken) return false; + for (int k = 0; k < RDMA_NBUF; k++) if (!send_busy[k]) { pend_buf = k; pend_len = 0; return true; } + if (progress() < 0) return false; + } +} + +// Post the pending frame. The whole STRIDE goes out even when only partly filled: +// TN3205 requires a SEND and its matching RECV to cover the same number of +// Thunderbolt frames, so a short send would fail the peer's receive. +bool apple_rdma::impl::post_pending() { + if (pend_buf < 0) return true; + int i = pend_buf; + rdma_seg_hdr * h = (rdma_seg_hdr *)(send_mem + (size_t)i * RDMA_STRIDE); + h->magic = RDMA_SEG_MAGIC; + h->len = pend_len; + if (!post_send(i, RDMA_STRIDE)) { broken = true; return false; } + send_busy[i] = 1; + pend_buf = -1; + pend_len = 0; + return true; +} + +// Coalescing write: append into the pending frame, posting a full frame when it +// fills. The trailing partial is posted by flush() at each message boundary. +bool apple_rdma::send(const void * data, size_t size) { + impl * c = pimpl.get(); + const uint8_t * p = (const uint8_t *)data; + while (size > 0) { + if (c->broken) return false; + if (!c->acquire_pending()) return false; + uint8_t * sb = c->send_mem + (size_t)c->pend_buf * RDMA_STRIDE; + size_t space = RDMA_PAYLOAD - c->pend_len; + size_t chunk = size < space ? size : space; + memcpy(sb + sizeof(rdma_seg_hdr) + c->pend_len, p, chunk); + c->pend_len += (uint32_t)chunk; + p += chunk; + size -= chunk; + if (c->pend_len == RDMA_PAYLOAD) { if (!c->post_pending()) return false; } + } + return true; +} + +bool apple_rdma::recv(void * data, size_t size) { + impl * c = pimpl.get(); + uint8_t * p = (uint8_t *)data; + if (!c->post_pending()) return false; // turnaround: flush the coalesced request + unsigned idle = 0; + while (size > 0) { + if (c->inq_count == 0) { + if (c->broken) return false; + int n = c->progress(); + if (n < 0) return false; + if (n == 0) { + // UC gives no disconnect notification, so the bootstrap TCP fd is + // the liveness anchor: nothing crosses it once RDMA is up, so any + // readability means the peer's FIN (macOS has no POLLRDHUP). + // Same idle interval as the Linux path. + if ((++idle & 0xFFFFF) == 0) { + struct pollfd pfd = { c->fd, POLLIN, 0 }; + if (poll(&pfd, 1, 0) > 0 && + (pfd.revents & (POLLIN | POLLHUP | POLLERR | POLLNVAL))) { + return false; + } + } + } else { + idle = 0; + } + continue; + } + idle = 0; + int slot = c->inq_head; + int b = c->inq[slot].buf; + uint32_t avail = c->inq[slot].len - c->inq[slot].off; + uint32_t take = (size < (size_t)avail) ? (uint32_t)size : avail; + memcpy(p, c->recv_mem + (size_t)b * RDMA_STRIDE + sizeof(rdma_seg_hdr) + c->inq[slot].off, take); + p += take; + size -= take; + c->inq[slot].off += take; + if (c->inq[slot].off == c->inq[slot].len) { + if (!c->post_recv(b)) { c->broken = true; return false; } + c->inq_head = (c->inq_head + 1) % RDMA_NBUF; + c->inq_count--; + } + } + return true; +} + +bool apple_rdma::flush() { + return pimpl->post_pending(); +} diff --git a/ggml/src/ggml-rpc/transport-apple.h b/ggml/src/ggml-rpc/transport-apple.h new file mode 100644 index 00000000..7968d38a --- /dev/null +++ b/ggml/src/ggml-rpc/transport-apple.h @@ -0,0 +1,27 @@ +#pragma once + +#include +#include +#include + +struct apple_rdma { + // target_gid is 16 bytes in, caps is RPC_CONN_CAPS_SIZE bytes out. + static std::unique_ptr probe(int fd, const uint8_t * target_gid, uint8_t * caps); + ~apple_rdma(); + + // Peer endpoint from its caps, which must be non-zero: this blocks on a + // readiness handshake over fd that the peer only joins if it also has RDMA. + bool activate(const uint8_t * caps); + + bool send(const void * data, size_t size); + bool recv(void * data, size_t size); + // Post the trailing partial frame; must be called at every message boundary. + bool flush(); + // True once the connection has failed; the caller should drop the socket. + bool broken() const; + +private: + struct impl; + explicit apple_rdma(std::unique_ptr p); + std::unique_ptr pimpl; +}; diff --git a/ggml/src/ggml-rpc/transport.cpp b/ggml/src/ggml-rpc/transport.cpp index a7281524..5ec15dc8 100644 --- a/ggml/src/ggml-rpc/transport.cpp +++ b/ggml/src/ggml-rpc/transport.cpp @@ -18,15 +18,20 @@ # include #endif #include +#include #include #include #ifdef GGML_RPC_RDMA # include +# include # include # ifndef _WIN32 # include # endif +# ifdef GGML_RPC_RDMA_APPLE +# include "transport-apple.h" +# endif #endif // GGML_RPC_RDMA #ifdef _WIN32 @@ -42,10 +47,13 @@ static const char * RPC_DEBUG = std::getenv("GGML_RPC_DEBUG"); do { if (RPC_DEBUG) GGML_LOG_DEBUG(__VA_ARGS__); } while (0) #ifdef GGML_RPC_RDMA -static constexpr size_t RDMA_CHUNK = 256 * 1024; // 256 KiB per send/recv (fits default 8 MiB memlock) -static constexpr int RDMA_RX_DEPTH = 24; // pre-posted recv ring: 24 × 256 KiB = 6 MiB static constexpr size_t RDMA_GID_SIZE = 16; // RoCE GID / IB GID is always 16 bytes using rdma_gid_t = std::array; +#endif // GGML_RPC_RDMA + +#if defined(GGML_RPC_RDMA) && !defined(GGML_RPC_RDMA_APPLE) +static constexpr size_t RDMA_CHUNK = 256 * 1024; // 256 KiB per send/recv (fits default 8 MiB memlock) +static constexpr int RDMA_RX_DEPTH = 24; // pre-posted recv ring: 24 × 256 KiB = 6 MiB struct rdma_conn { struct ibv_context * ctx = nullptr; @@ -111,27 +119,33 @@ struct rdma_caps { static_assert(sizeof(rdma_caps) == RPC_CONN_CAPS_SIZE, "rdma_caps must match conn_caps size"); -#endif // GGML_RPC_RDMA +#endif // GGML_RPC_RDMA && !GGML_RPC_RDMA_APPLE struct socket_t::impl { impl(sockfd_t fd) : use_rdma(false), fd(fd) {} ~impl(); bool send_data(const void * data, size_t size); bool recv_data(void * data, size_t size); + bool flush(); void get_caps(uint8_t * local_caps); void update_caps(const uint8_t * remote_caps); #ifdef GGML_RPC_RDMA - bool tcp_peer_closed(); std::optional rdma_build_target_gid(); + +# ifdef GGML_RPC_RDMA_APPLE + std::unique_ptr rdma; +# else bool rdma_probe(); - bool rdma_activate(uint32_t remote_qpn, uint32_t remote_psn, const uint8_t * remote_gid); - bool rdma_poll(struct ibv_cq * cq, struct ibv_wc * wc); bool rdma_send(const void * data, size_t size); bool rdma_recv(void * data, size_t size); + bool tcp_peer_closed(); + bool rdma_activate(uint32_t remote_qpn, uint32_t remote_psn, const uint8_t * remote_gid); + bool rdma_poll(struct ibv_cq * cq, struct ibv_wc * wc); std::unique_ptr rdma; rdma_local_info rdma_local = {}; +# endif #endif // GGML_RPC_RDMA bool use_rdma; sockfd_t fd; @@ -151,17 +165,6 @@ socket_t::impl::~impl() { #ifdef GGML_RPC_RDMA -bool socket_t::impl::tcp_peer_closed() { - if (fd < 0) return false; -#ifndef _WIN32 - struct pollfd pfd = { fd, POLLIN | POLLRDHUP, 0 }; - int r = poll(&pfd, 1, 0); - return r > 0 && (pfd.revents & (POLLHUP | POLLERR | POLLRDHUP)); -#else - return false; -#endif -} - // Build a RoCE GID-shaped 16-byte target from a TCP socket's local address. // Used to match the socket's local IP against the kernel's GID table so that // a single memcmp handles IPv4, IPv4-mapped IPv6, and native IPv6 uniformly: @@ -191,6 +194,19 @@ std::optional socket_t::impl::rdma_build_target_gid() { return std::nullopt; } +#ifndef GGML_RPC_RDMA_APPLE + +bool socket_t::impl::tcp_peer_closed() { + if (fd < 0) return false; +#ifndef _WIN32 + struct pollfd pfd = { fd, POLLIN | POLLRDHUP, 0 }; + int r = poll(&pfd, 1, 0); + return r > 0 && (pfd.revents & (POLLHUP | POLLERR | POLLRDHUP)); +#else + return false; +#endif +} + bool socket_t::impl::rdma_probe() { const char * dev_env = std::getenv("GGML_RDMA_DEV"); const char * gid_env = std::getenv("GGML_RDMA_GID"); @@ -457,10 +473,16 @@ bool socket_t::impl::rdma_recv(void * data, size_t size) { return true; } +#endif // !GGML_RPC_RDMA_APPLE (Linux RC transport) + #endif // GGML_RPC_RDMA bool socket_t::impl::send_data(const void * data, size_t size) { -#ifdef GGML_RPC_RDMA +#ifdef GGML_RPC_RDMA_APPLE + if (use_rdma) { + return rdma->send(data, size); + } +#elif defined(GGML_RPC_RDMA) if (use_rdma) { return rdma_send(data, size); } @@ -480,7 +502,11 @@ bool socket_t::impl::send_data(const void * data, size_t size) { } bool socket_t::impl::recv_data(void * data, size_t size) { -#ifdef GGML_RPC_RDMA +#ifdef GGML_RPC_RDMA_APPLE + if (use_rdma) { + return rdma->recv(data, size); + } +#elif defined(GGML_RPC_RDMA) if (use_rdma) { return rdma_recv(data, size); } @@ -506,6 +532,15 @@ bool socket_t::impl::recv_data(void * data, size_t size) { void socket_t::impl::get_caps(uint8_t * local_caps) { memset(local_caps, 0, RPC_CONN_CAPS_SIZE); #ifdef GGML_RPC_RDMA + if (std::getenv("GGML_RPC_NO_RDMA")) { + return; + } +# ifdef GGML_RPC_RDMA_APPLE + auto target_gid = rdma_build_target_gid(); + if (target_gid) { + rdma = apple_rdma::probe(fd, target_gid->data(), local_caps); + } +# else rdma_local = {}; if (rdma_probe()) { rdma_caps rc = {}; @@ -516,21 +551,30 @@ void socket_t::impl::get_caps(uint8_t * local_caps) { } else { rdma.reset(); } +# endif #endif // GGML_RPC_RDMA } void socket_t::impl::update_caps(const uint8_t * remote_caps) { #ifdef GGML_RPC_RDMA - if (!rdma) { - return; + // a peer that has no RDMA advertises all-zero caps and takes no further part + // in the negotiation, so drop to TCP without reporting a failure + bool remote_rdma = false; + for (size_t i = 0; i < RPC_CONN_CAPS_SIZE; i++) { + remote_rdma |= remote_caps[i] != 0; } - rdma_caps rc = {}; - memcpy(&rc, remote_caps, sizeof(rc)); - if (rc.qpn == 0) { + if (!rdma || !remote_rdma) { rdma.reset(); return; } - if (rdma_activate(rc.qpn, rc.psn, rc.gid)) { +# ifdef GGML_RPC_RDMA_APPLE + bool activated = rdma->activate(remote_caps); +# else + rdma_caps rc = {}; + memcpy(&rc, remote_caps, sizeof(rc)); + bool activated = rdma_activate(rc.qpn, rc.psn, rc.gid); +# endif + if (activated) { use_rdma = true; } else { GGML_LOG_ERROR("RDMA activate failed, staying on TCP\n"); @@ -541,6 +585,14 @@ void socket_t::impl::update_caps(const uint8_t * remote_caps) { #endif // GGML_RPC_RDMA } +bool socket_t::impl::flush() { +#ifdef GGML_RPC_RDMA_APPLE + if (use_rdma) { + return rdma->flush(); + } +#endif + return true; +} ///////////////////////////////////////////////////////////////////////////// @@ -556,6 +608,10 @@ bool socket_t::recv_data(void * data, size_t size) { return pimpl->recv_data(data, size); } +bool socket_t::flush() { + return pimpl->flush(); +} + void socket_t::get_caps(uint8_t * local_caps) { return pimpl->get_caps(local_caps); } diff --git a/ggml/src/ggml-rpc/transport.h b/ggml/src/ggml-rpc/transport.h index 73b85cc5..3f747ecf 100644 --- a/ggml/src/ggml-rpc/transport.h +++ b/ggml/src/ggml-rpc/transport.h @@ -15,6 +15,10 @@ struct socket_t { bool send_data(const void * data, size_t size); bool recv_data(void * data, size_t size); + // Must be called at every message boundary: the RDMA transport coalesces + // writes into fixed-size frames and posts the trailing partial frame only + // here. No-op on TCP. + bool flush(); socket_ptr accept(); diff --git a/ggml/src/ggml-sycl/CMakeLists.txt b/ggml/src/ggml-sycl/CMakeLists.txt index a8d9c0d8..d2196f74 100644 --- a/ggml/src/ggml-sycl/CMakeLists.txt +++ b/ggml/src/ggml-sycl/CMakeLists.txt @@ -110,15 +110,21 @@ if (GGML_SYCL_SUPPORT_LEVEL_ZERO_API) # Link against Level Zero loader for direct device memory allocation. # Avoids sycl::malloc_device triggering DMA-buf/TTM system RAM staging # in the xe kernel driver during multi-GPU inference. - find_path(LEVEL_ZERO_INCLUDE_DIR level_zero/ze_api.h HINTS ${ONEAPI_ROOT}/include ${LEVEL_ZERO_V1_SDK_PATH}/include) + find_path(LEVEL_ZERO_DEV_INCLUDE_DIR level_zero/ze_api.h HINTS ${ONEAPI_ROOT}/include ${LEVEL_ZERO_V1_SDK_PATH}/include) find_library(ZE_LOADER_LIB ze_loader HINTS ${ONEAPI_ROOT}/lib ${LEVEL_ZERO_V1_SDK_LIB_PATH} ENV LD_LIBRARY_PATH) - if(ZE_LOADER_LIB AND LEVEL_ZERO_INCLUDE_DIR) + if(ZE_LOADER_LIB AND LEVEL_ZERO_DEV_INCLUDE_DIR) target_link_libraries(ggml-sycl PRIVATE ${ZE_LOADER_LIB}) target_compile_definitions(ggml-sycl PRIVATE GGML_SYCL_SUPPORT_LEVEL_ZERO_API) message(STATUS "Level Zero loader found: ${ZE_LOADER_LIB}") - message(STATUS "Level Zero headers found: ${LEVEL_ZERO_INCLUDE_DIR}") + message(STATUS "Level Zero development headers found: ${LEVEL_ZERO_DEV_INCLUDE_DIR}") else() - message(WARNING "Level Zero loader or headers not found, Level Zero support disabled") + message(WARNING "Level Zero loader or development headers not found, " + "Level Zero API support disabled. " + "Please install the Level Zero SDK/development package " + "to support Level Zero API features. " + "Level Zero API is not mandatory for SYCL backend, " + "but it is required by the special features for better " + "function & performance on Intel GPUs.") endif() endif() diff --git a/ggml/src/ggml-sycl/backend.hpp b/ggml/src/ggml-sycl/backend.hpp index 51ab6f93..ab80a2a3 100644 --- a/ggml/src/ggml-sycl/backend.hpp +++ b/ggml/src/ggml-sycl/backend.hpp @@ -44,6 +44,7 @@ #include "ssm_conv.hpp" #include "softmax.hpp" #include "topk-moe.hpp" +#include "topk-radix.hpp" #include "tsembd.hpp" #include "upscale.hpp" #include "wkv.hpp" diff --git a/ggml/src/ggml-sycl/base.hpp b/ggml/src/ggml-sycl/base.hpp new file mode 100644 index 00000000..fe96c4ab --- /dev/null +++ b/ggml/src/ggml-sycl/base.hpp @@ -0,0 +1,43 @@ +#ifndef GGML_SYCL_BASE_HPP +#define GGML_SYCL_BASE_HPP + +/** + * Module: base + * + * Description: + * Provides zero-dependency, foundational primitives, core abstractions, + * and low-level system interfaces. This module acts as the lowest layer + * of the architecture and is consumed globally across all subsystems. + * + * Constraints: + * - STRICTLY zero upstream dependencies (leaf module). + * - High stability and backward compatibility required. + */ + +#include + +extern int g_ggml_sycl_debug; +extern int g_ggml_sycl_dev_debug; + +#if defined(__clang__) && __has_builtin(__builtin_expect) +// Hint the optimizer to pipeline the more likely following instruction in branches +# define LIKELY(expr) __builtin_expect(expr, true) +# define UNLIKELY(expr) __builtin_expect(expr, false) +#else +# define LIKELY(expr) (expr) +# define UNLIKELY(expr) (expr) +#endif + +#define GGML_SYCL_DEBUG(...) \ + do { \ + if (UNLIKELY(g_ggml_sycl_debug)) \ + fprintf(stderr, __VA_ARGS__); \ + } while (0) + +#define GGML_SYCL_DEV_DEBUG(...) \ + do { \ + if (UNLIKELY(g_ggml_sycl_dev_debug)) \ + fprintf(stderr, __VA_ARGS__); \ + } while (0) + +#endif // GGML_SYCL_BASE_HPP diff --git a/ggml/src/ggml-sycl/binbcast.cpp b/ggml/src/ggml-sycl/binbcast.cpp index 306eeddc..f2f7c4cd 100644 --- a/ggml/src/ggml-sycl/binbcast.cpp +++ b/ggml/src/ggml-sycl/binbcast.cpp @@ -1,5 +1,6 @@ #include "binbcast.hpp" +#include #include #include #include @@ -356,3 +357,294 @@ void ggml_sycl_repeat(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { ggml_sycl_op_repeat(ctx, dst); } +// fused ADD+ADD: dst = (src0 + src1) + src2. Same indexing as k_bin_bcast, so mixed +// types, broadcast, and non-contiguous layouts that add() already handles also fuse. +template +static void k_bin_bcast3(const src0_t * src0, const src1_t * src1, const src2_t * src2, dst_t * dst, + int ne0, int ne1, int ne2, int ne3, + int ne10, int ne11, int ne12, int ne13, + int ne20, int ne21, int ne22, int ne23, + int s1, int s2, int s3, + int s00, int s01, int s02, int s03, + int s10, int s11, int s12, int s13, + int s20, int s21, int s22, int s23, + const sycl::nd_item<3> & item_ct1) { + const int i0s = item_ct1.get_local_range(2) * item_ct1.get_group(2) + + item_ct1.get_local_id(2); + const int i1 = (item_ct1.get_local_range(1) * item_ct1.get_group(1) + + item_ct1.get_local_id(1)); + const int i2 = (item_ct1.get_local_range(0) * item_ct1.get_group(0) + + item_ct1.get_local_id(0)) / + ne3; + const int i3 = (item_ct1.get_local_range(0) * item_ct1.get_group(0) + + item_ct1.get_local_id(0)) % + ne3; + + if (i0s >= ne0 || i1 >= ne1 || i2 >= ne2 || i3 >= ne3) { + return; + } + + const int i11 = i1 % ne11; + const int i12 = i2 % ne12; + const int i13 = i3 % ne13; + const int i21 = i1 % ne21; + const int i22 = i2 % ne22; + const int i23 = i3 % ne23; + + const size_t i_src0 = i3 * s03 + i2 * s02 + i1 * s01; + const size_t i_src1 = i13 * s13 + i12 * s12 + i11 * s11; + const size_t i_src2 = i23 * s23 + i22 * s22 + i21 * s21; + const size_t i_dst = i3 * s3 + i2 * s2 + i1 * s1; + + const src0_t * src0_row = src0 + i_src0; + const src1_t * src1_row = src1 + i_src1; + const src2_t * src2_row = src2 + i_src2; + dst_t * dst_row = dst + i_dst; + + for (int i0 = i0s; i0 < ne0; + i0 += item_ct1.get_local_range(2) * item_ct1.get_group_range(2)) { + const int i10 = i0 % ne10; + const int i20 = i0 % ne20; + const float acc = bin_op((float) src0_row[i0 * s00], (float) src1_row[i10 * s10]); + dst_row[i0] = (dst_t) bin_op(acc, (float) src2_row[i20 * s20]); + } +} + +template +static void k_bin_bcast3_unravel(const src0_t * src0, const src1_t * src1, const src2_t * src2, dst_t * dst, + int ne0, int ne1, int ne2, int ne3, + int ne10, int ne11, int ne12, int ne13, + int ne20, int ne21, int ne22, int ne23, + int s1, int s2, int s3, + int s00, int s01, int s02, int s03, + int s10, int s11, int s12, int s13, + int s20, int s21, int s22, int s23, + const sycl::nd_item<3> & item_ct1) { + const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) + + item_ct1.get_local_id(2); + + const int i3 = i / (ne2 * ne1 * ne0); + const int i2 = (i / (ne1 * ne0)) % ne2; + const int i1 = (i / ne0) % ne1; + const int i0 = i % ne0; + + if (i0 >= ne0 || i1 >= ne1 || i2 >= ne2 || i3 >= ne3) { + return; + } + + const int i11 = i1 % ne11; + const int i12 = i2 % ne12; + const int i13 = i3 % ne13; + const int i21 = i1 % ne21; + const int i22 = i2 % ne22; + const int i23 = i3 % ne23; + + const size_t i_src0 = i3 * s03 + i2 * s02 + i1 * s01; + const size_t i_src1 = i13 * s13 + i12 * s12 + i11 * s11; + const size_t i_src2 = i23 * s23 + i22 * s22 + i21 * s21; + const size_t i_dst = i3 * s3 + i2 * s2 + i1 * s1; + + const int i10 = i0 % ne10; + const int i20 = i0 % ne20; + const float acc = bin_op((float) src0[i_src0 + i0 * s00], (float) src1[i_src1 + i10 * s10]); + dst[i_dst + i0] = (dst_t) bin_op(acc, (float) src2[i_src2 + i20 * s20]); +} + +template +static void launch_bin_bcast3(ggml_backend_sycl_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, + const ggml_tensor * src2, ggml_tensor * dst) { + dpct::queue_ptr stream = ctx.stream(); + SYCL_CHECK(ggml_sycl_set_device(ctx.device)); + + GGML_TENSOR_TERNARY_OP_LOCALS + + int nr1[4] = { (int) (ne10 / ne0), (int) (ne11 / ne1), (int) (ne12 / ne2), (int) (ne13 / ne3) }; + int nr2[4] = { (int) (ne20 / ne0), (int) (ne21 / ne1), (int) (ne22 / ne2), (int) (ne23 / ne3) }; + + int64_t cne[] = { ne0, ne1, ne2, ne3 }; + int64_t cne0[] = { ne00, ne01, ne02, ne03 }; + int64_t cne1[] = { ne10, ne11, ne12, ne13 }; + int64_t cne2[] = { ne20, ne21, ne22, ne23 }; + size_t cnb[] = { nb0, nb1, nb2, nb3 }; + size_t cnb0[] = { nb00, nb01, nb02, nb03 }; + size_t cnb1[] = { nb10, nb11, nb12, nb13 }; + size_t cnb2[] = { nb20, nb21, nb22, nb23 }; + + auto collapse = [](int64_t cne[]) { + cne[0] *= cne[1]; + cne[1] = cne[2]; + cne[2] = cne[3]; + cne[3] = 1; + }; + + auto collapse_nb = [](size_t cnb[], int64_t cne[]) { + cnb[1] *= cne[1]; + cnb[2] *= cne[2]; + cnb[3] *= cne[3]; + }; + + const bool can_collapse = ggml_is_contiguous(src0) && ggml_is_contiguous(src1) && ggml_is_contiguous(src2) && + !ggml_is_permuted(src0) && !ggml_is_permuted(src1) && !ggml_is_permuted(src2); + if (can_collapse) { + for (int i = 0; i < 4; i++) { + if (nr1[i] != 1 || nr2[i] != 1) { + break; + } + if (i > 0) { + collapse_nb(cnb, cne); + collapse_nb(cnb0, cne0); + collapse_nb(cnb1, cne1); + collapse_nb(cnb2, cne2); + collapse(cne); + collapse(cne0); + collapse(cne1); + collapse(cne2); + } + } + } + + { + int64_t ne0 = cne[0]; + int64_t ne1 = cne[1]; + int64_t ne2 = cne[2]; + int64_t ne3 = cne[3]; + + int64_t ne10 = cne1[0]; + int64_t ne11 = cne1[1]; + int64_t ne12 = cne1[2]; + int64_t ne13 = cne1[3]; + + int64_t ne20 = cne2[0]; + int64_t ne21 = cne2[1]; + int64_t ne22 = cne2[2]; + int64_t ne23 = cne2[3]; + + size_t s1 = cnb[1] / sizeof(dst_t); + size_t s2 = cnb[2] / sizeof(dst_t); + size_t s3 = cnb[3] / sizeof(dst_t); + + size_t s00 = cnb0[0] / sizeof(src0_t); + size_t s01 = cnb0[1] / sizeof(src0_t); + size_t s02 = cnb0[2] / sizeof(src0_t); + size_t s03 = cnb0[3] / sizeof(src0_t); + + size_t s10 = cnb1[0] / sizeof(src1_t); + size_t s11 = cnb1[1] / sizeof(src1_t); + size_t s12 = cnb1[2] / sizeof(src1_t); + size_t s13 = cnb1[3] / sizeof(src1_t); + + size_t s20 = cnb2[0] / sizeof(src2_t); + size_t s21 = cnb2[1] / sizeof(src2_t); + size_t s22 = cnb2[2] / sizeof(src2_t); + size_t s23 = cnb2[3] / sizeof(src2_t); + + GGML_ASSERT(cnb[0] % sizeof(dst_t) == 0 && cnb[1] % sizeof(dst_t) == 0 && cnb[2] % sizeof(dst_t) == 0 && + cnb[3] % sizeof(dst_t) == 0); + GGML_ASSERT(cnb0[0] % sizeof(src0_t) == 0 && cnb0[1] % sizeof(src0_t) == 0 && cnb0[2] % sizeof(src0_t) == 0 && + cnb0[3] % sizeof(src0_t) == 0); + GGML_ASSERT(cnb1[0] % sizeof(src1_t) == 0 && cnb1[1] % sizeof(src1_t) == 0 && cnb1[2] % sizeof(src1_t) == 0 && + cnb1[3] % sizeof(src1_t) == 0); + GGML_ASSERT(cnb2[0] % sizeof(src2_t) == 0 && cnb2[1] % sizeof(src2_t) == 0 && cnb2[2] % sizeof(src2_t) == 0 && + cnb2[3] % sizeof(src2_t) == 0); + + const src0_t * src0_dd = (const src0_t *) src0->data; + const src1_t * src1_dd = (const src1_t *) src1->data; + const src2_t * src2_dd = (const src2_t *) src2->data; + dst_t * dst_dd = (dst_t *) dst->data; + + const int block_size = 128; + int64_t hne0 = std::max(ne0 / 2LL, 1LL); + + sycl::range<3> block_dims(1, 1, 1); + block_dims[2] = std::min(hne0, block_size); + block_dims[1] = std::min(ne1, block_size / (unsigned int) block_dims[2]); + block_dims[0] = std::min(std::min(ne2 * ne3, + block_size / (unsigned int) block_dims[2] / + (unsigned int) block_dims[1]), + 64U); + + sycl::range<3> block_nums((ne2 * ne3 + block_dims[0] - 1) / block_dims[0], + (ne1 + block_dims[1] - 1) / block_dims[1], + (hne0 + block_dims[2] - 1) / block_dims[2]); + + dpct::has_capability_or_fail(stream->get_device(), { sycl::aspect::fp16 }); + + if (block_nums[0] > 65535) { + int block_num = (ne0 * ne1 * ne2 * ne3 + block_size - 1) / block_size; + stream->parallel_for( + sycl::nd_range<3>(sycl::range<3>(1, 1, block_num) * sycl::range<3>(1, 1, block_size), + sycl::range<3>(1, 1, block_size)), + [=](sycl::nd_item<3> item_ct1) { + k_bin_bcast3_unravel(src0_dd, src1_dd, src2_dd, dst_dd, ne0, ne1, ne2, ne3, ne10, ne11, + ne12, ne13, ne20, ne21, ne22, ne23, s1, s2, s3, s00, s01, s02, s03, + s10, s11, s12, s13, s20, s21, s22, s23, item_ct1); + }); + } else { + stream->parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) { + k_bin_bcast3(src0_dd, src1_dd, src2_dd, dst_dd, ne0, ne1, ne2, ne3, ne10, + ne11, ne12, ne13, ne20, ne21, ne22, ne23, s1, s2, s3, s00, + s01, s02, s03, s10, s11, s12, s13, s20, s21, s22, s23, + item_ct1); + }); + } + } +} + +void ggml_sycl_op_add_add_fused(ggml_backend_sycl_context & ctx, ggml_tensor * add0, ggml_tensor * add1) { + const ggml_tensor * src0 = add0->src[0]; + const ggml_tensor * src1 = add0->src[1]; + const ggml_tensor * src2 = add1->src[1]; + ggml_tensor * dst = add1; + + GGML_ASSERT(add1->src[0] == add0); + GGML_ASSERT(ggml_sycl_add_kernel_supports(src0->type, src1->type, add0->type)); + GGML_ASSERT(ggml_sycl_add_kernel_supports(add0->type, src2->type, dst->type)); + + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_F32 && + dst->type == GGML_TYPE_F32) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F16 && src2->type == GGML_TYPE_F16 && + dst->type == GGML_TYPE_F16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_F32 && + dst->type == GGML_TYPE_F16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F16 && src2->type == GGML_TYPE_F32 && + dst->type == GGML_TYPE_F16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_F16 && + dst->type == GGML_TYPE_F16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_I32 && src1->type == GGML_TYPE_I32 && src2->type == GGML_TYPE_I32 && + dst->type == GGML_TYPE_I32) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_I16 && src1->type == GGML_TYPE_I16 && src2->type == GGML_TYPE_I16 && + dst->type == GGML_TYPE_I16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); +#ifdef GGML_SYCL_HAS_BF16 + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && src2->type == GGML_TYPE_BF16 && + dst->type == GGML_TYPE_BF16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_F32 && + dst->type == GGML_TYPE_BF16) { + launch_bin_bcast3( + ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_BF16 && src2->type == GGML_TYPE_F32 && + dst->type == GGML_TYPE_BF16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); + } else if (src0->type == GGML_TYPE_BF16 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_BF16 && + dst->type == GGML_TYPE_BF16) { + launch_bin_bcast3(ctx, src0, src1, src2, dst); +#endif + } else { + fprintf(stderr, "%s: unsupported types: dst: %s, src0: %s, src1: %s, src2: %s\n", __func__, + ggml_type_name(dst->type), ggml_type_name(src0->type), ggml_type_name(src1->type), + ggml_type_name(src2->type)); + GGML_ABORT("fatal error"); + } +} + diff --git a/ggml/src/ggml-sycl/binbcast.hpp b/ggml/src/ggml-sycl/binbcast.hpp index 9cce0f05..0e5a5ca1 100644 --- a/ggml/src/ggml-sycl/binbcast.hpp +++ b/ggml/src/ggml-sycl/binbcast.hpp @@ -34,6 +34,36 @@ void ggml_sycl_div(ggml_backend_sycl_context & ctx, ggml_tensor * dst); void ggml_sycl_repeat(ggml_backend_sycl_context & ctx, ggml_tensor * dst); +void ggml_sycl_op_add_add_fused(ggml_backend_sycl_context & ctx, ggml_tensor * add0, ggml_tensor * add1); + +// Type combinations the standalone SYCL add() kernel can run. Fused ADD+ADD +// uses the same set; anything else falls back to two add() launches. +inline bool ggml_sycl_add_kernel_supports(enum ggml_type src0, enum ggml_type src1, enum ggml_type dst) { + if (src0 == GGML_TYPE_F32 && src1 == GGML_TYPE_F32 && dst == GGML_TYPE_F32) { + return true; + } + if (src0 == GGML_TYPE_F16 && src1 == GGML_TYPE_F16 && dst == GGML_TYPE_F16) { + return true; + } + if (src0 == GGML_TYPE_F16 && src1 == GGML_TYPE_F32 && dst == GGML_TYPE_F16) { + return true; + } + if (src0 == GGML_TYPE_I32 && src1 == GGML_TYPE_I32 && dst == GGML_TYPE_I32) { + return true; + } + if (src0 == GGML_TYPE_I16 && src1 == GGML_TYPE_I16 && dst == GGML_TYPE_I16) { + return true; + } +#ifdef GGML_SYCL_HAS_BF16 + if (src0 == GGML_TYPE_BF16 && src1 == GGML_TYPE_BF16 && dst == GGML_TYPE_BF16) { + return true; + } + if (src0 == GGML_TYPE_BF16 && src1 == GGML_TYPE_F32 && dst == GGML_TYPE_BF16) { + return true; + } +#endif + return false; +} #endif //GGML_SYCL_BINBCAST_HPP diff --git a/ggml/src/ggml-sycl/common.cpp b/ggml/src/ggml-sycl/common.cpp index e1b6db13..89400694 100644 --- a/ggml/src/ggml-sycl/common.cpp +++ b/ggml/src/ggml-sycl/common.cpp @@ -94,7 +94,7 @@ static bool ggml_sycl_use_level_zero_device_alloc(sycl::queue &q) { // Use Level Zero zeMemAllocDevice to avoid sycl::malloc_device triggering // DMA-buf/TTM system RAM staging in the xe kernel driver during multi-GPU inference. -void * ggml_sycl_malloc_device(size_t size, sycl::queue &q) { +void * ggml_sycl_malloc_device(size_t size, sycl::queue &q, ggml_sycl_mem_type type) { #ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API if (ggml_sycl_use_level_zero_device_alloc(q)) { void *ptr = nullptr; @@ -117,16 +117,25 @@ void * ggml_sycl_malloc_device(size_t size, sycl::queue &q) { #endif ze_result_t r = zeMemAllocDevice(ze_ctx, &alloc_desc, size, 64, ze_dev, &ptr); if (r == ZE_RESULT_SUCCESS && ptr) { + ggml_sycl_memtrace_add(type, ptr, size); return ptr; } + ggml_sycl_memtrace_fail(type, size); return nullptr; } #endif - return sycl::malloc_device(size, q); + void * ptr = sycl::malloc_device(size, q); + if (ptr == nullptr) { + ggml_sycl_memtrace_fail(type, size); + return nullptr; + } + ggml_sycl_memtrace_add(type, ptr, size); + return ptr; } void ggml_sycl_free_device(void *ptr, sycl::queue &q) { if (!ptr) return; + ggml_sycl_memtrace_del(ptr); #ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API if (ggml_sycl_use_level_zero_device_alloc(q)) { auto ze_ctx = sycl::get_native(q.get_context()); diff --git a/ggml/src/ggml-sycl/common.hpp b/ggml/src/ggml-sycl/common.hpp index 34de284d..dc6cdd3d 100644 --- a/ggml/src/ggml-sycl/common.hpp +++ b/ggml/src/ggml-sycl/common.hpp @@ -18,6 +18,7 @@ #include #include +#include "base.hpp" #include "dpct/helper.hpp" #include "ggml.h" #include "ggml-impl.h" @@ -26,6 +27,7 @@ #include "type.hpp" #include "sycl_hw.hpp" #include "fattn-buffers.hpp" +#include "memtrace.hpp" namespace syclexp = sycl::ext::oneapi::experimental; @@ -67,23 +69,11 @@ extern int g_ggml_sycl_enable_flash_attention; extern int g_ggml_sycl_dev2dev_memcpy; extern int g_ggml_sycl_fa_onednn; extern int g_ggml_sycl_fa_onednn_max_kv; +extern int g_ggml_sycl_enable_mkl_fa; +extern int g_ggml_sycl_memtrace; +extern int g_ggml_sycl_memtrace_step; -#if defined(__clang__) && __has_builtin(__builtin_expect) -// Hint the optimizer to pipeline the more likely following instruction in branches -# define LIKELY(expr) __builtin_expect(expr, true) -# define UNLIKELY(expr) __builtin_expect(expr, false) -#else -# define LIKELY(expr) (expr) -# define UNLIKELY(expr) (expr) -#endif - -#define GGML_SYCL_DEBUG(...) \ - do { \ - if (UNLIKELY(g_ggml_sycl_debug)) \ - fprintf(stderr, __VA_ARGS__); \ - } while (0) - #define CHECK_TRY_ERROR(expr) \ [&]() { \ try { \ @@ -331,7 +321,8 @@ struct ggml_tensor_extra_gpu { }; extern int g_ggml_sycl_use_level_zero_api; -void * ggml_sycl_malloc_device(size_t size, sycl::queue &q); +void * ggml_sycl_malloc_device(size_t size, sycl::queue &q, + ggml_sycl_mem_type type = GGML_SYCL_MEM_DIRECT); void ggml_sycl_free_device(void *ptr, sycl::queue &q); void release_extra_gpu(ggml_tensor_extra_gpu * extra, std::vector streams={}); @@ -410,29 +401,10 @@ struct ggml_backend_sycl_context { dnnl::stream stream_dnnl() { return stream_dnnl(device, 0); } - dnnl::memory get_scratchpad_mem(const dnnl::memory::desc & scratchpad_md, - const dnnl::engine & eng, const queue_ptr q) { - ggml_sycl_pool_alloc * pool; - auto it = scratchpad_map.find(q); - if (it == scratchpad_map.end()) { - scratchpad_map[q] = std::make_unique>(this->pool()); - pool = scratchpad_map[q].get(); - } else { - pool = it->second.get(); - } - - size_t scratchpad_size = scratchpad_md.get_size(); - if (scratchpad_size > pool->actual_size) { - pool->realloc(scratchpad_size); - } - void * mem_ptr = pool->get(); - return dnnl::memory(scratchpad_md, eng, mem_ptr); - } #endif // pool std::unique_ptr pools[GGML_SYCL_MAX_DEVICES]; - std::unordered_map>> scratchpad_map; std::unique_ptr fattn_bufs[GGML_SYCL_MAX_DEVICES]; diff --git a/ggml/src/ggml-sycl/convert.cpp b/ggml/src/ggml-sycl/convert.cpp index 9ec92769..b660b56a 100644 --- a/ggml/src/ggml-sycl/convert.cpp +++ b/ggml/src/ggml-sycl/convert.cpp @@ -76,6 +76,19 @@ static void dequantize_row_q2_K_sycl(const void *vx, dst_t *y, const int64_t k, #endif } +template +static void dequantize_row_q2_K_sycl_reorder(const void *vx, dst_t *y, const int64_t k, + dpct::queue_ptr stream) { + const int64_t nb = k / QK_K; + + dpct::has_capability_or_fail(stream->get_device(), { sycl::aspect::fp16 }); + stream->parallel_for( + sycl::nd_range<3>(sycl::range<3>(1, 1, nb) * sycl::range<3>(1, 1, 64), sycl::range<3>(1, 1, 64)), + [=](sycl::nd_item<3> item_ct1) { + dequantize_block_q2_K_reorder(vx, y, item_ct1, nb); + }); +} + template static void dequantize_row_q3_K_sycl(const void *vx, dst_t *y, const int64_t k, dpct::queue_ptr stream) { @@ -667,7 +680,11 @@ to_fp16_sycl_t ggml_get_to_fp16_sycl(ggml_type type, ggml_tensor * dst) { return dequantize_block_sycl; } case GGML_TYPE_Q2_K: - return dequantize_row_q2_K_sycl; + if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { + return dequantize_row_q2_K_sycl_reorder; + } else { + return dequantize_row_q2_K_sycl; + } case GGML_TYPE_Q3_K: if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { return dequantize_row_q3_K_sycl_reorder; @@ -753,7 +770,11 @@ to_fp32_sycl_t ggml_get_to_fp32_sycl(ggml_type type, ggml_tensor *dst) { return dequantize_block_sycl; } case GGML_TYPE_Q2_K: - return dequantize_row_q2_K_sycl; + if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { + return dequantize_row_q2_K_sycl_reorder; + } else { + return dequantize_row_q2_K_sycl; + } case GGML_TYPE_Q3_K: if (dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { return dequantize_row_q3_K_sycl_reorder; diff --git a/ggml/src/ggml-sycl/dequantize.hpp b/ggml/src/ggml-sycl/dequantize.hpp index 876ba1b4..1b13e0f1 100644 --- a/ggml/src/ggml-sycl/dequantize.hpp +++ b/ggml/src/ggml-sycl/dequantize.hpp @@ -943,6 +943,47 @@ static void dequantize_block_q2_K(const void * __restrict__ vx, dst_t * __restri } +template +static void dequantize_block_q2_K_reorder(const void * __restrict__ vx, dst_t * __restrict__ yy, + const sycl::nd_item<3> & item_ct1, int64_t n_blocks) { +#if QK_K == 256 + const int64_t i = item_ct1.get_group(2); + if (i >= n_blocks) { + return; + } + + const uint8_t * base = static_cast(vx); + const size_t qs_offset = i * (QK_K / 4); + const size_t scales_offset = n_blocks * (QK_K / 4) + i * (QK_K / 16); + const size_t dm_offset = n_blocks * (QK_K / 4) + n_blocks * (QK_K / 16) + i * sizeof(ggml_half2); + + const uint8_t * qs = base + qs_offset; + const uint8_t * scales = base + scales_offset; + const ggml_half2 * dm = reinterpret_cast(base + dm_offset); + + const int64_t tid = item_ct1.get_local_id(2); + const int64_t n = tid / 32; + const int64_t l = tid - 32 * n; + const int64_t is = 8 * n + l / 16; + + const uint8_t q = qs[32 * n + l]; + dst_t * y = yy + i * QK_K + 128 * n; + + const float dall = (*dm)[0]; + const float dmin = (*dm)[1]; + y[l+ 0] = dall * (scales[is+0] & 0xF) * ((q >> 0) & 3) - dmin * (scales[is+0] >> 4); + y[l+32] = dall * (scales[is+2] & 0xF) * ((q >> 2) & 3) - dmin * (scales[is+2] >> 4); + y[l+64] = dall * (scales[is+4] & 0xF) * ((q >> 4) & 3) - dmin * (scales[is+4] >> 4); + y[l+96] = dall * (scales[is+6] & 0xF) * ((q >> 6) & 3) - dmin * (scales[is+6] >> 4); +#else + GGML_UNUSED(vx); + GGML_UNUSED(yy); + GGML_UNUSED(item_ct1); + GGML_UNUSED(n_blocks); + GGML_ABORT("Q2_K reorder dequantize not supported for QK_K != 256"); +#endif +} + template static void dequantize_block_q3_K(const void * __restrict__ vx, dst_t * __restrict__ yy, const sycl::nd_item<3> &item_ct1) { diff --git a/ggml/src/ggml-sycl/dmmv.cpp b/ggml/src/ggml-sycl/dmmv.cpp index d8da0a16..d47d6831 100644 --- a/ggml/src/ggml-sycl/dmmv.cpp +++ b/ggml/src/ggml-sycl/dmmv.cpp @@ -1921,6 +1921,23 @@ ESIMD_INLINE void dequantize_mul_mat_vec_reorder_esimd( } } +static void dequantize_mul_mat_vec_q2_K_sycl_reorder_esimd(const void *vx, const float *y, + float *dst, const int ncols, + const int nrows, + dpct::queue_ptr stream) { + GGML_ASSERT(ncols % QK_K == 0); + const int workgroups = (nrows + 1) / 2; + stream->submit([&](sycl::handler &h) { + sycl::local_accessor lmem(sycl::range<1>(GGML_SYCL_DMMV_ESIMD_WG_SIZE * 2), h); + h.parallel_for( + sycl::nd_range<1>(sycl::range<1>((size_t)workgroups * GGML_SYCL_DMMV_ESIMD_WG_SIZE), sycl::range<1>(GGML_SYCL_DMMV_ESIMD_WG_SIZE)), + [=](sycl::nd_item<1> it) [[intel::sycl_explicit_simd]] { + dequantize_mul_mat_vec_reorder_esimd( + vx, y, dst, ncols, nrows, lmem, it); + }); + }); +} + static void dequantize_mul_mat_vec_q3_K_sycl_reorder_esimd(const void *vx, const float *y, float *dst, const int ncols, const int nrows, @@ -1955,6 +1972,23 @@ static void dequantize_mul_mat_vec_q4_K_sycl_reorder_esimd(const void *vx, const }); } +static void dequantize_mul_mat_vec_q5_K_sycl_reorder_esimd(const void *vx, const float *y, + float *dst, const int ncols, + const int nrows, + dpct::queue_ptr stream) { + GGML_ASSERT(ncols % QK_K == 0); + const int workgroups = (nrows + 1) / 2; + stream->submit([&](sycl::handler &h) { + sycl::local_accessor lmem(sycl::range<1>(GGML_SYCL_DMMV_ESIMD_WG_SIZE * 2), h); + h.parallel_for( + sycl::nd_range<1>(sycl::range<1>((size_t)workgroups * GGML_SYCL_DMMV_ESIMD_WG_SIZE), sycl::range<1>(GGML_SYCL_DMMV_ESIMD_WG_SIZE)), + [=](sycl::nd_item<1> it) [[intel::sycl_explicit_simd]] { + dequantize_mul_mat_vec_reorder_esimd( + vx, y, dst, ncols, nrows, lmem, it); + }); + }); +} + static void dequantize_mul_mat_vec_q6_K_sycl_reorder_esimd(const void *vx, const float *y, float *dst, const int ncols, const int nrows, @@ -2094,7 +2128,15 @@ void ggml_sycl_op_dequantize_mul_mat_vec( case GGML_TYPE_Q2_K: if ((ggml_tensor_extra_gpu *) dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { - dequantize_mul_mat_vec_q2_K_sycl_reorder(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); +#ifdef GGML_SYCL_DMMV_HAS_ESIMD + if (g_ggml_sycl_enable_esimd) { + dequantize_mul_mat_vec_q2_K_sycl_reorder_esimd(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); + } + else +#endif + { + dequantize_mul_mat_vec_q2_K_sycl_reorder(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); + } } else { dequantize_mul_mat_vec_q2_K_sycl(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); } @@ -2134,7 +2176,15 @@ void ggml_sycl_op_dequantize_mul_mat_vec( case GGML_TYPE_Q5_K: if ((ggml_tensor_extra_gpu *) dst->src[0]->extra && ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { - dequantize_mul_mat_vec_q5_K_sycl_reorder(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); +#ifdef GGML_SYCL_DMMV_HAS_ESIMD + if (g_ggml_sycl_enable_esimd) { + dequantize_mul_mat_vec_q5_K_sycl_reorder_esimd(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); + } + else +#endif + { + dequantize_mul_mat_vec_q5_K_sycl_reorder(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); + } } else { dequantize_mul_mat_vec_q5_K_sycl(src0_dd_i, src1_ddf_i, dst_dd_i, ne00, row_diff, stream); } diff --git a/ggml/src/ggml-sycl/dpct/helper.hpp b/ggml/src/ggml-sycl/dpct/helper.hpp index 664b8e96..85af4cab 100644 --- a/ggml/src/ggml-sycl/dpct/helper.hpp +++ b/ggml/src/ggml-sycl/dpct/helper.hpp @@ -62,7 +62,7 @@ #define DPCT_UNUSED(x) (void)(x) -inline void _abort(const char * str) { +[[noreturn]] inline void _abort(const char * str) { std::cerr << str << std::endl; std::abort(); } diff --git a/ggml/src/ggml-sycl/dsv4-hc.cpp b/ggml/src/ggml-sycl/dsv4-hc.cpp index bb66e8c1..337f4af4 100644 --- a/ggml/src/ggml-sycl/dsv4-hc.cpp +++ b/ggml/src/ggml-sycl/dsv4-hc.cpp @@ -2,22 +2,30 @@ #include "dsv4-hc.hpp" #include +#include static constexpr int DSV4_HC = 4; +// tunable: one work-item per (embedding element, token) +static constexpr int dsv4_hc_pre_block_size = 256; + +// gated: the weight is a per-element gate [n_embd, hc, n_tokens] passed through a sigmoid. +// otherwise it is one weight per (stream, token). +template static void dsv4_hc_pre_f32_sycl( const float * x, const float * weights, float * dst, int64_t n_embd, int64_t hc, int64_t n_tokens, int64_t sx0, int64_t sx1, int64_t sx2, - int64_t sw0, int64_t sw1, + int64_t sw0, int64_t sw1, int64_t sw2, int64_t sd0, int64_t sd1, + float scale, queue_ptr stream) { const int64_t nr = n_embd * n_tokens; - const int64_t block_size = 256; - const int64_t num_blocks = (nr + block_size - 1) / block_size; + const int64_t num_blocks = (nr + dsv4_hc_pre_block_size - 1) / dsv4_hc_pre_block_size; stream->parallel_for( - sycl::nd_range<1>(sycl::range<1>(num_blocks * block_size), sycl::range<1>(block_size)), + sycl::nd_range<1>(sycl::range<1>(num_blocks * dsv4_hc_pre_block_size), + sycl::range<1>(dsv4_hc_pre_block_size)), [=](sycl::nd_item<1> item) { const int64_t ir = item.get_global_id(0); if (ir >= nr) { @@ -27,14 +35,20 @@ static void dsv4_hc_pre_f32_sycl( const int64_t i0 = ir % n_embd; const int64_t it = ir / n_embd; - float sum = x[i0*sx0 + it*sx2] * weights[it*sw1]; - for (int64_t ih = 1; ih < hc; ++ih) { + float sum = 0.0f; + for (int64_t ih = 0; ih < hc; ++ih) { const float xv = x[i0*sx0 + ih*sx1 + it*sx2]; - const float wv = weights[ih*sw0 + it*sw1]; + float wv; + if constexpr (gated) { + const float gv = weights[i0*sw0 + ih*sw1 + it*sw2]; + wv = 1.0f / (1.0f + sycl::exp(-gv)); + } else { + wv = weights[ih*sw0 + it*sw1]; + } sum += xv * wv; } - dst[i0*sd0 + it*sd1] = sum; + dst[i0*sd0 + it*sd1] = scale * sum; }); } @@ -138,6 +152,12 @@ static void dsv4_hc_comb_f32_sycl( }); } +// tunable: one work-item per (embedding element, stream, token) +static constexpr int dsv4_hc_post_block_size = 256; + +// comb == nullptr is identity mixing: each destination stream keeps its own residual +// instead of summing across the streams. +template static void dsv4_hc_post_f32_sycl( const float * x, const float * residual, const float * post, const float * comb, float * dst, int64_t n_embd, int64_t hc, int64_t n_tokens, @@ -148,7 +168,7 @@ static void dsv4_hc_post_f32_sycl( int64_t sd0, int64_t sd1, int64_t sd2, queue_ptr stream) { const int64_t nr = n_embd * hc * n_tokens; - const int64_t block_size = 256; + const int64_t block_size = dsv4_hc_post_block_size; const int64_t num_blocks = (nr + block_size - 1) / block_size; stream->parallel_for( @@ -164,8 +184,12 @@ static void dsv4_hc_post_f32_sycl( const int64_t it = ir / (n_embd * hc); float sum = x[i0*sx0 + it*sx1] * post[idst*sp0 + it*sp1]; - for (int64_t isrc = 0; isrc < hc; ++isrc) { - sum += residual[i0*sr0 + isrc*sr1 + it*sr2] * comb[idst*sc0 + isrc*sc1 + it*sc2]; + if constexpr (has_comb) { + for (int64_t isrc = 0; isrc < hc; ++isrc) { + sum += residual[i0*sr0 + isrc*sr1 + it*sr2] * comb[idst*sc0 + isrc*sc1 + it*sc2]; + } + } else { + sum += residual[i0*sr0 + idst*sr1 + it*sr2]; } dst[i0*sd0 + idst*sd1 + it*sd2] = sum; @@ -189,15 +213,33 @@ void ggml_sycl_op_dsv4_hc_pre(ggml_backend_sycl_context & ctx, ggml_tensor * dst const int64_t hc = x->ne[1]; const int64_t n_tokens = x->ne[2]; + const float scale = ggml_get_op_params_f32(dst, 0); + const bool gated = ggml_get_op_params_i32(dst, 1) != 0; + queue_ptr stream = ctx.stream(); - dsv4_hc_pre_f32_sycl( - (const float *) x->data, (const float *) weights->data, (float *) dst->data, - n_embd, hc, n_tokens, - nbx0 / sizeof(float), nbx1 / sizeof(float), nbx2 / sizeof(float), - nbw0 / sizeof(float), nbw1 / sizeof(float), - nbd0 / sizeof(float), nbd1 / sizeof(float), - stream); + if (gated) { + GGML_ASSERT(weights->ne[0] == n_embd); + GGML_ASSERT(weights->ne[1] == hc); + GGML_ASSERT(weights->ne[2] == n_tokens); + dsv4_hc_pre_f32_sycl( + (const float *) x->data, (const float *) weights->data, (float *) dst->data, + n_embd, hc, n_tokens, + nbx0 / sizeof(float), nbx1 / sizeof(float), nbx2 / sizeof(float), + nbw0 / sizeof(float), nbw1 / sizeof(float), nbw2 / sizeof(float), + nbd0 / sizeof(float), nbd1 / sizeof(float), + scale, stream); + } else { + GGML_ASSERT(weights->ne[0] == hc); + GGML_ASSERT(weights->ne[1] == n_tokens); + dsv4_hc_pre_f32_sycl( + (const float *) x->data, (const float *) weights->data, (float *) dst->data, + n_embd, hc, n_tokens, + nbx0 / sizeof(float), nbx1 / sizeof(float), nbx2 / sizeof(float), + nbw0 / sizeof(float), nbw1 / sizeof(float), /*sw2=*/ 0, + nbd0 / sizeof(float), nbd1 / sizeof(float), + scale, stream); + } } void ggml_sycl_op_dsv4_hc_comb(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { @@ -252,24 +294,33 @@ void ggml_sycl_op_dsv4_hc_post(ggml_backend_sycl_context & ctx, ggml_tensor * ds GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(residual->type == GGML_TYPE_F32); GGML_ASSERT(post->type == GGML_TYPE_F32); - GGML_ASSERT(comb->type == GGML_TYPE_F32); GGML_ASSERT(dst->type == GGML_TYPE_F32); GGML_TENSOR_LOCALS(size_t, nbx, x, nb); GGML_TENSOR_LOCALS(size_t, nbr, residual, nb); GGML_TENSOR_LOCALS(size_t, nbp, post, nb); - GGML_TENSOR_LOCALS(size_t, nbc, comb, nb); GGML_TENSOR_LOCALS(size_t, nbd, dst, nb); + size_t nbc0 = 0; + size_t nbc1 = 0; + size_t nbc2 = 0; + if (comb) { + GGML_ASSERT(comb->type == GGML_TYPE_F32); + nbc0 = comb->nb[0]; + nbc1 = comb->nb[1]; + nbc2 = comb->nb[2]; + } + const int64_t n_embd = x->ne[0]; const int64_t n_tokens = x->ne[1]; const int64_t hc = residual->ne[1]; queue_ptr stream = ctx.stream(); - dsv4_hc_post_f32_sycl( + const auto launch = [&](auto has_comb) { + dsv4_hc_post_f32_sycl( (const float *) x->data, (const float *) residual->data, - (const float *) post->data, (const float *) comb->data, (float *) dst->data, + (const float *) post->data, comb ? (const float *) comb->data : nullptr, (float *) dst->data, n_embd, hc, n_tokens, nbx0 / sizeof(float), nbx1 / sizeof(float), nbr0 / sizeof(float), nbr1 / sizeof(float), nbr2 / sizeof(float), @@ -277,4 +328,11 @@ void ggml_sycl_op_dsv4_hc_post(ggml_backend_sycl_context & ctx, ggml_tensor * ds nbc0 / sizeof(float), nbc1 / sizeof(float), nbc2 / sizeof(float), nbd0 / sizeof(float), nbd1 / sizeof(float), nbd2 / sizeof(float), stream); + }; + + if (comb) { + launch(std::true_type{}); + } else { + launch(std::false_type{}); + } } diff --git a/ggml/src/ggml-sycl/element_wise.cpp b/ggml/src/ggml-sycl/element_wise.cpp index 8619ed6f..2e926abe 100644 --- a/ggml/src/ggml-sycl/element_wise.cpp +++ b/ggml/src/ggml-sycl/element_wise.cpp @@ -10,7 +10,7 @@ (ITEM.get_local_range(IDX) * ITEM.get_group(IDX) + ITEM.get_local_id(IDX)) static void acc_f32(const char * x, const char * y, float * dst, const int64_t ne, - const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t ne3, + const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t nb00, const int64_t nb01, const int64_t nb02, const int64_t nb03, const int64_t ne10, const int64_t ne11, const int64_t ne12, const int64_t ne13, const int64_t nb10, const int64_t nb11, const int64_t nb12, const int64_t nb13, @@ -455,7 +455,7 @@ static void unary_mul_sycl(const T * x, const T * g, T * dst, const int64_t k, c namespace ggml_sycl_detail { static void acc_f32_sycl(const char *x, const char *y, float *dst, const int64_t n_elements, - const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t ne3, + const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t nb00, const int64_t nb01, const int64_t nb02, const int64_t nb03, const int64_t ne10, const int64_t ne11, const int64_t ne12, const int64_t ne13, const int64_t nb10, const int64_t nb11, const int64_t nb12, const int64_t nb13, @@ -466,7 +466,7 @@ static void acc_f32_sycl(const char *x, const char *y, float *dst, sycl::range<3>(1, 1, SYCL_ACC_BLOCK_SIZE)), [=](sycl::nd_item<3> /*item_ct1*/) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { acc_f32(x, y, dst, n_elements, - ne0, ne1, ne2, ne3, + ne0, ne1, ne2, nb00, nb01, nb02, nb03, ne10, ne11, ne12, ne13, nb10, nb11, nb12, nb13, @@ -970,7 +970,7 @@ static inline void ggml_sycl_op_acc(ggml_backend_sycl_context & ctx, ggml_tensor const int64_t offset = (int64_t) ((const int32_t *) dst->op_params)[3] / (int64_t) sizeof(float); ggml_sycl_detail::acc_f32_sycl(src0_d, src1_d, dst_d, ggml_nelements(dst), - dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], + dst->ne[0], dst->ne[1], dst->ne[2], src0->nb[0], src0->nb[1], src0->nb[2], src0->nb[3], src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], src1->nb[0], src1->nb[1], src1->nb[2], src1->nb[3], @@ -1132,6 +1132,102 @@ void ggml_sycl_op_swiglu_oai(ggml_backend_sycl_context & ctx, ggml_tensor * dst) swiglu_oai_sycl(src0_p, src1_p, (float *)dst_d, ggml_nelements(dst), nc, src0_o / sizeof(float), src1_o / sizeof(float), alpha, limit, stream); } +template +static void swiglu_clamp_kernel(const T * gate, + const T * up, + T * dst, + const int64_t k, + const int64_t n, + const int64_t o0, + const int64_t o1, + float limit, + sycl::nd_item<3> item_ct1) { + const int64_t i = int64_t(item_ct1.get_local_range(2)) * item_ct1.get_group(2) + item_ct1.get_local_id(2); + + if (i >= k) { + return; + } + + const int64_t j0 = (i / n) * o0 + (i % n); + const int64_t j1 = o0 == o1 ? j0 : (i / n) * o1 + (i % n); + + const float gate_value = sycl::fmin((float) gate[j0], limit); + const float up_value = sycl::fmax(sycl::fmin((float) up[j1], limit), -limit); + dst[i] = (T) (gate_value / (1.0f + sycl::native::exp(-gate_value)) * up_value); +} + +template +static void swiglu_clamp_sycl(const T * gate, + const T * up, + T * dst, + const int64_t k, + const int64_t n, + const int64_t o0, + const int64_t o1, + float limit, + dpct::queue_ptr stream) { + const int64_t num_blocks = (k + SYCL_GLU_BLOCK_SIZE - 1) / SYCL_GLU_BLOCK_SIZE; + stream->parallel_for(sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) * sycl::range<3>(1, 1, SYCL_GLU_BLOCK_SIZE), + sycl::range<3>(1, 1, SYCL_GLU_BLOCK_SIZE)), + [=](sycl::nd_item<3> item_ct1) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + swiglu_clamp_kernel(gate, up, dst, k, n, o0, o1, limit, item_ct1); + }); +} + +static void ggml_sycl_op_swiglu_clamp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + void * src0_d = src0->data; + void * src1_d = src1 ? src1->data : src0->data; + const int64_t src0_o = src0->nb[1]; + const int64_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; + void * dst_d = dst->data; + const int64_t nc = src1 ? src0->ne[0] : src0->ne[0] / 2; + dpct::queue_ptr stream = ctx.stream(); + + GGML_ASSERT(ggml_is_contiguous_1(src0)); + GGML_ASSERT(src0->nb[0] == ggml_element_size(src0)); + GGML_ASSERT(ggml_is_contiguous(dst)); + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT(src0->type == dst->type); + GGML_ASSERT(dst->ne[0] == nc); + GGML_ASSERT(ggml_nrows(dst) == ggml_nrows(src0)); + + if (src1) { + GGML_ASSERT(ggml_is_contiguous_1(src1)); + GGML_ASSERT(src1->nb[0] == ggml_element_size(src1)); + GGML_ASSERT(src1->ne[0] == nc); + GGML_ASSERT(src0->type == src1->type); + } + + const int32_t swapped = ggml_get_op_params_i32(dst, 1); + const float limit = ggml_get_op_params_f32(dst, 3); + + if (src0->type == GGML_TYPE_F16) { + sycl::half * src0_p = (sycl::half *) src0_d; + sycl::half * src1_p = (sycl::half *) src1_d; + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + swiglu_clamp_sycl(src0_p, src1_p, (sycl::half *) dst_d, ggml_nelements(dst), nc, src0_o / sizeof(sycl::half), + src1_o / sizeof(sycl::half), limit, stream); + } else { + float * src0_p = (float *) src0_d; + float * src1_p = (float *) src1_d; + + if (!src1) { + src0_p += swapped ? nc : 0; + src1_p += swapped ? 0 : nc; + } + + swiglu_clamp_sycl(src0_p, src1_p, (float *) dst_d, ggml_nelements(dst), nc, src0_o / sizeof(float), + src1_o / sizeof(float), limit, stream); + } +} + static inline void ggml_sycl_op_geglu_erf(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { ggml_sycl_detail::ggml_sycl_op_unary_gated(ctx, dst, [](auto x) { return op_gelu_erf(x); @@ -1295,6 +1391,11 @@ void ggml_sycl_swiglu_oai(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { ggml_sycl_op_swiglu_oai(ctx, dst); } +void ggml_sycl_swiglu_clamp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { + scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/1); + ggml_sycl_op_swiglu_clamp(ctx, dst); +} + void ggml_sycl_geglu_erf(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/1); ggml_sycl_op_geglu_erf(ctx, dst); diff --git a/ggml/src/ggml-sycl/element_wise.hpp b/ggml/src/ggml-sycl/element_wise.hpp index 67bf422d..d280066e 100644 --- a/ggml/src/ggml-sycl/element_wise.hpp +++ b/ggml/src/ggml-sycl/element_wise.hpp @@ -77,6 +77,7 @@ void ggml_sycl_silu(ggml_backend_sycl_context & ctx, ggml_tensor * dst); void ggml_sycl_gelu_quick(ggml_backend_sycl_context & ctx, ggml_tensor * dst); void ggml_sycl_swiglu_oai(ggml_backend_sycl_context & ctx, ggml_tensor * dst); +void ggml_sycl_swiglu_clamp(ggml_backend_sycl_context & ctx, ggml_tensor * dst); void ggml_sycl_gelu_erf(ggml_backend_sycl_context & ctx, ggml_tensor * dst); diff --git a/ggml/src/ggml-sycl/esimd.hpp b/ggml/src/ggml-sycl/esimd.hpp index d7609b11..0485ff0c 100644 --- a/ggml/src/ggml-sycl/esimd.hpp +++ b/ggml/src/ggml-sycl/esimd.hpp @@ -1,15 +1,3 @@ -// -// MIT license -// Copyright (C) 2026 Intel Corporation -// SPDX-License-Identifier: MIT -// - -// -// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. -// See https://llvm.org/LICENSE.txt for license information. -// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception -// - #ifndef GGML_SYCL_ESIMD_HPP #define GGML_SYCL_ESIMD_HPP @@ -73,6 +61,93 @@ static ESIMD_INLINE void unpack_scale_min_k4( min_f = convert(m) * (-dmin); } +// --------------------------------------------------------------------------- +// Q2_K, SOA reorder layout produced by reorder_qw_q2_k: +// [qs: nb*(QK_K/4)] [scales: nb*(QK_K/16)] [dm: nb*sizeof(half2)] +// with nb = nrows*num_blocks_per_row. +// +// 2 bits per weight. The 8 output chunks of 32 (matching dequantize_row_q2_K) +// map to super-chunk s (0..7): byte base 32*(s/4) into the 64-byte qs array, +// bit shift 2*(s%4); the low 16 lanes use scales[2s], the high 16 use +// scales[2s+1], with dl = d*(sc & 0xF), ml = dmin*(sc >> 4), deq = dl*q - ml. +// --------------------------------------------------------------------------- +template <> struct esimd_reorder_q_traits { + struct ptrs { + const uint8_t * qs; + const uint8_t * scales; + const sycl::half * dm; + }; + + static ESIMD_INLINE ptrs make_ptrs(const void * vx, size_t nb) { + const uint8_t * qs = (const uint8_t *) vx; + const uint8_t * scales = qs + nb * (QK_K / 4); + const sycl::half * dm = (const sycl::half *) (scales + nb * (QK_K / 16)); + return { qs, scales, dm }; + } + + static ESIMD_INLINE void mac_pair( + const ptrs & pa, size_t bia, + const ptrs & pb, size_t bib, bool has_b, + sycl::ext::intel::esimd::simd & y_vec, + sycl::ext::intel::esimd::simd & acc_a, + sycl::ext::intel::esimd::simd & acc_b) { + using namespace sycl::ext::intel::esimd; + + simd qs_a = block_load(pa.qs + bia * (QK_K / 4)); + simd qs_b = 0; + simd scales_a = block_load(pa.scales + bia * (QK_K / 16)); + simd scales_b = 0; + + const float dall_a = (float) pa.dm[bia * 2 + 0]; + const float dmin_a = (float) pa.dm[bia * 2 + 1]; + float dall_b = 0.0f; + float dmin_b = 0.0f; + if (has_b) { + qs_b = block_load(pb.qs + bib * (QK_K / 4)); + scales_b = block_load(pb.scales + bib * (QK_K / 16)); + dall_b = (float) pb.dm[bib * 2 + 0]; + dmin_b = (float) pb.dm[bib * 2 + 1]; + } + + // per-chunk scale (d * (sc & 0xF)) and min (-dmin * (sc >> 4)), all 16 codes; + // min carries the negation so the dequant epilogue adds (matches Q4_K/Q5_K) + simd scale_f_a = convert(scales_a & simd(0x0F)) * dall_a; + simd min_f_a = convert(scales_a >> simd(4)) * (-dmin_a); + simd scale_f_b = convert(scales_b & simd(0x0F)) * dall_b; + simd min_f_b = convert(scales_b >> simd(4)) * (-dmin_b); + +#pragma unroll + for (int s = 0; s < 8; ++s) { + const int byte_base = 32 * (s / 4); + const uint8_t shift = (uint8_t) (2 * (s % 4)); + simd y_s = y_vec.select<32, 1>(s * 32); + + simd qa = (qs_a.select<32, 1>(byte_base) >> shift) & simd(3); + simd qb = (qs_b.select<32, 1>(byte_base) >> shift) & simd(3); + + const float scale_a_lo = scale_f_a[2 * s + 0]; + const float scale_a_hi = scale_f_a[2 * s + 1]; + const float min_a_lo = min_f_a[2 * s + 0]; + const float min_a_hi = min_f_a[2 * s + 1]; + const float scale_b_lo = scale_f_b[2 * s + 0]; + const float scale_b_hi = scale_f_b[2 * s + 1]; + const float min_b_lo = min_f_b[2 * s + 0]; + const float min_b_hi = min_f_b[2 * s + 1]; + + simd scale_vec_a = splat_lo_hi(scale_a_lo, scale_a_hi); + simd min_vec_a = splat_lo_hi(min_a_lo, min_a_hi); + simd scale_vec_b = splat_lo_hi(scale_b_lo, scale_b_hi); + simd min_vec_b = splat_lo_hi(min_b_lo, min_b_hi); + + simd deq_a = convert(qa) * scale_vec_a + min_vec_a; + simd deq_b = convert(qb) * scale_vec_b + min_vec_b; + + acc_a += y_s * deq_a; + acc_b += y_s * deq_b; + } + } +}; + // --------------------------------------------------------------------------- // Q3_K, SOA reorder layout produced by reorder_qw_q3_k: // [qs: nb*(QK_K/4)] [hmask: nb*(QK_K/8)] [scales: nb*12] [d: nb*sizeof(half)] @@ -287,6 +362,128 @@ template <> struct esimd_reorder_q_traits { } }; +// --------------------------------------------------------------------------- +// Q5_K, SOA reorder layout produced by reorder_qw_q5_k: +// [qs: nb*(QK_K/2)] [qh: nb*(QK_K/8)] [scales: nb*K_SCALE_SIZE] [dm: nb*sizeof(half2)] +// with nb = nrows*num_blocks_per_row. +// +// Identical to Q4_K except each 4-bit quant gains a 5th (high) bit from qh: +// output chunk c (0..7) adds 16 when bit c of qh[l] is set, where qh[l] indexes +// the same 32 bytes for every chunk (matches dequantize_row_q5_K). +// --------------------------------------------------------------------------- +template <> struct esimd_reorder_q_traits { + struct ptrs { + const uint8_t * qs; + const uint8_t * qh; + const uint8_t * scales; + const sycl::half * dm; + }; + + static ESIMD_INLINE ptrs make_ptrs(const void * vx, size_t nb) { + const uint8_t * qs = (const uint8_t *) vx; + const uint8_t * qh = qs + nb * (QK_K / 2); + const uint8_t * scales = qh + nb * (QK_K / 8); + const sycl::half * dm = (const sycl::half *) (scales + nb * K_SCALE_SIZE); + return { qs, qh, scales, dm }; + } + + // extract bit `bit` (0..7) of each lane and move it to bit position 4, + // e.g. for the 4-bit base quant's 5th (high) bit. `bit` is always a + // compile-time-known unrolled loop constant at call sites, so this folds + // to a single mask (bit==4), mask+left-shift (bit<4), or mask+right-shift + // (bit>4) instead of the shift+mask+shift a naive `(qh>>bit & 1) << 4` emits. + static ESIMD_INLINE sycl::ext::intel::esimd::simd extract_bit_to_pos4( + sycl::ext::intel::esimd::simd qh, int bit) { + using namespace sycl::ext::intel::esimd; + simd masked = convert(qh & simd((uint8_t) (1u << bit))); + if (bit < 4) { + return masked << simd((uint16_t) (4 - bit)); + } else if (bit > 4) { + return masked >> simd((uint16_t) (bit - 4)); + } + return masked; + } + + static ESIMD_INLINE void mac_pair( + const ptrs & pa, size_t bia, + const ptrs & pb, size_t bib, bool has_b, + sycl::ext::intel::esimd::simd & y_vec, + sycl::ext::intel::esimd::simd & acc_a, + sycl::ext::intel::esimd::simd & acc_b) { + using namespace sycl::ext::intel::esimd; + + simd qs_a = block_load(pa.qs + bia * (QK_K / 2)); + simd qs_b = 0; + simd qh_a = block_load(pa.qh + bia * (QK_K / 8)); + simd qh_b = 0; + simd scales_a = block_load(pa.scales + bia * K_SCALE_SIZE); + simd scales_b = 0; + + const float dall_a = (float) pa.dm[bia * 2 + 0]; + const float dmin_a = (float) pa.dm[bia * 2 + 1]; + float dall_b = 0.0f; + float dmin_b = 0.0f; + if (has_b) { + qs_b = block_load(pb.qs + bib * (QK_K / 2)); + qh_b = block_load(pb.qh + bib * (QK_K / 8)); + scales_b = block_load(pb.scales + bib * K_SCALE_SIZE); + dall_b = (float) pb.dm[bib * 2 + 0]; + dmin_b = (float) pb.dm[bib * 2 + 1]; + } + + simd scale_f_a, min_f_a, scale_f_b, min_f_b; + unpack_scale_min_k4(scales_a, dall_a, dmin_a, scale_f_a, min_f_a); + unpack_scale_min_k4(scales_b, dall_b, dmin_b, scale_f_b, min_f_b); + + simd qs_lo_a = qs_a & simd(0x0F); + simd qs_hi_a = qs_a >> simd(4); + simd qs_lo_b = qs_b & simd(0x0F); + simd qs_hi_b = qs_b >> simd(4); + +#pragma unroll + for (int sb = 0; sb < 8; sb += 2) { + const int q_offset = sb * 16; + simd y_lo = y_vec.select<32, 1>(sb * 32); + simd y_hi = y_vec.select<32, 1>((sb + 1) * 32); + + const float scale_a_lo = scale_f_a[sb]; + const float scale_a_hi = scale_f_a[sb + 1]; + const float min_a_lo = min_f_a[sb]; + const float min_a_hi = min_f_a[sb + 1]; + const float scale_b_lo = scale_f_b[sb]; + const float scale_b_hi = scale_f_b[sb + 1]; + const float min_b_lo = min_f_b[sb]; + const float min_b_hi = min_f_b[sb + 1]; + + simd qa_lo_u8 = qs_lo_a.select<32, 1>(q_offset); + simd qa_hi_u8 = qs_hi_a.select<32, 1>(q_offset); + simd qb_lo_u8 = qs_lo_b.select<32, 1>(q_offset); + simd qb_hi_u8 = qs_hi_b.select<32, 1>(q_offset); + simd qa_lo = convert(qa_lo_u8); + simd qa_hi = convert(qa_hi_u8); + simd qb_lo = convert(qb_lo_u8); + simd qb_hi = convert(qb_hi_u8); + + // add the 5th bit: chunk sb uses qh bit sb, chunk sb+1 uses qh bit sb+1; + // qh always indexes the same 32 bytes regardless of chunk + qa_lo += extract_bit_to_pos4(qh_a, sb); + qa_hi += extract_bit_to_pos4(qh_a, sb + 1); + qb_lo += extract_bit_to_pos4(qh_b, sb); + qb_hi += extract_bit_to_pos4(qh_b, sb + 1); + + simd deq_a_lo = convert(qa_lo) * scale_a_lo + min_a_lo; + simd deq_a_hi = convert(qa_hi) * scale_a_hi + min_a_hi; + simd deq_b_lo = convert(qb_lo) * scale_b_lo + min_b_lo; + simd deq_b_hi = convert(qb_hi) * scale_b_hi + min_b_hi; + + acc_a += y_lo * deq_a_lo; + acc_b += y_lo * deq_b_lo; + acc_a += y_hi * deq_a_hi; + acc_b += y_hi * deq_b_hi; + } + } +}; + // --------------------------------------------------------------------------- // Q6_K, SOA reorder layout: // [ql: nb*(QK_K/2)] [qh: nb*(QK_K/4)] [scales(int8): nb*(QK_K/16)] [d: nb*half] diff --git a/ggml/src/ggml-sycl/fattn-buffers.cpp b/ggml/src/ggml-sycl/fattn-buffers.cpp index 46cf6d55..78a52d2a 100644 --- a/ggml/src/ggml-sycl/fattn-buffers.cpp +++ b/ggml/src/ggml-sycl/fattn-buffers.cpp @@ -21,6 +21,7 @@ sycl::half * ggml_sycl_fattn_kv_buffers::kv_buffer::ensure_half(size_t n_elems) if (ptr) { SYCL_CHECK(CHECK_TRY_ERROR(qptr->wait())); + ggml_sycl_memtrace_del(ptr); SYCL_CHECK(CHECK_TRY_ERROR(sycl::free(ptr, *qptr))); ptr = nullptr; capacity = 0; @@ -38,11 +39,13 @@ sycl::half * ggml_sycl_fattn_kv_buffers::kv_buffer::ensure_half(size_t n_elems) if (!dev_ptr) { GGML_LOG_ERROR("%s: can't allocate %lu Bytes of memory on device\n", __func__, cap); + ggml_sycl_memtrace_fail(GGML_SYCL_MEM_FATTN_KV, cap); GGML_ABORT("fattn buffer alloc failed"); } ptr = static_cast(dev_ptr); capacity = cap; + ggml_sycl_memtrace_add(GGML_SYCL_MEM_FATTN_KV, ptr, cap); return ptr; } @@ -51,6 +54,7 @@ ggml_sycl_fattn_kv_buffers::kv_buffer::~kv_buffer() { GGML_LOG_INFO("ggml_sycl_fattn_kv_buffer[%d]: %.2f MiB\n", device, capacity / 1024.0 / 1024.0); #endif if (ptr) { + ggml_sycl_memtrace_del(ptr); SYCL_CHECK(CHECK_TRY_ERROR(sycl::free(ptr, *qptr))); } } diff --git a/ggml/src/ggml-sycl/fattn-common.hpp b/ggml/src/ggml-sycl/fattn-common.hpp index c6cc13cf..82813f7a 100644 --- a/ggml/src/ggml-sycl/fattn-common.hpp +++ b/ggml/src/ggml-sycl/fattn-common.hpp @@ -6,6 +6,7 @@ #include "convert.hpp" #include "vecdotq.hpp" #include "fattn-buffers.hpp" +#include "fattn.hpp" #include "ggml.h" @@ -926,6 +927,7 @@ void launch_fattn( ggml_sycl_fattn_alloc K_f16(fbuf.K); ggml_sycl_fattn_alloc V_f16(fbuf.V); + const ggml_sycl_fattn_extra extra = ggml_sycl_fattn_get_extra(dst); ggml_sycl_pool_alloc KV_max(pool); ggml_sycl_pool_alloc dst_tmp(pool); ggml_sycl_pool_alloc dst_tmp_meta(pool); @@ -944,10 +946,11 @@ void launch_fattn( const size_t bs = ggml_blck_size(K->type); const size_t ts = ggml_type_size(K->type); - K_f16.alloc(ggml_nelements(K)); + sycl::half * K_f16_ptr = extra.K_buffer_ptr ? (sycl::half *) extra.K_buffer_ptr + : K_f16.alloc(ggml_nelements(K)); if (ggml_is_contiguously_allocated(K)) { to_fp16_sycl_t to_fp16 = ggml_get_to_fp16_sycl(K->type, dst); - to_fp16(K_data, K_f16.ptr, ggml_nelements(K), main_stream); + to_fp16(K_data, K_f16_ptr, ggml_nelements(K), main_stream); nb11 = nb11 * bs * sizeof(sycl::half) / ts; nb12 = nb12 * bs * sizeof(sycl::half) / ts; @@ -958,13 +961,13 @@ void launch_fattn( const int64_t s01 = nb11 / ts; const int64_t s02 = nb12 / ts; const int64_t s03 = nb13 / ts; - to_fp16(K_data, K_f16.ptr, K->ne[0], K->ne[1], K->ne[2], K->ne[3], s01, s02, s03, main_stream); + to_fp16(K_data, K_f16_ptr, K->ne[0], K->ne[1], K->ne[2], K->ne[3], s01, s02, s03, main_stream); nb11 = K->ne[0] * sizeof(sycl::half); nb12 = K->ne[1] * nb11; nb13 = K->ne[2] * nb12; } - K_data = (char *) K_f16.ptr; + K_data = (char *) K_f16_ptr; } if (need_f16_V && V->type != GGML_TYPE_F16) { @@ -977,11 +980,12 @@ void launch_fattn( const size_t bs = ggml_blck_size(V->type); const size_t ts = ggml_type_size(V->type); - V_f16.alloc(ggml_nelements(V)); + sycl::half * V_f16_ptr = extra.V_buffer_ptr ? (sycl::half *) extra.V_buffer_ptr + : V_f16.alloc(ggml_nelements(V)); if (ggml_is_contiguously_allocated(V)) { to_fp16_sycl_t to_fp16 = ggml_get_to_fp16_sycl(V->type, dst); - to_fp16(V_data, V_f16.ptr, ggml_nelements(V), main_stream); - V_data = (char *) V_f16.ptr; + to_fp16(V_data, V_f16_ptr, ggml_nelements(V), main_stream); + V_data = (char *) V_f16_ptr; nb21 = nb21 * bs * sizeof(sycl::half) / ts; nb22 = nb22 * bs * sizeof(sycl::half) / ts; @@ -992,13 +996,13 @@ void launch_fattn( const int64_t s01 = nb21 / ts; const int64_t s02 = nb22 / ts; const int64_t s03 = nb23 / ts; - to_fp16(V_data, V_f16.ptr, V->ne[0], V->ne[1], V->ne[2], V->ne[3], s01, s02, s03, main_stream); + to_fp16(V_data, V_f16_ptr, V->ne[0], V->ne[1], V->ne[2], V->ne[3], s01, s02, s03, main_stream); nb21 = V->ne[0] * sizeof(sycl::half); nb22 = V->ne[1] * nb21; nb23 = V->ne[2] * nb22; } - V_data = (char *) V_f16.ptr; + V_data = (char *) V_f16_ptr; } } diff --git a/ggml/src/ggml-sycl/fattn-mkl.cpp b/ggml/src/ggml-sycl/fattn-mkl.cpp index fc22b7bd..30947b17 100644 --- a/ggml/src/ggml-sycl/fattn-mkl.cpp +++ b/ggml/src/ggml-sycl/fattn-mkl.cpp @@ -43,7 +43,7 @@ static void mkl_fa_pack_q_fp16( dpct::queue_ptr stream, sycl::half * __restrict dst, const float * __restrict q_src, - int n_queries, int n_query_rows, int DKQ, + int n_queries, int DKQ, int gqa_ratio, int kvh_base_head, float q_scale, int64_t q_row_stride, int64_t q_head_stride, int64_t wg_size) { @@ -110,8 +110,15 @@ static void mkl_fa_init_softmax_state( // The tile spans absolute rows [q0, q0 + q_rows). Score buffers // (KQ_f32/S_f16) are indexed RELATIVE to the tile; the persistent state // (VKQ_accum/KQ_max/KQ_sum) and mask are indexed by ABSOLUTE row. -// For each row: find local max → rescale previous VKQ_accum → -// compute exp(s - max) → write S_f16 → update running max/sum. +// One WORK-GROUP per query row (local size = wg_size): work-items stride +// over the chunk so adjacent items touch adjacent elements (coalesced), +// the row max/sum come from group reductions, and the DV-long VKQ +// rescale is spread across the items. Item 0 is the sole writer of +// KQ_max/KQ_sum; its writes are ordered after every other item's reads +// by the second group reduction (a collective). Per-element math is +// identical to the original one-item-per-row kernel: softcap before +// mask, native::exp, -1e30 sentinel, half-precision S. Only the float +// summation order differs (tree vs serial), i.e. last-ulp level. static void mkl_fa_online_softmax_chunk( dpct::queue_ptr stream, float * __restrict KQ_f32, @@ -121,30 +128,32 @@ static void mkl_fa_online_softmax_chunk( float * __restrict VKQ_accum, int q0, int q_rows, int n_queries, int DV, int chunk_size, int chunk_start, - int kvh_head, int gqa_ratio, + int kvh_head, const sycl::half * mask_data, int64_t mask_head_stride, int64_t mask_row_stride, int mask_n_heads, float logit_softcap, int64_t wg_size) { - const int64_t wg = ((q_rows + wg_size - 1) / wg_size) * wg_size; - + // One work-group per query row: exactly q_rows groups of wg_size + // items. q_rows * wg_size is already a multiple of wg_size, so unlike + // the one-item-per-row kernels there is no round-up / tail guard. + const int64_t wg = q_rows * wg_size; + const int local_size = (int) wg_size; // stride in the loops below stream->submit([&](sycl::handler & cgh) { cgh.parallel_for(sycl::nd_range<1>(wg, wg_size), [=](sycl::nd_item<1> item) { - int jc_rel = item.get_global_id(0); - if (jc_rel >= q_rows) return; - int jc_abs = q0 + jc_rel; - + const int local_id = (int)item.get_local_id(0); + const int row = (int)item.get_group(0); // tile-relative + const int jc_abs = q0 + row; const int gqa_group = jc_abs / n_queries; const int q_row = jc_abs % n_queries; - // Score buffers are tile-local (relative index). const float * __restrict KQ_row = KQ_f32 - + jc_rel * (int64_t)chunk_size; + + row * (int64_t)chunk_size; + sycl::half * __restrict S_row = S_f16 + + row * (int64_t)chunk_size; // Persistent accumulator is full-sized (absolute index). float * __restrict vkq = VKQ_accum + jc_abs * (int64_t)DV; - const sycl::half * mask_h = nullptr; int64_t m_stride = 0; if (mask_data) { @@ -153,10 +162,8 @@ static void mkl_fa_online_softmax_chunk( mask_h = mask_data + (int64_t)m_head * mask_head_stride; m_stride = mask_row_stride; } - - // Row-wise local maximum (softcap before mask) - float local_max = -1e30f; - for (int i = 0; i < chunk_size; i++) { + // Score at chunk offset i — original per-element math. + auto score = [&](int i) { float s = KQ_row[i]; if (logit_softcap != 0.0f) { s = logit_softcap * sycl::tanh(s); @@ -165,40 +172,38 @@ static void mkl_fa_online_softmax_chunk( s += (float)mask_h[q_row * m_stride + (chunk_start + i)]; } + return s; + }; + // Pass 1: strided (coalesced) row-wise local maximum. + float local_max = -1e30f; + for (int i = local_id; i < chunk_size; i += local_size) { + float s = score(i); if (s > local_max) local_max = s; } - + const float final_local_max = sycl::reduce_over_group( + item.get_group(), local_max, sycl::maximum()); // Rescale previous accumulator by exp(old_max - new_max) float old_max = KQ_max[jc_abs]; - float new_max = (old_max > local_max) ? old_max : local_max; + float new_max = (old_max > final_local_max) ? old_max : final_local_max; float rescale = (old_max < -1e29f) ? 1.0f : sycl::native::exp(old_max - new_max); - - for (int v = 0; v < DV; v++) { + for (int v = local_id; v < DV; v += local_size) { vkq[v] *= rescale; } - - // Softmax and write S_f16 (tile-local index) + // Pass 2: softmax numerators, strided; S row written once. float local_sum = 0.0f; - sycl::half * __restrict S_row = S_f16 - + jc_rel * (int64_t)chunk_size; - - for (int i = 0; i < chunk_size; i++) { - float s = KQ_row[i]; - if (logit_softcap != 0.0f) { - s = logit_softcap * sycl::tanh(s); - } - if (mask_h) { - s += (float)mask_h[q_row * m_stride - + (chunk_start + i)]; - } + for (int i = local_id; i < chunk_size; i += local_size) { + float s = score(i); float val = sycl::native::exp(s - new_max); S_row[i] = sycl::half(val); local_sum += val; } - - KQ_sum[jc_abs] = KQ_sum[jc_abs] * rescale + local_sum; - KQ_max[jc_abs] = new_max; + const float total_sum = sycl::reduce_over_group( + item.get_group(), local_sum, sycl::plus()); + if (local_id == 0) { + KQ_sum[jc_abs] = KQ_sum[jc_abs] * rescale + total_sum; + KQ_max[jc_abs] = new_max; + } }); }); } @@ -473,7 +478,6 @@ void ggml_sycl_flash_attn_ext_mkl(ggml_backend_sycl_context & ctx, ggml_tensor * MKL_ACCUM(dequant_time_us, t_deq); // --- Resolve mask pointers --- - const sycl::half * mask_data = nullptr; int64_t mask_head_stride = 0; int64_t mask_row_stride = 0; int mask_n_heads = 0; @@ -547,7 +551,7 @@ void ggml_sycl_flash_attn_ext_mkl(ggml_backend_sycl_context & ctx, ggml_tensor * // 1. Pack all GQA Q heads into fp16 (full n_query_rows) mkl_fa_pack_q_fp16(stream, Q_head_f16_ptr, Q_batch, - n_queries, n_query_rows, DKQ, + n_queries, DKQ, gqa_ratio, kvh_base_head, q_scale, q_row_stride, q_head_stride, wg_size); @@ -605,7 +609,7 @@ void ggml_sycl_flash_attn_ext_mkl(ggml_backend_sycl_context & ctx, ggml_tensor * KQ_max_ptr, KQ_sum_ptr, VKQ_accum_ptr, q0, q_rows, n_queries, DV, this_chunk, chunk_start, - kvh_base_head, gqa_ratio, + kvh_base_head, mask_batch, mask_head_stride, mask_row_stride, mask_n_heads, logit_softcap, wg_size); diff --git a/ggml/src/ggml-sycl/fattn-onednn.cpp b/ggml/src/ggml-sycl/fattn-onednn.cpp index fd17a25d..4349363a 100644 --- a/ggml/src/ggml-sycl/fattn-onednn.cpp +++ b/ggml/src/ggml-sycl/fattn-onednn.cpp @@ -1,3 +1,4 @@ +#include #include #include #include @@ -13,22 +14,26 @@ // set minimum query length to treat as prefill (32) #define GGML_SYCL_FA_ONEDNN_MIN_Q 32 -bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst) { +bool ggml_sycl_fattn_onednn_binds_kv(const ggml_tensor * K, const ggml_tensor * V) { + if (K->type != GGML_TYPE_F16 || V->type != GGML_TYPE_F16) { + return false; + } + auto bindable = [](const ggml_tensor * t) { + return t->nb[0] == sizeof(sycl::half) && t->nb[1] % sizeof(sycl::half) == 0 && + t->nb[2] % sizeof(sycl::half) == 0 && t->nb[3] % sizeof(sycl::half) == 0; + }; + return bindable(K) && bindable(V); +} + +bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst, bool use_shape_limit) { #if !GGML_SYCL_DNNL GGML_UNUSED(dst); + GGML_UNUSED(use_shape_limit); return false; #else if (!g_ggml_sycl_fa_onednn) { return false; } - // Battlemage (Xe2) only, for now. On other Intel archs oneDNN's fused SDPA returns wrong results - // for some shapes (e.g. head_dim=64 on Arc / xe_hpg) -- an oneDNN bug tracked upstream at - // https://github.com/uxlfoundation/oneDNN/issues/5510. Remove this hardware limitation once that - // is fixed; until then non-BMG archs fall back to the existing FA kernel. - const gpu_arch arch = ggml_sycl_info().devices[ggml_sycl_get_device()].hw_info.arch; - if (arch != gpu_arch::intel_gpu_bmg_g21 && arch != gpu_arch::intel_gpu_bmg_g31) { - return false; - } const ggml_tensor * Q = dst->src[0]; const ggml_tensor * K = dst->src[1]; const ggml_tensor * V = dst->src[2]; @@ -51,7 +56,7 @@ bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst) { if (!k_ok || !v_ok) { return false; } - if (Q->ne[1] < 32 || K->ne[1] < 1024) { + if (use_shape_limit && (Q->ne[1] < 32 || K->ne[1] < 1024)) { return false; } for (const ggml_tensor * t : {K, V}) { @@ -60,6 +65,17 @@ bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst) { } } } + // This is the improved SPDA gate. Rather than gating Alchemist GPUs from all SPDA features, we instead target only the failing shapes. + // If the GPU being assessed isn't in the grouping below, it has full access to all SPDA shapes. Otherwise, if it's an Alchemist GPU, we block only the shapes with head sizes that fail. + // It is much easier to compare the device to a small list of failing cases than to define all the passing ones. + const gpu_arch arch = ggml_sycl_info().devices[ggml_sycl_get_device()].hw_info.arch; + bool support_spda = !(arch == gpu_arch::intel_gpu_dg2_g10 || + arch == gpu_arch::intel_gpu_dg2_g11 || + arch == gpu_arch::intel_gpu_dg2_g12); + + if (!support_spda && K->ne[0] == 64) { + return false; + } // Optional KV-length ceiling (GGML_SYCL_FA_ONEDNN_MAX_KV, 0 = unlimited). Escape hatch: // very long sequences make the fused SDPA slow enough to risk the xe driver watchdog on // some stacks; past the cap we fall back to the native FA kernel instead. @@ -90,7 +106,7 @@ bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst) { return false; } // Prefill only. - if (Q->ne[1] < GGML_SYCL_FA_ONEDNN_MIN_Q) { + if (use_shape_limit && Q->ne[1] < GGML_SYCL_FA_ONEDNN_MIN_Q) { return false; } return true; @@ -147,7 +163,8 @@ struct sdpa_partition { // Build + compile the contiguous-input GQA SDPA graph (MatMul->Divide->Add->SoftMax->MatMul), f32 out. // Mirrors the hardware-verified scratch/onednn_sdpa_probe.cpp build_gqa (partitions=1, sdp_primitive_kernel_t). -static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int seq, int d) { +static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int seq, int d, + const std::array & k_str, const std::array & v_str) try { using ltype = logical_tensor::layout_type; using dt = logical_tensor::data_type; using ldims = logical_tensor::dims; @@ -155,11 +172,12 @@ static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int const int rep = H / Hkv; const ldims q_sz = {1, Hkv, rep, q, d}, kv_sz = {1, Hkv, 1, seq, d}, s_sz = {1, Hkv, rep, q, seq}, sc = {1, 1, 1, 1, 1}, msk = {1, 1, 1, q, seq}, o_sz = {1, Hkv, rep, q, d}; + const ldims k_st(k_str.begin(), k_str.end()), v_st(v_str.begin(), v_str.end()); int64_t id = 0; sdpa_partition E; auto query = logical_tensor(id++, t, q_sz, ltype::strided); - auto key = logical_tensor(id++, t, kv_sz, ltype::strided); + auto key = logical_tensor(id++, t, kv_sz, k_st); auto score = logical_tensor(id++, fi, s_sz, ltype::strided); auto bmm1 = op(id++, op::kind::MatMul, "bmm1"); bmm1.set_attr(op::attr::transpose_b, true); // key is [.., seq, d] @@ -181,7 +199,7 @@ static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int smax.set_attr(op::attr::mode, "inf_as_zero"); smax.add_inputs({masked}); smax.add_outputs({probs}); - auto value = logical_tensor(id++, t, kv_sz, ltype::strided); + auto value = logical_tensor(id++, t, kv_sz, v_st); // f16 output is REQUIRED to hit sdp_primitive_kernel_t (the systolic micro-kernel); an f32 output // falls to larger_partition_kernel_t which materializes N^2 (confirmed: scratch/onednn_sdpa_kernel_probe.cpp). // converted to the f32 ggml dst in the permute below. @@ -195,6 +213,7 @@ static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int auto parts = g.get_partitions(); if (parts.size() != 1 || !parts[0].is_supported()) { + GGML_LOG_WARN("%s: oneDNN did not fuse the SDPA graph; falling back to TILE kernel\n", __func__); return E; // ok stays false -> caller falls back to TILE } E.ins = parts[0].get_input_ports(); @@ -206,6 +225,12 @@ static sdpa_partition build_sdpa(const engine & eng, int H, int Hkv, int q, int E.ok = true; return E; } +catch (const std::exception & e) { + // compile() can reject a stride set the partitioner never inspects; memoise the failure so the + // fallback costs one build rather than one per call. + GGML_LOG_WARN("%s: oneDNN SDPA partition build failed (%s); falling back to TILE kernel\n", __func__, e.what()); + return {}; +} void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tensor * dst) try { const ggml_tensor * Q = dst->src[0]; @@ -227,27 +252,53 @@ void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tenso dnnl::engine eng = ctx.engine_dnnl(stream); dnnl::stream strm = ctx.stream_dnnl(stream); + const ggml_sycl_fattn_extra extra = ggml_sycl_fattn_get_extra(dst); + // Q: always f32 -- copy to dense f16. - ggml_sycl_pool_alloc Qf(ctx.pool(), (size_t) H * q * d); - cont_to_f16_sycl((const char *) Q->data, Qf.get(), d, q, H, mb, Q->nb[1], Q->nb[2], Q->nb[3], stream); + std::optional> Qf_pool; + sycl::half * Qf_ptr = (sycl::half *) extra.Q_buffer_ptr; + if (!Qf_ptr) { + Qf_pool.emplace(ctx.pool(), (size_t) H * q * d); + Qf_ptr = Qf_pool->get(); + } + cont_to_f16_sycl((const char *) Q->data, Qf_ptr, d, q, H, mb, Q->nb[1], Q->nb[2], Q->nb[3], stream); - // K/V: use pool-alloc for both F16 and dequant paths. + // K/V: bind the f16 cache in place. llama.cpp permutes it to [token][head][dim], so its head + // plane is strided rather than dense, which is what an explicit stride vector expresses. + // Quantized and f32 KV still stage a dense copy -- the layout the k_str/v_str defaults describe. sycl::half * K_ptr = nullptr; sycl::half * V_ptr = nullptr; + std::array k_str{ Hkv * seq * d, seq * d, seq * d, d, 1 }; + std::array v_str = k_str; std::optional> Kf_pool; std::optional> Vf_pool; + // Helper: hand out reserved space, or fall back to the pool. + auto stage_k = [&](size_t n) { if (extra.K_buffer_ptr) { return (sycl::half *) extra.K_buffer_ptr; } + Kf_pool.emplace(ctx.pool(), n); return Kf_pool->get(); }; + auto stage_v = [&](size_t n) { if (extra.V_buffer_ptr) { return (sycl::half *) extra.V_buffer_ptr; } + Vf_pool.emplace(ctx.pool(), n); return Vf_pool->get(); }; - if (K->type == GGML_TYPE_F16 && V->type == GGML_TYPE_F16) { - Kf_pool.emplace(ctx.pool(), (size_t) Hkv * seq * d); - Vf_pool.emplace(ctx.pool(), (size_t) Hkv * seq * d); - cont_to_f16_sycl((const char *) K->data, Kf_pool->get(), d, seq, Hkv, mb, K->nb[1], K->nb[2], K->nb[3], stream); - cont_to_f16_sycl((const char *) V->data, Vf_pool->get(), d, seq, Hkv, mb, V->nb[1], V->nb[2], V->nb[3], stream); - K_ptr = Kf_pool->get(); - V_ptr = Vf_pool->get(); + auto elem_strides = [](const ggml_tensor * t) { + const int64_t s1 = (int64_t) (t->nb[1] / t->nb[0]); + const int64_t s2 = (int64_t) (t->nb[2] / t->nb[0]); + const int64_t s3 = (int64_t) (t->nb[3] / t->nb[0]); + // dims are {mb=1, Hkv, rep=1, seq, d}; the size-1 dims at 0 and 2 never advance an address. + return std::array{ s3, s2, s2, s1, 1 }; + }; + + if (ggml_sycl_fattn_onednn_binds_kv(K, V)) { + K_ptr = (sycl::half *) K->data; + V_ptr = (sycl::half *) V->data; + k_str = elem_strides(K); + v_str = elem_strides(V); + } else if (K->type == GGML_TYPE_F16 && V->type == GGML_TYPE_F16) { + K_ptr = stage_k((size_t) Hkv * seq * d); + V_ptr = stage_v((size_t) Hkv * seq * d); + cont_to_f16_sycl((const char *) K->data, K_ptr, d, seq, Hkv, mb, K->nb[1], K->nb[2], K->nb[3], stream); + cont_to_f16_sycl((const char *) V->data, V_ptr, d, seq, Hkv, mb, V->nb[1], V->nb[2], V->nb[3], stream); } else if (ggml_is_quantized(K->type)) { // Quantized K/V: dequant to dense F16 using pool, same lifetime as F16 path. - Kf_pool.emplace(ctx.pool(), ggml_nelements(K)); - K_ptr = Kf_pool->get(); + K_ptr = stage_k((size_t) ggml_nelements(K)); { const char * K_data = (const char *)K->data; const bool k_non_dense = ((int64_t)K->ne[1] * K->nb[1] != K->nb[2]) && K->ne[2] > 1; @@ -281,8 +332,7 @@ void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tenso // data pointer), their logical values differ because the quantized // elements at different positions/offsets represent different K/V // data. Master's F16 path also never aliases K and V. - Vf_pool.emplace(ctx.pool(), ggml_nelements(V)); - V_ptr = Vf_pool->get(); + V_ptr = stage_v((size_t) ggml_nelements(V)); { const char * V_data = (const char *)V->data; const bool v_non_dense = ((int64_t)V->ne[1] * V->nb[1] != V->nb[2]) && V->ne[2] > 1; @@ -313,12 +363,10 @@ void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tenso } } else { // F32: strided copy to dense F16 via cont_to_f16_sycl. - Kf_pool.emplace(ctx.pool(), ggml_nelements(K)); - K_ptr = Kf_pool->get(); + K_ptr = stage_k((size_t) ggml_nelements(K)); cont_to_f16_sycl((const char *) K->data, K_ptr, K->ne[0], K->ne[1], K->ne[2], K->ne[3], K->nb[1], K->nb[2], K->nb[3], stream); - Vf_pool.emplace(ctx.pool(), ggml_nelements(V)); - V_ptr = Vf_pool->get(); + V_ptr = stage_v((size_t) ggml_nelements(V)); cont_to_f16_sycl((const char *) V->data, V_ptr, V->ne[0], V->ne[1], V->ne[2], V->ne[3], V->nb[1], V->nb[2], V->nb[3], stream); } @@ -332,28 +380,43 @@ void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tenso // instead -- the value is captured into the command, so no host memory has to outlive the // call, and the enqueue stays async. const sycl::half scale_h = (sycl::half) (1.0f / kq_scale); - ggml_sycl_pool_alloc scbuf(ctx.pool(), 1); - sycl::half * const scale_dev = scbuf.get(); + std::optional> scbuf; + sycl::half * scale_dev = (sycl::half *) extra.scale_buffer_ptr; + if (!scale_dev) { + scbuf.emplace(ctx.pool(), 1); + scale_dev = scbuf->get(); + } stream->single_task([=]() { *scale_dev = scale_h; }); - ggml_sycl_pool_alloc outf(ctx.pool(), (size_t) H * q * d); // f16 contiguous SDPA out [mb,H,q,d] + // f16 contiguous SDPA out [mb,H,q,d] + std::optional> outf_pool; + sycl::half * outf_ptr = (sycl::half *) extra.out_buffer_ptr; + if (!outf_ptr) { + outf_pool.emplace(ctx.pool(), (size_t) H * q * d); + outf_ptr = outf_pool->get(); + } - // compile once per (device, shape), reuse across layers/calls. + // compile once per (device, shape, KV strides), reuse across layers/calls. Stride 2 always + // repeats stride 1 and stride 4 is always 1, so the key covers every entry that can differ. static std::unordered_map cache; - char keyb[96]; - snprintf(keyb, sizeof(keyb), "%d:%lld:%lld:%lld:%lld:%lld", ggml_sycl_get_device(), - (long long) H, (long long) Hkv, (long long) q, (long long) seq, (long long) d); + char keyb[256]; + snprintf(keyb, sizeof(keyb), "%d:%lld:%lld:%lld:%lld:%lld:%lld:%lld:%lld:%lld:%lld:%lld", ggml_sycl_get_device(), + (long long) H, (long long) Hkv, (long long) q, (long long) seq, (long long) d, + (long long) k_str[0], (long long) k_str[1], (long long) k_str[3], + (long long) v_str[0], (long long) v_str[1], (long long) v_str[3]); auto it = cache.find(keyb); if (it == cache.end()) { - it = cache.emplace(keyb, build_sdpa(eng, (int) H, (int) Hkv, (int) q, (int) seq, (int) d)).first; + it = cache.emplace(keyb, build_sdpa(eng, (int) H, (int) Hkv, (int) q, (int) seq, (int) d, k_str, v_str)).first; } sdpa_partition & E = it->second; - // _supported() is authoritative: if it accepted this op the partition must build. - // A failure here is a gap in _supported() -- surface it, don't mask it with a fallback. - GGML_ASSERT(E.ok && "oneDNN SDPA partition failed to build for a _supported() shape"); + if (!E.ok) { + // oneDNN can decline a shape or a stride set that _supported() never sees; build_sdpa warns per key. + ggml_sycl_flash_attn_ext_tile(ctx, dst); + return; + } auto id2ptr = [&](size_t r) -> void * { - if (r == E.id_q) return Qf.get(); + if (r == E.id_q) return Qf_ptr; if (r == E.id_k) return K_ptr; if (r == E.id_v) return V_ptr; if (r == E.id_scale) return scale_dev; @@ -365,10 +428,10 @@ void ggml_sycl_flash_attn_ext_onednn(ggml_backend_sycl_context & ctx, ggml_tenso for (auto & lt : E.ins) { ti.emplace_back(lt, eng, id2ptr(lt.get_id())); } - tensor to(E.out, eng, outf.get()); + tensor to(E.out, eng, outf_ptr); E.cp.execute(strm, ti, {to}); - permute_sdpa_out_sycl(outf.get(), (float *) dst->data, mb, H, q, d, stream); + permute_sdpa_out_sycl(outf_ptr, (float *) dst->data, mb, H, q, d, stream); // Single device needs no sync: the dnnl stream wraps this same in-order queue, so the SDPA // serializes with the staging kernels before it and the permute/pool reuse after it. The // garbage output formerly blamed on the missing sync here was the scale use-after-return diff --git a/ggml/src/ggml-sycl/fattn-onednn.hpp b/ggml/src/ggml-sycl/fattn-onednn.hpp index d3019e87..9669d1bd 100644 --- a/ggml/src/ggml-sycl/fattn-onednn.hpp +++ b/ggml/src/ggml-sycl/fattn-onednn.hpp @@ -5,7 +5,11 @@ // Static-only check: fused-XMX oneDNN Graph SDPA path==flash-attn op // (f16 KV, no softcap/ALiBi, single stream, tuned head_dim, prefill-sized q.) -bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst); +bool ggml_sycl_flash_attn_ext_onednn_supported(const ggml_tensor * dst, bool use_shape_limit = true); + +// True when the oneDNN path binds an F16 KV cache in place instead of staging a dense copy of +// it. Depends only on the types and strides of K and V, so the answer holds for every call. +bool ggml_sycl_fattn_onednn_binds_kv(const ggml_tensor * K, const ggml_tensor * V); // Run flash attention through oneDNN's fused xmx SDPA // execute the cached SDPA partition, write the f32 dst. Falls back to the TILE kernel on any failure. diff --git a/ggml/src/ggml-sycl/fattn-tile.hpp b/ggml/src/ggml-sycl/fattn-tile.hpp index 9ba52969..dcdcad88 100644 --- a/ggml/src/ggml-sycl/fattn-tile.hpp +++ b/ggml/src/ggml-sycl/fattn-tile.hpp @@ -1173,6 +1173,10 @@ static void launch_fattn_tile_switch_ncols2(ggml_backend_sycl_context & ctx, ggm launch_fattn_tile_switch_ncols1(ctx, dst); return; } + if (use_gqa_opt && gqa_ratio % 8 == 0) { + launch_fattn_tile_switch_ncols1(ctx, dst); + return; + } if (use_gqa_opt && gqa_ratio % 4 == 0) { launch_fattn_tile_switch_ncols1(ctx, dst); return; diff --git a/ggml/src/ggml-sycl/fattn.cpp b/ggml/src/ggml-sycl/fattn.cpp index a85eb721..394cda59 100644 --- a/ggml/src/ggml-sycl/fattn.cpp +++ b/ggml/src/ggml-sycl/fattn.cpp @@ -104,7 +104,6 @@ enum best_fattn_kernel { static best_fattn_kernel ggml_sycl_get_best_fattn_kernel(const int device, const ggml_tensor * dst) { - GGML_UNUSED(device); #ifndef SYCL_FLASH_ATTN GGML_UNUSED(dst); return BEST_FATTN_KERNEL_NONE; @@ -147,14 +146,13 @@ static best_fattn_kernel ggml_sycl_get_best_fattn_kernel(const int device, const // Set GGML_SYCL_ENABLE_MKL_FA=0 to force TILE/VEC path for A/B testing. // Example: GGML_SYCL_ENABLE_MKL_FA=0 llama-cli -m model.gguf -fa -ngl 99 ... // Note: MKL GEMM calls are incompatible with SYCL graph capture replay. - static int mkl_enable = ggml_sycl_get_env("GGML_SYCL_ENABLE_MKL_FA", 1); // MKL is validated for the mainstream GQA envelope: grouped-query // (gqa_ratio >= 2), head_dim a multiple of 64 in [64,512] with matching // K/V head size, mask, no sinks/ALiBi/softcap. Gemma's global layers use // head_dim 512, so the cap must include it. Head sizes not a multiple of // 64 (72/80/96), MHA (gqa_ratio == 1), and MLA (DKQ != DV, e.g. 576/512) // fall through to TILE/VEC; see follow-up work. - if (mkl_enable == 1 && mask && !sinks && gqa_ratio >= 2 && + if (g_ggml_sycl_enable_mkl_fa == 1 && mask && !sinks && gqa_ratio >= 2 && Q->ne[0] >= 64 && Q->ne[0] <= 512 && Q->ne[0] % 64 == 0 && Q->ne[0] == V->ne[0] && Q->ne[1] >= 32 && K->ne[1] >= 1024 && @@ -263,6 +261,11 @@ static best_fattn_kernel ggml_sycl_get_best_fattn_kernel(const int device, const } } else { if (Q->ne[1] <= 2) { + // TILE is faster for quantized KV decode on Xe2 (BMG); keep VEC on untested archs + const gpu_arch arch = ggml_sycl_info().devices[device].hw_info.arch; + if (arch == gpu_arch::intel_gpu_bmg_g21 || arch == gpu_arch::intel_gpu_bmg_g31) { + return BEST_FATTN_KERNEL_TILE; + } return BEST_FATTN_KERNEL_VEC; } } @@ -374,3 +377,76 @@ void ggml_sycl_flash_attn_ext(ggml_backend_sycl_context & ctx, ggml_tensor * dst bool ggml_sycl_flash_attn_ext_supported(int device, const ggml_tensor * dst) { return ggml_sycl_get_best_fattn_kernel(device, dst) != BEST_FATTN_KERNEL_NONE; } + +static uintptr_t ggml_sycl_fattn_reserve_halves(ggml_sycl_fattn_extra & extra, size_t n_halves) { + if (n_halves == 0) { + return 0; + } + extra.end = GGML_PAD(extra.end, SYCL_BUFFER_ALIGNMENT); + const uintptr_t block = extra.end; + extra.end += n_halves * sizeof(sycl::half); + return block; +} + +ggml_sycl_fattn_extra ggml_sycl_fattn_get_extra(const ggml_tensor * dst) { + ggml_sycl_fattn_extra extra; + + extra.end = (uintptr_t) dst->data + ggml_nbytes(dst); + + if (dst->op != GGML_OP_FLASH_ATTN_EXT) { + return extra; + } + + const ggml_tensor * Q = dst->src[0]; + const ggml_tensor * K = dst->src[1]; + const ggml_tensor * V = dst->src[2]; + if (!Q || !K || !V) { + return extra; + } + + const int64_t d = K->ne[0]; + const int64_t H = Q->ne[2]; + const int64_t q = Q->ne[1]; + + // calculate the worst-case memory consumption across all kernels + const bool onednn_supported = ggml_sycl_flash_attn_ext_onednn_supported(dst, /* use_shape_limit */ false); + + const bool tile_needs_K = K->type != GGML_TYPE_F16; + const bool tile_needs_V = V->type != GGML_TYPE_F16; + + const bool V_is_K_view = V->view_src && + (V->view_src == K || (V->view_src == K->view_src && V->view_offs == K->view_offs)); + + size_t need_K = 0, need_V = 0, need_Q = 0, need_out = 0, need_scale = 0; + if (onednn_supported) { + need_Q = (size_t) H * q * d; + need_out = (size_t) H * q * d; + need_scale = 1; + // an f16 cache is bound in place, so it needs no staging copy + if (!ggml_sycl_fattn_onednn_binds_kv(K, V)) { + need_K = (size_t) ggml_nelements(K); + need_V = (size_t) ggml_nelements(V); + } + } + if (tile_needs_K) { + need_K = std::max(need_K, (size_t) ggml_nelements(K)); + } + if (tile_needs_V) { + need_V = std::max(need_V, (size_t) ggml_nelements(V)); + } + + extra.Q_buffer_ptr = ggml_sycl_fattn_reserve_halves(extra, need_Q); + extra.K_buffer_ptr = ggml_sycl_fattn_reserve_halves(extra, need_K); + extra.V_buffer_ptr = (V_is_K_view && !onednn_supported && need_V) + ? extra.K_buffer_ptr + : ggml_sycl_fattn_reserve_halves(extra, need_V); + extra.scale_buffer_ptr = ggml_sycl_fattn_reserve_halves(extra, need_scale); + extra.out_buffer_ptr = ggml_sycl_fattn_reserve_halves(extra, need_out); + + return extra; +} + +size_t ggml_sycl_flash_attn_ext_get_alloc_size(const ggml_tensor * dst) { + const ggml_sycl_fattn_extra extra = ggml_sycl_fattn_get_extra(dst); + return (size_t) (extra.end - (uintptr_t) dst->data); +} diff --git a/ggml/src/ggml-sycl/fattn.hpp b/ggml/src/ggml-sycl/fattn.hpp index c093970a..f803aa2a 100644 --- a/ggml/src/ggml-sycl/fattn.hpp +++ b/ggml/src/ggml-sycl/fattn.hpp @@ -19,6 +19,24 @@ void ggml_sycl_flash_attn_ext(ggml_backend_sycl_context & ctx, ggml_tensor * dst bool ggml_sycl_flash_attn_ext_supported(int device, const ggml_tensor * dst); +// Scratch that flash attention needs beyond the output tensor +struct ggml_sycl_fattn_extra { + uintptr_t K_buffer_ptr = 0; // F16 copy of the K cache + uintptr_t V_buffer_ptr = 0; // F16 copy of the V cache + uintptr_t Q_buffer_ptr = 0; // dense F16 copy of Q, oneDNN only + uintptr_t scale_buffer_ptr = 0; // the softmax scale as an F16 scalar, oneDNN only + uintptr_t out_buffer_ptr = 0; // F16 SDPA output before conversion to F32, oneDNN only + uintptr_t end = 0; // one past the last reserved byte; sizes the allocation +}; + +// ggml_sycl_fattn_get_extra() is the single source of truth for the layout: it both sizes +// the reservation and hands out the pointers, so the two cannot disagree. +// Each field is the address of one reserved block, or 0 if that block was not reserved, +// in which case the caller allocates from the scratch pool instead. +ggml_sycl_fattn_extra ggml_sycl_fattn_get_extra(const ggml_tensor * dst); + +size_t ggml_sycl_flash_attn_ext_get_alloc_size(const ggml_tensor * dst); + void ggml_sycl_flash_attn_ext_mkl(ggml_backend_sycl_context & ctx, ggml_tensor * dst); #endif // GGML_SYCL_FATTN_HPP diff --git a/ggml/src/ggml-sycl/fusion.cpp b/ggml/src/ggml-sycl/fusion.cpp index 709bc8ca..d3e99523 100644 --- a/ggml/src/ggml-sycl/fusion.cpp +++ b/ggml/src/ggml-sycl/fusion.cpp @@ -1,4 +1,5 @@ #include "fusion.hpp" +#include "binbcast.hpp" #include @@ -21,16 +22,22 @@ static bool ggml_sycl_should_fuse_mul_mat_glu(const ggml_tensor * gate, const gg const ggml_tensor * wg = gate->src[0]; const ggml_tensor * act = up->src[1]; - // one set of block offsets and one quantized activation must serve both weights - if (wu->type != wg->type || !ggml_are_same_shape(wu, wg) || !ggml_are_same_stride(wu, wg)) { + // one activation and one output indexing must serve both weights; the block types + // may differ, since the plain-layout fused kernel runs each operand's own vec_dot + // (different types then imply different byte strides, so only the shape must agree) + if (!ggml_are_same_shape(wu, wg)) { return false; } if (act != gate->src[1]) { return false; } - // only q4_K has a fused reorder GEMV so far, and it walks whole super-blocks - if (wu->type != GGML_TYPE_Q4_K || wu->ne[0] % QK_K != 0) { + // fused GEMVs walk whole QK_K super-blocks: the reorder kernel covers same-type + // q4_K, the plain-layout kernel covers q5_K / iq4_xs pairs incl. mixed gate/up types + const bool reorder_pair = wu->type == GGML_TYPE_Q4_K && wg->type == GGML_TYPE_Q4_K; + const bool plain_pair = (wu->type == GGML_TYPE_Q5_K || wu->type == GGML_TYPE_IQ4_XS) && + (wg->type == GGML_TYPE_Q5_K || wg->type == GGML_TYPE_IQ4_XS); + if ((!reorder_pair && !plain_pair) || wu->ne[0] % QK_K != 0) { return false; } @@ -94,9 +101,14 @@ bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializ return false; } - if (ops.size() == 2 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { + if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { + if (ops.size() == 3 && ops.begin()[2] != GGML_OP_ADD) { + return false; + } + const ggml_tensor * rms_norm = cgraph->nodes[node_idx]; const ggml_tensor * mul = cgraph->nodes[node_idx + 1]; + const ggml_tensor * add = ops.size() == 3 ? cgraph->nodes[node_idx + 2] : nullptr; GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); GGML_ASSERT(rms_norm->type == GGML_TYPE_F32); @@ -122,6 +134,43 @@ bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializ return false; } + if (add != nullptr) { + if (add->src[0]->type != GGML_TYPE_F32 || + add->src[1]->type != GGML_TYPE_F32 || + add->type != GGML_TYPE_F32) { + return false; + } + + // the fused kernel indexes the residual as add[col] and does not broadcast it + const ggml_tensor * add_w = (add->src[0] == mul) ? add->src[1] : add->src[0]; + if (!ggml_are_same_shape(add_w, add)) { + return false; + } + + if (!ggml_is_contiguous(add->src[0]) || !ggml_is_contiguous_rows(add->src[1])) { + return false; + } + } + + return true; + } + + if (ops.size() == 2 && ops.begin()[0] == GGML_OP_ADD && ops.begin()[1] == GGML_OP_ADD) { + const ggml_tensor * add0 = cgraph->nodes[node_idx]; + const ggml_tensor * add1 = cgraph->nodes[node_idx + 1]; + // ggml_can_fuse already guarantees add1 consumes add0 and that add0 has a single use. + // Keep the CUDA association: the running sum is src0 of the next ADD so the fused + // float fold matches two sequential add() launches. + if (add1->src[0] != add0) { + return false; + } + + const ggml_tensor * c = add1->src[1]; + if (!ggml_sycl_add_kernel_supports(add0->src[0]->type, add0->src[1]->type, add0->type) || + !ggml_sycl_add_kernel_supports(add0->type, c->type, add1->type)) { + return false; + } + return true; } @@ -165,5 +214,67 @@ bool ggml_sycl_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializ return true; } + if (ops.size() == 2 && ops.begin()[0] == GGML_OP_SSM_CONV && ops.begin()[1] == GGML_OP_UNARY && + unary_ops.size() == 1 && unary_ops.begin()[0] == GGML_UNARY_OP_SILU) { + const ggml_tensor * ssm_conv = cgraph->nodes[node_idx]; + const ggml_tensor * silu = cgraph->nodes[node_idx + 1]; + + if (ggml_get_unary_op(silu) != unary_ops.begin()[0]) { + return false; + } + if (ssm_conv->type != GGML_TYPE_F32 || silu->type != GGML_TYPE_F32) { + return false; + } + // the fused kernel writes the SiLU output with dense strides, so it must be contiguous + if (!ggml_is_contiguous(silu)) { + return false; + } + + return true; + } + + if (ops.size() == 3 && ops.begin()[0] == GGML_OP_SSM_CONV && ops.begin()[1] == GGML_OP_ADD && + ops.begin()[2] == GGML_OP_UNARY && unary_ops.size() == 1 && unary_ops.begin()[0] == GGML_UNARY_OP_SILU) { + const ggml_tensor * ssm_conv = cgraph->nodes[node_idx]; + const ggml_tensor * add = cgraph->nodes[node_idx + 1]; + const ggml_tensor * silu = cgraph->nodes[node_idx + 2]; + + if (ggml_get_unary_op(silu) != unary_ops.begin()[0]) { + return false; + } + if (ssm_conv->type != GGML_TYPE_F32 || add->type != GGML_TYPE_F32 || silu->type != GGML_TYPE_F32) { + return false; + } + // the fused kernel writes the SiLU output with dense strides, so it must be contiguous + if (!ggml_is_contiguous(silu)) { + return false; + } + + // ADD must consume ssm_conv's output and broadcast a 1-D channel-wise bias + const ggml_tensor * bias = (add->src[0] == ssm_conv) ? add->src[1] : add->src[0]; + if (bias->type != GGML_TYPE_F32 || !ggml_is_contiguous(bias)) { + return false; + } + if (ggml_nelements(bias) != ssm_conv->ne[0] || bias->ne[0] != ssm_conv->ne[0]) { + return false; + } + + return true; + } + + if (ops.size() == 2 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_SCALE) { + const ggml_tensor * rms_norm = cgraph->nodes[node_idx]; + const ggml_tensor * scale = cgraph->nodes[node_idx + 1]; + GGML_ASSERT(rms_norm->src[0]->type == GGML_TYPE_F32); + GGML_ASSERT(rms_norm->type == GGML_TYPE_F32); + if (scale->src[0]->type != GGML_TYPE_F32 || scale->type != GGML_TYPE_F32) { + return false; + } + // the fused kernel reads/writes rows flat like the unfused pair + if (!ggml_is_contiguous_rows(rms_norm) || !ggml_is_contiguous_rows(scale)) { + return false; + } + return true; + } return false; } diff --git a/ggml/src/ggml-sycl/fwht.cpp b/ggml/src/ggml-sycl/fwht.cpp new file mode 100644 index 00000000..39f273be --- /dev/null +++ b/ggml/src/ggml-sycl/fwht.cpp @@ -0,0 +1,291 @@ +#include "fwht.hpp" + +#include +#define P 1.0f +#define N -1.0f + +// constant Hadamard matrix via Paley I construction +static constexpr float H12[12][12] = { + { P, P, P, P, P, P, P, P, P, P, P, P }, + { P, N, P, N, P, P, P, N, N, N, P, N }, + { P, N, N, P, N, P, P, P, N, N, N, P }, + { P, P, N, N, P, N, P, P, P, N, N, N }, + { P, N, P, N, N, P, N, P, P, P, N, N }, + { P, N, N, P, N, N, P, N, P, P, P, N }, + { P, N, N, N, P, N, N, P, N, P, P, P }, + { P, P, N, N, N, P, N, N, P, N, P, P }, + { P, P, P, N, N, N, P, N, N, P, N, P }, + { P, P, P, P, N, N, N, P, N, N, P, N }, + { P, N, P, P, P, N, N, N, P, N, N, P }, + { P, P, N, P, P, P, N, N, N, P, N, N } +}; + +static constexpr float H20[20][20] = { + { P, P, P, P, P, P, P, P, P, P, P, P, P, P, P, P, P, P, P, P }, + { P, N, P, N, N, P, P, P, P, N, P, N, P, N, N, N, N, P, P, N }, + { P, N, N, P, N, N, P, P, P, P, N, P, N, P, N, N, N, N, P, P }, + { P, P, N, N, P, N, N, P, P, P, P, N, P, N, P, N, N, N, N, P }, + { P, P, P, N, N, P, N, N, P, P, P, P, N, P, N, P, N, N, N, N }, + { P, N, P, P, N, N, P, N, N, P, P, P, P, N, P, N, P, N, N, N }, + { P, N, N, P, P, N, N, P, N, N, P, P, P, P, N, P, N, P, N, N }, + { P, N, N, N, P, P, N, N, P, N, N, P, P, P, P, N, P, N, P, N }, + { P, N, N, N, N, P, P, N, N, P, N, N, P, P, P, P, N, P, N, P }, + { P, P, N, N, N, N, P, P, N, N, P, N, N, P, P, P, P, N, P, N }, + { P, N, P, N, N, N, N, P, P, N, N, P, N, N, P, P, P, P, N, P }, + { P, P, N, P, N, N, N, N, P, P, N, N, P, N, N, P, P, P, P, N }, + { P, N, P, N, P, N, N, N, N, P, P, N, N, P, N, N, P, P, P, P }, + { P, P, N, P, N, P, N, N, N, N, P, P, N, N, P, N, N, P, P, P }, + { P, P, P, N, P, N, P, N, N, N, N, P, P, N, N, P, N, N, P, P }, + { P, P, P, P, N, P, N, P, N, N, N, N, P, P, N, N, P, N, N, P }, + { P, P, P, P, P, N, P, N, P, N, N, N, N, P, P, N, N, P, N, N }, + { P, N, P, P, P, P, N, P, N, P, N, N, N, N, P, P, N, N, P, N }, + { P, N, N, P, P, P, P, N, P, N, P, N, N, N, N, P, P, N, N, P }, + { P, P, N, N, P, P, P, P, N, P, N, P, N, N, N, N, P, P, N, N } +}; + +#undef P +#undef N + +template +static void fwht_kernel(const float * __restrict__ src, float * __restrict__ dst, const int64_t n_rows, + const float scale, const sycl::nd_item<2> & item) { + const sycl::sub_group sg = item.get_sub_group(); + + const int64_t r = item.get_global_id(0); + if (r >= n_rows) { + return; + } + + src += r * N; + dst += r * N; + + constexpr int el_w = N / WARP_SIZE; + static_assert(el_w >= 1 && N % WARP_SIZE == 0, "row must be a whole number of sub-group widths"); + + float reg[el_w]; + const int lane = sg.get_local_linear_id(); + +#pragma unroll + for (int i = 0; i < el_w; ++i) { + reg[i] = src[i * WARP_SIZE + lane] * scale; + } + + // Butterflies inside the sub-group. The partner of a lane with bit h clear is the + // lower index of the pair, so it takes the sum and the upper takes lower - upper. +#pragma unroll + for (int h = 1; h < WARP_SIZE; h *= 2) { +#pragma unroll + for (int j = 0; j < el_w; ++j) { + const float val = reg[j]; + const float val2 = dpct::permute_sub_group_by_xor(sg, val, h, WARP_SIZE); + + reg[j] = (lane & h) == 0 ? val + val2 : val2 - val; + } + } + + // Butterflies across registers: h is a multiple of WARP_SIZE, so the partner of + // element i*WARP_SIZE + lane lives in reg[i + h/WARP_SIZE] on the same lane. +#pragma unroll + for (int h = WARP_SIZE; h < N; h *= 2) { + const int step = h / WARP_SIZE; +#pragma unroll + for (int j = 0; j < el_w; j += 2 * step) { +#pragma unroll + for (int k = 0; k < step; ++k) { + const float x = reg[j + k]; + const float y = reg[j + k + step]; + + reg[j + k] = x + y; + reg[j + k + step] = x - y; + } + } + } + +#pragma unroll + for (int i = 0; i < el_w; ++i) { + dst[i * WARP_SIZE + lane] = reg[i]; + } +} + +template +static void launch_fwht(const float * src, float * dst, const int64_t n_rows, const float scale, + dpct::queue_ptr stream) { + constexpr int rows_per_block = 4; + + const int64_t num_blocks = (n_rows + rows_per_block - 1) / rows_per_block; + + // dim 1 is the fastest-varying, so a sub-group is exactly one row's WARP_SIZE lanes. + const sycl::range<2> global(num_blocks * rows_per_block, WARP_SIZE); + const sycl::range<2> local(rows_per_block, WARP_SIZE); + + stream->parallel_for(sycl::nd_range<2>(global, local), + [=](sycl::nd_item<2> item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + fwht_kernel(src, dst, n_rows, scale, item); + }); +} + +template +static void kronecker_kernel(const float * __restrict__ src, + float * __restrict__ dst, + const int64_t n_rows, + const float scale, + const sycl::nd_item<2> & item) { + static_assert(m == 12 || m == 20, "block size has to be 12 or 20."); + + const sycl::sub_group sg = item.get_sub_group(); + + const int64_t r = item.get_global_id(0); + if (r >= n_rows) { + return; + } + + src += r * N; + dst += r * N; + + constexpr int blocks_per_group = N / m; + constexpr int el_w = blocks_per_group / WARP_SIZE; + static_assert(el_w >= 1 && blocks_per_group % WARP_SIZE == 0, "blocks_per_group must be a multiple of WARP_SIZE"); + float reg[el_w * m]; + const int lane = sg.get_local_linear_id(); + +#pragma unroll + for (int i = 0; i < el_w; ++i) { + const int b_idx = i * WARP_SIZE + lane; + +#pragma unroll + for (int j = 0; j < m; ++j) { + reg[i * m + j] = src[b_idx * m + j] * scale; + } + } + +#pragma unroll + for (int b = 0; b < el_w; ++b) { + float z[m] = { 0.0f }; + +#pragma unroll + for (int i = 0; i < m; ++i) { +#pragma unroll + for (int j = 0; j < m; ++j) { + const float h = (m == 12 ? H12[j][i] : H20[j][i]); + z[i] += reg[b * m + j] * h; + } + } + +#pragma unroll + for (int i = 0; i < m; ++i) { + reg[b * m + i] = z[i]; + } + } + +#pragma unroll + for (int h = 1; h < WARP_SIZE; h *= 2) { +#pragma unroll + for (int j = 0; j < el_w; ++j) { +#pragma unroll + for (int k = 0; k < m; ++k) { + const float val = reg[j * m + k]; + const float val2 = dpct::permute_sub_group_by_xor(sg, val, h, WARP_SIZE); + + reg[j * m + k] = (lane & h) == 0 ? val + val2 : val2 - val; + } + } + } + +#pragma unroll + for (int h = WARP_SIZE; h < blocks_per_group; h *= 2) { + const int step = h / WARP_SIZE; +#pragma unroll + for (int j = 0; j < el_w; j += 2 * step) { +#pragma unroll + for (int s = 0; s < step; ++s) { +#pragma unroll + for (int k = 0; k < m; ++k) { + const float x = reg[(j + s) * m + k]; + const float y = reg[(j + s + step) * m + k]; + + reg[(j + s) * m + k] = x + y; + reg[(j + s + step) * m + k] = x - y; + } + } + } + } + +#pragma unroll + for (int i = 0; i < el_w; ++i) { + const int b_idx = i * WARP_SIZE + lane; +#pragma unroll + for (int k = 0; k < m; ++k) { + dst[b_idx * m + k] = reg[i * m + k]; + } + } +} + +template +static void launch_kronecker(const float * src, + float * dst, + const int64_t n_rows, + const float scale, + dpct::queue_ptr stream) { + constexpr int rows_per_block = 4; + + const int64_t num_blocks = (n_rows + rows_per_block - 1) / rows_per_block; + + // dim 1 is the fastest-varying, so a sub-group is exactly one row's WARP_SIZE lanes. + const sycl::range<2> global(num_blocks * rows_per_block, WARP_SIZE); + const sycl::range<2> local(rows_per_block, WARP_SIZE); + + stream->parallel_for(sycl::nd_range<2>(global, local), + [=](sycl::nd_item<2> item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + kronecker_kernel(src, dst, n_rows, scale, item); + }); +} + +bool ggml_sycl_op_fwht(ggml_backend_sycl_context & ctx, const ggml_tensor * src, ggml_tensor * dst) { + if (src->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { + return false; + } + if (!ggml_are_same_shape(src, dst)) { + return false; + } + if (!ggml_is_contiguous(src) || !ggml_is_contiguous(dst)) { + return false; + } + + const int n = (int) src->ne[0]; + const int64_t rows = ggml_nrows(src); + + const float * src_d = (const float *) src->data; + float * dst_d = (float *) dst->data; + dpct::queue_ptr stream = ctx.stream(); + + const float scale = 1.0f / std::sqrt((float) n); + + switch (n) { + case 64: + launch_fwht<64>(src_d, dst_d, rows, scale, stream); + return true; + case 128: + launch_fwht<128>(src_d, dst_d, rows, scale, stream); + return true; + case 256: + launch_fwht<256>(src_d, dst_d, rows, scale, stream); + return true; + case 512: + launch_fwht<512>(src_d, dst_d, rows, scale, stream); + return true; + case 384: + launch_kronecker<384, 12>(src_d, dst_d, rows, scale, stream); + return true; + case 768: + launch_kronecker<768, 12>(src_d, dst_d, rows, scale, stream); + return true; + case 640: + launch_kronecker<640, 20>(src_d, dst_d, rows, scale, stream); + return true; + case 1280: + launch_kronecker<1280, 20>(src_d, dst_d, rows, scale, stream); + return true; + default: + return false; + } +} diff --git a/ggml/src/ggml-sycl/fwht.hpp b/ggml/src/ggml-sycl/fwht.hpp new file mode 100644 index 00000000..cd238cfa --- /dev/null +++ b/ggml/src/ggml-sycl/fwht.hpp @@ -0,0 +1,12 @@ +#ifndef GGML_SYCL_FWHT_HPP +#define GGML_SYCL_FWHT_HPP + +#include "common.hpp" + +// Fast Walsh-Hadamard transform, the fast path for a MUL_MAT whose src0 ggml has +// tagged GGML_HINT_SRC0_IS_HADAMARD. src0 is not read at all. Returns false if the +// shape is not one this can serve, in which case the caller must fall through to the +// ordinary mat-mul dispatch. +bool ggml_sycl_op_fwht(ggml_backend_sycl_context & ctx, const ggml_tensor * src, ggml_tensor * dst); + +#endif // GGML_SYCL_FWHT_HPP diff --git a/ggml/src/ggml-sycl/gemm.hpp b/ggml/src/ggml-sycl/gemm.hpp index c202da11..81bc5c2e 100644 --- a/ggml/src/ggml-sycl/gemm.hpp +++ b/ggml/src/ggml-sycl/gemm.hpp @@ -66,8 +66,10 @@ class DnnlGemmWrapper { auto matmul_pd = dnnl::matmul::primitive_desc(eng, a_in_md, b_in_md, c_md, primitive_attr); auto c_mem = dnnl::memory(matmul_pd.dst_desc(), eng, c); - auto scratchpad_md = matmul_pd.scratchpad_desc(); - auto scratchpad_mem = ctx.get_scratchpad_mem(scratchpad_md, eng, q); + const auto scratchpad_md = matmul_pd.scratchpad_desc(); + ggml_sycl_pool_alloc scratchpad(ctx.pool()); + void * scratchpad_ptr = scratchpad_md.get_size() > 0 ? scratchpad.alloc(scratchpad_md.get_size()) : nullptr; + auto scratchpad_mem = dnnl::memory(scratchpad_md, eng, scratchpad_ptr); auto matmul_prim = dnnl::matmul(matmul_pd); diff --git a/ggml/src/ggml-sycl/getrows.cpp b/ggml/src/ggml-sycl/getrows.cpp index 2113f356..4e84bd22 100644 --- a/ggml/src/ggml-sycl/getrows.cpp +++ b/ggml/src/ggml-sycl/getrows.cpp @@ -245,6 +245,84 @@ static void get_rows_sycl_float(ggml_backend_sycl_context & ctx, const ggml_tens GGML_UNUSED(ctx); } +template +static void k_get_rows_back_float(const src0_t * src0, const int32_t * src1, float * dst, + const int64_t ncols, const int64_t nrows_grad_10, const int64_t nrows_grad_11, const int64_t nrows_dst, + const size_t s01, const size_t s02, + const size_t s10, const size_t s11, + const size_t s1, + const int64_t block_num_y, + const sycl::nd_item<3> & item_ct1) { + const int64_t col = item_ct1.get_group(2) * item_ct1.get_local_range(2) + item_ct1.get_local_id(2); + if (col >= ncols) { + return; + } + + // block_num_y is clamped, so stride over destination rows like CUDA k_get_rows_back_float + for (int64_t dst_row = item_ct1.get_group(1); dst_row < nrows_dst; dst_row += block_num_y) { + float sum = 0.0f; + + const int64_t nrows_grad_total = nrows_grad_10 * nrows_grad_11; + for (int64_t i = 0; i < nrows_grad_total; ++i) { + const int64_t i10 = i % nrows_grad_10; + const int64_t i11 = i / nrows_grad_10; + if (src1[i10*s10 + i11*s11] != dst_row) { + continue; + } + sum += (float) src0[col + i10*s01 + i11*s02]; + } + + dst[col + dst_row*s1] = sum; + } +} + +template +static void get_rows_back_sycl_float(ggml_backend_sycl_context & ctx, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst, + const src0_t * src0_dd, const int32_t * src1_dd, + float * dst_dd, queue_ptr stream) { + + GGML_TENSOR_BINARY_OP_LOCALS + + GGML_ASSERT(ne02*ne03 == 1); + GGML_ASSERT(ne12*ne13 == 1); + GGML_ASSERT(ne2*ne3 == 1); + GGML_ASSERT(src0->nb[0] == ggml_type_size(src0->type)); + GGML_ASSERT(src1->nb[0] == ggml_type_size(src1->type)); + GGML_ASSERT(dst->nb[0] == ggml_type_size(dst->type)); + + const int64_t ncols = ne00; + const int64_t nrows_grad_10 = ne10; + const int64_t nrows_grad_11 = ne11; + const int64_t nrows_dst = ne1; + + const size_t s01 = nb01 / sizeof(src0_t); + const size_t s02 = nb02 / sizeof(src0_t); + + const size_t s10 = nb10 / sizeof(int32_t); + const size_t s11 = nb11 / sizeof(int32_t); + + const size_t s1 = nb1 / sizeof(float); + + const sycl::range<3> block_dims(1, 1, SYCL_GET_ROWS_BLOCK_SIZE); + const int64_t block_num_x = (ncols + SYCL_GET_ROWS_BLOCK_SIZE - 1) / SYCL_GET_ROWS_BLOCK_SIZE; + const int64_t block_num_y = std::min(nrows_dst, (int64_t) UINT16_MAX); + const sycl::range<3> block_nums(1, block_num_y, block_num_x); + + stream->parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) { + k_get_rows_back_float(src0_dd, src1_dd, dst_dd, + ncols, nrows_grad_10, nrows_grad_11, nrows_dst, + s01, s02, s10, s11, s1, + block_num_y, item_ct1); + }); + + GGML_UNUSED(src0); + GGML_UNUSED(src1); + GGML_UNUSED(dst); + GGML_UNUSED(ctx); +} + void ggml_sycl_op_get_rows(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { GGML_ASSERT(dst->src[1]->type == GGML_TYPE_I32); GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_I32 ); @@ -366,3 +444,30 @@ void ggml_sycl_op_get_rows(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { GGML_ABORT("fatal error"); } } + +void ggml_sycl_op_get_rows_back(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT(src1->type == GGML_TYPE_I32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + + GGML_ASSERT(ggml_is_contiguous(dst)); + + switch (src0->type) { + case GGML_TYPE_F16: + get_rows_back_sycl_float(ctx, src0, src1, dst, (const sycl::half *) src0->data, + (const int32_t *) src1->data, (float *) dst->data, + ctx.stream()); + break; + case GGML_TYPE_F32: + get_rows_back_sycl_float(ctx, src0, src1, dst, (const float *) src0->data, + (const int32_t *) src1->data, (float *) dst->data, + ctx.stream()); + break; + default: + GGML_ABORT("%s: unsupported src0 type: %s\n", __func__, ggml_type_name(src0->type)); + break; + } +} diff --git a/ggml/src/ggml-sycl/getrows.hpp b/ggml/src/ggml-sycl/getrows.hpp index 1c560cd9..0388e6c7 100644 --- a/ggml/src/ggml-sycl/getrows.hpp +++ b/ggml/src/ggml-sycl/getrows.hpp @@ -16,5 +16,6 @@ #include "common.hpp" void ggml_sycl_op_get_rows(ggml_backend_sycl_context & ctx, ggml_tensor *dst); +void ggml_sycl_op_get_rows_back(ggml_backend_sycl_context & ctx, ggml_tensor *dst); #endif // GGML_SYCL_GETROWS_HPP diff --git a/ggml/src/ggml-sycl/ggml-sycl.cpp b/ggml/src/ggml-sycl/ggml-sycl.cpp index 5416d4f0..e13ec852 100644 --- a/ggml/src/ggml-sycl/ggml-sycl.cpp +++ b/ggml/src/ggml-sycl/ggml-sycl.cpp @@ -35,6 +35,7 @@ #include #ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API #include +#include #endif #if defined(GGML_SYCL_GRAPH) && SYCL_EXT_ONEAPI_ASYNC_MEMORY_ALLOC # include @@ -58,8 +59,10 @@ #include "ggml-sycl/backend.hpp" #include "ggml-sycl/common.hpp" #include "ggml-sycl/element_wise.hpp" +#include "ggml-sycl/fwht.hpp" #include "ggml-sycl/gemm.hpp" #include "ggml-sycl/getrows.hpp" +#include "ggml-sycl/mem.hpp" #include "ggml-sycl/norm.hpp" #include "ggml-sycl/presets.hpp" #include "ggml-sycl/quantize.hpp" @@ -88,11 +91,15 @@ static bool g_sycl_loaded = false; int g_ggml_sycl_debug = 0; +int g_ggml_sycl_dev_debug = 0; int g_ggml_sycl_enable_optimize = 1; int g_ggml_sycl_enable_graph = 0; int g_ggml_sycl_enable_dnn = 1; int g_ggml_sycl_fa_onednn = 1; int g_ggml_sycl_fa_onednn_max_kv = 0; +int g_ggml_sycl_enable_mkl_fa = 1; +int g_ggml_sycl_memtrace = 0; +int g_ggml_sycl_memtrace_step = 64; int g_ggml_sycl_enable_vmm = 1; int g_ggml_sycl_enable_fusion = 1; int g_ggml_sycl_enable_esimd = 1; @@ -104,11 +111,21 @@ int g_ggml_sycl_enable_flash_attention = 1; int g_ggml_sycl_dev2dev_memcpy = DEV2DEV_MEMCPY_SYCL; int g_ggml_sycl_usm_system = 0; int g_ggml_sycl_enable_host_pinned_mem = 1; +int g_ggml_sycl_host_pinned_mem_2g = 0; +int g_ggml_sycl_get_mem_api = MEMORY_API_TYPE_LEVEL_ZERO; static ggml_sycl_device_info ggml_sycl_init() { + GGML_SYCL_DEBUG("[SYCL] call ggml_sycl_init\n"); ggml_sycl_device_info info = {}; - info.device_count = dpct::dev_mgr::instance().device_count(); + // Do not hard crash when there exists no SYCL devices. + // We want to allow the user to use non-SYCL tools when SYCL is compiled (such as llama-quantize) + try { + info.device_count = dpct::dev_mgr::instance().device_count(); + } catch (sycl::exception const &exc) { + GGML_LOG_INFO("%s: no SYCL device available: %s\n", __func__, exc.what()); + info.device_count = 0; + } if (info.device_count == 0) { GGML_LOG_ERROR("%s: failed to initialize: %s\n", GGML_SYCL_NAME, __func__); return info; @@ -189,12 +206,9 @@ static ggml_sycl_device_info ggml_sycl_init() { } #ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API - // Large buffers can be allocated before ggml_check_sycl() initializes other - // g_ggml_sycl_enable_* globals, so initialize this one as early as we can. + //update g_ggml_sycl_use_level_zero_api according to the device support g_ggml_sycl_use_level_zero_api = - info.ext_oneapi_level_zero && ggml_sycl_get_env("GGML_SYCL_USE_LEVEL_ZERO_API", 1); -#else - g_ggml_sycl_use_level_zero_api = 0; + info.ext_oneapi_level_zero && g_ggml_sycl_use_level_zero_api; #endif return info; @@ -293,24 +307,68 @@ static const char* dev2dev_int2str(int dev2dev) { } } +/* +* There are several entry APIs to be called as first function in SYCL backend in different cases. +* It's the first internal function to be called by them in SYCL backend. +* This function is used to do initialize work for the SYCL backend and set the global variables. +*/ +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API +static ze_result_t init_zes() { + ze_result_t res = zesInit(0); + if (res != ZE_RESULT_SUCCESS) { + GGML_SYCL_DEBUG("Warning: [%s] zesInit failed with code %d. Sysman free-memory query be unavailable.\n", + __func__, (int) res); + } + return res; +} + +ze_result_t get_zes_init_res() { + static ze_result_t zes_init_res = init_zes(); + GGML_SYCL_DEBUG("[SYCL] call %s: zesInit result: %d\n", __func__, (int) zes_init_res); + return zes_init_res; +} +#endif + +void initialize_sycl_begining() { +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API + //must be called in initialization stage, before any other Level Zero API calls + GGML_SYCL_DEBUG("[SYCL] call %s\n", __func__); + get_zes_init_res(); +#endif +} + static void ggml_check_sycl() try { + GGML_SYCL_DEBUG("[SYCL] ggml_check_sycl()\n"); static bool initialized = false; if (!initialized) { + initialize_sycl_begining(); + g_ggml_sycl_debug = ggml_sycl_get_env("GGML_SYCL_DEBUG", 0); + g_ggml_sycl_dev_debug = ggml_sycl_get_env("GGML_SYCL_DEV_DEBUG", 0); g_ggml_sycl_enable_optimize = ggml_sycl_get_env("GGML_SYCL_ENABLE_OPT", 1); g_ggml_sycl_enable_graph = ggml_sycl_get_env("GGML_SYCL_ENABLE_GRAPH", 0); g_ggml_sycl_enable_dnn = ggml_sycl_get_env("GGML_SYCL_ENABLE_DNN", 1); g_ggml_sycl_fa_onednn = ggml_sycl_get_env("GGML_SYCL_FA_ONEDNN", 1); g_ggml_sycl_fa_onednn_max_kv = ggml_sycl_get_env("GGML_SYCL_FA_ONEDNN_MAX_KV", 0); + g_ggml_sycl_enable_mkl_fa = ggml_sycl_get_env("GGML_SYCL_ENABLE_MKL_FA", 1); + g_ggml_sycl_memtrace = ggml_sycl_get_env("GGML_SYCL_MEMTRACE", 0); + g_ggml_sycl_memtrace_step = ggml_sycl_get_env("GGML_SYCL_MEMTRACE_STEP", 64); g_ggml_sycl_enable_vmm = ggml_sycl_get_env("GGML_SYCL_ENABLE_VMM", 1); g_ggml_sycl_enable_fusion = ggml_sycl_get_env("GGML_SYCL_ENABLE_FUSION", 1); g_ggml_sycl_enable_esimd = ggml_sycl_get_env("GGML_SYCL_ENABLE_ESIMD", 1); g_ggml_sycl_prioritize_dmmv = ggml_sycl_get_env("GGML_SYCL_PRIORITIZE_DMMV", 0); +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API + g_ggml_sycl_use_level_zero_api = ggml_sycl_get_env("GGML_SYCL_USE_LEVEL_ZERO_API", 1); +#else + g_ggml_sycl_use_level_zero_api = 0; +#endif g_ggml_sycl_dev2dev_memcpy = ggml_sycl_get_env("GGML_SYCL_DEV2DEV_MEMCPY", DEV2DEV_MEMCPY_SYCL); + g_ggml_sycl_get_mem_api = ggml_sycl_get_env("GGML_SYCL_GET_MEM_API", MEMORY_API_TYPE_LEVEL_ZERO); if (g_ggml_sycl_use_level_zero_api == 0) { g_ggml_sycl_dev2dev_memcpy = DEV2DEV_MEMCPY_SYCL; + g_ggml_sycl_get_mem_api = MEMORY_API_TYPE_SYCL; } #ifdef SYCL_FLASH_ATTN @@ -323,6 +381,9 @@ static void ggml_check_sycl() try { g_ggml_sycl_enable_host_pinned_mem = ggml_sycl_get_env("GGML_SYCL_ENABLE_HOST_PINNED_MEM", 1); + g_ggml_sycl_host_pinned_mem_2g = + ggml_sycl_get_env("GGML_SYCL_HOST_PINNED_MEM_2G", 0) & g_ggml_sycl_enable_host_pinned_mem; + GGML_SYCL_DEBUG("[SYCL] call ggml_check_sycl\n"); GGML_LOG_INFO("Build with Macros:\n"); @@ -363,12 +424,16 @@ static void ggml_check_sycl() try { GGML_LOG_INFO("Running with Environment Variables:\n"); GGML_LOG_INFO(" GGML_SYCL_DEBUG: %d\n", g_ggml_sycl_debug); + GGML_LOG_INFO(" GGML_SYCL_DEV_DEBUG: %d\n", g_ggml_sycl_dev_debug); #ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API GGML_LOG_INFO(" GGML_SYCL_DEV2DEV_MEMCPY: %d (%s)\n", g_ggml_sycl_dev2dev_memcpy, dev2dev_int2str(g_ggml_sycl_dev2dev_memcpy)); + GGML_LOG_INFO(" GGML_SYCL_GET_MEM_API: %d (%s)\n", g_ggml_sycl_get_mem_api, mem_api_int2str(g_ggml_sycl_get_mem_api)); #else GGML_LOG_INFO(" GGML_SYCL_DEV2DEV_MEMCPY: %d (%s), enable to SYCL API since missing GGML_SYCL_SUPPORT_LEVEL_ZERO_API\n", g_ggml_sycl_dev2dev_memcpy, dev2dev_int2str(g_ggml_sycl_dev2dev_memcpy)); + GGML_LOG_INFO(" GGML_SYCL_GET_MEM_API: %d (%s), enable to SYCL API since missing GGML_SYCL_SUPPORT_LEVEL_ZERO_API\n", + g_ggml_sycl_get_mem_api, mem_api_int2str(g_ggml_sycl_get_mem_api)); #endif #if defined(GGML_SYCL_DNNL) @@ -379,6 +444,9 @@ static void ggml_check_sycl() try { GGML_LOG_INFO(" GGML_SYCL_FA_ONEDNN: %d\n", g_ggml_sycl_fa_onednn); #endif GGML_LOG_INFO(" GGML_SYCL_FA_ONEDNN_MAX_KV: %d\n", g_ggml_sycl_fa_onednn_max_kv); + GGML_LOG_INFO(" GGML_SYCL_ENABLE_MKL_FA: %d\n", g_ggml_sycl_enable_mkl_fa); + GGML_LOG_INFO(" GGML_SYCL_MEMTRACE: %d\n", g_ggml_sycl_memtrace); + GGML_LOG_INFO(" GGML_SYCL_MEMTRACE_STEP: %d\n", g_ggml_sycl_memtrace_step); #ifdef SYCL_FLASH_ATTN GGML_LOG_INFO(" GGML_SYCL_ENABLE_FLASH_ATTN: %d\n", g_ggml_sycl_enable_flash_attention); #else @@ -421,6 +489,7 @@ static void ggml_check_sycl() try { GGML_LOG_INFO(" GGML_SYCL_USM_SYSTEM: %d\n", g_ggml_sycl_usm_system); GGML_LOG_INFO(" GGML_SYCL_ENABLE_HOST_PINNED_MEM: %d\n", g_ggml_sycl_enable_host_pinned_mem); + GGML_LOG_INFO(" GGML_SYCL_HOST_PINNED_MEM_2G: %d\n", g_ggml_sycl_host_pinned_mem_2g); /* NOT REMOVE, keep it for next optimize for XMX. #if defined(SYCL_USE_XMX) @@ -702,6 +771,7 @@ static void dev2dev_memcpy(int device_dst, sycl::queue &q_dst, int device_src, s if (q_dst.get_device().ext_oneapi_can_access_peer(q_src.get_device(), sycl::ext::oneapi::peer_access::access_supported)) { GGML_SYCL_DEBUG("[SYCL] dev2dev memcpy by SYCL\n"); + q_dst.get_device().ext_oneapi_enable_peer_access(q_src.get_device()); SYCL_CHECK(CHECK_TRY_ERROR(q_dst.memcpy(ptr_dst, ptr_src, size).wait())); return; } @@ -895,6 +965,7 @@ inline void * aligned_malloc_host(size_t alignment, size_t size) { static ggml_backend_buffer_t ggml_backend_sycl_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) try { + GGML_SYCL_DEBUG("[SYCL] call %s: size=%zu\n", __func__, size); ggml_check_sycl(); ggml_backend_sycl_buffer_type_context * buft_ctx = (ggml_backend_sycl_buffer_type_context *)buft->context; @@ -913,16 +984,16 @@ ggml_backend_sycl_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, void * dev_ptr; if (use_usm_system) { - GGML_SYCL_DEBUG("[SYCL] allocating %lu Bytes with USM system\n", size); + GGML_SYCL_DEBUG("[SYCL] allocating %zu Bytes with USM system\n", size); dev_ptr = (void *)aligned_malloc_host(alignment, aligned_size); if (!dev_ptr) { - GGML_LOG_ERROR("%s: can't allocate %lu Bytes of memory on host\n", __func__, size); + GGML_LOG_ERROR("%s: can't allocate %zu Bytes of memory on host\n", __func__, size); return nullptr; } } else { - SYCL_CHECK(CHECK_TRY_ERROR(dev_ptr = (void *)ggml_sycl_malloc_device(size, *stream))); + SYCL_CHECK(CHECK_TRY_ERROR(dev_ptr = (void *)ggml_sycl_malloc_device(size, *stream, GGML_SYCL_MEM_BUFFER))); if (!dev_ptr) { - GGML_LOG_ERROR("%s: can't allocate %lu Bytes of memory on device\n", __func__, size); + GGML_LOG_ERROR("%s: can't allocate %zu Bytes of memory on device\n", __func__, size); return nullptr; } } @@ -940,14 +1011,34 @@ static size_t ggml_backend_sycl_buffer_type_get_alignment(ggml_backend_buffer_ty GGML_UNUSED(buft); } +bool is_bmg_g31_arch(int device) { + return ggml_sycl_info().devices[device].hw_info.arch == gpu_arch::intel_gpu_bmg_g31; +} + static size_t ggml_backend_sycl_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) { - return dpct::get_current_device().get_max_mem_alloc_size(); + size_t max_alloc_size = dpct::get_current_device().get_max_mem_alloc_size(); + if (g_ggml_sycl_host_pinned_mem_2g) { + return std::min(max_alloc_size, (size_t) 2LL*1024*1024*1024); + } else { + ggml_backend_sycl_buffer_type_context * ctx = (ggml_backend_sycl_buffer_type_context *)buft->context; + int device = ctx->device; + if(is_bmg_g31_arch(device)) { + //Todo, it's workaround for BMG-G31, which has a known issue with large allocations. + //The max alloc size is reduced to 60% of the reported max alloc size. + //remove it after https://github.com/intel/compute-runtime/issues/998 is fixed. + max_alloc_size = max_alloc_size*0.6; + } + return max_alloc_size; + } GGML_UNUSED(buft); } static size_t ggml_backend_sycl_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor) { - size_t size = ggml_nbytes(tensor); + // Reserve the additional scratch so it's visible to the graph allocator + size_t size = tensor->op == GGML_OP_FLASH_ATTN_EXT + ? ggml_sycl_flash_attn_ext_get_alloc_size(tensor) + : ggml_nbytes(tensor); int64_t ne0 = tensor->ne[0]; if (ggml_is_quantized(tensor->type)) { @@ -1166,10 +1257,10 @@ ggml_backend_sycl_split_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_sycl_set_device(i); const queue_ptr stream = ctx->streams[i]; char * buf; - SYCL_CHECK(CHECK_TRY_ERROR(buf = (char *)ggml_sycl_malloc_device(size, *stream))); + SYCL_CHECK(CHECK_TRY_ERROR(buf = (char *)ggml_sycl_malloc_device(size, *stream, GGML_SYCL_MEM_BUFFER))); if (!buf) { char err_buf[1024]; - snprintf(err_buf, 1023, "%s: can't allocate %lu Bytes of memory on device\n", __func__, size); + snprintf(err_buf, 1023, "%s: can't allocate %zu Bytes of memory on device\n", __func__, size); throw std::runtime_error(err_buf); } // set padding to 0 to avoid possible NaN values @@ -1406,11 +1497,14 @@ static ggml_backend_buffer_type_i ggml_backend_sycl_split_buffer_type_interface /* .is_host = */ ggml_backend_sycl_split_buffer_type_is_host, }; -ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(const float * tensor_split) { +ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(int main_device, const float * tensor_split) { + GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_split_buffer_type\n"); + + GGML_UNUSED(main_device); + static std::mutex mutex; std::lock_guard lock(mutex); - GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_split_buffer_type\n"); ggml_check_sycl(); // FIXME: this is not thread safe static std::map, struct ggml_backend_buffer_type> buft_map; @@ -1461,13 +1555,18 @@ static const char * ggml_backend_sycl_host_buffer_type_name(ggml_backend_buffer_ GGML_UNUSED(buft); } +static int ggml_backend_sycl_host_buffer_type_device(ggml_backend_buffer_type_t buft) { + return static_cast(buft->device->context)->device; +} + //host pinned memory -static void * ggml_backend_sycl_host_malloc(size_t size) { +static void * ggml_backend_sycl_host_malloc(int device, size_t size) { + GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_host_malloc of size %.2f MiB on device %d\n", size / 1024.0 / 1024.0, device); void * ptr = nullptr; try { ggml_check_sycl(); // USM host memory is page-locked and device-accessible by construction - auto & q = dpct::dev_mgr::instance().get_device(0).default_queue(); + auto & q = dpct::dev_mgr::instance().get_device(device).default_queue(); ptr = sycl::malloc_host(size, q, sycl::property_list{}); } catch (...) { ptr = nullptr; @@ -1485,7 +1584,8 @@ static void ggml_backend_sycl_host_buffer_free_buffer(ggml_backend_buffer_t buff return; } if (g_ggml_sycl_enable_host_pinned_mem) { - auto & q = dpct::dev_mgr::instance().get_device(0).default_queue(); + const int device = ggml_backend_sycl_host_buffer_type_device(buffer->buft); + auto & q = dpct::dev_mgr::instance().get_device(device).default_queue(); SYCL_CHECK(CHECK_TRY_ERROR(sycl::free(buffer->context, q))); } else { free_aligned_mem_host((void *) buffer->context); @@ -1493,8 +1593,9 @@ static void ggml_backend_sycl_host_buffer_free_buffer(ggml_backend_buffer_t buff } static ggml_backend_buffer_t ggml_backend_sycl_host_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) { - void * ptr = g_ggml_sycl_enable_host_pinned_mem ? ggml_backend_sycl_host_malloc(size) : - aligned_malloc_host(TENSOR_ALIGNMENT, size); + void * ptr = g_ggml_sycl_enable_host_pinned_mem ? + ggml_backend_sycl_host_malloc(ggml_backend_sycl_host_buffer_type_device(buft), size) : + aligned_malloc_host(TENSOR_ALIGNMENT, size); if (ptr == nullptr) { // fallback to cpu buffer return ggml_backend_buft_alloc_buffer(ggml_backend_cpu_buffer_type(), size); @@ -1509,26 +1610,51 @@ static ggml_backend_buffer_t ggml_backend_sycl_host_buffer_type_alloc_buffer(ggm } static size_t ggml_backend_sycl_host_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) { - ggml_backend_sycl_device_context * dev_ctx = (ggml_backend_sycl_device_context *) buft->device->context; - return dpct::dev_mgr::instance().get_device(dev_ctx->device).get_max_mem_alloc_size(); + + if (g_ggml_sycl_enable_host_pinned_mem) { + const int device = ggml_backend_sycl_host_buffer_type_device(buft); + size_t max_alloc_size = dpct::dev_mgr::instance().get_device(device).get_max_mem_alloc_size(); + if (g_ggml_sycl_host_pinned_mem_2g) { + return std::min(max_alloc_size, (size_t) 2LL*1024*1024*1024); + } else { + return max_alloc_size; + } + } else { + return SIZE_MAX; + } +} + +static ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type_for_device(int device) { + GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_host_buffer_type_for_device on device %d\n", device); + + // the vector is never resized after this, so the returned pointers stay valid + static std::vector buffer_types_host = [] { + std::vector bufts(ggml_backend_sycl_get_device_count()); + for (size_t i = 0; i < bufts.size(); i++) { + bufts[i] = { + /* .iface = */ { + /* .get_name = */ ggml_backend_sycl_host_buffer_type_name, + /* .alloc_buffer = */ ggml_backend_sycl_host_buffer_type_alloc_buffer, + /* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment, + /* .get_max_size = */ ggml_backend_sycl_host_buffer_type_get_max_size, + /* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size, + /* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host, + }, + /* .device = */ ggml_backend_reg_dev_get(ggml_backend_sycl_reg(), i), + /* .context = */ nullptr, + }; + } + return bufts; + }(); + + GGML_ASSERT(device >= 0 && device < (int) buffer_types_host.size()); + + return &buffer_types_host[device]; } +// TODO: this function is unused and is a temporary hack to avoid breaking changes ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type() { - GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_host_buffer_type\n"); - static struct ggml_backend_buffer_type ggml_backend_sycl_buffer_type_host = { - /* .iface = */ { - /* .get_name = */ ggml_backend_sycl_host_buffer_type_name, - /* .alloc_buffer = */ ggml_backend_sycl_host_buffer_type_alloc_buffer, - /* .get_alignment = */ ggml_backend_cpu_buffer_type()->iface.get_alignment, - /* .get_max_size = */ ggml_backend_sycl_host_buffer_type_get_max_size, - /* .get_alloc_size = */ ggml_backend_cpu_buffer_type()->iface.get_alloc_size, - /* .is_host = */ ggml_backend_cpu_buffer_type()->iface.is_host, - }, - /* .device = */ ggml_backend_reg_dev_get(ggml_backend_sycl_reg(), 0), - /* .context = */ nullptr, - }; - - return &ggml_backend_sycl_buffer_type_host; + return ggml_backend_sycl_host_buffer_type_for_device(0); } // buffer pool for sycl (legacy) @@ -1636,9 +1762,9 @@ struct ggml_sycl_pool_leg : public ggml_sycl_pool { void * ptr; size_t look_ahead_size = (size_t) (1.05 * size); - SYCL_CHECK(CHECK_TRY_ERROR(ptr = (void *)ggml_sycl_malloc_device(look_ahead_size, *qptr))); + SYCL_CHECK(CHECK_TRY_ERROR(ptr = (void *)ggml_sycl_malloc_device(look_ahead_size, *qptr, GGML_SYCL_MEM_POOL_LEG))); if (!ptr) { - GGML_LOG_ERROR("%s: can't allocate %lu Bytes of memory on device/GPU\n", __func__, look_ahead_size); + GGML_LOG_ERROR("%s: can't allocate %zu Bytes of memory on device/GPU\n", __func__, look_ahead_size); return nullptr; } @@ -1650,7 +1776,7 @@ struct ggml_sycl_pool_leg : public ggml_sycl_pool { (uint32_t)(max_size/1024/1024), (uint32_t)(g_sycl_pool_size[id]/1024/1024), (uint32_t)(size/1024/1024)); #endif - // GGML_SYCL_DEBUG("ggml_sycl_pool_malloc_leg look_ahead_size=%lu, return %p\n", look_ahead_size, ptr); + // GGML_SYCL_DEBUG("ggml_sycl_pool_malloc_leg look_ahead_size=%zu, return %p\n", look_ahead_size, ptr); return ptr; } @@ -1725,6 +1851,13 @@ struct ggml_sycl_pool_vmm : public ggml_sycl_pool { GGML_ASSERT(pool_size + reserve_size <= SYCL_POOL_VMM_MAX_SIZE); + if (ggml_sycl_memtrace_enabled()) { + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " pool_vmm[%d] committing %5zu MiB (pool %5zu -> %5zu MiB)\n", + device, reserve_size / (1024 * 1024), pool_size / (1024 * 1024), + (pool_size + reserve_size) / (1024 * 1024)); + ggml_sycl_memtrace_report("before pool_vmm commit"); + } + // allocate more physical memory std::optional phys; SYCL_CHECK(CHECK_TRY_ERROR(phys.emplace(dev, ctx, reserve_size))); @@ -1750,6 +1883,7 @@ struct ggml_sycl_pool_vmm : public ggml_sycl_pool { // add to the pool pool_size += reserve_size; + ggml_sycl_memtrace_add(GGML_SYCL_MEM_POOL_VMM, map_ptr, reserve_size); #ifdef DEBUG_SYCL_MALLOC GGML_LOG_INFO("sycl pool[%d]: size increased to %llu MB (reserved %llu MB)\n", @@ -1830,7 +1964,7 @@ struct ggml_sycl_pool_host : public ggml_sycl_pool { SYCL_CHECK(CHECK_TRY_ERROR(ptr = (void *) sycl::malloc_host(size, *qptr))); if (!ptr) { - GGML_LOG_ERROR("%s: can't allocate %lu Bytes of memory on host\n", __func__, size); + GGML_LOG_ERROR("%s: can't allocate %zu Bytes of memory on host\n", __func__, size); return nullptr; } pool_size += size; @@ -2386,7 +2520,138 @@ static void argsort_f32_i32_sycl(const float *x, int *dst, const int ncols, } } +// Scan and block merge, shared by every launch shape below so a partitioned row uses the +// same insertion order as an unpartitioned one. +// +// src_map != nullptr: report src_map[col] instead of col, so a merge pass can carry the +// original column index through. +// out_vals != nullptr: also emit the k winning values, for a later merge pass. +// swap01: emit in the output order the single-pass path uses. +static void top_k_scan_merge_f32( + const float * src_vals, + const int32_t * src_map, + const int begin, + const int end, + const int k, + const int block_size, + float * shared_vals, + int * shared_idx, + float * out_vals, + int32_t * out_idx, + const bool swap01, + const sycl::nd_item<1> & item_ct1 +) { + const int tid = item_ct1.get_local_id(0); + + // The running top-k lives in SLM (shared local memory) rather than a private array: + // an array indexed by a runtime position cannot be register-allocated, so a private + // one lands in scratch, i.e. device memory, and insertion is this kernel's dominant + // cost. + // + // Lane-strided (lv[i * block_size]) rather than lane-blocked (lv[i]) so a given i is + // contiguous across lanes; a k-strided layout would put every lane of a shift step in + // the same SLM bank. + float * lv = shared_vals + tid; + int * li = shared_idx + tid; + + for (int i = 0; i < k; i++) { + lv[i * block_size] = -FLT_MAX; + li[i * block_size] = -1; + } + + // The k-th best, cached in a register. The reject test is taken for the large + // majority of elements scanned, and in that case touches no memory. + float kth = -FLT_MAX; + + for (int col = begin + tid; col < end; col += block_size) { + float val = src_vals[col]; + + if (val > kth) { + int pos = k - 1; + while (pos > 0 && val > lv[(pos - 1) * block_size]) { + pos--; + } + + for (int i = k - 1; i > pos; i--) { + lv[i * block_size] = lv[(i - 1) * block_size]; + li[i * block_size] = li[(i - 1) * block_size]; + } + lv[pos * block_size] = val; + li[pos * block_size] = src_map ? src_map[col] : col; + + kth = lv[(k - 1) * block_size]; + } + } + + item_ct1.barrier(sycl::access::fence_space::local_space); + + if (tid != 0) { + return; + } + + // Same treatment for the merge accumulator, past the per-lane region. + float * fv = shared_vals + (size_t) k * block_size; + int * fi = shared_idx + (size_t) k * block_size; + + for (int i = 0; i < k; i++) { + fv[i] = -FLT_MAX; + fi[i] = -1; + } + + float fkth = -FLT_MAX; + + // Candidates are visited in the same (t, i) order as before, so tie-breaking is + // unchanged. + for (int t = 0; t < block_size; t++) { + for (int i = 0; i < k; i++) { + float val = shared_vals[i * block_size + t]; + + if (val <= fkth) { + // Lane t's list is sorted descending, so once one of its entries loses + // to the k-th best, every later entry loses too. fkth only rises, so + // that stays true for the rest of the merge. This turns the merge from + // block_size*k steps into roughly block_size plus the candidates + // accepted. + break; + } + + int idx = shared_idx[i * block_size + t]; + + int pos = k - 1; + while (pos > 0 && val > fv[pos - 1]) { + pos--; + } + + for (int j = k - 1; j > pos; j--) { + fv[j] = fv[j - 1]; + fi[j] = fi[j - 1]; + } + fv[pos] = val; + fi[pos] = idx; + + fkth = fv[k - 1]; + } + } + + if (out_vals) { + for (int i = 0; i < k; i++) { + out_vals[i] = fv[i]; + } + } + + for (int i = 0; i < k; i++) { + out_idx[i] = fi[i]; + } + + if (swap01 && k > 1) { + int32_t temp = out_idx[0]; + out_idx[0] = out_idx[1]; + out_idx[1] = temp; + } +} + static void top_k_f32_sycl( + ggml_backend_sycl_context & ctx, const float * src, int32_t * dst_indices, const int64_t ncols, @@ -2394,98 +2659,107 @@ static void top_k_f32_sycl( const int k, dpct::queue_ptr main_stream ) { + // A row is scanned by exactly one work-group, so a vocabulary-sized row leaves the + // rest of the device idle. What the scan is short of is memory requests in flight, + // not bandwidth or per-request latency, so lanes in flight is the lever: split the + // row across independent work-groups, have each emit its partition's top-k, and + // merge those nsplit*k candidates in a second launch. + // + // split_block trades parallelism against SLM residency. Its cost is + // (split_block + 1) * k * 8 bytes of SLM per group, so at the k <= 32 ceiling 128 + // lanes need about 33 KB, which leaves a single resident group per Xe-core. Revisit + // if the supported k ever grows. + constexpr int split_block = 128; + constexpr int max_splits = 128; + constexpr int min_cols = 8192; + + int nsplit = 1; + if (ncols >= min_cols) { + // A partition is then always >= split_block = 128 columns, hence always more than + // the k <= 32 ceiling, so no pass is ever padded with -FLT_MAX sentinels. + const int64_t want = ncols / split_block; + nsplit = (int) (want > max_splits ? max_splits : want); + } + + if (nsplit > 1) { + const int nchunk = (int) ((ncols + nsplit - 1) / nsplit); + const size_t ncand = (size_t) nrows * nsplit * k; + + ggml_sycl_pool_alloc part_vals(ctx.pool(), ncand); + ggml_sycl_pool_alloc part_idx(ctx.pool(), ncand); + + float * pv = part_vals.get(); + int32_t * pi = part_idx.get(); + + const sycl::range<1> block_dims(split_block); + + main_stream->submit([&](sycl::handler &cgh) { + sycl::local_accessor shared_vals(sycl::range<1>((split_block + 1) * k), cgh); + sycl::local_accessor shared_idx(sycl::range<1>((split_block + 1) * k), cgh); + + cgh.parallel_for( + sycl::nd_range<1>(sycl::range<1>(nrows * nsplit) * block_dims, block_dims), + [=](sycl::nd_item<1> item_ct1) { + const int grp = item_ct1.get_group(0); + const int row = grp / nsplit; + const int part = grp % nsplit; + + const int begin = part * nchunk; + int end = begin + nchunk; + if (end > (int) ncols) { + end = (int) ncols; + } + + top_k_scan_merge_f32( + src + (int64_t) row * ncols, nullptr, begin, end, k, split_block, + shared_vals.get_multi_ptr().get(), + shared_idx.get_multi_ptr().get(), + pv + (size_t) grp * k, pi + (size_t) grp * k, false, item_ct1); + }); + }); + + main_stream->submit([&](sycl::handler &cgh) { + sycl::local_accessor shared_vals(sycl::range<1>((split_block + 1) * k), cgh); + sycl::local_accessor shared_idx(sycl::range<1>((split_block + 1) * k), cgh); + + cgh.parallel_for( + sycl::nd_range<1>(sycl::range<1>(nrows) * block_dims, block_dims), + [=](sycl::nd_item<1> item_ct1) { + const int row = item_ct1.get_group(0); + const size_t off = (size_t) row * nsplit * k; + + top_k_scan_merge_f32( + pv + off, pi + off, 0, nsplit * k, k, split_block, + shared_vals.get_multi_ptr().get(), + shared_idx.get_multi_ptr().get(), + nullptr, dst_indices + (int64_t) row * k, true, item_ct1); + }); + }); + + return; + } + const int block_size = 128; const sycl::range<1> block_dims(block_size); const sycl::range<1> grid_dims(nrows); main_stream->submit([&](sycl::handler &cgh) { - sycl::local_accessor shared_vals(sycl::range<1>(block_size * k), cgh); - sycl::local_accessor shared_idx(sycl::range<1>(block_size * k), cgh); + sycl::local_accessor shared_vals(sycl::range<1>((block_size + 1) * k), cgh); + sycl::local_accessor shared_idx(sycl::range<1>((block_size + 1) * k), cgh); cgh.parallel_for( sycl::nd_range<1>(grid_dims * block_dims, block_dims), [=](sycl::nd_item<1> item_ct1) { const int row = item_ct1.get_group(0); - const int tid = item_ct1.get_local_id(0); if (row >= nrows) return; - const float * src_row = src + row * ncols; - int32_t * dst_idx_row = dst_indices + row * k; - - float local_vals[32]; - int local_idx[32]; - - for (int i = 0; i < k; i++) { - local_vals[i] = -FLT_MAX; - local_idx[i] = -1; - } - - for (int col = tid; col < ncols; col += block_size) { - float val = src_row[col]; - - if (val > local_vals[k-1]) { - int pos = k - 1; - while (pos > 0 && val > local_vals[pos - 1]) { - pos--; - } - - for (int i = k - 1; i > pos; i--) { - local_vals[i] = local_vals[i - 1]; - local_idx[i] = local_idx[i - 1]; - } - local_vals[pos] = val; - local_idx[pos] = col; - } - } - - for (int i = 0; i < k; i++) { - shared_vals[tid * k + i] = local_vals[i]; - shared_idx[tid * k + i] = local_idx[i]; - } - item_ct1.barrier(sycl::access::fence_space::local_space); - - if (tid == 0) { - float final_vals[32]; - int final_idx[32]; - - for (int i = 0; i < k; i++) { - final_vals[i] = -FLT_MAX; - final_idx[i] = -1; - } - - for (int t = 0; t < block_size; t++) { - for (int i = 0; i < k; i++) { - float val = shared_vals[t * k + i]; - int idx = shared_idx[t * k + i]; - - if (val > final_vals[k-1]) { - int pos = k - 1; - while (pos > 0 && val > final_vals[pos - 1]) { - pos--; - } - - for (int j = k - 1; j > pos; j--) { - final_vals[j] = final_vals[j - 1]; - final_idx[j] = final_idx[j - 1]; - } - final_vals[pos] = val; - final_idx[pos] = idx; - } - } - } - - for (int i = 0; i < k; i++) { - dst_idx_row[i] = final_idx[i]; - } - - if (k > 1) { - int32_t temp = dst_idx_row[0]; - dst_idx_row[0] = dst_idx_row[1]; - dst_idx_row[1] = temp; - } - } + top_k_scan_merge_f32( + src + (int64_t) row * ncols, nullptr, 0, (int) ncols, k, block_size, + shared_vals.get_multi_ptr().get(), + shared_idx.get_multi_ptr().get(), + nullptr, dst_indices + (int64_t) row * k, true, item_ct1); }); }); } @@ -2766,9 +3040,9 @@ inline void ggml_sycl_op_mul_mat_sycl( const float * src1_ddf1_i = src1->type == GGML_TYPE_F32 ? (const float *) src1_ddf_i : src1_ddq_as_f32.get(); { +#if GGML_SYCL_DNNL const int64_t gemm_flops = (int64_t)row_diff * src1_ncols * ne10; const bool use_mkl_direct = gemm_flops < 256 * 256 * 256; -#if GGML_SYCL_DNNL if (g_ggml_sycl_enable_dnn && !use_mkl_direct) { DnnlGemmWrapper::row_gemm(ctx, row_diff, src1_ncols, ne10, src0_ddf_i, DnnlGemmWrapper::to_dt(), src1_ddf1_i, DnnlGemmWrapper::to_dt(), @@ -2883,10 +3157,14 @@ static void ggml_sycl_op_top_k(ggml_backend_sycl_context & ctx, ggml_tensor * ds const int64_t ncols = src0->ne[0]; const int64_t nrows = ggml_nrows(src0); - GGML_ASSERT(k > 0 && k <= 32); + GGML_ASSERT(k > 0); GGML_ASSERT(k <= ncols); - top_k_f32_sycl(src0_dd, dst_dd, ncols, nrows, k, main_stream); + if (k <= SYCL_TOP_K_SCAN_MERGE_MAX_K) { + top_k_f32_sycl(ctx, src0_dd, dst_dd, ncols, nrows, k, main_stream); + } else { + ggml_sycl_top_k_radix(ctx, src0_dd, dst_dd, ncols, nrows, k, main_stream); + } } inline void ggml_sycl_op_argmax(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { @@ -3001,7 +3279,7 @@ inline void ggml_sycl_op_scale(ggml_backend_sycl_context & ctx, ggml_tensor * ds SYCL_CHECK(0); } -static void ggml_sycl_set_peer_access(const int n_tokens, int main_device) { +static void ggml_sycl_set_peer_access(const int n_tokens, [[maybe_unused]] int main_device) { static bool peer_access_enabled = false; const bool enable_peer_access = n_tokens <= GGML_SYCL_PEER_MAX_BATCH_SIZE; @@ -3062,7 +3340,6 @@ static void ggml_sycl_op_mul_mat(ggml_backend_sycl_context & ctx, const ggml_ten GGML_ASSERT(!ggml_backend_buffer_is_sycl_split(dst->buffer)); GGML_ASSERT(!ggml_backend_buffer_is_sycl_split(src1->buffer)); - GGML_ASSERT(src1->type == GGML_TYPE_F32 || (src1->ne[2] == 1 && src1->ne[3] == 1)); GGML_ASSERT(ne12 >= ne02 && ne12 % ne02 == 0); @@ -3230,7 +3507,8 @@ static void ggml_sycl_op_mul_mat(ggml_backend_sycl_context & ctx, const ggml_ten // for split tensors the data begins at i0 == i0_offset_low char * src0_dd_i = dev[i].src0_dd + (i0/i02_divisor) * (ne01*ne00*src0_ts)/src0_bs; - float * src1_ddf_i = dev[i].src1_ddf + (i0*ne11 + src1_col_0) * ne10; + float * src1_ddf_i = (float *) ((char *) dev[i].src1_ddf + + (i0*ne11 + src1_col_0) * ne10 * ggml_type_size(src1->type)); char * src1_ddq_i = dev[i].src1_ddq + src1_ddq_i_offset; float * dst_dd_i = dev[i].dst_dd + (i0*ne1 + src1_col_0) * (dst_on_device ? ne0 : row_diff); @@ -3251,12 +3529,12 @@ static void ggml_sycl_op_mul_mat(ggml_backend_sycl_context & ctx, const ggml_ten src1_ncols * src1_padded_col_size * q8_1_ts / q8_1_bs) .wait())); } else { - float * src1_ddf_i_source = (float *) src1_extra->data_device[ctx.device]; - src1_ddf_i_source += (i0 * ne11 + src1_col_0) * ne10; + const char * src1_ddf_i_source = (const char *) src1_extra->data_device[ctx.device] + + (i0 * ne11 + src1_col_0) * ne10 * ggml_type_size(src1->type); SYCL_CHECK( CHECK_TRY_ERROR(dev2dev_memcpy(i, *stream, ctx.device, *main_stream, src1_ddf_i, src1_ddf_i_source, - src1_ncols * ne10 * sizeof(float)))); + src1_ncols * ne10 * ggml_type_size(src1->type)))); } } } else { @@ -3362,6 +3640,11 @@ static void ggml_sycl_get_rows(ggml_backend_sycl_context & ctx, ggml_tensor * ds ggml_sycl_op_get_rows(ctx, dst); } +static void ggml_sycl_get_rows_back(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { + scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/2); + ggml_sycl_op_get_rows_back(ctx, dst); +} + static void ggml_sycl_norm(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/1); ggml_sycl_op_norm(ctx, dst); @@ -3505,7 +3788,9 @@ static void ggml_sycl_mul_mat_batched_sycl(ggml_backend_sycl_context & ctx, cons float * dst_ddf = static_cast(dst->data); const sycl::half * src1_f16 = static_cast(src1->data); +#if GGML_SYCL_DNNL const size_t type_size_src0 = ggml_type_size(src0->type); +#endif const size_t type_size_src1 = ggml_type_size(src1->type); bool is_src0_cont_2 = ggml_is_contiguous_2(src0); @@ -3522,6 +3807,7 @@ static void ggml_sycl_mul_mat_batched_sycl(ggml_backend_sycl_context & ctx, cons scope_op_debug_print scope_dbg_print(__func__, "/to_fp16_nc_sycl", dst, /*num_src=*/2, " : converting src1 to fp16"); +#if GGML_SYCL_DNNL // iterate tensor dims and find the slowest moving dim and stride int last_dim=0; int last_str=0; @@ -3541,7 +3827,6 @@ static void ggml_sycl_mul_mat_batched_sycl(ggml_backend_sycl_context & ctx, cons } } -#if GGML_SYCL_DNNL // oneDNN handles strided data and does not need overhead of ggml_get_to_fp16_nc_sycl const int64_t ne_src1 = src1->nb[last_str] * src1->ne[last_dim] / type_size_src1; src1_f16_alloc.alloc(ne_src1); @@ -3781,6 +4066,7 @@ inline bool ggml_sycl_supports_reorder_mmvq(enum ggml_type type) { case GGML_TYPE_Q1_0: case GGML_TYPE_Q4_0: case GGML_TYPE_Q8_0: + case GGML_TYPE_Q2_K: case GGML_TYPE_Q3_K: case GGML_TYPE_Q4_K: case GGML_TYPE_Q5_K: @@ -3794,8 +4080,10 @@ inline bool ggml_sycl_supports_reorder_mmvq(enum ggml_type type) { static bool ggml_sycl_supports_reorder_esimd(enum ggml_type type) { #ifdef GGML_SYCL_DMMV_HAS_ESIMD switch (type) { + case GGML_TYPE_Q2_K: case GGML_TYPE_Q3_K: case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: case GGML_TYPE_Q6_K: return true; default: @@ -3833,7 +4121,9 @@ static inline void * sycl_ext_malloc_device(dpct::queue_ptr stream, size_t size) bool use_async = g_ggml_sycl_use_async_mem_op; #if defined(GGML_SYCL_GRAPH) && SYCL_EXT_ONEAPI_ASYNC_MEMORY_ALLOC if (use_async) { - return syclex::async_malloc(*stream, sycl::usm::alloc::device, size); + void * ptr = syclex::async_malloc(*stream, sycl::usm::alloc::device, size); + ggml_sycl_memtrace_add(GGML_SYCL_MEM_ASYNC, ptr, size); + return ptr; } #else // If async allocation extension is not available, use_async should always be false. @@ -3846,6 +4136,7 @@ static inline void sycl_ext_free(dpct::queue_ptr stream, void * ptr) { bool use_async = g_ggml_sycl_use_async_mem_op; #if defined(GGML_SYCL_GRAPH) && SYCL_EXT_ONEAPI_ASYNC_MEMORY_ALLOC if (use_async) { + ggml_sycl_memtrace_del(ptr); syclex::async_free(*stream, ptr); return; } @@ -4473,6 +4764,18 @@ static bool can_use_mul_mat_vec_q(const ggml_tensor * src0, const ggml_tensor * static void ggml_sycl_mul_mat(ggml_backend_sycl_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/2); + + // Handle HADAMARAD hint given from further up the pipeline and pass it to the correct + // kernel. + // + // The op check is not redundant: this backend also routes MUL_MAT_ID through here with a + // stack copy of dst, which carries MUL_MAT_ID's own op_params. ggml_mul_mat_set_hint() + // asserts GGML_OP_MUL_MAT for the same reason. + if (dst->op == GGML_OP_MUL_MAT && ggml_get_op_params_i32(dst, 1) == GGML_HINT_SRC0_IS_HADAMARD && + ggml_sycl_op_fwht(ctx, src1, dst)) { + return; + } + const bool split = ggml_backend_buffer_is_sycl_split(src0->buffer); int64_t min_compute_capability = INT_MAX; @@ -4562,6 +4865,43 @@ static void ggml_sycl_mul_mat(ggml_backend_sycl_context & ctx, const ggml_tensor } } +// {mul_mat(gate), mul_mat(up), GLU} over the standard (non-reorder) weight layout, +// for quant pairs the reorder kernel does not cover (mixed gate/up types, e.g. UD-Q4_K_XL's +// iq4_xs gate + q5_K up). Two launches replace five: one shared q8_1 quantization and one +// dual-GEMV+GLU. +static bool ggml_sycl_mul_mat_glu_mmvq_plain(ggml_backend_sycl_context & ctx, ggml_tensor * glu, + ggml_tensor * gate, ggml_tensor * up, const ggml_tensor * wu, + const ggml_tensor * wg, const ggml_tensor * act) { + // weights already migrated to the reorder layout would be misread by the plain kernel + const auto * extra_u = static_cast(wu->extra); + const auto * extra_g = static_cast(wg->extra); + if ((extra_u && extra_u->optimized_feature.reorder) || (extra_g && extra_g->optimized_feature.reorder)) { + return false; + } + + // log the up mat-mul: glu's own srcs are the two intermediates the fusion never materialises + scope_op_debug_print scope_dbg_print(__func__, up, /*num_src=*/2, " : fused with gate + GLU (plain layout)"); + + const int64_t ne00 = wu->ne[0]; + const int64_t ne11 = act->ne[1]; + + const queue_ptr stream = ctx.stream(); + const int src1_padded_cols = GGML_PAD((int) ne00, MATRIX_ROW_PADDING); + + ggml_sycl_pool_alloc src1_q8_alloc(ctx.pool(), + (size_t) ne11 * src1_padded_cols * sizeof(block_q8_1) / QK8_1); + char * src1_ddq = src1_q8_alloc.get(); + + quantize_row_q8_1_sycl((const float *) act->data, src1_ddq, (int) ne00, (int) ne11, + src1_padded_cols, stream); + + return ggml_sycl_mul_mat_vec_q_glu_plain(wg->type, wu->type, ggml_get_glu_op(glu), wg->data, wu->data, + src1_ddq, (float *) glu->data, (int) ne00, (int) wu->ne[1], + (int) ne11, + /*stride_col_y=*/src1_padded_cols / QK8_1, + /*stride_col_dst=*/(int) glu->ne[0], stream); +} + // Fused dense-FFN mat-vec for the {mul_mat(gate), mul_mat(up), GLU} subgraph at node_idx. // Returns false if it declined, in which case the caller runs the three nodes normally. static bool ggml_sycl_mul_mat_glu_mmvq_fused(ggml_backend_sycl_context & ctx, ggml_cgraph * cgraph, int node_idx) { @@ -4587,6 +4927,12 @@ static bool ggml_sycl_mul_mat_glu_mmvq_fused(ggml_backend_sycl_context & ctx, gg return false; } + // quant pairs the reorder kernel cannot serve (mixed gate/up types) take the + // standard-layout fused path instead; q4_K keeps the reorder path below + if (wg->type != GGML_TYPE_Q4_K || wu->type != GGML_TYPE_Q4_K) { + return ggml_sycl_mul_mat_glu_mmvq_plain(ctx, glu, gate, up, wu, wg, act); + } + // install the reorder (SoA) layout the fused kernel needs, as the unfused mmvq path would; // a no-op once done. after the bail checks so a declined op does not pay for it. opt_for_reorder(&ctx, wu, act, up, mul_mat_algo::MMVQ); @@ -4623,6 +4969,78 @@ static bool ggml_sycl_mul_mat_glu_mmvq_fused(ggml_backend_sycl_context & ctx, gg /*stride_col_dst=*/(int) glu->ne[0], stream); } +// Batch the run of consecutive L2_NORM siblings starting at node_idx into one launch. +// Returns the number of extra graph nodes consumed, or 0 if the run is shorter than two +// (the caller then runs the norm through the per-tensor kernel). +static int ggml_sycl_l2_norm_batch_fused(ggml_backend_sycl_context & ctx, ggml_cgraph * cgraph, int node_idx) { + const ggml_tensor * node = cgraph->nodes[node_idx]; + if (ggml_sycl_info().device_count != 1 || node->type != GGML_TYPE_F32 || + node->src[0]->type != GGML_TYPE_F32 || node->src[0]->ne[0] >= 1024) { + return 0; + } + + ggml_tensor * batch[GGML_SYCL_L2_BATCH_MAX]; + int count = 0; + int last = node_idx; + float eps0; + memcpy(&eps0, node->op_params, sizeof(float)); + + // Conservative aliasing test: the batched norms run concurrently in one kernel, + // so none may read what another writes, and none may write where another writes. + auto overlaps = [](const ggml_tensor * a, const ggml_tensor * b) { + const char * ab = (const char *) a->data; + const char * bb = (const char *) b->data; + return ab < bb + ggml_nbytes(b) && bb < ab + ggml_nbytes(a); + }; + + for (int j = node_idx; j < cgraph->n_nodes && count < GGML_SYCL_L2_BATCH_MAX; ++j) { + ggml_tensor * nj = cgraph->nodes[j]; + if (ggml_is_empty(nj) || nj->op == GGML_OP_RESHAPE || nj->op == GGML_OP_TRANSPOSE || + nj->op == GGML_OP_VIEW || nj->op == GGML_OP_PERMUTE || nj->op == GGML_OP_NONE || + (nj->flags & GGML_TENSOR_FLAG_COMPUTE) == 0) { + continue; // not a launch; cannot break a run of adjacent norms + } + if (nj->op != GGML_OP_L2_NORM || nj->type != GGML_TYPE_F32 || + nj->src[0]->type != GGML_TYPE_F32 || !ggml_are_same_shape(nj, node) || + !ggml_are_same_shape(nj->src[0], node->src[0])) { + break; // any other launch ends the run + } + bool same_nb = true; + for (int d = 0; d < GGML_MAX_DIMS; ++d) { + if (nj->nb[d] != node->nb[d] || nj->src[0]->nb[d] != node->src[0]->nb[d]) { + same_nb = false; + break; + } + } + if (!same_nb) { + break; // one nb[] stride set is shared by the whole batch + } + float epsj; + memcpy(&epsj, nj->op_params, sizeof(float)); + if (epsj != eps0) { + break; // eps mismatch ends the run + } + bool indep = true; + for (int k = 0; k < count; ++k) { + if (overlaps(nj->src[0], batch[k]) || overlaps(nj, batch[k])) { + indep = false; + break; + } + } + if (!indep) { + break; // an overlapping tensor would race inside one launch + } + batch[count++] = nj; + last = j; + } + if (count < 2) { + return 0; // a lone norm falls through to the per-tensor kernel + } + ggml_sycl_l2_norm_batch(ctx, batch, count); + return last - node_idx; +} + + __dpct_inline__ static void k_copy_src1_to_contiguous( const char *__restrict__ src1_original, char *__restrict__ src1_contiguous, const mmid_row_mapping *__restrict__ row_mapping, @@ -5034,6 +5452,7 @@ catch (sycl::exception const &exc) { } static bool ggml_sycl_compute_forward(ggml_backend_sycl_context & ctx, struct ggml_tensor * dst) try { + GGML_SYCL_DEBUG("[SYCL] ggml_sycl_compute_forward: dst=%s, op=%s\n", dst->name, ggml_op_name(dst->op)); if (!g_sycl_loaded) return false; if (dst->src[0] != nullptr && ggml_backend_buffer_is_sycl_split(dst->src[0]->buffer)) { @@ -5068,6 +5487,9 @@ static bool ggml_sycl_compute_forward(ggml_backend_sycl_context & ctx, struct gg case GGML_OP_GET_ROWS: ggml_sycl_get_rows(ctx, dst); break; + case GGML_OP_GET_ROWS_BACK: + ggml_sycl_get_rows_back(ctx, dst); + break; case GGML_OP_SET: ggml_sycl_op_set(ctx, dst); break; @@ -5200,6 +5622,9 @@ static bool ggml_sycl_compute_forward(ggml_backend_sycl_context & ctx, struct gg case GGML_GLU_OP_SWIGLU_OAI: ggml_sycl_swiglu_oai(ctx, dst); break; + case GGML_GLU_OP_SWIGLU_CLAMP: + ggml_sycl_swiglu_clamp(ctx, dst); + break; case GGML_GLU_OP_GEGLU_ERF: ggml_sycl_geglu_erf(ctx, dst); break; @@ -5414,18 +5839,33 @@ catch (sycl::exception const &exc) { std::exit(1); } -void ggml_backend_sycl_get_device_memory(int device, size_t *free, - size_t *total) try { - GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_get_device_memory\n"); - ggml_sycl_set_device(device); +bool sycl_get_mem_info(int device, size_t * free, size_t * total) { + GGML_SYCL_DEBUG("[SYCL] [%s] g_ggml_sycl_get_mem_api=%d\n", + __func__, g_ggml_sycl_get_mem_api); - SYCL_CHECK(CHECK_TRY_ERROR( - dpct::dev_mgr::instance().get_device(device).get_memory_info(*free, *total))); + MemoryAPIType mem_api_type = MemoryAPIType::MEMORY_API_TYPE_SYCL; + +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API + mem_api_type = get_zes_init_res() == ZE_RESULT_SUCCESS ? + (MemoryAPIType) g_ggml_sycl_get_mem_api : MemoryAPIType::MEMORY_API_TYPE_SYCL; +#else + mem_api_type = MemoryAPIType::MEMORY_API_TYPE_SYCL; +#endif + bool res = get_memory_size(dpct::dev_mgr::instance().get_device(device), + *free, *total, mem_api_type); + GGML_SYCL_DEBUG("[SYCL] [%s] total = %zu free = %zu\n", __func__, *total, *free); + return res; } -catch (sycl::exception const &exc) { - std::cerr << exc.what() << "Exception caught at file:" << __FILE__ - << ", line:" << __LINE__ << std::endl; - std::exit(1); + +void ggml_backend_sycl_get_device_memory(int device, size_t * free, size_t * total) try { + GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_get_device_memory\n"); + if (!sycl_get_mem_info(device, free, total)) { + GGML_ABORT("[%s] failed to get device memory size", __func__); + } + ggml_sycl_memtrace_report_device("device memory query", device, *free, *total); +} catch (const sycl::exception & exc) { + std::cerr << exc.what() << "Exception caught at file:" << __FILE__ << ", line:" << __LINE__ << std::endl; + std::exit(1); } //////////////////////////////////////////////////////////////////////////////// @@ -5645,12 +6085,32 @@ static void ggml_backend_sycl_graph_compute_impl(ggml_backend_sycl_context * syc continue; } } + if (node->op == GGML_OP_RMS_NORM && + ggml_sycl_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) { + ggml_sycl_op_rms_norm_fused_add(*sycl_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); + i += 2; + continue; + } if (node->op == GGML_OP_RMS_NORM && ggml_sycl_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL }, {})) { ggml_sycl_op_rms_norm_fused(*sycl_ctx, node, cgraph->nodes[i + 1]); i++; continue; } + // qwen35 GDN l2 norms are emitted as rms_norm + scalar scale (models.h + // build_gdn_l2_norm), which the rms_norm+mul fusion above cannot match + if (node->op == GGML_OP_RMS_NORM && + ggml_sycl_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_SCALE }, {})) { + ggml_sycl_op_rms_norm_scale_fused(*sycl_ctx, node, cgraph->nodes[i + 1]); + i++; + continue; + } + if (node->op == GGML_OP_ADD && + ggml_sycl_can_fuse(cgraph, i, { GGML_OP_ADD, GGML_OP_ADD }, {})) { + ggml_sycl_op_add_add_fused(*sycl_ctx, node, cgraph->nodes[i + 1]); + i++; + continue; + } if (node->op == GGML_OP_UNARY && ggml_sycl_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_MUL }, { ggml_get_unary_op(node) })) { ggml_sycl_op_unary_mul_fused(*sycl_ctx, node, cgraph->nodes[i + 1]); @@ -5658,6 +6118,31 @@ static void ggml_backend_sycl_graph_compute_impl(ggml_backend_sycl_context * syc continue; } + // Batch consecutive independent same-shape F32 L2_NORM siblings (the GDN q/k + // norms) into one launch; sources are strided views of the fused qkv buffer, so + // the scan skips the interleaved view nodes instead of breaking on them. + if (node->op == GGML_OP_L2_NORM) { + const int l2_batch_skip = ggml_sycl_l2_norm_batch_fused(*sycl_ctx, cgraph, i); + if (l2_batch_skip > 0) { + i += l2_batch_skip; + continue; + } + } + + if (node->op == GGML_OP_SSM_CONV && + ggml_sycl_can_fuse(cgraph, i, { GGML_OP_SSM_CONV, GGML_OP_ADD, GGML_OP_UNARY }, { GGML_UNARY_OP_SILU })) { + ggml_sycl_ssm_conv_fused(*sycl_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); + i += 2; + continue; + } + + if (node->op == GGML_OP_SSM_CONV && + ggml_sycl_can_fuse(cgraph, i, { GGML_OP_SSM_CONV, GGML_OP_UNARY }, { GGML_UNARY_OP_SILU })) { + ggml_sycl_ssm_conv_fused(*sycl_ctx, node, nullptr, cgraph->nodes[i + 1]); + i++; + continue; + } + if (node->op == GGML_OP_MUL_MAT && ggml_sycl_mul_mat_glu_mmvq_fused(*sycl_ctx, cgraph, i)) { i += 2; continue; @@ -5826,7 +6311,7 @@ bool ggml_backend_is_sycl(ggml_backend_t backend) { return backend != NULL && ggml_guid_matches(backend->guid, ggml_backend_sycl_guid()); } -int ggml_backend_sycl_get_device_count() { +int ggml_backend_sycl_get_device_count(void) { return ggml_sycl_info().device_count; } @@ -5844,10 +6329,13 @@ static const char * ggml_backend_sycl_device_get_description(ggml_backend_dev_t } static void ggml_backend_sycl_device_get_memory(ggml_backend_dev_t dev, size_t * free, size_t * total) { - ggml_backend_sycl_device_context * ctx = (ggml_backend_sycl_device_context *)dev->context; - ggml_sycl_set_device(ctx->device); - SYCL_CHECK(CHECK_TRY_ERROR( - dpct::dev_mgr::instance().get_device(ctx->device).get_memory_info(*free, *total))); + GGML_SYCL_DEBUG("[SYCL] call %s\n", __func__); + ggml_backend_sycl_device_context * ctx = (ggml_backend_sycl_device_context *) dev->context; + if (!sycl_get_mem_info(ctx->device, free, total)) { + GGML_ABORT("[%s] failed to get device memory size", __func__); + } + GGML_SYCL_DEBUG("[SYCL] call %s total %zu free %zu\n", __func__, *total, *free); + ggml_sycl_memtrace_report_device("device memory query (dev)", ctx->device, *free, *total); } static enum ggml_backend_dev_type ggml_backend_sycl_device_get_type(ggml_backend_dev_t dev) { @@ -5893,8 +6381,8 @@ static ggml_backend_buffer_type_t ggml_backend_sycl_device_get_buffer_type(ggml_ } static ggml_backend_buffer_type_t ggml_backend_sycl_device_get_host_buffer_type(ggml_backend_dev_t dev) { - GGML_UNUSED(dev); - return ggml_backend_sycl_host_buffer_type(); + ggml_backend_sycl_device_context * ctx = (ggml_backend_sycl_device_context *) dev->context; + return ggml_backend_sycl_host_buffer_type_for_device(ctx->device); } static ggml_backend_buffer_t ggml_backend_sycl_device_buffer_from_host_ptr(ggml_backend_dev_t dev, void * ptr, size_t size, size_t max_tensor_size) { @@ -5960,6 +6448,7 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return ggml_is_contiguous_1(op->src[0]); default: return false; @@ -5988,6 +6477,11 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons a->ne[0] > 128 && a->ne[2] == 1 && src0_type == GGML_TYPE_F16) { return false; } + + if (src0_type == GGML_TYPE_TQ2_0 || src0_type == GGML_TYPE_TQ1_0) { + return false; + } + return true; } case GGML_OP_OUT_PROD: @@ -6030,6 +6524,12 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons return false; } } + case GGML_OP_GET_ROWS_BACK: + // return true; + return op->type == GGML_TYPE_F32 && + (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16) && + op->src[1]->type == GGML_TYPE_I32 && + op->ne[2] == 1 && op->ne[3] == 1; case GGML_OP_SET: return (op->type == GGML_TYPE_F32) && (op->src[0] && op->src[1]) && @@ -6038,6 +6538,9 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons case GGML_OP_SET_ROWS: { + if (op->type == GGML_TYPE_TQ2_0 || op->type == GGML_TYPE_TQ1_0) { + return false; + } auto res = (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16 || op->src[0]->type == GGML_TYPE_BF16) && (op->src[1]->type == GGML_TYPE_I64 || op->src[1]->type == GGML_TYPE_I32); @@ -6052,7 +6555,8 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons op->src[2]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32; case GGML_OP_DSV4_HC_POST: return op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32 && - op->src[2]->type == GGML_TYPE_F32 && op->src[3]->type == GGML_TYPE_F32 && + op->src[2]->type == GGML_TYPE_F32 && + (op->src[3] == nullptr || op->src[3]->type == GGML_TYPE_F32) && op->type == GGML_TYPE_F32; case GGML_OP_LIGHTNING_INDEXER: return op->src[0]->type == GGML_TYPE_F32 && @@ -6156,11 +6660,18 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons src1_type == GGML_TYPE_IQ3_XXS || src1_type == GGML_TYPE_IQ3_S || src1_type == GGML_TYPE_IQ1_S || - src1_type == GGML_TYPE_IQ1_M) { + src1_type == GGML_TYPE_IQ1_M || + src1_type == GGML_TYPE_TQ2_0 || + src1_type == GGML_TYPE_TQ1_0) { return false; } } + if (src0_type == GGML_TYPE_TQ2_0 || src1_type == GGML_TYPE_TQ2_0 || + src0_type == GGML_TYPE_TQ1_0 || src1_type == GGML_TYPE_TQ1_0) { + return false; + } + return true; } case GGML_OP_REPEAT_BACK: @@ -6253,7 +6764,7 @@ static bool do_ggml_backend_sycl_device_supports_op(ggml_backend_dev_t dev, cons op->type == GGML_TYPE_I32 && src0->type == GGML_TYPE_F32 && ggml_is_contiguous(src0) && - k > 0 && k <= 32; + k > 0 && k <= src0->ne[0]; } case GGML_OP_POOL_2D: case GGML_OP_POOL_1D: @@ -6709,6 +7220,7 @@ static const ggml_backend_reg_i ggml_backend_sycl_reg_interface = { // backend registry ggml_backend_reg_t ggml_backend_sycl_reg() { + GGML_SYCL_DEBUG("[SYCL] call ggml_backend_sycl_reg\n"); static ggml_backend_reg reg; static bool initialized = false; @@ -6716,6 +7228,7 @@ ggml_backend_reg_t ggml_backend_sycl_reg() { static std::mutex mutex; std::lock_guard lock(mutex); if (!initialized) { + ggml_check_sycl(); ggml_backend_sycl_reg_context * ctx = new ggml_backend_sycl_reg_context; const int min_batch_size = getenv("GGML_OP_OFFLOAD_MIN_BATCH") ? atoi(getenv("GGML_OP_OFFLOAD_MIN_BATCH")) : 32; diff --git a/ggml/src/ggml-sycl/im2col.cpp b/ggml/src/ggml-sycl/im2col.cpp index 7bf3584f..e6661675 100644 --- a/ggml/src/ggml-sycl/im2col.cpp +++ b/ggml/src/ggml-sycl/im2col.cpp @@ -85,7 +85,7 @@ static void im2col_sycl(const float * x, */ stream->parallel_for(sycl::nd_range<3>(block_nums * sycl::range<3>(1, 1, MIN(IC_KH_KW, SYCL_IM2COL_BLOCK_SIZE)), sycl::range<3>(1, 1, MIN(IC_KH_KW, SYCL_IM2COL_BLOCK_SIZE))), - [=](sycl::nd_item<3> item_ct1) { + [=](sycl::nd_item<3>) { im2col_kernel(x, dst, IC, IW, IH, OH, OW, KW, KH, IC_IH_IW, IH_IW, N_OH, KH_KW, IC_KH_KW, s0, s1, p0, p1, d0, d1); }); @@ -271,7 +271,7 @@ static void im2col_3d_sycl(const float * src, */ stream->parallel_for(sycl::nd_range<3>(block_nums * sycl::range<3>(1, 1, MIN(IC_KD_KH_KW, SYCL_IM2COL_BLOCK_SIZE)), sycl::range<3>(1, 1, MIN(IC_KD_KH_KW, SYCL_IM2COL_BLOCK_SIZE))), - [=](sycl::nd_item<3> item_ct1) { + [=](sycl::nd_item<3>) { im2col_3d_kernel(src, dst, N, IC, ID, IH, IW, OC, KD, KH, KW, OD, OH, OW, OH_OW, KD_KH_KW, ID_IH_IW, KH_KW, IH_IW, IC_ID_IH_IW, IC_KD_KH_KW, OW_KD_KH_KW, OD_OH_OW_IC_KD_KH_KW, OH_OW_IC_KD_KH_KW, OW_IC_KD_KH_KW, N_OD_OH, OD_OH, diff --git a/ggml/src/ggml-sycl/mem.cpp b/ggml/src/ggml-sycl/mem.cpp new file mode 100644 index 00000000..ad5bfe0f --- /dev/null +++ b/ggml/src/ggml-sycl/mem.cpp @@ -0,0 +1,151 @@ +#include +#include + +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API +#include +#include +#endif + +#include +#include +#include + +#include "base.hpp" +#include "mem.hpp" + +const char * mem_api_int2str(int mem_api) { + if (mem_api == MEMORY_API_TYPE_SYCL) { + return "SYCL API"; + } else if (mem_api == MEMORY_API_TYPE_LEVEL_ZERO) { + return "Level Zero API"; + } else { + return "Unknown"; + } +} + +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API +/* +* Depend on to call zesInit(0) before any other Level Zero API calls, otherwise the Level Zero API calls may fail. +*/ +bool query_free_memory_by_ze(sycl::device dev, size_t & free_bytes, size_t & total_bytes) { + GGML_SYCL_DEBUG("[SYCL] call %s: Querying free memory using Level Zero API.\n", __func__); + + free_bytes = 0; + total_bytes = 0; + + uint32_t module_count = 0; + +#if defined(SYCL_EXT_ONEAPI_BACKEND_LEVEL_ZERO) + constexpr sycl::backend kL0Backend = sycl::backend::ext_oneapi_level_zero; +#else + constexpr sycl::backend kL0Backend = sycl::backend::level_zero; +#endif + + try { + + if (dev.get_platform().get_backend() != kL0Backend) { + GGML_SYCL_DEBUG("Device backend is not Level Zero.\n"); + return false; + } + + ze_device_handle_t ze_dev = sycl::get_native(dev); + if (ze_dev == nullptr) { + GGML_SYCL_DEBUG("Level Zero device handle is null.\n"); + return false; + } + + ze_result_t r = zesDeviceEnumMemoryModules(ze_dev, &module_count, nullptr); + if (r != ZE_RESULT_SUCCESS || module_count == 0) { + GGML_SYCL_DEBUG("Failed to enumerate Level Zero memory modules.\n"); + return false; + } + + std::vector modules(module_count); + r = zesDeviceEnumMemoryModules(ze_dev, &module_count, modules.data()); + if (r != ZE_RESULT_SUCCESS || module_count == 0) { + GGML_SYCL_DEBUG("Failed to enumerate Level Zero memory modules.\n"); + return false; + } + + for (uint32_t i = 0; i < module_count; ++i) { + zes_mem_state_t state = {}; + state.stype = ZES_STRUCTURE_TYPE_MEM_STATE; + state.pNext = nullptr; + + r = zesMemoryGetState(modules[i], &state); + if (r != ZE_RESULT_SUCCESS) { + continue; + } + + free_bytes += state.free; + total_bytes += state.size; + } + + if (total_bytes == 0) { + GGML_SYCL_DEBUG("Level Zero memory query returned zero total bytes.\n"); + return false; + } + return total_bytes >= free_bytes; + + } catch (const sycl::exception & e) { + GGML_SYCL_DEBUG("Level Zero memory query failed: %s\n", e.what()); + return false; + } +} +#endif + +bool get_memory_size_by_sycl_api(sycl::device dev, size_t & free_bytes, size_t & total_bytes) { + GGML_SYCL_DEBUG("[SYCL] call %s: Querying free memory using SYCL API.\n", __func__); + total_bytes = dev.get_info(); + +#if (defined(__SYCL_COMPILER_VERSION) && __SYCL_COMPILER_VERSION >= 20221105) + if (dev.has(sycl::aspect::ext_intel_free_memory)) { + try { + GGML_SYCL_DEBUG("Querying free memory using SYCL aspect::ext_intel_free_memory.\n"); + free_bytes = dev.get_info(); + return true; + } catch (const sycl::exception &) { + GGML_SYCL_DEBUG( + "Failed to query free memory using SYCL aspect::ext_intel_free_memory.\n"); + return false; + } + } else { + GGML_SYCL_DEBUG( + "Device does not support SYCL aspect::ext_intel_free_memory.\n"); + } +#else + GGML_SYCL_DEBUG("SYCL Compiler version is older than 20221105.\n"); +#endif + return false; +} + +bool get_memory_size(sycl::device dev, size_t & free_bytes, size_t & total_bytes, MemoryAPIType api_type) { + + GGML_SYCL_DEBUG("[%s]GPU Name: %s\n", __func__, + dev.get_info().c_str()); + GGML_SYCL_DEBUG("[%s]GPU Vendor: %s\n", __func__, + dev.get_info().c_str()); + + if (api_type == MEMORY_API_TYPE_LEVEL_ZERO) { +#ifdef GGML_SYCL_SUPPORT_LEVEL_ZERO_API + GGML_SYCL_DEBUG("[%s] Querying free memory using Level Zero API.\n", __func__); + if (query_free_memory_by_ze(dev, free_bytes, total_bytes)) { + return true; + } + //fallback to SYCL API if Level Zero API fails + GGML_SYCL_DEBUG("[%s] Falling back to SYCL API for memory query.\n", __func__); +#endif + } + + //MEMORY_API_TYPE_SYCL + if(get_memory_size_by_sycl_api(dev, free_bytes, total_bytes)){ + return true; + } + + //Todo, fallback to other methods to get free memory size, such as using OS-specific APIs (e.g., /proc/meminfo on Linux, GlobalMemoryStatusEx on Windows, etc.) + GGML_SYCL_DEBUG( + "[%s] Can't get free mem size by Level Zero and SYCL API. Using total memory as free memory.\n", __func__); + free_bytes = total_bytes; + + return true; +} diff --git a/ggml/src/ggml-sycl/mem.hpp b/ggml/src/ggml-sycl/mem.hpp new file mode 100644 index 00000000..b3e45cfe --- /dev/null +++ b/ggml/src/ggml-sycl/mem.hpp @@ -0,0 +1,16 @@ +#ifndef GGML_SYCL_MEM_HPP +#define GGML_SYCL_MEM_HPP + +#include + +enum MemoryAPIType { + MEMORY_API_TYPE_LEVEL_ZERO = 0, + MEMORY_API_TYPE_SYCL = 1, +}; + +const char* mem_api_int2str(int mem_api); + +bool get_memory_size(sycl::device dev, size_t & free_bytes, size_t & total_bytes, + MemoryAPIType api_type); + +#endif // GGML_SYCL_MEM_HPP diff --git a/ggml/src/ggml-sycl/memtrace.cpp b/ggml/src/ggml-sycl/memtrace.cpp new file mode 100644 index 00000000..9c4f8853 --- /dev/null +++ b/ggml/src/ggml-sycl/memtrace.cpp @@ -0,0 +1,194 @@ +#include "memtrace.hpp" + +#include "common.hpp" +#include "ggml-impl.h" + +#include +#include +#include + +constexpr size_t MIB = 1024 * 1024; + +static const char * mem_type_name(ggml_sycl_mem_type type) { + switch (type) { + case GGML_SYCL_MEM_BUFFER: return "buffer"; + case GGML_SYCL_MEM_POOL_LEG: return "pool_leg"; + case GGML_SYCL_MEM_POOL_VMM: return "pool_vmm"; + case GGML_SYCL_MEM_ASYNC: return "async"; + case GGML_SYCL_MEM_FATTN_KV: return "fattn_kv"; + case GGML_SYCL_MEM_DIRECT: return "direct"; + default: GGML_ABORT("[%s] The type value %d is not supported\n", __func__, (int) type); + } +} + +struct mem_tracker { + std::mutex mutex; + std::unordered_map> live_by_ptr; + size_t live[GGML_SYCL_MEM_TYPE_COUNT] = {}; + size_t peak[GGML_SYCL_MEM_TYPE_COUNT] = {}; + size_t total_live = 0; + size_t total_peak = 0; + size_t last_logged_peak = 0; +}; + +static mem_tracker & get_tracker() { + static mem_tracker t; + return t; +} + +static size_t step_bytes() { + const int mib = g_ggml_sycl_memtrace_step > 0 ? g_ggml_sycl_memtrace_step : 64; + return (size_t) mib * MIB; +} + +static void report_sites_locked() { + mem_tracker & t = get_tracker(); + for (int i = 0; i < GGML_SYCL_MEM_TYPE_COUNT; i++) { + if (t.peak[i] == 0) { + continue; + } + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " %-9s allocated %5zu MiB, peak %5zu MiB\n", + mem_type_name((ggml_sycl_mem_type) i), t.live[i] / MIB, t.peak[i] / MIB); + } +} + +static void report_locked(const char * tag) { + mem_tracker & t = get_tracker(); + + const size_t allocated = t.total_live / MIB; + const size_t buffers = t.live[GGML_SYCL_MEM_BUFFER] / MIB; + + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " %s: allocated %5zu MiB (buffers %5zu + scratch %5zu)," + " peak %5zu MiB\n", + tag, allocated, buffers, allocated - buffers, t.total_peak / MIB); + report_sites_locked(); +} + +static void log_event_locked(const char * op, ggml_sycl_mem_type type, const void * ptr, size_t bytes) { + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " allocated %5zu MiB %-5s %-9s %9.3f MiB ptr=%p\n", + get_tracker().total_live / MIB, op, mem_type_name(type), + (double) bytes / MIB, ptr); +} + +bool ggml_sycl_memtrace_enabled() { + return g_ggml_sycl_memtrace > 0; +} + +void ggml_sycl_memtrace_add(ggml_sycl_mem_type type, const void * ptr, size_t bytes) { + if (!ggml_sycl_memtrace_enabled()) { + return; + } + GGML_ASSERT(ptr != nullptr); + GGML_ASSERT(bytes != 0); + + mem_tracker & t = get_tracker(); + std::lock_guard lock(t.mutex); + + auto it = t.live_by_ptr.find(ptr); + if (it != t.live_by_ptr.end()) { + t.live[it->second.first] -= it->second.second; + t.total_live -= it->second.second; + } + + t.live_by_ptr[ptr] = { type, bytes }; + t.live[type] += bytes; + t.total_live += bytes; + + if (t.live[type] > t.peak[type]) { + t.peak[type] = t.live[type]; + } + if (t.total_live > t.total_peak) { + t.total_peak = t.total_live; + } + + if (g_ggml_sycl_memtrace >= 2) { + log_event_locked("alloc", type, ptr, bytes); + } + + static const size_t step = step_bytes(); + if (t.total_peak >= t.last_logged_peak + step) { + t.last_logged_peak = t.total_peak; + char tag[96]; + std::snprintf(tag, sizeof(tag), "peak grew (+%zu MiB from %s)", bytes / MIB, + mem_type_name(type)); + report_locked(tag); + } +} + +void ggml_sycl_memtrace_del(const void * ptr) { + if (!ggml_sycl_memtrace_enabled() || ptr == nullptr) { + return; + } + mem_tracker & t = get_tracker(); + std::lock_guard lock(t.mutex); + + auto it = t.live_by_ptr.find(ptr); + if (it == t.live_by_ptr.end()) { + return; + } + const ggml_sycl_mem_type type = it->second.first; + const size_t bytes = it->second.second; + t.live[type] -= bytes; + t.total_live -= bytes; + t.live_by_ptr.erase(it); + + if (g_ggml_sycl_memtrace >= 2) { + log_event_locked("free", type, ptr, bytes); + } +} + +void ggml_sycl_memtrace_fail(ggml_sycl_mem_type type, size_t bytes) { + GGML_LOG_ERROR(GGML_SYCL_MEMTRACE_TAG " alloc FAILED: %9.3f MiB %s\n", + (double) bytes / MIB, mem_type_name(type)); + if (!ggml_sycl_memtrace_enabled()) { + return; + } + mem_tracker & t = get_tracker(); + std::lock_guard lock(t.mutex); + report_locked("at allocation failure"); +} + +void ggml_sycl_memtrace_report(const char * tag) { + if (!ggml_sycl_memtrace_enabled()) { + return; + } + mem_tracker & t = get_tracker(); + std::lock_guard lock(t.mutex); + report_locked(tag); +} + +static bool device_memory_is_dedicated(int device) { + if (device < 0 || device >= ggml_sycl_info().device_count) { + return false; + } + const sycl_device_info & info = ggml_sycl_info().devices[device]; + return info.l0_device_type_valid && info.l0_discrete_gpu; +} + +void ggml_sycl_memtrace_report_device(const char * tag, int device, size_t dev_free, size_t dev_total) { + if (!ggml_sycl_memtrace_enabled()) { + return; + } + mem_tracker & t = get_tracker(); + std::lock_guard lock(t.mutex); + + const size_t in_use = dev_total > dev_free ? dev_total - dev_free : 0; + const size_t total = dev_total / MIB; + const size_t freed = dev_free / MIB; + const size_t allocated = t.total_live / MIB; + const size_t buffers = t.live[GGML_SYCL_MEM_BUFFER] / MIB; + const size_t peak = t.total_peak / MIB; + + if (in_use >= t.total_live && device_memory_is_dedicated(device) && total >= freed + allocated) { + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " %s: total %5zu MiB = free %5zu + allocated %5zu" + " (buffers %5zu + scratch %5zu) + other %5zu, peak %5zu MiB\n", + tag, total, freed, allocated, buffers, allocated - buffers, + total - freed - allocated, peak); + } else { + GGML_LOG_INFO(GGML_SYCL_MEMTRACE_TAG " %s: total %5zu MiB, free %5zu, in use %5zu;" + " allocated %5zu (buffers %5zu + scratch %5zu), peak %5zu MiB\n", + tag, total, freed, in_use / MIB, allocated, buffers, + allocated - buffers, peak); + } + report_sites_locked(); +} diff --git a/ggml/src/ggml-sycl/memtrace.hpp b/ggml/src/ggml-sycl/memtrace.hpp new file mode 100644 index 00000000..426d9096 --- /dev/null +++ b/ggml/src/ggml-sycl/memtrace.hpp @@ -0,0 +1,28 @@ +#ifndef GGML_SYCL_MEMTRACE_HPP +#define GGML_SYCL_MEMTRACE_HPP + +#include + +#define GGML_SYCL_MEMTRACE_TAG "[SYCL-MEMTRACE]" + +enum ggml_sycl_mem_type { + GGML_SYCL_MEM_BUFFER = 0, + GGML_SYCL_MEM_POOL_LEG, + GGML_SYCL_MEM_POOL_VMM, + GGML_SYCL_MEM_ASYNC, + GGML_SYCL_MEM_FATTN_KV, + GGML_SYCL_MEM_DIRECT, + + GGML_SYCL_MEM_TYPE_COUNT, +}; + +bool ggml_sycl_memtrace_enabled(); + +void ggml_sycl_memtrace_add(ggml_sycl_mem_type type, const void * ptr, size_t bytes); +void ggml_sycl_memtrace_del(const void * ptr); + +void ggml_sycl_memtrace_report(const char * tag); +void ggml_sycl_memtrace_report_device(const char * tag, int device, size_t dev_free, size_t dev_total); +void ggml_sycl_memtrace_fail(ggml_sycl_mem_type type, size_t bytes); + +#endif // GGML_SYCL_MEMTRACE_HPP diff --git a/ggml/src/ggml-sycl/mmvq.cpp b/ggml/src/ggml-sycl/mmvq.cpp index 123b2a2f..7e4f22dd 100644 --- a/ggml/src/ggml-sycl/mmvq.cpp +++ b/ggml/src/ggml-sycl/mmvq.cpp @@ -6,6 +6,24 @@ #include "quants.hpp" #include "vecdotq.hpp" +// Minimum weight-row count at which the Q4_K multi-column MMVQ kernel handles two output rows per +// subgroup (rows_per_sg == 2) instead of one, when ncols_dst == 2. +// +// Pairing rows lets a subgroup load each activation block once and apply it to two rows, at the cost +// of halving the number of subgroups in the launch. With only two destination columns there is too +// little work per row to hide that loss of parallelism, so pairing only pays off once there are +// enough rows to keep the device occupied. This is a measured performance crossover, not a +// correctness or hardware limit - both variants compute the same result for any nrows. +// +// Derived on Intel Arc Pro B70 with `test-backend-ops perf -o MUL_MAT` (Q4_K, ncols_dst == 2), +// sweeping nrows over 5120..6912 at ncols 17408 and 19968: one row per subgroup was up to 9% faster +// below the crossover, two rows per subgroup 8-15% faster above it, and the crossover fell inside +// (6144, 6272] for both ncols with no measurable ncols dependence. A later 32-row granularity sweep +// narrowed it to (6144, 6176], so 6272 is a conservative gate rather than the exact crossover. +// ncols_dst >= 3 amortizes the activation loads over more columns and is faster with two rows at +// every row count, so it does not consult this threshold. +static constexpr int Q4_K_MMVQ_ROW_PAIR_MIN_NROWS = 6272; + template static void mul_mat_vec_q_reorder(const void * __restrict__ vx, const void * __restrict__ vy, float * __restrict__ dst, const int ncols, const int nrows, const sycl::nd_item<3> & nd_item) { @@ -59,7 +77,7 @@ static void mul_mat_vec_q_reorder(const void * __restrict__ vx, const void * __r // With has_fusion, `vgate` is a second weight matrix sharing vx's shape, stride and reorder // layout: one pass computes both row dot products and the epilogue writes glu(gate, up). -template +template static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void * __restrict__ vgate, const void * __restrict__ vy, float * __restrict__ dst, const int ncols, const int nrows, const int stride_col_y_bytes, const int stride_col_dst, @@ -71,14 +89,17 @@ static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void const int sg_range = sg.get_group_linear_range(); const int workgroup_id = nd_item.get_group_linear_id(); const int sg_id = sg.get_group_linear_id(); - const int row = workgroup_id * sg_range + sg_id; + const int row0 = (workgroup_id * sg_range + sg_id) * rows_per_sg; // row is sub-group uniform, so this retires whole sub-groups and the collectives below // stay convergent - if (row >= nrows) { + if (row0 >= nrows) { return; } + static_assert(rows_per_sg == 1 || + reorder_vec_dot_shared_activations::value); + const int blocks_per_row = ncols / block_traits::qk; constexpr int blocks_per_subgroup = ceil_div(block_traits::vdr_mmvq * WARP_SIZE, block_traits::qi); constexpr int block_elements_per_subgroup = block_traits::qi / block_traits::vdr_mmvq; @@ -87,34 +108,96 @@ static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void static_assert(blocks_per_subgroup > 0); static_assert(block_elements_per_subgroup > 0); - float partial_sum[ncols_dst] = { 0.0f }; + float partial_sum[ncols_dst][rows_per_sg] = {}; // sized 1 rather than 0 when unused: zero-length arrays are not standard C++, and the // array is dead and eliminated in that case - [[maybe_unused]] float partial_gate[has_fusion ? ncols_dst : 1] = { 0.0f }; + [[maybe_unused]] float partial_gate[has_fusion ? ncols_dst : 1][has_fusion ? rows_per_sg : 1] = {}; for (int i = sg.get_local_linear_id() / block_elements_per_subgroup; i < blocks_per_row; i += blocks_per_subgroup) { - const int ibx = row * blocks_per_row + i; - - // the offsets depend only on the block index and the matrix shape, never on the base - // pointer, which is what lets vgate reuse them - const auto bx_offset = block_type::get_block_offset(ibx, nblocks); - const auto d_offset = block_type::get_d_offset(nrows, ncols, ibx); const int iby = i * block_type::block_to_q8_1_ratio(); #pragma unroll for (int elem = 0; elem < block_elements_per_subgroup; elem += WARP_SIZE) { const int iqs = elem + block_traits::vdr_mmvq * (sg.get_local_linear_id() % block_elements_per_subgroup); + if constexpr (rows_per_sg > 1) { + typename reorder_vec_dot_q_sycl::weights wx[rows_per_sg]; + [[maybe_unused]] typename reorder_vec_dot_q_sycl::weights wg[rows_per_sg]; #pragma unroll - for (int j = 0; j < ncols_dst; ++j) { - const char * vy_j = (const char *) vy + j * stride_col_y_bytes; - const int8_t * q8_1_quant_ptr = (const int8_t *) vy_j + iby * QK8_1; - const sycl::half2 * q8_1_ds_ptr = (const sycl::half2 *) (vy_j + ncols + iby * sizeof(sycl::half2)); + for (int r = 0; r < rows_per_sg; ++r) { + const int row = sycl::min(row0 + r, nrows - 1); + const int ibx = row * blocks_per_row + i; + const auto bx_offset = block_type::get_block_offset(ibx, nblocks); + const auto d_offset = block_type::get_d_offset(nrows, ncols, ibx); + wx[r] = reorder_vec_dot_q_sycl::load(vx, bx_offset, d_offset, iqs); + if constexpr (has_fusion) { + wg[r] = reorder_vec_dot_q_sycl::load(vgate, bx_offset, d_offset, iqs); + } + } +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + const char * vy_j = (const char *) vy + j * stride_col_y_bytes; + const int8_t * q8_1_quant_ptr = (const int8_t *) vy_j + iby * QK8_1; + const sycl::half2 * q8_1_ds_ptr = + (const sycl::half2 *) (vy_j + ncols + iby * sizeof(sycl::half2)); + const auto a = reorder_vec_dot_q_sycl::load_activations(q8_1_quant_ptr, q8_1_ds_ptr, iqs); +#pragma unroll + for (int r = 0; r < rows_per_sg; ++r) { + partial_sum[j][r] += reorder_vec_dot_q_sycl::apply(wx[r], a); + if constexpr (has_fusion) { + partial_gate[j][r] += reorder_vec_dot_q_sycl::apply(wg[r], a); + } + } + } + } else if constexpr (reorder_vec_dot_shared_weights::value) { + const int ibx = row0 * blocks_per_row + i; + const auto bx_offset = block_type::get_block_offset(ibx, nblocks); + const auto d_offset = block_type::get_d_offset(nrows, ncols, ibx); + const auto wx = reorder_vec_dot_q_sycl::load(vx, bx_offset, d_offset, iqs); + if constexpr (has_fusion) { + const auto wg = reorder_vec_dot_q_sycl::load(vgate, bx_offset, d_offset, iqs); - partial_sum[j] += reorder_vec_dot_q_sycl()(vx, bx_offset, d_offset, q8_1_quant_ptr, q8_1_ds_ptr, iqs); +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + const char * vy_j = (const char *) vy + j * stride_col_y_bytes; + const int8_t * q8_1_quant_ptr = (const int8_t *) vy_j + iby * QK8_1; + const sycl::half2 * q8_1_ds_ptr = + (const sycl::half2 *) (vy_j + ncols + iby * sizeof(sycl::half2)); - if constexpr (has_fusion) { - partial_gate[j] += - reorder_vec_dot_q_sycl()(vgate, bx_offset, d_offset, q8_1_quant_ptr, q8_1_ds_ptr, iqs); + // up and gate share the activation, so load it once and apply it twice + const auto a = reorder_vec_dot_q_sycl::load_activations(q8_1_quant_ptr, q8_1_ds_ptr, iqs); + + partial_sum[j][0] += reorder_vec_dot_q_sycl::apply(wx, a); + partial_gate[j][0] += reorder_vec_dot_q_sycl::apply(wg, a); + } + } else { +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + const char * vy_j = (const char *) vy + j * stride_col_y_bytes; + const int8_t * q8_1_quant_ptr = (const int8_t *) vy_j + iby * QK8_1; + const sycl::half2 * q8_1_ds_ptr = + (const sycl::half2 *) (vy_j + ncols + iby * sizeof(sycl::half2)); + + partial_sum[j][0] += reorder_vec_dot_q_sycl::dot(wx, q8_1_quant_ptr, q8_1_ds_ptr, iqs); + } + } + } else { + const int ibx = row0 * blocks_per_row + i; + const auto bx_offset = block_type::get_block_offset(ibx, nblocks); + const auto d_offset = block_type::get_d_offset(nrows, ncols, ibx); +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + const char * vy_j = (const char *) vy + j * stride_col_y_bytes; + const int8_t * q8_1_quant_ptr = (const int8_t *) vy_j + iby * QK8_1; + const sycl::half2 * q8_1_ds_ptr = + (const sycl::half2 *) (vy_j + ncols + iby * sizeof(sycl::half2)); + + partial_sum[j][0] += + reorder_vec_dot_q_sycl()(vx, bx_offset, d_offset, q8_1_quant_ptr, q8_1_ds_ptr, iqs); + + if constexpr (has_fusion) { + partial_gate[j][0] += + reorder_vec_dot_q_sycl()(vgate, bx_offset, d_offset, q8_1_quant_ptr, q8_1_ds_ptr, iqs); + } } } } @@ -122,17 +205,20 @@ static void mul_mat_vec_q_reorder_ncols(const void * __restrict__ vx, const void #pragma unroll for (int j = 0; j < ncols_dst; ++j) { - float sum = sycl::reduce_over_group(nd_item.get_sub_group(), partial_sum[j], std::plus<>()); +#pragma unroll + for (int r = 0; r < rows_per_sg; ++r) { + float sum = sycl::reduce_over_group(nd_item.get_sub_group(), partial_sum[j][r], std::plus<>()); - if constexpr (has_fusion) { - const float gate = sycl::reduce_over_group(nd_item.get_sub_group(), partial_gate[j], std::plus<>()); + if constexpr (has_fusion) { + const float gate = sycl::reduce_over_group(nd_item.get_sub_group(), partial_gate[j][r], std::plus<>()); - // uniform across the launch; the launcher only instantiates SWIGLU and GEGLU - sum *= glu_op == GGML_GLU_OP_SWIGLU ? op_silu(gate) : op_gelu(gate); - } + // uniform across the launch; the launcher only instantiates SWIGLU and GEGLU + sum *= glu_op == GGML_GLU_OP_SWIGLU ? op_silu(gate) : op_gelu(gate); + } - if (sg.leader()) { - dst[j * stride_col_dst + row] = sum; + if (sg.leader() && row0 + r < nrows) { + dst[j * stride_col_dst + row0 + r] = sum; + } } } } @@ -1401,6 +1487,65 @@ static void mul_mat_vec_q2_K_q8_1_sycl_switch_ncols( } } +static void reorder_mul_mat_vec_q2_k_q8_1_sycl(const void * vx, const void * vy, float * dst, const int ncols, + const int nrows, dpct::queue_ptr stream) { + GGML_ASSERT(ncols % QK_K == 0); + + // Round up to a whole number of subgroup-sized workgroups; out-of-range rows are skipped inside the kernel. + constexpr size_t num_subgroups = WARP_SIZE; + const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups); + const sycl::range<3> block_nums(1, 1, block_num_y); + const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE); + + stream->submit([&](sycl::handler & cgh) { + cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), + [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + mul_mat_vec_q_reorder>(vx, vy, dst, ncols, nrows, + nd_item); + }); + }); +} + +template +static void reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols( + const void * vx, const void * vy, float * dst, + const int ncols, const int nrows, + const int stride_col_y_bytes, const int stride_col_dst, + dpct::queue_ptr stream) { + GGML_ASSERT(ncols % QK_K == 0); + constexpr size_t num_subgroups = WARP_SIZE; + const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups); + const sycl::range<3> block_nums(1, 1, block_num_y); + const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE); + + stream->submit([&](sycl::handler & cgh) { + cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), + [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + mul_mat_vec_q_reorder_ncols, ncols_dst>( + vx, /*vgate=*/ nullptr, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, + /*glu_op=*/ GGML_GLU_OP_SWIGLU, nd_item); + }); + }); +} + +static void reorder_mul_mat_vec_q2_k_q8_1_sycl_switch_ncols( + const void * vx, const void * vy, float * dst, + const int ncols, const int nrows, const int ncols_dst, + const int stride_col_y_bytes, const int stride_col_dst, + dpct::queue_ptr stream) { + switch (ncols_dst) { + case 1: reorder_mul_mat_vec_q2_k_q8_1_sycl(vx, vy, dst, ncols, nrows, stream); break; + case 2: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<2>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 3: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<3>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 4: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<4>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 5: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<5>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 6: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<6>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 7: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<7>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 8: reorder_mul_mat_vec_q2_k_q8_1_sycl_ncols<8>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + default: GGML_ABORT("unsupported ncols_dst=%d for Q2_K reorder multi-col MMVQ", ncols_dst); + } +} + static void mul_mat_vec_q3_K_q8_1_sycl(const void *vx, const void *vy, float *dst, const int ncols, const int nrows, @@ -1612,8 +1757,8 @@ static void reorder_mul_mat_vec_q4_k_q8_1_sycl(const void * vx, const void * vy, }); } -template -static void reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols( +template +static void reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols_impl( const void * vx, const void * vy, float * dst, const int ncols, const int nrows, const int stride_col_y_bytes, const int stride_col_dst, @@ -1621,20 +1766,31 @@ static void reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols( GGML_ASSERT(ncols % QK_K == 0); constexpr size_t num_subgroups = WARP_SIZE; - const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups); + const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups * rows_per_sg); const sycl::range<3> block_nums(1, 1, block_num_y); const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE); stream->submit([&](sycl::handler & cgh) { cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { - mul_mat_vec_q_reorder_ncols, ncols_dst>( + mul_mat_vec_q_reorder_ncols, ncols_dst, + /*has_fusion=*/ false, rows_per_sg>( vx, /*vgate=*/ nullptr, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, /*glu_op=*/ GGML_GLU_OP_SWIGLU, nd_item); }); }); } +template +static void reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols( + const void * vx, const void * vy, float * dst, + const int ncols, const int nrows, + const int stride_col_y_bytes, const int stride_col_dst, + dpct::queue_ptr stream) { + constexpr int rows_per_sg = ncols_dst >= 3 && ncols_dst <= 4 ? 2 : 1; + reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols_impl(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); +} + static void reorder_mul_mat_vec_q4_k_q8_1_sycl_switch_ncols( const void * vx, const void * vy, float * dst, const int ncols, const int nrows, const int ncols_dst, @@ -1642,7 +1798,13 @@ static void reorder_mul_mat_vec_q4_k_q8_1_sycl_switch_ncols( dpct::queue_ptr stream) { switch (ncols_dst) { case 1: reorder_mul_mat_vec_q4_k_q8_1_sycl(vx, vy, dst, ncols, nrows, stream); break; - case 2: reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols<2>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; + case 2: + if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) { + reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols_impl<2, 2>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); + } else { + reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols_impl<2, 1>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); + } + break; case 3: reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols<3>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; case 4: reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols<4>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; case 5: reorder_mul_mat_vec_q4_k_q8_1_sycl_ncols<5>(vx, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, stream); break; @@ -2297,7 +2459,21 @@ void ggml_sycl_op_mul_mat_vec_q(ggml_backend_sycl_context & ctx, const ggml_tens } break; case GGML_TYPE_Q2_K: - if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) { + if ((ggml_tensor_extra_gpu *) dst->src[0]->extra && + ((ggml_tensor_extra_gpu *) dst->src[0]->extra)->optimized_feature.reorder) { + if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) { + const int stride_col_y_bytes = src1_padded_col_size * q8_1_ts / q8_1_bs; + const int stride_col_dst = dst->ne[0]; + GGML_SYCL_DEBUG("Calling reorder_mul_mat_vec_q2_k_q8_1_sycl_switch_ncols ncols=%d\n", (int)src1_ncols); + reorder_mul_mat_vec_q2_k_q8_1_sycl_switch_ncols( + src0_dd_i, src1_ddq_i, dst_dd_i, ne00, row_diff, + src1_ncols, stride_col_y_bytes, stride_col_dst, stream); + return; + } else { + GGML_SYCL_DEBUG("Calling reorder_mul_mat_vec_q2_k_q8_1_sycl\n"); + reorder_mul_mat_vec_q2_k_q8_1_sycl(src0_dd_i, src1_ddq_i_bs, dst_dd_i_bs, ne00, row_diff, stream); + } + } else if (i == 0 && src1_ncols > 1 && src1_ncols <= 8) { const int stride_col_y = src1_padded_col_size / QK8_1; const int stride_col_dst = dst->ne[0]; GGML_SYCL_DEBUG("Calling mul_mat_vec_q2_K_q8_1_sycl_switch_ncols ncols=%d\n", (int)src1_ncols); @@ -2306,6 +2482,7 @@ void ggml_sycl_op_mul_mat_vec_q(ggml_backend_sycl_context & ctx, const ggml_tens src1_ncols, stride_col_y, stride_col_dst, stream); return; } else if (i == 0 || src1_ncols == 1) { + GGML_SYCL_DEBUG("Calling mul_mat_vec_q2_K_q8_1_sycl\n"); mul_mat_vec_q2_K_q8_1_sycl(src0_dd_i, src1_ddq_i_bs, dst_dd_i_bs, ne00, row_diff, stream); } break; @@ -2485,7 +2662,7 @@ void ggml_sycl_op_mul_mat_vec_q(ggml_backend_sycl_context & ctx, const ggml_tens } break; default: - GGML_ABORT("fatal error: unsupport data type=%s\n", ggml_type_name(src0->type)); + GGML_ABORT("fatal error: unsupport src0 data type %s\n", ggml_type_name(src0->type)); } } GGML_UNUSED(src1); @@ -2494,6 +2671,34 @@ void ggml_sycl_op_mul_mat_vec_q(ggml_backend_sycl_context & ctx, const ggml_tens GGML_UNUSED(ctx); } +// vec_dot_q_sycl_t adapters for the IQ vec_dots that take their codebook tables as extra +// arguments: bind the constant tables here (as vec_dot_iq2_s_q8_1 / vec_dot_iq1_m_q8_1 already do +// internally) so they can be used as template arguments of mul_mat_vec_q_moe. +static __dpct_inline__ float vec_dot_iq2_xxs_q8_1_moe(const void * __restrict__ vbq, + const block_q8_1 * __restrict__ bq8_1, const int & iqs) { + return vec_dot_iq2_xxs_q8_1(vbq, bq8_1, iqs, iq2xxs_grid, ksigns_iq2xs, kmask_iq2xs); +} + +static __dpct_inline__ float vec_dot_iq2_xs_q8_1_moe(const void * __restrict__ vbq, + const block_q8_1 * __restrict__ bq8_1, const int & iqs) { + return vec_dot_iq2_xs_q8_1(vbq, bq8_1, iqs, iq2xs_grid, ksigns64); +} + +static __dpct_inline__ float vec_dot_iq3_xxs_q8_1_moe(const void * __restrict__ vbq, + const block_q8_1 * __restrict__ bq8_1, const int & iqs) { + return vec_dot_iq3_xxs_q8_1(vbq, bq8_1, iqs, iq3xxs_grid, ksigns64); +} + +static __dpct_inline__ float vec_dot_iq3_s_q8_1_moe(const void * __restrict__ vbq, + const block_q8_1 * __restrict__ bq8_1, const int & iqs) { + return vec_dot_iq3_s_q8_1(vbq, bq8_1, iqs, iq3s_grid); +} + +static __dpct_inline__ float vec_dot_iq1_s_q8_1_moe(const void * __restrict__ vbq, + const block_q8_1 * __restrict__ bq8_1, const int & iqs) { + return vec_dot_iq1_s_q8_1(vbq, bq8_1, iqs, iq1s_grid_gpu); +} + // src1_row_stride: 0 for shared src1 (gate/up proj), else per-expert stride (down proj). template static void mul_mat_vec_q_moe( @@ -2645,6 +2850,51 @@ bool ggml_sycl_mul_mat_vec_q_id( vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, expert_weight_stride, dst_row_stride, src1_row_stride, stream); return true; + case GGML_TYPE_IQ2_XXS: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ2_XS: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ2_S: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ3_XXS: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ3_S: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ1_S: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ1_M: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ4_NL: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; + case GGML_TYPE_IQ4_XS: + launch_mul_mat_vec_q_moe( + vx_base, vy, ids_dev, dst_base, ncols, nrows, n_experts_used, + expert_weight_stride, dst_row_stride, src1_row_stride, stream); + return true; default: return false; } @@ -2765,8 +3015,8 @@ bool ggml_sycl_mul_mat_vec_q_id_reorder( } } -template -static void launch_mul_mat_vec_q_reorder_glu(const void * vx, const void * vgate, const void * vy, float * dst, +template +static void launch_mul_mat_vec_q_reorder_glu_impl(const void * vx, const void * vgate, const void * vy, float * dst, const int ncols, const int nrows, const int stride_col_y_bytes, const int stride_col_dst, const ggml_glu_op glu_op, dpct::queue_ptr stream) { @@ -2774,20 +3024,218 @@ static void launch_mul_mat_vec_q_reorder_glu(const void * vx, const void * vgate constexpr size_t num_subgroups = WARP_SIZE; - const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups); + const int block_num_y = ceil_div(nrows, GGML_SYCL_MMV_Y * (int) num_subgroups * rows_per_sg); const sycl::range<3> block_nums(1, 1, block_num_y); const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, num_subgroups * WARP_SIZE); stream->submit([&](sycl::handler & cgh) { cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { - mul_mat_vec_q_reorder_ncols( + mul_mat_vec_q_reorder_ncols( vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, nd_item); }); }); } +template +static void launch_mul_mat_vec_q_reorder_glu(const void * vx, const void * vgate, const void * vy, float * dst, + const int ncols, const int nrows, const int stride_col_y_bytes, + const int stride_col_dst, const ggml_glu_op glu_op, + dpct::queue_ptr stream) { + constexpr int rows_per_sg = + reorder_vec_dot_shared_activations::value && ncols_dst >= 3 && ncols_dst <= 4 + ? 2 + : 1; + launch_mul_mat_vec_q_reorder_glu_impl(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream); +} + +// --------------------------------------------------------------------------- +// Fused dense-FFN GEMV + GLU over the standard (non-reorder) weight layout. +// +// Unlike the reorder variant below, the two weights may carry different block +// types (e.g. an unsloth UD mix with an iq4_xs gate and a q5_K up), as long as +// both quantize in QK_K-sized super-blocks so that one q8_1 activation +// quantization serves both dots. Per-operand accumulation order matches +// mul_mat_vec_q exactly, so results are bit-identical to running the three +// nodes separately. +// --------------------------------------------------------------------------- +template +static void mul_mat_vec_q_glu(const void * __restrict__ vxg, const void * __restrict__ vxu, + const void * __restrict__ vy, float * __restrict__ dst, const int ncols, + const int nrows, const int stride_col_y, const int stride_col_dst, + const ggml_glu_op glu_op, const sycl::nd_item<3> & item_ct1) { + static_assert(QK_K % QK8_1 == 0); + const int row = item_ct1.get_group(2) * item_ct1.get_local_range(1) + item_ct1.get_local_id(1); + if (row >= nrows) { + return; + } + const int blocks_per_row = ncols / QK_K; + constexpr int blocks_per_warp_g = (vdr_g * WARP_SIZE + qi_g - 1) / qi_g; + constexpr int blocks_per_warp_u = (vdr_u * WARP_SIZE + qi_u - 1) / qi_u; + // one partial sum per output column, per operand + float tmpg[ncols_dst] = {0.0f}; + float tmpu[ncols_dst] = {0.0f}; + const block_g_t * xg = (const block_g_t *) vxg; + const block_u_t * xu = (const block_u_t *) vxu; + const block_q8_1 * y = (const block_q8_1 *) vy; + for (int i = item_ct1.get_local_id(2) / (qi_g / vdr_g); i < blocks_per_row; i += blocks_per_warp_g) { + const int ibx = row * blocks_per_row + i; + const int iby = i * (QK_K / QK8_1); + for (size_t elem = 0; elem < qi_g / vdr_g; elem += WARP_SIZE) { + const int iqs = elem + vdr_g * (item_ct1.get_local_id(2) % (qi_g / vdr_g)); +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + tmpg[j] += vec_dot_g(&xg[ibx], &y[j * stride_col_y + iby], iqs); + } + } + } + for (int i = item_ct1.get_local_id(2) / (qi_u / vdr_u); i < blocks_per_row; i += blocks_per_warp_u) { + const int ibx = row * blocks_per_row + i; + const int iby = i * (QK_K / QK8_1); + for (size_t elem = 0; elem < qi_u / vdr_u; elem += WARP_SIZE) { + const int iqs = elem + vdr_u * (item_ct1.get_local_id(2) % (qi_u / vdr_u)); +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + tmpu[j] += vec_dot_u(&xu[ibx], &y[j * stride_col_y + iby], iqs); + } + } + } + // sum up partial sums and write back the activated product +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { +#pragma unroll + for (int mask = WARP_SIZE / 2; mask > 0; mask >>= 1) { + tmpg[j] += dpct::permute_sub_group_by_xor(item_ct1.get_sub_group(), tmpg[j], mask); + tmpu[j] += dpct::permute_sub_group_by_xor(item_ct1.get_sub_group(), tmpu[j], mask); + } + } + if (item_ct1.get_local_id(2) == 0) { +#pragma unroll + for (int j = 0; j < ncols_dst; ++j) { + // uniform across the launch; the dispatcher only accepts SWIGLU and GEGLU + const float gate = glu_op == GGML_GLU_OP_SWIGLU ? op_silu(tmpg[j]) : op_gelu(tmpg[j]); + dst[j * stride_col_dst + row] = gate * tmpu[j]; + } + } +} + +template +static void launch_mul_mat_vec_q_glu(const void * vxg, const void * vxu, const void * vy, float * dst, + const int ncols, const int nrows, const int stride_col_y, + const int stride_col_dst, const ggml_glu_op glu_op, + dpct::queue_ptr stream) { + GGML_ASSERT(ncols % QK_K == 0); + const int block_num_y = (nrows + GGML_SYCL_MMV_Y - 1) / GGML_SYCL_MMV_Y; + const sycl::range<3> block_nums(1, 1, block_num_y); + const sycl::range<3> block_dims(1, GGML_SYCL_MMV_Y, WARP_SIZE); + stream->submit([&](sycl::handler & cgh) { + cgh.parallel_for(sycl::nd_range<3>(block_nums * block_dims, block_dims), + [=](sycl::nd_item<3> nd_item) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + mul_mat_vec_q_glu( + vxg, vxu, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, + glu_op, nd_item); + }); + }); +} + +// Dispatch the plain-layout fused GLU GEMV over the activation batch: ncols_dst +// selects the kernel's per-column template parameter. Returns false when the +// batch exceeds the instantiated range; the caller falls back to unfused nodes. +template +static bool dispatch_mul_mat_vec_q_glu_plain(const void * vgate, const void * vup, const void * vy, + float * dst, const int ncols, const int nrows, + const int stride_col_y, const int stride_col_dst, + const ggml_glu_op glu_op, dpct::queue_ptr stream, + const int ncols_dst) { + switch (ncols_dst) { + case 1: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 2: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 3: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 4: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 5: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 6: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 7: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + case 8: + launch_mul_mat_vec_q_glu( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream); + return true; + default: + return false; + } +} + +// Fused dense-FFN GEMV + GLU for weight pairs the reorder kernel does not cover. +// vgate/vup must be in the standard block layout; vy must be quantized with plain +// quantize_q8_1 (padded rows). stride_col_y is in block_q8_1 units. +// Returns false if the type pair or batch is unhandled; caller should fall back. +bool ggml_sycl_mul_mat_vec_q_glu_plain(enum ggml_type gate_type, enum ggml_type up_type, + enum ggml_glu_op glu_op, const void * vgate, const void * vup, + const void * vy, float * dst, int ncols, int nrows, int ncols_dst, + int stride_col_y, int stride_col_dst, dpct::queue_ptr stream) { + if (glu_op != GGML_GLU_OP_SWIGLU && glu_op != GGML_GLU_OP_GEGLU) { + return false; + } + if (ncols % QK_K != 0) { + return false; + } + if (gate_type == GGML_TYPE_Q5_K && up_type == GGML_TYPE_Q5_K) { + return dispatch_mul_mat_vec_q_glu_plain( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream, ncols_dst); + } + if (gate_type == GGML_TYPE_IQ4_XS && up_type == GGML_TYPE_IQ4_XS) { + return dispatch_mul_mat_vec_q_glu_plain( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream, ncols_dst); + } + if (gate_type == GGML_TYPE_IQ4_XS && up_type == GGML_TYPE_Q5_K) { + return dispatch_mul_mat_vec_q_glu_plain( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream, ncols_dst); + } + if (gate_type == GGML_TYPE_Q5_K && up_type == GGML_TYPE_IQ4_XS) { + return dispatch_mul_mat_vec_q_glu_plain( + vgate, vup, vy, dst, ncols, nrows, stride_col_y, stride_col_dst, glu_op, stream, ncols_dst); + } + return false; +} + bool ggml_sycl_mul_mat_vec_q_glu_reorder(enum ggml_type src0_type, enum ggml_glu_op glu_op, const void * vx, const void * vgate, const void * vy, float * dst, int ncols, int nrows, int ncols_dst, int stride_col_y_bytes, int stride_col_dst, @@ -2807,8 +3255,11 @@ bool ggml_sycl_mul_mat_vec_q_glu_reorder(enum ggml_type src0_type, enum ggml_glu stride_col_dst, glu_op, stream); return true; case 2: - launch_mul_mat_vec_q_reorder_glu(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, - stride_col_dst, glu_op, stream); + if (nrows >= Q4_K_MMVQ_ROW_PAIR_MIN_NROWS) { + launch_mul_mat_vec_q_reorder_glu_impl(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream); + } else { + launch_mul_mat_vec_q_reorder_glu_impl(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, stride_col_dst, glu_op, stream); + } return true; case 3: launch_mul_mat_vec_q_reorder_glu(vx, vgate, vy, dst, ncols, nrows, stride_col_y_bytes, diff --git a/ggml/src/ggml-sycl/mmvq.hpp b/ggml/src/ggml-sycl/mmvq.hpp index 9d2f5645..7fb9cf6f 100644 --- a/ggml/src/ggml-sycl/mmvq.hpp +++ b/ggml/src/ggml-sycl/mmvq.hpp @@ -73,4 +73,24 @@ bool ggml_sycl_mul_mat_vec_q_glu_reorder( int stride_col_dst, // floats between output columns in dst dpct::queue_ptr stream); + +// Fused dense-FFN GEMV + GLU over the standard (non-reorder) layout; the gate and up +// weights may carry different block types (q5_K / iq4_xs, mixed included). +// vy: src1 quantized with plain quantize_q8_1 (padded rows). stride_col_y is in +// block_q8_1 units. Returns false if the pair or batch is unhandled; caller falls back. +bool ggml_sycl_mul_mat_vec_q_glu_plain( + enum ggml_type gate_type, + enum ggml_type up_type, + enum ggml_glu_op glu_op, + const void * vgate, + const void * vup, + const void * vy, + float * dst, + int ncols, // K, shared by both weights + int nrows, // output rows, i.e. weight ne[1] + int ncols_dst, // activation columns, 1..MMVQ_MAX_BATCH_SIZE + int stride_col_y, // block_q8_1 units between activation columns + int stride_col_dst, // floats between output columns in dst + dpct::queue_ptr stream); + #endif // GGML_SYCL_MMVQ_HPP diff --git a/ggml/src/ggml-sycl/norm.cpp b/ggml/src/ggml-sycl/norm.cpp index 682a9f51..36576e9c 100644 --- a/ggml/src/ggml-sycl/norm.cpp +++ b/ggml/src/ggml-sycl/norm.cpp @@ -7,9 +7,6 @@ static void norm_f32(const float* x, float* dst, const int ncols, const int64_t dst_stride_col, const int64_t dst_stride_row, const int64_t dst_stride_channel, const int64_t dst_stride_sample, const float eps, const sycl::nd_item<3>& item_ct1, sycl::float2* s_sum, int block_size) { - const int nrows = item_ct1.get_group_range(2); - const int nchannels = item_ct1.get_group_range(1); - const int nthreads = item_ct1.get_local_range(2); const int sample = item_ct1.get_group(0); const int channel = item_ct1.get_group(1); @@ -147,16 +144,18 @@ static void group_norm_f32(const float* x, float* dst, const int group_size, con } } -template +template static void rms_norm_f32(const float* x, float* dst, const int ncols, const int64_t src_stride_col, const int64_t src_stride_row, const int64_t src_stride_channel, const int64_t src_stride_sample, const int64_t dst_stride_col, const int64_t dst_stride_row, const int64_t dst_stride_channel, const int64_t dst_stride_sample, const float eps, const sycl::nd_item<3>& item_ct1, float* s_sum, int block_size, const float* mul = nullptr, const int64_t mul_stride_row = 0, const int64_t mul_stride_channel = 0, - const int64_t mul_stride_sample = 0, const int mul_nrows = 0, const int mul_nchannels = 0, const int mul_nsamples = 0) { + const int64_t mul_stride_sample = 0, const int mul_nrows = 0, const int mul_nchannels = 0, const int mul_nsamples = 0, + const float* add = nullptr, const int64_t add_stride_row = 0, const int64_t add_stride_channel = 0, + const int64_t add_stride_sample = 0, const int add_nrows = 0, const int add_nchannels = 0, const int add_nsamples = 0, + const float scale_mul = 1.0f) { - const int nrows = item_ct1.get_group_range(2); - const int nchannels = item_ct1.get_group_range(1); + static_assert(!do_add || do_multiply, "fusing add is not supported without multiplying"); const int sample = item_ct1.get_group(0); const int channel = item_ct1.get_group(1); @@ -180,6 +179,13 @@ static void rms_norm_f32(const float* x, float* dst, const int ncols, mul += mul_sample * mul_stride_sample + mul_channel * mul_stride_channel + mul_row * mul_stride_row; } + if constexpr (do_add) { + const int add_row = row % add_nrows; + const int add_channel = channel % add_nchannels; + const int add_sample = sample % add_nsamples; + add += add_sample * add_stride_sample + add_channel * add_stride_channel + add_row * add_stride_row; + } + float tmp = 0.0f; // partial sum for thread in warp for (int col = tid; col < ncols; col += block_size) { @@ -211,10 +217,16 @@ static void rms_norm_f32(const float* x, float* dst, const int ncols, const float scale = sycl::rsqrt(mean + eps); for (int col = tid; col < ncols; col += block_size) { - if constexpr (do_multiply) { + if constexpr (do_multiply && do_add) { + dst[col * dst_stride_col] = scale * x[col * src_stride_col] * mul[col] + add[col]; + } else if constexpr (do_multiply) { dst[col * dst_stride_col] = scale * x[col * src_stride_col] * mul[col]; } else { - dst[col * dst_stride_col] = scale * x[col * src_stride_col]; + // folded epilogue of a fused GGML_OP_SCALE consumer (qwen35 GDN l2 norms); + // the explicit temporary keeps the float evaluation order identical to + // running rms_norm and scale as two separate kernels + const float v = scale * x[col * src_stride_col]; + dst[col * dst_stride_col] = v * scale_mul; } } } @@ -225,8 +237,6 @@ static void l2_norm_f32(const float * x, float * dst, const int ncols, const int64_t src_stride_sample, const int64_t dst_stride_col, const int64_t dst_stride_row, const int64_t dst_stride_channel, const int64_t dst_stride_sample, const float eps, const sycl::nd_item<3>& item_ct1, float* s_sum, const int block_size) { - const int nrows = item_ct1.get_group_range(2); - const int nchannels = item_ct1.get_group_range(1); const int row = item_ct1.get_group(2); const int channel = item_ct1.get_group(1); @@ -389,6 +399,52 @@ static void rms_norm_f32_sycl(const float* x, float* dst, const int ncols, const } } +static void rms_norm_scale_f32_sycl(const float* x, float* dst, const int ncols, const int nrows, + const int nchannels, const int nsamples, + const int64_t src_stride_col, const int64_t src_stride_row, const int64_t src_stride_channel, const int64_t src_stride_sample, + const int64_t dst_stride_col, const int64_t dst_stride_row, const int64_t dst_stride_channel, const int64_t dst_stride_sample, + const float eps, const float scale_mul, queue_ptr stream, int device) { + const sycl::range<3> global_dims(nsamples, nchannels, nrows); + if (ncols < 1024) { + const sycl::range<3> block_dims(1, 1, WARP_SIZE); + stream->submit([&](sycl::handler& cgh) { + cgh.parallel_for( + sycl::nd_range<3>(global_dims * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) + [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + rms_norm_f32(x, dst, ncols, + src_stride_col, src_stride_row, src_stride_channel, src_stride_sample, + dst_stride_col, dst_stride_row, dst_stride_channel, dst_stride_sample, + eps, item_ct1, nullptr, WARP_SIZE, + nullptr, 0, 0, 0, 0, 0, 0, + nullptr, 0, 0, 0, 0, 0, 0, + scale_mul); + }); + }); + } + else { + const int work_group_size = ggml_sycl_info().max_work_group_sizes[device]; + assert(work_group_size % (WARP_SIZE * WARP_SIZE) == 0); + const sycl::range<3> block_dims(1, 1, work_group_size); + stream->submit([&](sycl::handler& cgh) { + sycl::local_accessor s_sum_acc_ct1(sycl::range<1>(work_group_size / WARP_SIZE), + cgh); + cgh.parallel_for( + sycl::nd_range<3>(global_dims * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) + [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + rms_norm_f32(x, dst, ncols, + src_stride_col, src_stride_row, src_stride_channel, src_stride_sample, + dst_stride_col, dst_stride_row, dst_stride_channel, dst_stride_sample, + eps, item_ct1, get_pointer(s_sum_acc_ct1), work_group_size, + nullptr, 0, 0, 0, 0, 0, 0, + nullptr, 0, 0, 0, 0, 0, 0, + scale_mul); + }); + }); + } +} + static void rms_norm_mul_f32_sycl(const float* x, const float* mul, float* dst, const int ncols, const int nrows, const int nchannels, const int nsamples, const int64_t src_stride_col, const int64_t src_stride_row, const int64_t src_stride_channel, const int64_t src_stride_sample, @@ -432,6 +488,53 @@ static void rms_norm_mul_f32_sycl(const float* x, const float* mul, float* dst, } } +static void rms_norm_mul_add_f32_sycl(const float* x, const float* mul, const float* add, float* dst, + const int ncols, const int nrows, const int nchannels, const int nsamples, + const int64_t src_stride_col, const int64_t src_stride_row, const int64_t src_stride_channel, const int64_t src_stride_sample, + const int64_t dst_stride_col, const int64_t dst_stride_row, const int64_t dst_stride_channel, const int64_t dst_stride_sample, + const int64_t mul_stride_row, const int64_t mul_stride_channel, const int64_t mul_stride_sample, + const int mul_nrows, const int mul_nchannels, const int mul_nsamples, + const int64_t add_stride_row, const int64_t add_stride_channel, const int64_t add_stride_sample, + const int add_nrows, const int add_nchannels, const int add_nsamples, + const float eps, queue_ptr stream, int device) { + const sycl::range<3> global_dims(nsamples, nchannels, nrows); + if (ncols < 1024) { + const sycl::range<3> block_dims(1, 1, WARP_SIZE); + stream->submit([&](sycl::handler& cgh) { + cgh.parallel_for( + sycl::nd_range<3>(global_dims * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) + [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + rms_norm_f32(x, dst, ncols, + src_stride_col, src_stride_row, src_stride_channel, src_stride_sample, + dst_stride_col, dst_stride_row, dst_stride_channel, dst_stride_sample, + eps, item_ct1, nullptr, WARP_SIZE, + mul, mul_stride_row, mul_stride_channel, mul_stride_sample, mul_nrows, mul_nchannels, mul_nsamples, + add, add_stride_row, add_stride_channel, add_stride_sample, add_nrows, add_nchannels, add_nsamples); + }); + }); + } + else { + const int work_group_size = ggml_sycl_info().max_work_group_sizes[device]; + assert(work_group_size % (WARP_SIZE * WARP_SIZE) == 0); + const sycl::range<3> block_dims(1, 1, work_group_size); + stream->submit([&](sycl::handler& cgh) { + sycl::local_accessor s_sum_acc_ct1(sycl::range<1>(work_group_size / WARP_SIZE), cgh); + cgh.parallel_for( + sycl::nd_range<3>(global_dims * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) + [[sycl::reqd_sub_group_size(WARP_SIZE)]] { + rms_norm_f32(x, dst, ncols, + src_stride_col, src_stride_row, src_stride_channel, src_stride_sample, + dst_stride_col, dst_stride_row, dst_stride_channel, dst_stride_sample, + eps, item_ct1, get_pointer(s_sum_acc_ct1), work_group_size, + mul, mul_stride_row, mul_stride_channel, mul_stride_sample, mul_nrows, mul_nchannels, mul_nsamples, + add, add_stride_row, add_stride_channel, add_stride_sample, add_nrows, add_nchannels, add_nsamples); + }); + }); + } +} + template static void l2_norm_f32_sycl(const float * x, float * dst, @@ -491,6 +594,62 @@ static void l2_norm_f32_sycl(const float * x, } } +// Batched L2 norm: N independent same-shape F32 tensors in one launch; the tensor +// index is folded into grid dim0 and each row's reduction is identical to the +// single-tensor kernel, so the result is bit-exact. +struct l2_batch_ptrs { + const float * src[GGML_SYCL_L2_BATCH_MAX]; + float * dst[GGML_SYCL_L2_BATCH_MAX]; +}; + +// One stride set shared by the whole batch: the caller only groups tensors whose nb[] +// all match, so per-tensor state stays two pointers. +struct l2_batch_strides { + int ne1, ne2; + int64_t ss0, ss1, ss2, ss3; + int64_t ds0, ds1, ds2, ds3; +}; + +template +static void l2_norm_f32_batch(l2_batch_ptrs p, l2_batch_strides st, const int ncols, const float eps, + const sycl::nd_item<3> & item_ct1) { + const int t = item_ct1.get_group(0); // tensor index + const int r = item_ct1.get_group(2); // flattened row over ne1*ne2*ne3 + const int tid = item_ct1.get_local_id(2); + + const int i1 = r % st.ne1; + const int i2 = (r / st.ne1) % st.ne2; + const int i3 = r / (st.ne1 * st.ne2); + + const float * x = p.src[t] + i3 * st.ss3 + i2 * st.ss2 + i1 * st.ss1; + float * dst = p.dst[t] + i3 * st.ds3 + i2 * st.ds2 + i1 * st.ds1; + + float tmp = 0.0f; + for (int col = tid; col < ncols; col += warp_size) { + const float xi = x[col * st.ss0]; + tmp += xi * xi; + } + tmp = block_reduce(tmp, (float *) nullptr, warp_size); + const float scale = sycl::rsqrt(sycl::fmax(tmp, eps * eps)); + for (int col = tid; col < ncols; col += warp_size) { + dst[col * st.ds0] = scale * x[col * st.ss0]; + } +} + +template +static void l2_norm_f32_batch_sycl(l2_batch_ptrs p, l2_batch_strides st, const int n_tensors, + const int ncols, const int nrows_total, const float eps, + queue_ptr stream) { + const dpct::dim3 blocks_num(nrows_total, 1, n_tensors); + const dpct::dim3 block_dims(warp_size, 1, 1); + stream->submit([&](sycl::handler & cgh) { + cgh.parallel_for(sycl::nd_range<3>(blocks_num * block_dims, block_dims), + [=](sycl::nd_item<3> item_ct1) [[sycl::reqd_sub_group_size(warp_size)]] { + l2_norm_f32_batch(p, st, ncols, eps, item_ct1); + }); + }); +} + void ggml_sycl_op_norm(ggml_backend_sycl_context& ctx, ggml_tensor* dst) { const ggml_tensor * src0 = dst->src[0]; @@ -574,6 +733,46 @@ void ggml_sycl_op_rms_norm(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { ss0, ss1, ss2, ss3, ds0, ds1, ds2, ds3, eps, main_stream, ctx.device); } +// Fused rms_norm + scale (the qwen35 GDN l2-norm pair build_gdn_l2_norm emits): +// the scale factor is a host scalar in the GGML_OP_SCALE node's op_params, so +// unlike the mul variants there is no second device tensor to wire up. +void ggml_sycl_op_rms_norm_scale_fused(ggml_backend_sycl_context & ctx, ggml_tensor * dst, + ggml_tensor * scale_tensor) { + const ggml_tensor * src0 = dst->src[0]; + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(scale_tensor->type == GGML_TYPE_F32); + + dpct::queue_ptr main_stream = ctx.stream(); + SYCL_CHECK(ggml_sycl_set_device(ctx.device)); + + const float * src0_dd = static_cast(src0->data); + float * dst_dd = static_cast(scale_tensor->data); + + float eps; + memcpy(&eps, dst->op_params, sizeof(float)); + float scale_mul; + memcpy(&scale_mul, scale_tensor->op_params, sizeof(float)); + GGML_ASSERT(scale_mul >= 0.0f); + + GGML_TENSOR_UNARY_OP_LOCALS + const size_t ts0 = ggml_type_size(src0->type); + const size_t tdst = ggml_type_size(scale_tensor->type); + GGML_ASSERT(nb00 % ts0 == 0 && nb01 % ts0 == 0 && nb02 % ts0 == 0 && nb03 % ts0 == 0); + GGML_ASSERT(scale_tensor->nb[0] % tdst == 0 && scale_tensor->nb[1] % tdst == 0 && + scale_tensor->nb[2] % tdst == 0 && scale_tensor->nb[3] % tdst == 0); + const int64_t ss0 = nb00 / ts0; + const int64_t ss1 = nb01 / ts0; + const int64_t ss2 = nb02 / ts0; + const int64_t ss3 = nb03 / ts0; + const int64_t ds0 = scale_tensor->nb[0] / tdst; + const int64_t ds1 = scale_tensor->nb[1] / tdst; + const int64_t ds2 = scale_tensor->nb[2] / tdst; + const int64_t ds3 = scale_tensor->nb[3] / tdst; + rms_norm_scale_f32_sycl(src0_dd, dst_dd, ne00, ne01, ne02, ne03, + ss0, ss1, ss2, ss3, ds0, ds1, ds2, ds3, eps, scale_mul, main_stream, ctx.device); +} + void ggml_sycl_op_rms_norm_fused(ggml_backend_sycl_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor) { const ggml_tensor * rms_norm_src = dst->src[0]; float eps = 0.0f; @@ -634,6 +833,91 @@ void ggml_sycl_op_rms_norm_fused(ggml_backend_sycl_context & ctx, ggml_tensor * mul_s01, mul_s02, mul_s03, mul_nrows, mul_nchannels, mul_nsamples, eps, main_stream, ctx.device); } +void ggml_sycl_op_rms_norm_fused_add(ggml_backend_sycl_context & ctx, ggml_tensor * dst, + ggml_tensor * mul_tensor, ggml_tensor * add_tensor) { + const ggml_tensor * rms_norm_src = dst->src[0]; + float eps = 0.0f; + memcpy(&eps, dst->op_params, sizeof(float)); + + const float * src0_dd = static_cast(rms_norm_src->data); + const float * mul_dd = nullptr; + const ggml_tensor * mul_src = nullptr; + if (mul_tensor->src[0] == dst) { + mul_dd = static_cast(mul_tensor->src[1]->data); + mul_src = mul_tensor->src[1]; + } else if (mul_tensor->src[1] == dst) { + mul_dd = static_cast(mul_tensor->src[0]->data); + mul_src = mul_tensor->src[0]; + } else { + GGML_ASSERT(false); + } + + const float * add_dd = nullptr; + const ggml_tensor * add_src = nullptr; + if (add_tensor->src[0] == mul_tensor) { + add_dd = static_cast(add_tensor->src[1]->data); + add_src = add_tensor->src[1]; + } else if (add_tensor->src[1] == mul_tensor) { + add_dd = static_cast(add_tensor->src[0]->data); + add_src = add_tensor->src[0]; + } else { + GGML_ASSERT(false); + } + + float * dst_dd = static_cast(add_tensor->data); + + dpct::queue_ptr main_stream = ctx.stream(); + SYCL_CHECK(ggml_sycl_set_device(ctx.device)); + + GGML_ASSERT(rms_norm_src->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(mul_tensor->type == GGML_TYPE_F32); + GGML_ASSERT(add_tensor->type == GGML_TYPE_F32); + GGML_ASSERT(eps >= 0.0f); + + const int64_t ne00 = rms_norm_src->ne[0]; + const int64_t ne01 = rms_norm_src->ne[1]; + const int64_t ne02 = rms_norm_src->ne[2]; + const int64_t ne03 = rms_norm_src->ne[3]; + + const size_t ts0 = ggml_type_size(rms_norm_src->type); + GGML_ASSERT(rms_norm_src->nb[0] == ts0); + const int64_t s00 = rms_norm_src->nb[0] / ts0; + const int64_t s01 = rms_norm_src->nb[1] / ts0; + const int64_t s02 = rms_norm_src->nb[2] / ts0; + const int64_t s03 = rms_norm_src->nb[3] / ts0; + + const size_t tdst = ggml_type_size(add_tensor->type); + GGML_ASSERT(add_tensor->nb[0] == tdst); + const int64_t d00 = add_tensor->nb[0] / tdst; + const int64_t d01 = add_tensor->nb[1] / tdst; + const int64_t d02 = add_tensor->nb[2] / tdst; + const int64_t d03 = add_tensor->nb[3] / tdst; + + const size_t ts_mul = ggml_type_size(mul_src->type); + GGML_ASSERT(mul_src->nb[0] == ts_mul); + const int64_t mul_s01 = mul_src->nb[1] / ts_mul; + const int64_t mul_s02 = mul_src->nb[2] / ts_mul; + const int64_t mul_s03 = mul_src->nb[3] / ts_mul; + const int mul_nrows = mul_src->ne[1]; + const int mul_nchannels = mul_src->ne[2]; + const int mul_nsamples = mul_src->ne[3]; + + const size_t ts_add = ggml_type_size(add_src->type); + GGML_ASSERT(add_src->nb[0] == ts_add); + const int64_t add_s01 = add_src->nb[1] / ts_add; + const int64_t add_s02 = add_src->nb[2] / ts_add; + const int64_t add_s03 = add_src->nb[3] / ts_add; + const int add_nrows = add_src->ne[1]; + const int add_nchannels = add_src->ne[2]; + const int add_nsamples = add_src->ne[3]; + + rms_norm_mul_add_f32_sycl(src0_dd, mul_dd, add_dd, dst_dd, ne00, ne01, ne02, ne03, + s00, s01, s02, s03, d00, d01, d02, d03, + mul_s01, mul_s02, mul_s03, mul_nrows, mul_nchannels, mul_nsamples, + add_s01, add_s02, add_s03, add_nrows, add_nchannels, add_nsamples, eps, main_stream, ctx.device); +} + void ggml_sycl_op_rms_norm_back(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/2); @@ -824,3 +1108,30 @@ void ggml_sycl_op_l2_norm(ggml_backend_sycl_context& ctx, ggml_tensor* dst) { l2_norm_f32_sycl(src0_d, dst_d, ne00, ne01, ne02, ne03, ss0, ss1, ss2, ss3, ds0, ds1, ds2, ds3, eps, stream, ctx.device); } + +// nodes[0..count) are independent, same-shape, same-eps, same-nb L2_NORM ops validated +// by the caller; requires ncols < 1024 (the warp reduction path). +void ggml_sycl_l2_norm_batch(ggml_backend_sycl_context & ctx, ggml_tensor ** nodes, int count) { + const ggml_tensor * s0 = nodes[0]->src[0]; + const int ncols = (int) s0->ne[0]; + const int nrows_total = (int) ggml_nrows(s0); + float eps; + memcpy(&eps, nodes[0]->op_params, sizeof(float)); + GGML_ASSERT(eps >= 0.0f); + + l2_batch_ptrs p{}; + for (int t = 0; t < count; ++t) { + p.src[t] = (const float *) nodes[t]->src[0]->data; + p.dst[t] = (float *) nodes[t]->data; + } + + const ggml_tensor * d0 = nodes[0]; + const size_t ts = ggml_type_size(GGML_TYPE_F32); + l2_batch_strides st{}; + st.ne1 = (int) s0->ne[1]; + st.ne2 = (int) s0->ne[2]; + st.ss0 = s0->nb[0] / ts; st.ss1 = s0->nb[1] / ts; st.ss2 = s0->nb[2] / ts; st.ss3 = s0->nb[3] / ts; + st.ds0 = d0->nb[0] / ts; st.ds1 = d0->nb[1] / ts; st.ds2 = d0->nb[2] / ts; st.ds3 = d0->nb[3] / ts; + + l2_norm_f32_batch_sycl(p, st, count, ncols, nrows_total, eps, ctx.stream()); +} diff --git a/ggml/src/ggml-sycl/norm.hpp b/ggml/src/ggml-sycl/norm.hpp index 51217c42..fb667ac6 100644 --- a/ggml/src/ggml-sycl/norm.hpp +++ b/ggml/src/ggml-sycl/norm.hpp @@ -21,10 +21,17 @@ void ggml_sycl_op_rms_norm(ggml_backend_sycl_context& ctx, ggml_tensor* dst); void ggml_sycl_op_rms_norm_fused(ggml_backend_sycl_context& ctx, ggml_tensor* dst, ggml_tensor* mul); +void ggml_sycl_op_rms_norm_scale_fused(ggml_backend_sycl_context& ctx, ggml_tensor* dst, ggml_tensor* scale_tensor); + +void ggml_sycl_op_rms_norm_fused_add(ggml_backend_sycl_context& ctx, ggml_tensor* dst, ggml_tensor* mul_tensor, ggml_tensor* add_tensor); + void ggml_sycl_op_rms_norm_back(ggml_backend_sycl_context& ctx, ggml_tensor* dst); void ggml_sycl_op_group_norm(ggml_backend_sycl_context& ctx, ggml_tensor* dst); void ggml_sycl_op_l2_norm(ggml_backend_sycl_context& ctx, ggml_tensor* dst); +#define GGML_SYCL_L2_BATCH_MAX 8 +void ggml_sycl_l2_norm_batch(ggml_backend_sycl_context & ctx, ggml_tensor ** nodes, int count); + #endif // GGML_SYCL_NORM_HPP diff --git a/ggml/src/ggml-sycl/quants.hpp b/ggml/src/ggml-sycl/quants.hpp index 95287f17..a26a6ce6 100644 --- a/ggml/src/ggml-sycl/quants.hpp +++ b/ggml/src/ggml-sycl/quants.hpp @@ -58,6 +58,29 @@ template <> struct block_q_t { static constexpr int block_to_q8_1_ratio() { return traits::qk / QK8_1; } }; +template <> struct block_q_t { + struct traits { + static constexpr uint32_t qk = QK_K; + static constexpr uint32_t qi = QI2_K; + static constexpr uint32_t qr = QR2_K; + static constexpr uint32_t vdr_mmvq = 1; + }; + + // Reordered layout: [qs (QK_K/4 per block)] [scales (QK_K/16 per block)] [dm] + static constexpr std::pair get_block_offset(const int block_index, const int /* n_blocks */) { + return { block_index * (QK_K / 4), 0 }; + } + + static constexpr std::pair get_d_offset(int nrows, int ncols, const int block_index) { + auto nblocks = (nrows * (ncols / QK_K)); + auto total_qs_bytes = nblocks * (QK_K / 4); + return { total_qs_bytes + block_index * (QK_K / 16), + total_qs_bytes + nblocks * (QK_K / 16) + block_index * sizeof(ggml_half2) }; + } + + static constexpr int block_to_q8_1_ratio() { return traits::qk / QK8_1; } +}; + template <> struct block_q_t { struct traits { static constexpr uint32_t qk = QK_K; diff --git a/ggml/src/ggml-sycl/rope.cpp b/ggml/src/ggml-sycl/rope.cpp index 9d83a1e9..b6d22559 100644 --- a/ggml/src/ggml-sycl/rope.cpp +++ b/ggml/src/ggml-sycl/rope.cpp @@ -41,7 +41,7 @@ template static void rope_norm(const T *x, D *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, - const int n_dims, const int32_t *pos, + const int n_dims, const int n_offs, const int32_t *pos, const float freq_scale, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float theta_scale, const float *freq_factors, @@ -78,19 +78,21 @@ static void rope_norm(const T *x, D *dst, const int ne00, const int ne01, ggml_sycl_memcpy_1<4>(dst + idst, &v); } }; - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { store_coaelsced(x[ix + 0], x[ix + 1]); return; } - const float theta_base = pos[i2] * dpct::pow(theta_scale, i0 / 2.0f); + const int iw = i0 - n_offs; // relative idx - const float freq_factor = has_ff ? freq_factors[i0 / 2] : 1.0f; + const float theta_base = pos[i2] * dpct::pow(theta_scale, iw / 2.0f); + + const float freq_factor = has_ff ? freq_factors[iw / 2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, i0, + rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); const float x0 = x[ix + 0]; @@ -104,7 +106,7 @@ template static void rope_neox(const T *x, D *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, - const int n_dims, const int32_t *pos, + const int n_dims, const int n_offs, const int32_t *pos, const float freq_scale, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float theta_scale, const float *freq_factors, @@ -132,35 +134,38 @@ static void rope_neox(const T *x, D *dst, const int ne00, const int ne01, idst += row_indices[i2] * set_rows_stride; } - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { dst[idst + i0 / 2 + 0] = ggml_sycl_cast(x[ix + i0 / 2 + 0]); dst[idst + i0 / 2 + 1] = ggml_sycl_cast(x[ix + i0 / 2 + 1]); return; } - const float theta_base = pos[i2] * dpct::pow(theta_scale, i0 / 2.0f); + const int iw = i0 - n_offs; // relative idx - const float freq_factor = has_ff ? freq_factors[i0 / 2] : 1.0f; + const float theta_base = pos[i2] * dpct::pow(theta_scale, iw / 2.0f); + + const float freq_factor = has_ff ? freq_factors[iw / 2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, i0, + rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); - const float x0 = x[ix + 0]; - const float x1 = x[ix + n_dims / 2]; + // idst/ix point at channel i0/2; the first channel of the rotated pair is n_offs + iw/2 = i0/2 + n_offs/2 + const float x0 = x[ix + n_offs / 2 + 0]; + const float x1 = x[ix + n_offs / 2 + n_dims / 2]; - dst[idst + 0] = ggml_sycl_cast(x0 * cos_theta - x1 * sin_theta); - dst[idst + n_dims / 2] = ggml_sycl_cast(x0 * sin_theta + x1 * cos_theta); + dst[idst + n_offs / 2 + 0] = ggml_sycl_cast(x0 * cos_theta - x1 * sin_theta); + dst[idst + n_offs / 2 + n_dims / 2] = ggml_sycl_cast(x0 * sin_theta + x1 * cos_theta); } template static void rope_multi(const T *x, T *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, - const int n_dims, const int32_t *pos, + const int n_dims, const int n_offs, const int32_t *pos, const float freq_scale, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float theta_scale, const float *freq_factors, @@ -183,54 +188,57 @@ static void rope_multi(const T *x, T *dst, const int ne00, const int ne01, int idst = i0 / 2 + i1 * s1 + i2 * s2 + i3 * s3; const int ix = i0 / 2 + i1 * s01 + i2 * s02 + i3 * s03; - if (i0 >= n_dims) { + if (i0 < n_offs || i0 >= n_offs + n_dims) { dst[idst + i0 / 2 + 0] = x[ix + i0 / 2 + 0]; dst[idst + i0 / 2 + 1] = x[ix + i0 / 2 + 1]; return; } + const int iw = i0 - n_offs; // relative idx + const int sect_dims = sections.v[0] + sections.v[1] + sections.v[2] + sections.v[3]; const int sec_w = sections.v[1] + sections.v[0]; - const int sector = (i0 / 2) % sect_dims; + const int sector = (iw / 2) % sect_dims; float theta_base = 0.0; if (is_imrope) { if (sector % 3 == 1 && sector < 3 * sections.v[1]) { // h - theta_base = pos[i2 + ne02 * 1] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 1] * dpct::pow(theta_scale, iw / 2.0f); } else if (sector % 3 == 2 && sector < 3 * sections.v[2]) { // w - theta_base = pos[i2 + ne02 * 2] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 2] * dpct::pow(theta_scale, iw / 2.0f); } else if (sector % 3 == 0 && sector < 3 * sections.v[0]) { // t - theta_base = pos[i2] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2] * dpct::pow(theta_scale, iw / 2.0f); } else { - theta_base = pos[i2 + ne02 * 3] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 3] * dpct::pow(theta_scale, iw / 2.0f); } } else { if (sector < sections.v[0]) { - theta_base = pos[i2] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2] * dpct::pow(theta_scale, iw / 2.0f); } else if (sector >= sections.v[0] && sector < sec_w) { - theta_base = pos[i2 + ne02 * 1] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 1] * dpct::pow(theta_scale, iw / 2.0f); } else if (sector >= sec_w && sector < sec_w + sections.v[2]) { - theta_base = pos[i2 + ne02 * 2] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 2] * dpct::pow(theta_scale, iw / 2.0f); } else if (sector >= sec_w + sections.v[2]) { - theta_base = pos[i2 + ne02 * 3] * dpct::pow(theta_scale, i0 / 2.0f); + theta_base = pos[i2 + ne02 * 3] * dpct::pow(theta_scale, iw / 2.0f); } } - const float freq_factor = has_ff ? freq_factors[i0 / 2] : 1.0f; + const float freq_factor = has_ff ? freq_factors[iw / 2] : 1.0f; float cos_theta; float sin_theta; - rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, i0, + rope_yarn(theta_base / freq_factor, freq_scale, corr_dims, iw, ext_factor, attn_factor, cos_theta, sin_theta); - const float x0 = x[ix + 0]; - const float x1 = x[ix + n_dims / 2]; + // idst/ix point at channel i0/2; the first channel of the rotated pair is n_offs + iw/2 = i0/2 + n_offs/2 + const float x0 = x[ix + n_offs / 2 + 0]; + const float x1 = x[ix + n_offs / 2 + n_dims / 2]; - dst[idst + 0] = x0 * cos_theta - x1 * sin_theta; - dst[idst + n_dims / 2] = x0 * sin_theta + x1 * cos_theta; + dst[idst + n_offs / 2 + 0] = x0 * cos_theta - x1 * sin_theta; + dst[idst + n_offs / 2 + n_dims / 2] = x0 * sin_theta + x1 * cos_theta; } template @@ -293,7 +301,7 @@ static void rope_norm_sycl(const T *x, D *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, const int n_dims, - const int nr, const int32_t *pos, const float freq_scale, + const int n_offs, const int nr, const int32_t *pos, const float freq_scale, const float freq_base, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float *freq_factors, const int64_t *row_indices, @@ -313,7 +321,7 @@ rope_norm_sycl(const T *x, D *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_norm( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); }); } else { @@ -323,7 +331,7 @@ rope_norm_sycl(const T *x, D *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_norm( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); }); } @@ -334,7 +342,7 @@ static void rope_neox_sycl(const T *x, D *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, const int n_dims, - const int nr, const int32_t *pos, const float freq_scale, + const int n_offs, const int nr, const int32_t *pos, const float freq_scale, const float freq_base, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float *freq_factors, const int64_t *row_indices, @@ -354,7 +362,7 @@ rope_neox_sycl(const T *x, D *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_neox( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); }); } else { @@ -364,7 +372,7 @@ rope_neox_sycl(const T *x, D *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_neox( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, row_indices, set_rows_stride); }); } @@ -375,7 +383,7 @@ static void rope_multi_sycl(const T *x, T *dst, const int ne00, const int ne01, const int ne02, const int s01, const int s02, const int s03, const int s1, const int s2, const int s3, const int n_dims, - const int nr, const int32_t *pos, const float freq_scale, + const int n_offs, const int nr, const int32_t *pos, const float freq_scale, const float freq_base, const float ext_factor, const float attn_factor, const rope_corr_dims corr_dims, const float *freq_factors, const mrope_sections sections, @@ -395,7 +403,7 @@ rope_multi_sycl(const T *x, T *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_multi( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope); }); } else { @@ -405,7 +413,7 @@ rope_multi_sycl(const T *x, T *dst, const int ne00, const int ne01, GGML_UNUSED(item_ct1); rope_multi( x, dst, ne00, ne01, ne02, s01, s02, s03, s1, s2, s3, n_dims, - pos, freq_scale, ext_factor, attn_factor, corr_dims, + n_offs, pos, freq_scale, ext_factor, attn_factor, corr_dims, theta_scale, freq_factors, sections, is_imrope); }); } @@ -497,6 +505,7 @@ void ggml_sycl_op_rope_impl(ggml_backend_sycl_context &ctx, ggml_tensor *dst, const int n_dims = ((int32_t *)dst->op_params)[1]; const int mode = ((int32_t *)dst->op_params)[2]; const int n_ctx_orig = ((int32_t *)dst->op_params)[4]; + const int n_offs = ((int32_t *)dst->op_params)[15]; mrope_sections sections; float freq_base; @@ -526,6 +535,7 @@ void ggml_sycl_op_rope_impl(ggml_backend_sycl_context &ctx, ggml_tensor *dst, if (is_vision) { GGML_ASSERT(n_dims == ne00 / 2); + GGML_ASSERT(n_offs == 0); // offset not supported for vision, as the rotated pairs span the whole row } const int32_t *pos = (const int32_t *)src1_d; @@ -545,19 +555,19 @@ void ggml_sycl_op_rope_impl(ggml_backend_sycl_context &ctx, ggml_tensor *dst, if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F32) { rope_neox_sycl( (const float *)src0_d, (float *)dst_d, ne00, ne01, ne02, s01, - s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F16) { rope_neox_sycl( (const float *)src0_d, (sycl::half *)dst_d, ne00, ne01, ne02, - s01, s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, + s01, s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else if (src0->type == GGML_TYPE_F16 && dst_type == GGML_TYPE_F16) { rope_neox_sycl( (const sycl::half *)src0_d, (sycl::half *)dst_d, ne00, ne01, - ne02, s01, s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, + ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else { @@ -568,13 +578,13 @@ void ggml_sycl_op_rope_impl(ggml_backend_sycl_context &ctx, ggml_tensor *dst, if (src0->type == GGML_TYPE_F32) { rope_multi_sycl((const float *)src0_d, (float *)dst_d, ne00, ne01, ne02, s01, s02, s03, s1, s2, - s3, n_dims, nr, pos, freq_scale, freq_base, + s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, sections, is_imrope, stream); } else if (src0->type == GGML_TYPE_F16) { rope_multi_sycl( (const sycl::half *)src0_d, (sycl::half *)dst_d, ne00, ne01, - ne02, s01, s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, + ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, sections, is_imrope, stream); } else { @@ -602,19 +612,19 @@ void ggml_sycl_op_rope_impl(ggml_backend_sycl_context &ctx, ggml_tensor *dst, if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F32) { rope_norm_sycl( (const float *)src0_d, (float *)dst_d, ne00, ne01, ne02, s01, - s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, freq_base, + s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else if (src0->type == GGML_TYPE_F32 && dst_type == GGML_TYPE_F16) { rope_norm_sycl( (const float *)src0_d, (sycl::half *)dst_d, ne00, ne01, ne02, - s01, s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, + s01, s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else if (src0->type == GGML_TYPE_F16 && dst_type == GGML_TYPE_F16) { rope_norm_sycl( (const sycl::half *)src0_d, (sycl::half *)dst_d, ne00, ne01, - ne02, s01, s02, s03, s1, s2, s3, n_dims, nr, pos, freq_scale, + ne02, s01, s02, s03, s1, s2, s3, n_dims, n_offs, nr, pos, freq_scale, freq_base, ext_factor, attn_factor, corr_dims, freq_factors, row_indices, set_rows_stride, stream); } else { diff --git a/ggml/src/ggml-sycl/set_rows.cpp b/ggml/src/ggml-sycl/set_rows.cpp index 52a0bcb6..c73ad8dc 100644 --- a/ggml/src/ggml-sycl/set_rows.cpp +++ b/ggml/src/ggml-sycl/set_rows.cpp @@ -291,7 +291,7 @@ static void set_rows_sycl( stream->parallel_for( sycl::nd_range<1>(grid_size * block_size, block_size), - [=](sycl::nd_item<1> item_ct1) [[intel::reqd_sub_group_size(WARP_SIZE)]] { + [=](sycl::nd_item<1> item_ct1) [[sycl::reqd_sub_group_size(WARP_SIZE)]] { k_set_rows( src0_d, src1_d, dst_d, ne00, ne01, ne02, @@ -546,7 +546,8 @@ static void set_rows_sycl(ggml_backend_sycl_context & ctx, const ggml_tensor * s stream); break; default: - GGML_ABORT("Unsupported tensor type!"); + GGML_ABORT("Unsupported tensor type: src0 %s src1 %s dst %s", ggml_type_name(dst->src[0]->type), + ggml_type_name(dst->src[1]->type), ggml_type_name(dst->type)); break; } } diff --git a/ggml/src/ggml-sycl/ssm_conv.cpp b/ggml/src/ggml-sycl/ssm_conv.cpp index 3eafa1a6..a8714351 100644 --- a/ggml/src/ggml-sycl/ssm_conv.cpp +++ b/ggml/src/ggml-sycl/ssm_conv.cpp @@ -1,11 +1,71 @@ #include "ssm_conv.hpp" #include "common.hpp" +#include "element_wise.hpp" #include using namespace sycl; -static void kernel_ssm_conv( +// One output element of the conv. DC is d_conv as a compile-time constant (0 keeps the +// runtime loop); unfused callers pass literal false/nullptr so the epilogue folds away. +template +static __dpct_inline__ void ssm_conv_element( + size_t idx, + const float *src_data, + const float *weights, + float *dst_data, + int d_conv, + int d_inner, + int n_t, + int src_stride_inner, + int src_stride_seq, + int dst_stride_token, + int dst_stride_seq, + bool apply_silu, + const float *bias +) { + // src is token-contiguous per channel, dst is channel-contiguous per token, + // so indexing token-fastest coalesces the d_conv loads. + const int token = static_cast(idx % n_t); + const int channel = static_cast((idx / n_t) % d_inner); + const int seq = static_cast(idx / (static_cast(n_t) * static_cast(d_inner))); + + const float *s = src_data + + static_cast(seq) * static_cast(src_stride_seq) + + static_cast(channel) * static_cast(src_stride_inner) + + static_cast(token); + + const float *c = weights + static_cast(channel) * static_cast(d_conv); + + float sumf = 0.0f; + if constexpr (DC > 0) { +#pragma unroll + for (int i0 = 0; i0 < DC; ++i0) { + sumf += s[i0] * c[i0]; + } + } else { + for (int i0 = 0; i0 < d_conv; ++i0) { + sumf += s[i0] * c[i0]; + } + } + + // fused bias add: the ADD node broadcasts a 1-D channel bias over tokens + if (bias != nullptr) { + sumf += bias[channel]; + } + + const size_t dst_idx = + static_cast(seq) * static_cast(dst_stride_seq) + + static_cast(token) * static_cast(dst_stride_token) + + static_cast(channel); + + dst_data[dst_idx] = apply_silu ? op_silu(sumf) : sumf; +} + +// FUSED=false keeps apply_silu/bias out of the kernel capture list, so the unfused launch +// takes the pre-fusion argument list; matters at n_t == 1, where the op is launch-bound. +template +static void kernel_ssm_conv_impl( queue &q, const float *src_data, const float *weights, @@ -18,7 +78,9 @@ static void kernel_ssm_conv( int src_stride_inner, int src_stride_seq, int dst_stride_token, - int dst_stride_seq + int dst_stride_seq, + bool apply_silu, + const float *bias ) { const size_t total_work = static_cast(d_inner) * static_cast(n_t) * static_cast(n_s); const size_t work_group_size = 256; @@ -27,53 +89,199 @@ static void kernel_ssm_conv( const range<1> global_range(num_work_groups * work_group_size); const range<1> local_range(work_group_size); - q.submit([&](handler &h) { - h.parallel_for( - nd_range<1>(global_range, local_range), - [=](nd_item<1> item) { - const size_t idx = item.get_global_id(0); - if (idx >= total_work) { - return; + if constexpr (FUSED) { + q.submit([&](handler &h) { + h.parallel_for( + nd_range<1>(global_range, local_range), + [=](nd_item<1> item) { + const size_t idx = item.get_global_id(0); + if (idx >= total_work) { + return; + } + + ssm_conv_element(idx, src_data, weights, dst_data, d_conv, d_inner, n_t, + src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); } + ); + }); + } else { + GGML_UNUSED(apply_silu); + GGML_UNUSED(bias); - // src has the tokens of one channel contiguous, dst has the channels of one - // token contiguous, so either the loads or the store must be strided. Indexing - // token-fastest coalesces the d_conv loads, which measured faster except for - // short, cache-resident rows. - const int token = static_cast(idx % n_t); - const int channel = static_cast((idx / n_t) % d_inner); - const int seq = static_cast(idx / (static_cast(n_t) * static_cast(d_inner))); + q.submit([&](handler &h) { + h.parallel_for( + nd_range<1>(global_range, local_range), + [=](nd_item<1> item) { + const size_t idx = item.get_global_id(0); + if (idx >= total_work) { + return; + } - const float *s = src_data - + static_cast(seq) * static_cast(src_stride_seq) - + static_cast(channel) * static_cast(src_stride_inner) - + static_cast(token); + ssm_conv_element(idx, src_data, weights, dst_data, d_conv, d_inner, n_t, + src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, false, nullptr); + } + ); + }); + } +} - const float *c = weights + static_cast(channel) * static_cast(d_conv); +// SLM transpose tile: coalesces both the loads and the stores. The +1 pad makes the row +// stride 33, coprime with 32 banks, so both phases are bank-conflict-free. +template +static __dpct_inline__ void ssm_conv_tile( + nd_item<1> it, local_accessor tile, const float *src_data, const float *weights, + float *dst_data, int n_t, int nt_tiles, int nc_tiles, int src_stride_inner, + int src_stride_seq, int dst_stride_token, int dst_stride_seq, bool apply_silu, + const float *bias +) { + const int lid = static_cast(it.get_local_id(0)); + const size_t g = it.get_group(0); + const int tt = static_cast(g % nt_tiles); + const int ct = static_cast((g / nt_tiles) % nc_tiles); + const int seq = static_cast(g / (static_cast(nt_tiles) * nc_tiles)); + const int t0 = tt * TT, c0 = ct * TC; - float sumf = 0.0f; - for (int i0 = 0; i0 < d_conv; ++i0) { - sumf += s[i0] * c[i0]; - } + const int ti = lid % TT; + const int cj = lid / TT; +#pragma unroll + for (int r = 0; r < TC / (WG / TT); ++r) { + const int c = cj + r * (WG / TT); + const int tok = t0 + ti; + float sumf = 0.0f; + if (tok < n_t) { + const float *s = src_data + static_cast(seq) * src_stride_seq + + static_cast(c0 + c) * src_stride_inner + tok; + const float *cw = weights + static_cast(c0 + c) * DC; +#pragma unroll + for (int i = 0; i < DC; ++i) sumf += s[i] * cw[i]; + if (bias != nullptr) sumf += bias[c0 + c]; + if (apply_silu) sumf = op_silu(sumf); + } + tile[c * (TT + 1) + ti] = sumf; + } + it.barrier(access::fence_space::local_space); + + const int cc = lid % TC; + const int tj = lid / TC; +#pragma unroll + for (int r = 0; r < TT / (WG / TC); ++r) { + const int t = tj + r * (WG / TC); + const int tok = t0 + t; + if (tok < n_t) { + dst_data[static_cast(seq) * dst_stride_seq + + static_cast(tok) * dst_stride_token + c0 + cc] + = tile[cc * (TT + 1) + t]; + } + } +} + +// Same FUSED split as kernel_ssm_conv_impl. The fused instantiation keeps the runtime +// apply_silu/bias branches: at n_t >= 32 they are amortized over the whole tile. +template +static void kernel_ssm_conv_tiled( + queue &q, const float *src_data, const float *weights, float *dst_data, + int d_inner, int n_t, int n_s, int src_stride_inner, int src_stride_seq, + int dst_stride_token, int dst_stride_seq, bool apply_silu, const float *bias +) { + constexpr int TT = 32, TC = 32, WG = 256; + const int nt_tiles = (n_t + TT - 1) / TT; + const int nc_tiles = d_inner / TC; + const size_t groups = static_cast(nt_tiles) * nc_tiles * n_s; - const size_t dst_idx = - static_cast(seq) * static_cast(dst_stride_seq) + - static_cast(token) * static_cast(dst_stride_token) + - static_cast(channel); + if constexpr (FUSED) { + q.submit([&](handler &h) { + local_accessor tile(range<1>(TC * (TT + 1)), h); + h.parallel_for(nd_range<1>(range<1>(groups * WG), range<1>(WG)), [=](nd_item<1> it) { + ssm_conv_tile(it, tile, src_data, weights, dst_data, n_t, nt_tiles, + nc_tiles, src_stride_inner, src_stride_seq, + dst_stride_token, dst_stride_seq, apply_silu, bias); + }); + }); + } else { + GGML_UNUSED(apply_silu); + GGML_UNUSED(bias); - dst_data[dst_idx] = sumf; - } - ); - }); + q.submit([&](handler &h) { + local_accessor tile(range<1>(TC * (TT + 1)), h); + h.parallel_for(nd_range<1>(range<1>(groups * WG), range<1>(WG)), [=](nd_item<1> it) { + ssm_conv_tile(it, tile, src_data, weights, dst_data, n_t, nt_tiles, + nc_tiles, src_stride_inner, src_stride_seq, + dst_stride_token, dst_stride_seq, false, nullptr); + }); + }); + } +} + +static void kernel_ssm_conv( + queue &q, + const float *src_data, + const float *weights, + float *dst_data, + int d_conv, + int d_inner, + int n_t, + int n_s, + int ncs, + int src_stride_inner, + int src_stride_seq, + int dst_stride_token, + int dst_stride_seq, + bool apply_silu, + const float *bias +) { + // Only the fused instantiations carry apply_silu/bias as kernel arguments; the plain + // ssm_conv launch keeps the argument list it had before the fusion landed. + const bool fused = apply_silu || bias != nullptr; + + // d_inner must be a multiple of 32 so the channel tiles are exact; the transpose is only + // worth it for n_t >= 32. d_conv == 4 is the only window with a DC-specialized kernel. + if (d_conv == 4 && n_t >= 32 && (d_inner % 32) == 0) { + if (fused) { + kernel_ssm_conv_tiled<4, true>(q, src_data, weights, dst_data, d_inner, n_t, n_s, + src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } else { + kernel_ssm_conv_tiled<4, false>(q, src_data, weights, dst_data, d_inner, n_t, n_s, + src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } + return; + } + + if (d_conv == 4) { + if (fused) { + kernel_ssm_conv_impl<4, true>(q, src_data, weights, dst_data, d_conv, d_inner, n_t, n_s, + ncs, src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } else { + kernel_ssm_conv_impl<4, false>(q, src_data, weights, dst_data, d_conv, d_inner, n_t, n_s, + ncs, src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } + return; + } + + if (fused) { + kernel_ssm_conv_impl<0, true>(q, src_data, weights, dst_data, d_conv, d_inner, n_t, n_s, + ncs, src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } else { + kernel_ssm_conv_impl<0, false>(q, src_data, weights, dst_data, d_conv, d_inner, n_t, n_s, + ncs, src_stride_inner, src_stride_seq, dst_stride_token, + dst_stride_seq, apply_silu, bias); + } } -inline void ggml_sycl_op_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { +inline void ggml_sycl_op_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * dst, ggml_tensor * silu_dst = nullptr, const float * bias = nullptr) { ggml_tensor * src0 = dst->src[0]; ggml_tensor * src1 = dst->src[1]; GGML_ASSERT(src0->type == GGML_TYPE_F32); GGML_ASSERT(src1->type == GGML_TYPE_F32); GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(bias == nullptr || silu_dst != nullptr); const int d_conv = src1->ne[0]; const int ncs = src0->ne[0]; @@ -104,7 +312,8 @@ inline void ggml_sycl_op_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * const float *src_data = static_cast(src0->data); const float *weights = static_cast(src1->data); - float *dst_data = static_cast(dst->data); + const bool apply_silu = silu_dst != nullptr; + float *dst_data = static_cast((silu_dst ? silu_dst : dst)->data); GGML_ASSERT(src_data && weights && dst_data); @@ -121,7 +330,9 @@ inline void ggml_sycl_op_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * src_stride_inner, src_stride_seq, dst_stride_token, - dst_stride_seq + dst_stride_seq, + apply_silu, + bias ); } catch (const std::exception &e) { @@ -134,3 +345,17 @@ void ggml_sycl_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * dst) { scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/2); ggml_sycl_op_ssm_conv(ctx, dst); } + +// Fused ssm_conv + ADD + SiLU: write silu(conv(x) + b) straight into silu_dst, eliding the +// standalone SiLU launch and its HBM round-trip of the conv output. +void ggml_sycl_ssm_conv_fused(ggml_backend_sycl_context & ctx, ggml_tensor * dst, ggml_tensor * add, ggml_tensor * silu_dst) { + scope_op_debug_print scope_dbg_print(__func__, dst, /*num_src=*/2); + GGML_ASSERT(silu_dst && ggml_are_same_shape(dst, silu_dst) && silu_dst->type == GGML_TYPE_F32); + // the fused kernel reads only the ADD's bias operand; the ADD result is never written + const float * bias = nullptr; + if (add != nullptr) { + const ggml_tensor * bias_t = (add->src[0] == dst) ? add->src[1] : add->src[0]; + bias = static_cast(bias_t->data); + } + ggml_sycl_op_ssm_conv(ctx, dst, silu_dst, bias); +} diff --git a/ggml/src/ggml-sycl/ssm_conv.hpp b/ggml/src/ggml-sycl/ssm_conv.hpp index 1a8ad05f..72c90662 100644 --- a/ggml/src/ggml-sycl/ssm_conv.hpp +++ b/ggml/src/ggml-sycl/ssm_conv.hpp @@ -3,3 +3,4 @@ #include "common.hpp" void ggml_sycl_ssm_conv(ggml_backend_sycl_context & ctx, ggml_tensor * dst); +void ggml_sycl_ssm_conv_fused(ggml_backend_sycl_context & ctx, ggml_tensor * dst, ggml_tensor * add, ggml_tensor * silu_dst); diff --git a/ggml/src/ggml-sycl/topk-radix.cpp b/ggml/src/ggml-sycl/topk-radix.cpp new file mode 100644 index 00000000..8cd0bd2f --- /dev/null +++ b/ggml/src/ggml-sycl/topk-radix.cpp @@ -0,0 +1,531 @@ +#include "topk-radix.hpp" + +#include "common.hpp" + +#include + +// Large-k top-k by radix select on an order-preserving unsigned key. +// +// The k-th largest key of a row is found by four most-significant-first passes over its +// 8-bit digits: histogram the digit over the candidate set, walk the buckets from the +// top, and recurse into the bucket where the running count reaches what is still +// needed. Everything strictly above that bucket is in the top-k. A final pass emits +// every column whose key beats the pivot, then exactly as many pivot-equal columns as +// are still missing, so duplicate keys yield exactly k distinct indices. +// +// SLM holds only the histogram, so unlike the scan-merge kernels the cost does not grow +// with k. One work-group owns a row and runs every pass, so a top-k is one launch and +// needs no pool scratch. The row is re-read once per pass rather than compacted, which +// keeps the candidate set implicit: (key & mask) == prefix. +// +// The output is the set of winning indices in no particular order, which is what the +// reference op provides (it swaps its first two outputs to say so) and what +// test-backend-ops compares. + +static constexpr int SYCL_TOP_K_RADIX_BITS = 8; +static constexpr int SYCL_TOP_K_RADIX_BUCKETS = 1 << SYCL_TOP_K_RADIX_BITS; +// Private histogram copies, interleaved per bucket so neighbouring lanes hit +// neighbouring banks. Lanes of one instruction spread over the copies, which is what +// bounds the atomic serialisation on tie-heavy rows. +static constexpr int SYCL_TOP_K_RADIX_HIST_COPIES = 8; +static constexpr int SYCL_TOP_K_RADIX_HIST_SIZE = SYCL_TOP_K_RADIX_BUCKETS * SYCL_TOP_K_RADIX_HIST_COPIES; +// Past the histogram: pivot digit, pivot bucket count, remaining need, then the two +// emit counters. +static constexpr int SYCL_TOP_K_RADIX_SLM_WORDS = SYCL_TOP_K_RADIX_HIST_SIZE + 5; + +// Larger float <=> larger key. The reference comparator is a plain float '>', under which +// -0.0 and +0.0 tie, so -0.0 is folded onto +0.0 first. NaN has no defined order in the +// reference (its comparator is not a strict weak order on NaN); here a positive NaN keys +// above +inf and a negative NaN below -inf, which at least makes the result deterministic. +static inline uint32_t top_k_radix_key(float f) { + uint32_t u = sycl::bit_cast(f); + if (u == 0x80000000u) { + u = 0u; + } + return (u & 0x80000000u) ? ~u : (u | 0x80000000u); +} + +static void top_k_radix_select_f32( + const float * src, + int32_t * dst_idx, + const int ncols, + const int k, + uint32_t * slm, + const sycl::nd_item<1> & item_ct1 +) { + using local_atomic = sycl::atomic_ref; + + const int tid = item_ct1.get_local_id(0); + const int block_size = item_ct1.get_local_range(0); + + uint32_t * hist = slm; + uint32_t * s_digit = slm + SYCL_TOP_K_RADIX_HIST_SIZE; + uint32_t * s_bucket = slm + SYCL_TOP_K_RADIX_HIST_SIZE + 1; + uint32_t * s_need = slm + SYCL_TOP_K_RADIX_HIST_SIZE + 2; + uint32_t * s_cnt_gt = slm + SYCL_TOP_K_RADIX_HIST_SIZE + 3; + uint32_t * s_cnt_eq = slm + SYCL_TOP_K_RADIX_HIST_SIZE + 4; + + if (tid == 0) { + *s_cnt_gt = 0; + *s_cnt_eq = 0; + } + + const int copy = tid & (SYCL_TOP_K_RADIX_HIST_COPIES - 1); + + uint32_t prefix = 0; // digits fixed so far, in place + uint32_t mask = 0; // which bits of prefix are fixed + uint32_t need = (uint32_t) k; + + for (int shift = 32 - SYCL_TOP_K_RADIX_BITS; shift >= 0; shift -= SYCL_TOP_K_RADIX_BITS) { + for (int i = tid; i < SYCL_TOP_K_RADIX_HIST_SIZE; i += block_size) { + hist[i] = 0; + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + for (int col = tid; col < ncols; col += block_size) { + const uint32_t key = top_k_radix_key(src[col]); + if ((key & mask) == prefix) { + const uint32_t bucket = (key >> shift) & (SYCL_TOP_K_RADIX_BUCKETS - 1); + local_atomic(hist[bucket * SYCL_TOP_K_RADIX_HIST_COPIES + copy]).fetch_add(1u); + } + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + // Lane t takes bucket 255 - t, so an inclusive scan over lanes counts from the top + // bucket downward. The pivot is the unique bucket whose cumulative count first + // reaches need; the previous cumulative count is what the higher buckets contribute. + uint32_t cnt = 0; + if (tid < SYCL_TOP_K_RADIX_BUCKETS) { + const uint32_t * h = hist + (SYCL_TOP_K_RADIX_BUCKETS - 1 - tid) * SYCL_TOP_K_RADIX_HIST_COPIES; + for (int c = 0; c < SYCL_TOP_K_RADIX_HIST_COPIES; c++) { + cnt += h[c]; + } + } + const uint32_t incl = sycl::inclusive_scan_over_group(item_ct1.get_group(), cnt, sycl::plus()); + + if (tid < SYCL_TOP_K_RADIX_BUCKETS && incl >= need && incl - cnt < need) { + *s_digit = (uint32_t) (SYCL_TOP_K_RADIX_BUCKETS - 1 - tid); + *s_bucket = cnt; + *s_need = need - (incl - cnt); + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + const uint32_t digit = *s_digit; + const uint32_t bucket_cnt = *s_bucket; + need = *s_need; + prefix |= digit << shift; + mask |= (uint32_t) (SYCL_TOP_K_RADIX_BUCKETS - 1) << shift; + + // Every candidate in the pivot bucket is wanted: the remaining digits cannot + // change the answer, and the masked emit below is exact as it stands. + if (bucket_cnt == need) { + break; + } + // The next pass rewrites hist and s_*; the reads above must land first. + item_ct1.barrier(sycl::access::fence_space::local_space); + } + + item_ct1.barrier(sycl::access::fence_space::local_space); + + // Exactly k - need columns have (key & mask) > prefix; the first need of the pivot-equal + // columns fill the tail. Both counters live in SLM since the whole row is this group. + const uint32_t base_eq = (uint32_t) k - need; + + for (int col = tid; col < ncols; col += block_size) { + const uint32_t kp = top_k_radix_key(src[col]) & mask; + if (kp > prefix) { + const uint32_t pos = local_atomic(*s_cnt_gt).fetch_add(1u); + dst_idx[pos] = col; + } else if (kp == prefix) { + const uint32_t pos = local_atomic(*s_cnt_eq).fetch_add(1u); + if (pos < need) { + dst_idx[base_eq + pos] = col; + } + } + } +} + +static void top_k_radix_f32_sycl( + ggml_backend_sycl_context & ctx, + const float * src, + int32_t * dst_indices, + const int64_t ncols, + const int64_t nrows, + const int k, + dpct::queue_ptr main_stream +) { + GGML_ASSERT(ncols <= INT32_MAX); + + // One group per row; every pass is a strided sweep of the row, so lanes in flight is the + // only lever, and the device's own limit is the answer -- there is nothing here that + // wants a smaller group. Must still cover the 256 buckets for the scan step. + const int block_size = ggml_sycl_info().max_work_group_sizes[ctx.device]; + GGML_ASSERT(block_size >= SYCL_TOP_K_RADIX_BUCKETS); + + const sycl::range<1> block_dims(block_size); + const sycl::range<1> grid_dims(nrows); + + main_stream->submit([&](sycl::handler &cgh) { + sycl::local_accessor slm(sycl::range<1>(SYCL_TOP_K_RADIX_SLM_WORDS), cgh); + + cgh.parallel_for( + sycl::nd_range<1>(grid_dims * block_dims, block_dims), + [=](sycl::nd_item<1> item_ct1) { + const int row = item_ct1.get_group(0); + + top_k_radix_select_f32( + src + (int64_t) row * ncols, dst_indices + (int64_t) row * k, + (int) ncols, k, + slm.get_multi_ptr().get(), + item_ct1); + }); + }); +} + +// One work-group owns a whole row above, which leaves the device idle whenever a graph +// has fewer rows than it has cores -- the common case at batch size 1, where the +// sparse-attention indexer and the backend sampler both top-k a single row. The kernels +// below spread one row over several groups instead. +// +// A digit pass now needs the whole row's histogram before any group can pick the pivot, +// so the per-pass state moves to global memory and the passes become separate launches: +// a work-group barrier no longer spans the row. Each group still accumulates into SLM +// and contributes 256 global atomics at the end, so global traffic is per-group, not +// per-element. The last group to finish a pass (the one whose fetch_add returns G - 1) +// does the scan for the row and clears the histogram for the next pass, which keeps the +// launch count at one per digit rather than two. +// +// Running all four digits unconditionally costs nothing in correctness: once a bucket +// holds exactly the elements still needed, later digits only extend the prefix, and the +// count of columns above that longer prefix grows by exactly as much as `need` shrinks. +// The emit below therefore stays exact whatever pass the answer settled on. + +static constexpr int SYCL_TOP_K_RADIX_ROW_DONE = SYCL_TOP_K_RADIX_BUCKETS + 0; +static constexpr int SYCL_TOP_K_RADIX_ROW_PREFIX = SYCL_TOP_K_RADIX_BUCKETS + 1; +static constexpr int SYCL_TOP_K_RADIX_ROW_MASK = SYCL_TOP_K_RADIX_BUCKETS + 2; +static constexpr int SYCL_TOP_K_RADIX_ROW_NEED = SYCL_TOP_K_RADIX_BUCKETS + 3; +static constexpr int SYCL_TOP_K_RADIX_ROW_CNT_GT = SYCL_TOP_K_RADIX_BUCKETS + 4; +static constexpr int SYCL_TOP_K_RADIX_ROW_CNT_EQ = SYCL_TOP_K_RADIX_BUCKETS + 5; +static constexpr int SYCL_TOP_K_RADIX_ROW_WORDS = SYCL_TOP_K_RADIX_BUCKETS + 6; + +// How wide the split goes is a property of the device, not of the model: enough groups to +// cover the cores, and no more. Past that the extra groups add histogram traffic without +// adding bandwidth (measured on this device: 20 and 40 groups tie, 60 and 160 lose). +// +// nsm is max_compute_units / 16, i.e. it counts an Xe core as 16 EUs. That is a core's +// width on Xe-HPG, but an Xe2 core is 8 XVEs wide, so on Battlemage the field reads half +// the cores actually present (10 for a 20-core B60). The measured curve is flat from one +// group per core to two and only falls off at three, so a factor of two covers the device +// on Xe2 and lands in the flat region on Xe-HPG. It is the one number here that a correct +// core count would remove; it was tuned on Xe2 and has not been measured on Xe-HPG. +static constexpr int SYCL_TOP_K_RADIX_GROUPS_PER_NSM = 2; +// Splitting trades one kernel for five. Below the width at which the single-group kernel +// runs longer than those four extra launches, it wins on its own; measured break-even on +// this device sits just under 64K columns. +static constexpr int SYCL_TOP_K_RADIX_MIN_SPLIT_COLS = 65536; +// A partition thinner than this cannot keep a group's sweep busy. +static constexpr int SYCL_TOP_K_RADIX_MIN_PART_COLS = 4096; + +static int top_k_radix_split_groups(const int device, const int64_t ncols, const int64_t nrows) { + const int64_t target = (int64_t) SYCL_TOP_K_RADIX_GROUPS_PER_NSM * ggml_sycl_info().devices[device].nsm; + + // One group per row already, so a graph with rows enough to cover the device gains + // nothing from splitting and would only pay the extra launches. + if (ncols < SYCL_TOP_K_RADIX_MIN_SPLIT_COLS || nrows >= target) { + return 1; + } + + const int64_t by_rows = target / nrows; // floor: never overshoot a row that is nearly covered + const int64_t by_cols = ncols / SYCL_TOP_K_RADIX_MIN_PART_COLS; + + return (int) std::max(1, std::min(by_rows, by_cols)); +} + +using top_k_radix_gatomic = sycl::atomic_ref; + +static void top_k_radix_split_pass_f32( + const float * src, + uint32_t * state, + const int ncols, + const int k, + const int shift, + const bool first, + const int part, + const int nparts, + uint32_t * slm, + const sycl::nd_item<1> & item_ct1 +) { + using local_atomic = sycl::atomic_ref; + + const int tid = item_ct1.get_local_id(0); + const int block_size = item_ct1.get_local_range(0); + + uint32_t * hist = slm; + uint32_t * s_last = slm + SYCL_TOP_K_RADIX_HIST_SIZE; + uint32_t * s_row = slm + SYCL_TOP_K_RADIX_HIST_SIZE + 1; // prefix, mask, need + + // The previous launch is the barrier that publishes these, so a plain load is enough. + // One lane reads them and the group takes them from SLM: a device-scope atomic load + // is uncached here, and having every work-item issue three of them off the same + // address costs more than the whole sweep below. + if (tid == 0) { + s_row[0] = first ? 0u : state[SYCL_TOP_K_RADIX_ROW_PREFIX]; + s_row[1] = first ? 0u : state[SYCL_TOP_K_RADIX_ROW_MASK]; + s_row[2] = first ? (uint32_t) k : state[SYCL_TOP_K_RADIX_ROW_NEED]; + } + + for (int i = tid; i < SYCL_TOP_K_RADIX_HIST_SIZE; i += block_size) { + hist[i] = 0; + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + const uint32_t prefix = s_row[0]; + const uint32_t mask = s_row[1]; + const uint32_t need = s_row[2]; + + const int copy = tid & (SYCL_TOP_K_RADIX_HIST_COPIES - 1); + const int chunk = (ncols + nparts - 1) / nparts; + const int col0 = part * chunk; + const int col1 = std::min(ncols, col0 + chunk); + + for (int col = col0 + tid; col < col1; col += block_size) { + const uint32_t key = top_k_radix_key(src[col]); + if ((key & mask) == prefix) { + const uint32_t bucket = (key >> shift) & (SYCL_TOP_K_RADIX_BUCKETS - 1); + local_atomic(hist[bucket * SYCL_TOP_K_RADIX_HIST_COPIES + copy]).fetch_add(1u); + } + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + // One global atomic per bucket per group, not per element. + for (int b = tid; b < SYCL_TOP_K_RADIX_BUCKETS; b += block_size) { + uint32_t sum = 0; + for (int c = 0; c < SYCL_TOP_K_RADIX_HIST_COPIES; c++) { + sum += hist[b * SYCL_TOP_K_RADIX_HIST_COPIES + c]; + } + if (sum) { + top_k_radix_gatomic(state[b]).fetch_add(sum); + } + } + + // Publish this group's bins, then claim the scan if this group is the row's last. + // The group-wide barrier flushes the atomics above; only the claiming lane needs the + // release, so the device-scope fence is paid once per group rather than per work-item. + item_ct1.barrier(sycl::access::fence_space::global_and_local); + if (tid == 0) { + sycl::atomic_fence(sycl::memory_order::release, sycl::memory_scope::device); + sycl::atomic_ref done(state[SYCL_TOP_K_RADIX_ROW_DONE]); + *s_last = (done.fetch_add(1u) == (uint32_t) (nparts - 1)) ? 1u : 0u; + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + if (*s_last == 0u) { + return; + } + sycl::atomic_fence(sycl::memory_order::acquire, sycl::memory_scope::device); + + // Lane t takes bucket 255 - t, so an inclusive scan counts down from the top bucket. + uint32_t cnt = 0; + if (tid < SYCL_TOP_K_RADIX_BUCKETS) { + cnt = top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_BUCKETS - 1 - tid]).load(); + } + const uint32_t incl = sycl::inclusive_scan_over_group(item_ct1.get_group(), cnt, sycl::plus()); + + if (tid < SYCL_TOP_K_RADIX_BUCKETS && incl >= need && incl - cnt < need) { + const uint32_t digit = (uint32_t) (SYCL_TOP_K_RADIX_BUCKETS - 1 - tid); + top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_PREFIX]).store(prefix | (digit << shift)); + top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_MASK]).store( + mask | ((uint32_t) (SYCL_TOP_K_RADIX_BUCKETS - 1) << shift)); + top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_NEED]).store(need - (incl - cnt)); + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + // Clear for the next pass; the next launch is the barrier that orders this. + for (int b = tid; b < SYCL_TOP_K_RADIX_BUCKETS; b += block_size) { + top_k_radix_gatomic(state[b]).store(0u); + } + if (tid == 0) { + top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_DONE]).store(0u); + } +} + +static void top_k_radix_split_emit_f32( + const float * src, + int32_t * dst_idx, + uint32_t * state, + const int ncols, + const int k, + const int part, + const int nparts, + uint32_t * slm, + const sycl::nd_item<1> & item_ct1 +) { + using local_atomic = sycl::atomic_ref; + + const int tid = item_ct1.get_local_id(0); + const int block_size = item_ct1.get_local_range(0); + + uint32_t * s_gt = slm; + uint32_t * s_eq = slm + 1; + uint32_t * s_base_gt = slm + 2; + uint32_t * s_base_eq = slm + 3; + + uint32_t * s_row = slm + 4; // prefix, mask, need + + if (tid == 0) { + *s_gt = 0; + *s_eq = 0; + s_row[0] = state[SYCL_TOP_K_RADIX_ROW_PREFIX]; + s_row[1] = state[SYCL_TOP_K_RADIX_ROW_MASK]; + s_row[2] = state[SYCL_TOP_K_RADIX_ROW_NEED]; + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + const uint32_t prefix = s_row[0]; + const uint32_t mask = s_row[1]; + const uint32_t need = s_row[2]; + + // Exactly k - need columns beat the pivot; the first need pivot-equal ones fill the tail. + const uint32_t base_eq = (uint32_t) k - need; + + const int chunk = (ncols + nparts - 1) / nparts; + const int col0 = part * chunk; + const int col1 = std::min(ncols, col0 + chunk); + + // Counting first and reserving one range per group keeps the row's two counters out of + // the inner loop: a per-element global atomic on a single address serialises the whole + // emit, and at k in the thousands that alone outweighs every read the kernel does. + for (int col = col0 + tid; col < col1; col += block_size) { + const uint32_t kp = top_k_radix_key(src[col]) & mask; + if (kp > prefix) { + local_atomic(*s_gt).fetch_add(1u); + } else if (kp == prefix) { + local_atomic(*s_eq).fetch_add(1u); + } + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + if (tid == 0) { + const uint32_t n_gt = *s_gt; + const uint32_t n_eq = *s_eq; + *s_base_gt = n_gt ? top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_CNT_GT]).fetch_add(n_gt) : 0u; + *s_base_eq = n_eq ? top_k_radix_gatomic(state[SYCL_TOP_K_RADIX_ROW_CNT_EQ]).fetch_add(n_eq) : 0u; + *s_gt = 0; + *s_eq = 0; + } + item_ct1.barrier(sycl::access::fence_space::local_space); + + const uint32_t base_gt_g = *s_base_gt; + const uint32_t base_eq_g = *s_base_eq; + + for (int col = col0 + tid; col < col1; col += block_size) { + const uint32_t kp = top_k_radix_key(src[col]) & mask; + if (kp > prefix) { + dst_idx[base_gt_g + local_atomic(*s_gt).fetch_add(1u)] = col; + } else if (kp == prefix) { + const uint32_t pos = base_eq_g + local_atomic(*s_eq).fetch_add(1u); + if (pos < need) { + dst_idx[base_eq + pos] = col; + } + } + } +} + +static void top_k_radix_split_f32_sycl( + ggml_backend_sycl_context & ctx, + const float * src, + int32_t * dst_indices, + const int64_t ncols, + const int64_t nrows, + const int k, + const int nparts, + dpct::queue_ptr main_stream +) { + GGML_ASSERT(ncols <= INT32_MAX); + GGML_ASSERT(nparts > 1); + + const int block_size = ggml_sycl_info().max_work_group_sizes[ctx.device]; + GGML_ASSERT(block_size >= SYCL_TOP_K_RADIX_BUCKETS); + + const size_t state_words = (size_t) nrows * SYCL_TOP_K_RADIX_ROW_WORDS; + ggml_sycl_pool_alloc state_alloc(ctx.pool(), state_words); + uint32_t * state = state_alloc.get(); + + // Zero histogram, done counter and both emit counters. prefix/mask/need are seeded by + // the first pass, which ignores the stored values. + // The queue is in-order, so the passes below are already ordered after this fill. + SYCL_CHECK(CHECK_TRY_ERROR(main_stream->memset(state, 0, state_words * sizeof(uint32_t)))); + + const sycl::range<1> block_dims(block_size); + const sycl::range<1> grid_dims(nrows * nparts); + + bool first = true; + for (int shift = 32 - SYCL_TOP_K_RADIX_BITS; shift >= 0; shift -= SYCL_TOP_K_RADIX_BITS) { + const bool is_first = first; + first = false; + main_stream->submit([&](sycl::handler &cgh) { + sycl::local_accessor slm(sycl::range<1>(SYCL_TOP_K_RADIX_HIST_SIZE + 4), cgh); + + cgh.parallel_for( + sycl::nd_range<1>(grid_dims * block_dims, block_dims), + [=](sycl::nd_item<1> item_ct1) { + const int g = item_ct1.get_group(0); + const int row = g / nparts; + const int part = g % nparts; + + top_k_radix_split_pass_f32( + src + (int64_t) row * ncols, + state + (int64_t) row * SYCL_TOP_K_RADIX_ROW_WORDS, + (int) ncols, k, shift, is_first, part, nparts, + slm.get_multi_ptr().get(), + item_ct1); + }); + }); + } + + main_stream->submit([&](sycl::handler &cgh) { + sycl::local_accessor slm(sycl::range<1>(8), cgh); + + cgh.parallel_for( + sycl::nd_range<1>(grid_dims * block_dims, block_dims), + [=](sycl::nd_item<1> item_ct1) { + const int g = item_ct1.get_group(0); + const int row = g / nparts; + const int part = g % nparts; + + top_k_radix_split_emit_f32( + src + (int64_t) row * ncols, + dst_indices + (int64_t) row * k, + state + (int64_t) row * SYCL_TOP_K_RADIX_ROW_WORDS, + (int) ncols, k, part, nparts, + slm.get_multi_ptr().get(), + item_ct1); + }); + }); +} + +void ggml_sycl_top_k_radix( + ggml_backend_sycl_context & ctx, + const float * src, + int32_t * dst_indices, + const int64_t ncols, + const int64_t nrows, + const int k, + dpct::queue_ptr main_stream +) { + const int nparts = top_k_radix_split_groups(ctx.device, ncols, nrows); + if (nparts > 1) { + top_k_radix_split_f32_sycl(ctx, src, dst_indices, ncols, nrows, k, nparts, main_stream); + } else { + top_k_radix_f32_sycl(ctx, src, dst_indices, ncols, nrows, k, main_stream); + } +} diff --git a/ggml/src/ggml-sycl/topk-radix.hpp b/ggml/src/ggml-sycl/topk-radix.hpp new file mode 100644 index 00000000..db479607 --- /dev/null +++ b/ggml/src/ggml-sycl/topk-radix.hpp @@ -0,0 +1,24 @@ +#pragma once + +#include "common.hpp" + +// The legacy implementation uses SLM to implement sorting and top_k selection. +// SLM is limited to 128KB on Xe, which limits how much can be sorted to k<32. +// After a k=8, the radix selection becomes beneficial for most cases, because +// scan-merge has (block + 1) * k pairs of (value, index). Given normal sorting of nlog(n), +// radix-select becomes beneficial quite early. This sets it to 8 - however, the other parameters +// (columns and rows) may also be a driving factor. +// We select the legacy implementation for k below this constant because the overhead of radix select +// exceeds the benefit for very small problems +constexpr int SYCL_TOP_K_SCAN_MERGE_MAX_K = 8; + +// Top-k of every row of src, k indices per row into dst_indices, in no particular order. +// Picks between the one-group-per-row and the split-row kernel from the shape and the device. +void ggml_sycl_top_k_radix( + ggml_backend_sycl_context & ctx, + const float * src, + int32_t * dst_indices, + const int64_t ncols, + const int64_t nrows, + const int k, + dpct::queue_ptr main_stream); diff --git a/ggml/src/ggml-sycl/vecdotq.hpp b/ggml/src/ggml-sycl/vecdotq.hpp index c11a6e8f..909f7a78 100644 --- a/ggml/src/ggml-sycl/vecdotq.hpp +++ b/ggml/src/ggml-sycl/vecdotq.hpp @@ -351,6 +351,25 @@ template struct reorder_vec_dot_q_sycl { static_assert(T != T, "ggml_type for reorder vecdot not implemented"); }; +// For some types the weight side of the dot product does not depend on the destination column, so a +// multi-column mul_mat_vec can unpack it once per block instead of once per column. Such a type adds +// load() and dot() next to operator() and opts in here. See reorder_vec_dot_q_sycl. +template struct reorder_vec_dot_shared_weights { + static constexpr bool value = false; +}; + +template <> struct reorder_vec_dot_shared_weights { + static constexpr bool value = true; +}; + +template struct reorder_vec_dot_shared_activations { + static constexpr bool value = false; +}; + +template <> struct reorder_vec_dot_shared_activations { + static constexpr bool value = true; +}; + template <> struct reorder_vec_dot_q_sycl { static constexpr ggml_type gtype = GGML_TYPE_Q4_0; @@ -429,6 +448,39 @@ template <> struct reorder_vec_dot_q_sycl { } }; +template <> struct reorder_vec_dot_q_sycl { + static constexpr ggml_type gtype = GGML_TYPE_Q2_K; + + using q2_k_block = ggml_sycl_reordered::block_q_t; + using q2_k_traits = typename q2_k_block::traits; + + __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair ibx_offset, + const std::pair d_offset, const int8_t * q8_1_quant_ptr, + const sycl::half2 * q8_1_ds, const int & iqs) { + const uint8_t * base = static_cast(vbq); + const uint8_t * qs = base + ibx_offset.first; + const uint8_t * scales = base + d_offset.first; + const ggml_half2 * dm = reinterpret_cast(base + d_offset.second); + + const int bq8_offset = QR2_K * (iqs / QI8_1); + const int scale_offset = iqs - iqs % QI8_1 + (iqs % QI8_1) / (QI8_1 / 2); + + const int v = get_int_from_uint8_aligned(qs, iqs); + + int u[QR2_K]; + float d8[QR2_K]; + +#pragma unroll + for (int i = 0; i < QR2_K; ++i) { + const int8_t * quant_base_ptr = q8_1_quant_ptr + (bq8_offset + i) * QK8_1; + u[i] = get_int_from_int8_aligned(quant_base_ptr, iqs % QI8_1); + d8[i] = (*(q8_1_ds + bq8_offset + i))[0]; + } + + return vec_dot_q2_K_q8_1_impl_mmvq(v, u, scales + scale_offset, *dm, d8); + } +}; + template <> struct reorder_vec_dot_q_sycl { static constexpr ggml_type gtype = GGML_TYPE_Q3_K; @@ -507,50 +559,84 @@ template <> struct reorder_vec_dot_q_sycl { using q4_k_block = ggml_sycl_reordered::block_q_t; using q4_k_traits = typename q4_k_block::traits; - __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair ibx_offset, - const std::pair d_offset, const int8_t * q8_1_quant_ptr, - const sycl::half2 * q8_1_ds, const int & iqs) { - const uint8_t * base = static_cast(vbq); - const uint8_t * qs = base + ibx_offset.first; - const uint8_t * scs = base + d_offset.first; - const ggml_half2 * dms = reinterpret_cast(base + d_offset.second); - - const int bq8_offset = QR4_K * ((iqs / 2) / (QI8_1 / 2)); - const int * q4 = (const int *) (qs + 16 * bq8_offset + 4 * ((iqs / 2) % 4)); - const uint16_t * scales = (const uint16_t *) scs; + struct weights { + int v[2]; + uint16_t aux[2]; + ggml_half2 dm; + int bq8_offset; + }; - int v[2]; + struct activations { int u[2 * QR4_K]; float d8[QR4_K]; + }; - v[0] = q4[0]; - v[1] = q4[4]; + __dpct_inline__ static weights load(const void * __restrict__ vbq, const std::pair ibx_offset, + const std::pair d_offset, const int & iqs) { + const uint8_t * base = static_cast(vbq); + const uint8_t * qs = base + ibx_offset.first; + const uint8_t * scs = base + d_offset.first; + const ggml_half2 * dms = reinterpret_cast(base + d_offset.second); + + weights w; + w.bq8_offset = QR4_K * ((iqs / 2) / (QI8_1 / 2)); + + const int * q4 = (const int *) (qs + 16 * w.bq8_offset + 4 * ((iqs / 2) % 4)); + const uint16_t * scales = (const uint16_t *) scs; + + w.v[0] = q4[0]; + w.v[1] = q4[4]; - uint16_t aux[2]; const int j = (QR4_K * ((iqs / 2) / (QI8_1 / 2))) / 2; if (j < 2) { - aux[0] = scales[j + 0] & 0x3f3f; - aux[1] = scales[j + 2] & 0x3f3f; + w.aux[0] = scales[j + 0] & 0x3f3f; + w.aux[1] = scales[j + 2] & 0x3f3f; } else { - aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2); - aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2); + w.aux[0] = ((scales[j + 2] >> 0) & 0x0f0f) | ((scales[j - 2] & 0xc0c0) >> 2); + w.aux[1] = ((scales[j + 2] >> 4) & 0x0f0f) | ((scales[j - 0] & 0xc0c0) >> 2); } - const uint8_t * sc = (const uint8_t *) aux; - const uint8_t * m = sc + 2; + w.dm = *dms; + + return w; + } + __dpct_inline__ static activations load_activations(const int8_t * q8_1_quant_ptr, + const sycl::half2 * q8_1_ds, const int & iqs) { + activations a; + const int bq8_offset = QR4_K * ((iqs / 2) / (QI8_1 / 2)); for (int i = 0; i < QR4_K; ++i) { - const int8_t* quant_base_ptr = q8_1_quant_ptr + (bq8_offset + i) * QK8_1; - sycl::half2 ds_values = *(q8_1_ds + bq8_offset + i); + const int8_t * quant_base_ptr = q8_1_quant_ptr + (bq8_offset + i) * QK8_1; + sycl::half2 ds_values = *(q8_1_ds + bq8_offset + i); - d8[i] = ds_values[0]; + a.d8[i] = ds_values[0]; const int * q8 = (const int *) quant_base_ptr + ((iqs / 2) % 4); - u[2 * i + 0] = q8[0]; - u[2 * i + 1] = q8[4]; + a.u[2 * i + 0] = q8[0]; + a.u[2 * i + 1] = q8[4]; } - return vec_dot_q4_K_q8_1_impl_vmmq(v, u, sc, m, *dms, d8); + return a; + } + + __dpct_inline__ static float apply(const weights & w, const activations & a) { + const uint8_t * sc = (const uint8_t *) w.aux; + const uint8_t * m = sc + 2; + + return vec_dot_q4_K_q8_1_impl_vmmq(w.v, a.u, sc, m, w.dm, a.d8); + } + + __dpct_inline__ static float dot(const weights & w, const int8_t * q8_1_quant_ptr, + const sycl::half2 * q8_1_ds, const int & iqs) { + const auto a = load_activations(q8_1_quant_ptr, q8_1_ds, iqs); + + return apply(w, a); + } + + __dpct_inline__ float operator()(const void * __restrict__ vbq, const std::pair ibx_offset, + const std::pair d_offset, const int8_t * q8_1_quant_ptr, + const sycl::half2 * q8_1_ds, const int & iqs) { + return dot(load(vbq, ibx_offset, d_offset, iqs), q8_1_quant_ptr, q8_1_ds, iqs); } }; @@ -1312,6 +1398,11 @@ vec_dot_q6_K_q8_1(const void *__restrict__ vbq, } +// NOTE: the VDR_IQ*_Q8_1_MMVQ values deliberately differ from the identically named CUDA constants +// (vecdotq.cuh): the SYCL kernels pair them with a halved qi (e.g. QI3_S/2), so the values are not +// interchangeable and must not be copied across backends. +#define VDR_IQ2_XXS_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq2_xxs_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs, @@ -1343,6 +1434,8 @@ vec_dot_iq2_xxs_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ2_XS_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq2_xs_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs, @@ -1393,6 +1486,8 @@ vec_dot_iq2_xs_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ2_S_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq2_s_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs) { @@ -1445,6 +1540,8 @@ vec_dot_iq2_s_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ3_XXS_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq3_xxs_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs, @@ -1485,6 +1582,8 @@ vec_dot_iq3_xxs_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ3_S_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq3_s_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs, @@ -1523,6 +1622,8 @@ vec_dot_iq3_s_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ1_S_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq1_s_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs, @@ -1551,6 +1652,8 @@ vec_dot_iq1_s_q8_1(const void *__restrict__ vbq, #endif } +#define VDR_IQ1_M_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq1_m_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs) { @@ -1585,6 +1688,8 @@ vec_dot_iq1_m_q8_1(const void *__restrict__ vbq, } +#define VDR_IQ4_NL_Q8_1_MMVQ 2 + static __dpct_inline__ float vec_dot_iq4_nl_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs) { @@ -1610,6 +1715,8 @@ vec_dot_iq4_nl_q8_1(const void *__restrict__ vbq, } +#define VDR_IQ4_XS_Q8_1_MMVQ 1 + static __dpct_inline__ float vec_dot_iq4_xs_q8_1(const void *__restrict__ vbq, const block_q8_1 *__restrict__ bq8_1, const int &iqs) { diff --git a/ggml/src/ggml-version.h.in b/ggml/src/ggml-version.h.in new file mode 100644 index 00000000..37de3629 --- /dev/null +++ b/ggml/src/ggml-version.h.in @@ -0,0 +1,4 @@ +#pragma once + +#define GGML_VERSION "@GGML_VERSION@" +#define GGML_COMMIT "@GGML_BUILD_COMMIT@" diff --git a/ggml/src/ggml-virtgpu/ggml-backend.cpp b/ggml/src/ggml-virtgpu/ggml-backend.cpp index 12756c92..996c57e3 100644 --- a/ggml/src/ggml-virtgpu/ggml-backend.cpp +++ b/ggml/src/ggml-virtgpu/ggml-backend.cpp @@ -17,7 +17,8 @@ static ggml_status ggml_backend_remoting_graph_compute(ggml_backend_t backend, g return apir_backend_graph_compute(gpu, cgraph); } -static void ggml_backend_remoting_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph) { +static void ggml_backend_remoting_graph_optimize(ggml_backend_t backend, ggml_cgraph * cgraph, ggml_backend_graph_optimize_params * params) { + UNUSED(params); virtgpu * gpu = DEV_TO_GPU(backend->device); #if true UNUSED(gpu); diff --git a/ggml/src/ggml-vulkan/CMakeLists.txt b/ggml/src/ggml-vulkan/CMakeLists.txt index 1dc6a145..d7dec92a 100644 --- a/ggml/src/ggml-vulkan/CMakeLists.txt +++ b/ggml/src/ggml-vulkan/CMakeLists.txt @@ -62,8 +62,18 @@ if (Vulkan_FOUND) ggml_add_backend_library(ggml-vulkan ggml-vulkan.cpp ../../include/ggml-vulkan.h + ggml-vulkan-types.h + ggml-vulkan-push-constants.h + ggml-vulkan-common.h + ggml-vulkan-buffers.cpp + ggml-vulkan-debug.cpp ) + # hide symbols, so dlclosed duplicate copies cannot interpose them + set_target_properties(ggml-vulkan PROPERTIES + CXX_VISIBILITY_PRESET hidden + VISIBILITY_INLINES_HIDDEN ON) + set(VULKAN_SHADER_GEN_CMAKE_ARGS "") # Test all shader extensions @@ -200,8 +210,11 @@ if (Vulkan_FOUND) set (_ggml_vk_header "${CMAKE_CURRENT_BINARY_DIR}/ggml-vulkan-shaders.hpp") set (_ggml_vk_input_dir "${CMAKE_CURRENT_SOURCE_DIR}/vulkan-shaders") set (_ggml_vk_output_dir "${CMAKE_CURRENT_BINARY_DIR}/vulkan-shaders.spv") + set (_ggml_vk_generated_shader_files ${_ggml_vk_header}) file(GLOB _ggml_vk_shader_files CONFIGURE_DEPENDS "${_ggml_vk_input_dir}/*.comp") + set_source_files_properties(${_ggml_vk_shader_files} PROPERTIES HEADER_FILE_ONLY TRUE) + target_sources(ggml-vulkan PRIVATE ${_ggml_vk_shader_files}) # Because external projects do not provide source-level tracking, # the vulkan-shaders-gen sources need to be explicitly added to @@ -241,8 +254,11 @@ if (Vulkan_FOUND) COMMENT "Generate vulkan shaders for ${file}" ) target_sources(ggml-vulkan PRIVATE ${_ggml_vk_target_cpp}) + list(APPEND _ggml_vk_generated_shader_files ${_ggml_vk_target_cpp}) endforeach() + source_group("Vulkan shaders" FILES ${_ggml_vk_shader_files}) + source_group("Generated Vulkan shaders" FILES ${_ggml_vk_generated_shader_files}) else() message(WARNING "Vulkan not found") endif() diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp new file mode 100644 index 00000000..4d4c8495 --- /dev/null +++ b/ggml/src/ggml-vulkan/ggml-vulkan-buffers.cpp @@ -0,0 +1,783 @@ +#include "ggml-vulkan-common.h" + +ggml_backend_buffer_type_i ggml_backend_vk_buffer_type_interface = { + /* .get_name = */ ggml_backend_vk_buffer_type_name, + /* .alloc_buffer = */ ggml_backend_vk_buffer_type_alloc_buffer, + /* .get_alignment = */ ggml_backend_vk_buffer_type_get_alignment, + /* .get_max_size = */ ggml_backend_vk_buffer_type_get_max_size, + /* .get_alloc_size = */ ggml_backend_vk_buffer_type_get_alloc_size, + /* .is_host = */ NULL, +}; + +static std::vector ggml_vk_find_memory_properties(const vk::PhysicalDeviceMemoryProperties* mem_props, vk::MemoryRequirements* mem_req, vk::MemoryPropertyFlags flags) { + std::vector indices; + + for (uint32_t i = 0; i < mem_props->memoryTypeCount; ++i) { + vk::MemoryType memory_type = mem_props->memoryTypes[i]; + if ((mem_req->memoryTypeBits & ((uint64_t)1 << i)) && + (flags & memory_type.propertyFlags) == flags && + mem_props->memoryHeaps[memory_type.heapIndex].size >= mem_req->size) { + indices.push_back(i); + } + } + return indices; +} + +static vk_buffer ggml_vk_create_buffer(vk_device& device, size_t size, const std::initializer_list & req_flags_list, + void *import_ptr = nullptr) { + VK_LOG_DEBUG("ggml_vk_create_buffer(" << device->name << ", " << size << ", " << to_string(req_flags_list.begin()[0]) << ", " << to_string(req_flags_list.begin()[req_flags_list.size()-1]) << ")"); + if (size > device->max_buffer_size) { + throw vk::OutOfDeviceMemoryError("Requested buffer size exceeds device buffer size limit"); + } + + vk_buffer buf = std::make_shared(); + + if (size == 0) { + buf->size = 0; + return buf; + } + + vk::BufferUsageFlags usage_flags = vk::BufferUsageFlagBits::eStorageBuffer | vk::BufferUsageFlagBits::eTransferSrc | vk::BufferUsageFlagBits::eTransferDst; + vk::MemoryAllocateFlags mem_flags {}; + if (device->buffer_device_address) { + usage_flags |= vk::BufferUsageFlagBits::eShaderDeviceAddress; + mem_flags |= vk::MemoryAllocateFlagBits::eDeviceAddress; + } + + vk::BufferCreateInfo buffer_create_info{ + vk::BufferCreateFlags(), + size, + usage_flags, + vk::SharingMode::eExclusive, + 0, + nullptr, + }; + + vk::ExternalMemoryBufferCreateInfo external_memory_bci; + if (import_ptr) { + external_memory_bci.handleTypes = vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT; + buffer_create_info.setPNext(&external_memory_bci); + } + + buf->buffer = device->device.createBuffer(buffer_create_info); + + vk::MemoryRequirements mem_req = device->device.getBufferMemoryRequirements(buf->buffer); + + vk::PhysicalDeviceMemoryProperties mem_props = device->physical_device.getMemoryProperties(); + + const vk::MemoryPriorityAllocateInfoEXT mem_priority_info { 1.0f }; + + vk::MemoryAllocateFlagsInfo mem_flags_info { mem_flags }; + + if (device->memory_priority) { + mem_flags_info.setPNext(&mem_priority_info); + } + + if (import_ptr) { + vk::MemoryHostPointerPropertiesEXT host_pointer_props; + try { + host_pointer_props = device->device.getMemoryHostPointerPropertiesEXT(vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT, import_ptr); + } catch (vk::SystemError& e) { + GGML_LOG_WARN("ggml_vulkan: Failed getMemoryHostPointerPropertiesEXT (%s)\n", e.what()); + device->device.destroyBuffer(buf->buffer); + return {}; + } + vk::PhysicalDeviceMemoryProperties mem_props = device->physical_device.getMemoryProperties(); + + uint32_t memory_type_idx; + vk::MemoryPropertyFlags property_flags = *req_flags_list.begin(); + for (memory_type_idx = 0; memory_type_idx < 32; ++memory_type_idx) { + if (!(host_pointer_props.memoryTypeBits & (1u << memory_type_idx))) { + continue; + } + if (!(mem_req.memoryTypeBits & (1u << memory_type_idx))) { + continue; + } + + vk::MemoryType memory_type = mem_props.memoryTypes[memory_type_idx]; + // check for visible+coherent+cached. Other flags (e.g. devicelocal) are allowed + if ((memory_type.propertyFlags & property_flags) == property_flags) { + property_flags = memory_type.propertyFlags; + break; + } + } + if (memory_type_idx == 32) { + GGML_LOG_WARN("ggml_vulkan: Memory type for host allocation not found\n"); + device->device.destroyBuffer(buf->buffer); + return {}; + } + + buf->memory_property_flags = mem_props.memoryTypes[memory_type_idx].propertyFlags; + try { + vk::ImportMemoryHostPointerInfoEXT import_info; + import_info.handleType = vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT; + import_info.pHostPointer = import_ptr; + import_info.setPNext(&mem_flags_info); + buf->device_memory = device->device.allocateMemory({ size, memory_type_idx, &import_info }); + } catch (const vk::SystemError& e) { + } + } else { + for (auto it = req_flags_list.begin(); it != req_flags_list.end(); it++) { + const auto & req_flags = *it; + + const std::vector memory_type_indices = ggml_vk_find_memory_properties(&mem_props, &mem_req, req_flags); + + if (memory_type_indices.empty()) { + continue; + } + + bool done = false; + + for (auto mtype_it = memory_type_indices.begin(); mtype_it != memory_type_indices.end(); mtype_it++) { + try { + buf->device_memory = device->device.allocateMemory({ mem_req.size, *mtype_it, &mem_flags_info }); + buf->memory_property_flags = mem_props.memoryTypes[*mtype_it].propertyFlags; + done = true; + break; + } catch (const vk::SystemError& e) { + // loop and retry + // during last attempt throw the exception + if (it + 1 == req_flags_list.end() && mtype_it + 1 == memory_type_indices.end()) { + device->device.destroyBuffer(buf->buffer); + throw e; + } + } + } + + if (done) { + break; + } + } + } + + if (!buf->device_memory) { + device->device.destroyBuffer(buf->buffer); + throw vk::OutOfDeviceMemoryError("No suitable memory type found"); + } + + buf->ptr = nullptr; + + if (import_ptr) { + buf->ptr = import_ptr; + } else { + if (buf->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { + buf->ptr = device->device.mapMemory(buf->device_memory, 0, VK_WHOLE_SIZE); + } + } + + device->device.bindBufferMemory(buf->buffer, buf->device_memory, 0); + + buf->device = device; + buf->size = size; + + if (device->buffer_device_address) { + const vk::BufferDeviceAddressInfo addressInfo(buf->buffer); + buf->bda_addr = device->device.getBufferAddress(addressInfo); + } + + device->memory_logger->log_allocation(buf, size); + + return buf; +} + +vk_buffer ggml_vk_create_buffer_check(vk_device& device, size_t size, vk::MemoryPropertyFlags req_flags, vk::MemoryPropertyFlags fallback_flags) { + try { + return ggml_vk_create_buffer(device, size, {req_flags, fallback_flags}); + } catch (const vk::SystemError& e) { + std::cerr << "ggml_vulkan: Memory allocation of size " << size << " failed." << std::endl; + std::cerr << "ggml_vulkan: " << e.what() << std::endl; + throw e; + } +} + +vk_buffer ggml_vk_create_buffer_device(vk_device& device, size_t size) { + vk_buffer buf; + try { + if (device->prefer_host_memory) { + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, + vk::MemoryPropertyFlagBits::eDeviceLocal}); + } else if (device->uma) { + // On UMA, prefer host-visible memory so direct tensor borrowing works. + // If unavailable, fall back to device-local memory. + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, + vk::MemoryPropertyFlagBits::eDeviceLocal, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + } else if (device->disable_host_visible_vidmem) { + if (device->allow_sysmem_fallback) { + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + } else { + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + } + } else { + // use rebar if available, otherwise fallback to device only visible memory + if (device->allow_sysmem_fallback) { + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, + vk::MemoryPropertyFlagBits::eDeviceLocal, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + } else { + buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, + vk::MemoryPropertyFlagBits::eDeviceLocal}); + } + } + } catch (const vk::SystemError& e) { + std::cerr << "ggml_vulkan: Device memory allocation of size " << size << " failed." << std::endl; + std::cerr << "ggml_vulkan: " << e.what() << std::endl; + throw e; + } + + return buf; +} + +void ggml_vk_destroy_buffer(vk_buffer& buf) { + if (buf == nullptr) { + return; + } + + if (buf->device != nullptr) { + buf->device->memory_logger->log_deallocation(buf); + } + + buf.reset(); +} + +void * ggml_vk_host_malloc(vk_device& device, size_t size) { + VK_LOG_MEMORY("ggml_vk_host_malloc(" << size << ")"); + vk_buffer buf = ggml_vk_create_buffer(device, size, + {vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + + if(!(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible)) { + fprintf(stderr, "WARNING: failed to allocate %.2f MB of pinned memory\n", + size/1024.0/1024.0); + device->device.freeMemory(buf->device_memory); + device->device.destroyBuffer(buf->buffer); + return nullptr; + } + + std::lock_guard guard(device->pinned_memory_mutex); + device->pinned_memory.push_back(std::make_tuple(buf->ptr, size, buf)); + + return buf->ptr; +} + +void ggml_vk_host_free(vk_device& device, void* ptr) { + if (ptr == nullptr) { + return; + } + VK_LOG_MEMORY("ggml_vk_host_free(" << ptr << ")"); + std::lock_guard guard(device->pinned_memory_mutex); + + vk_buffer buf; + size_t index; + for (size_t i = 0; i < device->pinned_memory.size(); i++) { + const uint8_t* addr = (const uint8_t*) std::get<0>(device->pinned_memory[i]); + const uint8_t* endr = addr + std::get<1>(device->pinned_memory[i]); + if (ptr >= addr && ptr < endr) { + buf = std::get<2>(device->pinned_memory[i]); + index = i; + break; + } + } + if (buf == nullptr) { + fprintf(stderr, "WARNING: failed to free pinned memory: memory not in map\n"); + return; + } + + ggml_vk_destroy_buffer(buf); + + device->pinned_memory.erase(device->pinned_memory.begin() + index); +} + +void ggml_vk_host_get(const vk_device& device, const void * ptr, vk_buffer& buf, size_t& buf_offset) { + std::shared_lock guard(device->pinned_memory_mutex); + buf = nullptr; + buf_offset = 0; + for (size_t i = 0; i < device->pinned_memory.size(); i++) { + const uint8_t* addr = (const uint8_t*) std::get<0>(device->pinned_memory[i]); + const uint8_t* endr = addr + std::get<1>(device->pinned_memory[i]); + if (ptr >= addr && ptr < endr) { + buf = std::get<2>(device->pinned_memory[i]); + buf_offset = ((const uint8_t *)ptr) - addr; + break; + } + } +} + +void ggml_vk_ensure_sync_staging_buffer(vk_device& device, size_t size) { + if (device->sync_staging == nullptr || device->sync_staging->size < size) { + VK_LOG_MEMORY("ggml_vk_ensure_sync_staging_buffer(" << size << ")"); + ggml_vk_destroy_buffer(device->sync_staging); + device->sync_staging = ggml_vk_create_buffer_check(device, size, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent); + } +} + +void ggml_vk_ensure_sync_staging_buffer(ggml_backend_vk_context * ctx, size_t size) { + if (ctx->sync_staging == nullptr || ctx->sync_staging->size < size) { + VK_LOG_MEMORY("ggml_vk_ensure_sync_staging_buffer(" << size << ")"); + ggml_vk_destroy_buffer(ctx->sync_staging); + ctx->sync_staging = ggml_vk_create_buffer_check(ctx->device, size, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, + vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent); + } +} + +static void ggml_vk_buffer_write_nc_async(ggml_backend_vk_context * ctx, vk_context& subctx, vk_buffer& dst, size_t offset, const ggml_tensor * tensor, bool sync_staging = false) { + VK_LOG_DEBUG("ggml_vk_buffer_write_nc_async(" << tensor << ")"); + GGML_ASSERT(!ggml_is_contiguous(tensor)); + // Buffer is already mapped + if(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { + std::cerr << "ggml_vulkan: buffer_write_nc_async dst buffer is host_visible. Use synchronous write." << std::endl; + GGML_ABORT("fatal error"); + } + // Check if src is pinned memory + vk_buffer buf = nullptr; + size_t buf_offset = 0; + ggml_vk_host_get(ctx->device, tensor->data, buf, buf_offset); + + const uint64_t ne0 = tensor->ne[0]; + const uint64_t ne1 = tensor->ne[1]; + const uint64_t ne2 = tensor->ne[2]; + const uint64_t ne3 = tensor->ne[3]; + const uint64_t nb0 = tensor->nb[0]; + const uint64_t nb1 = tensor->nb[1]; + const uint64_t nb2 = tensor->nb[2]; + const uint64_t nb3 = tensor->nb[3]; + const ggml_type type = tensor->type; + const uint64_t ts = ggml_type_size(type); + const uint64_t bs = ggml_blck_size(type); + + const uint64_t dstnb0 = ts; + const uint64_t dstnb1 = dstnb0*(ne0/bs); + const uint64_t dstnb2 = dstnb1*ne1; + const uint64_t dstnb3 = dstnb2*ne2; + + const uint64_t ne = ggml_nelements(tensor); + + if (buf != nullptr) { + // Memory is pinned, use as staging buffer + std::vector slices; + + for (uint64_t i3 = 0; i3 < ne3; i3++) { + for (uint64_t i2 = 0; i2 < ne2; i2++) { + // Find longest contiguous slice + if (ne1*nb1 == dstnb2) { + slices.push_back({ buf_offset + i3*nb3 + i2*nb2, offset + i3*dstnb3 + i2*dstnb2, dstnb2 }); + } else { + for (uint64_t i1 = 0; i1 < ne1; i1++) { + if (ne0*nb0/bs == dstnb1) { + slices.push_back({ buf_offset + i3*nb3 + i2*nb2 + i1*nb1, offset + i3*dstnb3 + i2*dstnb2 + i1*dstnb1, dstnb1 }); + } else { + const uint64_t s_off = buf_offset + i3*nb3 + i2*nb2 + i1*nb1; + const uint64_t d_off = offset + i3*dstnb3 + i2*dstnb2 + i1*dstnb1; + for (uint64_t i0 = 0; i0 < ne0; i0++) { + slices.push_back({ s_off + i0*nb0, d_off + i0*dstnb0, dstnb0 }); + } + } + } + } + } + } + + ggml_vk_sync_buffers(ctx, subctx); + subctx->s->buffer->buf.copyBuffer(buf->buffer, dst->buffer, slices); + return; + } + + if (!sync_staging) { + GGML_ABORT("Asynchronous write to non-pinned memory not supported"); + } + + // Staging buffer required + vk_buffer& staging = ctx->device->sync_staging; + const uint64_t copy_size = ts*ne/bs; + ggml_vk_ensure_sync_staging_buffer(ctx->device, copy_size); + VkBufferCopy buf_copy{ 0, offset, copy_size }; + + ggml_vk_sync_buffers(ctx, subctx); + vkCmdCopyBuffer(subctx->s->buffer->buf, (VkBuffer)staging->buffer, (VkBuffer)dst->buffer, 1, &buf_copy); + + for (uint64_t i3 = 0; i3 < ne3; i3++) { + for (uint64_t i2 = 0; i2 < ne2; i2++) { + // Find longest contiguous slice + if (ne1*nb1 == dstnb2) { + deferred_memcpy((uint8_t *)staging->ptr + i3*dstnb3 + i2*dstnb2, (const uint8_t *) tensor->data + buf_offset + i3*nb3 + i2*nb2, dstnb2, &subctx->in_memcpys); + } else { + for (uint64_t i1 = 0; i1 < ne1; i1++) { + if (ne0*nb0/bs == dstnb1) { + deferred_memcpy((uint8_t *)staging->ptr + i3*dstnb3 + i2*dstnb2 + i1*dstnb1, (const uint8_t *) tensor->data + buf_offset + i3*nb3 + i2*nb2 + i1*nb1, dstnb1, &subctx->in_memcpys); + } else { + const uint64_t s_off = buf_offset + i3*nb3 + i2*nb2 + i1*nb1; + const uint64_t d_off = i3*dstnb3 + i2*dstnb2 + i1*dstnb1; + for (uint64_t i0 = 0; i0 < ne0; i0++) { + deferred_memcpy((uint8_t *)staging->ptr + d_off + i0*dstnb0, (const uint8_t *) tensor->data + s_off + i0*nb0, dstnb0, &subctx->in_memcpys); + } + } + } + } + } + } +} + +bool ggml_vk_buffer_write_2d_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging) { + VK_LOG_DEBUG("ggml_vk_buffer_write_2d_async(" << width << ", " << height << ")"); + // Check if src is pinned memory + vk_buffer buf = nullptr; + size_t buf_offset = 0; + ggml_vk_host_get(dst->device, src, buf, buf_offset); + + if (buf != nullptr) { + // Memory is pinned, use as staging buffer + std::vector slices(1); + if (width == spitch && width == dpitch) { + // Only do single write if stride is equal + slices[0].srcOffset = buf_offset; + slices[0].dstOffset = offset; + slices[0].size = width * height; + } else { + slices.resize(height); + for (size_t i = 0; i < height; i++) { + slices[i].srcOffset = buf_offset + i * spitch; + slices[i].dstOffset = offset + i * dpitch; + slices[i].size = width; + } + } + + ggml_vk_sync_buffers(nullptr, subctx); + subctx->s->buffer->buf.copyBuffer(buf->buffer, dst->buffer, slices); + return true; + } + VK_LOG_DEBUG("STAGING"); + + if (!sync_staging) { + // copy was not handled caller needs to fall back + return false; + } + + // Staging buffer required + const size_t staging_size = width * height; + ggml_vk_ensure_sync_staging_buffer(dst->device, staging_size); + + vk_buffer& staging_buffer = dst->device->sync_staging; + + std::vector slices(1); + if (width == dpitch) { + slices[0].srcOffset = 0; + slices[0].dstOffset = offset; + slices[0].size = staging_size; + } else { + slices.resize(height); + for (size_t i = 0; i < height; i++) { + slices[i].srcOffset = i * width; + slices[i].dstOffset = offset + i * dpitch; + slices[i].size = width; + } + } + + ggml_vk_sync_buffers(nullptr, subctx); + subctx->s->buffer->buf.copyBuffer(staging_buffer->buffer, dst->buffer, slices); + + if (width == spitch) { + deferred_memcpy((uint8_t *)staging_buffer->ptr, src, staging_size, &subctx->in_memcpys); + } else { + for (size_t i = 0; i < height; i++) { + deferred_memcpy((uint8_t *)staging_buffer->ptr + i * width, (const uint8_t *) src + i * spitch, width, &subctx->in_memcpys); + } + } + return true; +} + +bool ggml_vk_buffer_write_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t size, bool sync_staging) { + VK_LOG_DEBUG("ggml_vk_buffer_write_async(" << size << ")"); + return ggml_vk_buffer_write_2d_async(subctx, dst, offset, src, size, size, size, 1, sync_staging); +} + +void ggml_vk_buffer_write_2d(vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height) { + VK_LOG_DEBUG("ggml_vk_buffer_write_2d(" << width << ", " << height << ")"); + // Buffer is already mapped + if(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { + GGML_ASSERT(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent); + + if (width == spitch && width == dpitch) { + memcpy((uint8_t *)dst->ptr + offset, src, width * height); + } else { + for (size_t i = 0; i < height; i++) { + memcpy((uint8_t *)dst->ptr + offset + i * dpitch, (const uint8_t *) src + i * spitch, width); + } + } + } else { + std::lock_guard guard(dst->device->mutex); + + vk_context subctx = ggml_vk_create_temporary_context(dst->device->transfer_queue->cmd_pool); + ggml_vk_ctx_begin(dst->device, subctx); + bool ret = ggml_vk_buffer_write_2d_async(subctx, dst, offset, src, spitch, dpitch, width, height, true); + GGML_ASSERT(ret); + ggml_vk_ctx_end(subctx); + + for (auto& cpy : subctx->in_memcpys) { + memcpy(cpy.dst, cpy.src, cpy.n); + } + + for (auto& mset : subctx->memsets) { + memset(mset.dst, mset.val, mset.n); + } + + ggml_vk_submit(subctx, dst->device->fence); + VK_CHECK(dst->device->device.waitForFences({ dst->device->fence }, true, UINT64_MAX), "vk_buffer_write_2d waitForFences", dst->device); + dst->device->device.resetFences({ dst->device->fence }); + ggml_vk_queue_command_pools_cleanup(dst->device); + } +} + +void ggml_vk_buffer_write(vk_buffer& dst, size_t offset, const void * src, size_t size) { + VK_LOG_DEBUG("ggml_vk_buffer_write(" << size << ")"); + ggml_vk_buffer_write_2d(dst, offset, src, size, size, size, 1); +} + +bool ggml_vk_buffer_read_2d_async(vk_context subctx, vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging) { + VK_LOG_DEBUG("ggml_vk_buffer_read_2d_async(offset=" << offset << ", width=" << width << ", height=" << height << ")"); + GGML_ASSERT(width > 0); + GGML_ASSERT(height > 0); + GGML_ASSERT(src != nullptr); + + // TODO: staging_offset is not used + + // Check if dst is pinned memory + vk_buffer buf = nullptr; + size_t buf_offset = 0; + ggml_vk_host_get(src->device, dst, buf, buf_offset); + + std::vector slices(1); + if (width == spitch && width == dpitch) { + // Only do single write if stride is equal + slices[0].srcOffset = offset; + slices[0].dstOffset = buf_offset; + slices[0].size = width * height; + } else { + slices.resize(height); + for (size_t i = 0; i < height; i++) { + slices[i].srcOffset = offset + i * spitch; + slices[i].dstOffset = buf_offset + i * dpitch; + slices[i].size = width; + } + } + + if (buf != nullptr) { + // Memory is pinned, use as staging buffer + ggml_vk_sync_buffers(nullptr, subctx); + subctx->s->buffer->buf.copyBuffer(src->buffer, buf->buffer, slices); + + return true; + } + VK_LOG_DEBUG("STAGING"); + + if (!sync_staging) { + // copy was not handled caller needs to fall back + return false; + } + + // Fall back to staging buffer + const size_t staging_size = width * height; + ggml_vk_ensure_sync_staging_buffer(src->device, staging_size); + + vk_buffer& staging_buffer = src->device->sync_staging; + + std::vector staging_slices(1); + if (width == spitch) { + staging_slices[0].srcOffset = offset; + staging_slices[0].dstOffset = 0; + staging_slices[0].size = staging_size; + } else { + staging_slices.resize(height); + for (size_t i = 0; i < height; i++) { + staging_slices[i].srcOffset = offset + i * spitch; + staging_slices[i].dstOffset = i * width; + staging_slices[i].size = width; + } + } + + ggml_vk_sync_buffers(nullptr, subctx); + subctx->s->buffer->buf.copyBuffer(src->buffer, staging_buffer->buffer, staging_slices); + + if (width == dpitch) { + deferred_memcpy(dst, staging_buffer->ptr, staging_size, &subctx->out_memcpys); + } else { + for (size_t i = 0; i < height; i++) { + deferred_memcpy((uint8_t *) dst + i * dpitch, (const uint8_t *) staging_buffer->ptr + i * width, width, &subctx->out_memcpys); + } + } + return true; +} + +static bool ggml_vk_buffer_read_async(vk_context subctx, vk_buffer& src, size_t offset, void * dst, size_t size, bool sync_staging = false) { + return ggml_vk_buffer_read_2d_async(subctx, src, offset, dst, size, size, size, 1, sync_staging); +} + +void ggml_vk_buffer_read_2d(vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height) { + VK_LOG_DEBUG("ggml_vk_buffer_read_2d(" << src->buffer << ", " << offset << ", " << width << ", " << height << ")"); + + // If the device is not an UMA device the memory is host-accessible through rebar. While writing + // through PCIe is sufficient fast reading back data from PCIe is slower than going through + // the HW device to host copy path. + if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma) { + GGML_ASSERT(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent); + + std::lock_guard guard(src->device->mutex); + vk_context subctx = ggml_vk_create_temporary_context(src->device->compute_queue->cmd_pool); + ggml_vk_ctx_begin(src->device, subctx); + subctx->s->buffer->buf.pipelineBarrier( + vk::PipelineStageFlagBits::eComputeShader | vk::PipelineStageFlagBits::eTransfer, + vk::PipelineStageFlagBits::eHost, + {}, + { { vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferWrite, + vk::AccessFlagBits::eHostRead } }, + {}, {}); + ggml_vk_ctx_end(subctx); + ggml_vk_submit(subctx, src->device->fence); + VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), + "vk_buffer_read_2d uma waitForFences", src->device); + src->device->device.resetFences({ src->device->fence }); + ggml_vk_queue_command_pools_cleanup(src->device); + + if (width == spitch && width == dpitch) { + memcpy(dst, (const uint8_t *) src->ptr + offset, width * height); + } else { + for (size_t i = 0; i < height; i++) { + memcpy((uint8_t *) dst + i * dpitch, (const uint8_t *) src->ptr + offset + i * spitch, width); + } + } + } else { + std::lock_guard guard(src->device->mutex); + + vk_context subctx = ggml_vk_create_temporary_context(src->device->transfer_queue->cmd_pool); + ggml_vk_ctx_begin(src->device, subctx); + bool ret = ggml_vk_buffer_read_2d_async(subctx, src, offset, dst, spitch, dpitch, width, height, true); + GGML_ASSERT(ret); + ggml_vk_ctx_end(subctx); + + ggml_vk_submit(subctx, src->device->fence); + VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), "vk_buffer_read_2d waitForFences", src->device); + src->device->device.resetFences({ src->device->fence }); + ggml_vk_queue_command_pools_cleanup(src->device); + + for (auto& cpy : subctx->out_memcpys) { + memcpy(cpy.dst, cpy.src, cpy.n); + } + } +} + +void ggml_vk_buffer_read(vk_buffer& src, size_t offset, void * dst, size_t size) { + VK_LOG_DEBUG("ggml_vk_buffer_read(" << src->buffer << ", " << offset << ", " << size << ")"); + ggml_vk_buffer_read_2d(src, offset, dst, size, size, size, 1); +} + +void ggml_vk_buffer_copy_async(vk_context& ctx, vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size) { + VK_LOG_DEBUG("ggml_vk_buffer_copy_async(" << size << ")"); + // Make sure both buffers are on same device + GGML_ASSERT(src->device == dst->device); + + VkBufferCopy bc{ src_offset, dst_offset, size }; + + vkCmdCopyBuffer(ctx->s->buffer->buf, (VkBuffer)src->buffer, (VkBuffer)dst->buffer, 1, &bc); +} + +void ggml_vk_buffer_copy(vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size) { + if (src->device == dst->device) { + std::lock_guard guard(src->device->mutex); + VK_LOG_DEBUG("ggml_vk_buffer_copy(SINGLE_DEVICE, " << size << ")"); + // Copy within the device + vk_context subctx = ggml_vk_create_temporary_context(src->device->transfer_queue->cmd_pool); + ggml_vk_ctx_begin(src->device, subctx); + ggml_vk_buffer_copy_async(subctx, dst, dst_offset, src, src_offset, size); + ggml_vk_ctx_end(subctx); + ggml_vk_submit(subctx, src->device->fence); + VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), "vk_buffer_copy waitForFences", src->device); + src->device->device.resetFences({ src->device->fence }); + ggml_vk_queue_command_pools_cleanup(src->device); + } else { + VK_LOG_DEBUG("ggml_vk_buffer_copy(MULTI_DEVICE, " << size << ")"); + // Copy device to device + ggml_vk_ensure_sync_staging_buffer(src->device, size); + + // Copy to src staging buffer + ggml_vk_buffer_copy(src->device->sync_staging, 0, src, src_offset, size); + // Copy to dst buffer + ggml_vk_buffer_write(dst, dst_offset, src->device->sync_staging->ptr, size); + } +} + +void ggml_vk_buffer_memset_async(vk_context& ctx, vk_buffer& dst, size_t offset, uint32_t c, size_t size) { + VK_LOG_DEBUG("ggml_vk_buffer_memset_async(" << offset << ", " << c << ", " << size << ")"); + + if (dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && + dst->device->uma) { + deferred_memset((uint8_t*)dst->ptr + offset, c, size, &ctx->memsets); + return; + } + + // Fall back to GPU fillBuffer for non-UMA or non-host-visible buffers + ctx->s->buffer->buf.fillBuffer(dst->buffer, offset, size, c); +} + +void ggml_vk_buffer_memset(vk_buffer& dst, size_t offset, uint32_t c, size_t size) { + VK_LOG_DEBUG("ggml_vk_buffer_memset(" << offset << ", " << c << ", " << size << ")"); + + if (dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && + dst->device->uma) { + memset((uint8_t*)dst->ptr + offset, c, size); + return; + } + + std::lock_guard guard(dst->device->mutex); + vk_context subctx = ggml_vk_create_temporary_context(dst->device->transfer_queue->cmd_pool); + ggml_vk_ctx_begin(dst->device, subctx); + subctx->s->buffer->buf.fillBuffer(dst->buffer, offset, size, c); + ggml_vk_ctx_end(subctx); + + ggml_vk_submit(subctx, dst->device->fence); + VK_CHECK(dst->device->device.waitForFences({ dst->device->fence }, true, UINT64_MAX), "vk_memset waitForFences", dst->device); + dst->device->device.resetFences({ dst->device->fence }); + ggml_vk_queue_command_pools_cleanup(dst->device); +} + +ggml_backend_buffer_i ggml_backend_vk_buffer_interface = { + /* .free_buffer = */ ggml_backend_vk_buffer_free_buffer, + /* .get_base = */ ggml_backend_vk_buffer_get_base, + /* .init_tensor = */ ggml_backend_vk_buffer_init_tensor, + /* .memset_tensor = */ ggml_backend_vk_buffer_memset_tensor, + /* .set_tensor = */ ggml_backend_vk_buffer_set_tensor, + /* .get_tensor = */ ggml_backend_vk_buffer_get_tensor, + /* .set_tensor_2d = */ ggml_backend_vk_buffer_set_tensor_2d, + /* .get_tensor_2d = */ ggml_backend_vk_buffer_get_tensor_2d, + /* .cpy_tensor = */ ggml_backend_vk_buffer_cpy_tensor, + /* .clear = */ ggml_backend_vk_buffer_clear, + /* .reset = */ NULL, +}; + +vk_buffer ggml_vk_buffer_from_host_ptr(vk_device & device, void * ptr, size_t size) { + if (!device->external_memory_host) { + return {}; + } + + uintptr_t uptr = reinterpret_cast(ptr); + if (uptr & (device->min_imported_host_pointer_alignment - 1)) { + return {}; + } + if (size & (device->min_imported_host_pointer_alignment - 1)) { + return {}; + } + + const vk::MemoryPropertyFlags property_flags = vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached; + + vk_buffer buf {}; + try { + buf = ggml_vk_create_buffer(device, size, { property_flags }, ptr); + } catch (vk::SystemError& e) { + GGML_LOG_WARN("ggml_vulkan: Failed ggml_vk_create_buffer (%s)\n", e.what()); + } + + return buf; +} + diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-common.h b/ggml/src/ggml-vulkan/ggml-vulkan-common.h new file mode 100644 index 00000000..4ae5fea7 --- /dev/null +++ b/ggml/src/ggml-vulkan/ggml-vulkan-common.h @@ -0,0 +1,282 @@ +#pragma once +#include "ggml-vulkan-push-constants.h" + +// shared globals +extern ggml_backend_buffer_type_i ggml_backend_vk_buffer_type_interface; +extern bool vk_memory_logger_enabled; +extern bool vk_perf_logger_enabled; +extern bool vk_perf_logger_concurrent; +extern bool vk_enable_sync_logger; +extern uint32_t vk_perf_logger_frequency; +extern std::string vk_pipeline_stats_filter; +extern void * const vk_ptr_base; +extern vk_instance_t vk_instance; +extern ggml_backend_buffer_i ggml_backend_vk_buffer_interface; + +// instance +vk_device ggml_vk_get_device(size_t idx); +DispatchLoaderDynamic & ggml_vk_default_dispatcher(); +void ggml_vk_instance_init(); +void ggml_vk_init(ggml_backend_vk_context * ctx, size_t idx); +int ggml_vk_get_device_count(); +void ggml_vk_get_device_description(int device, char * description, size_t description_size); +bool ggml_vk_instance_layer_settings_available(); +bool ggml_vk_instance_portability_enumeration_ext_available(const std::vector& instance_extensions); +bool ggml_vk_instance_debug_utils_ext_available(const std::vector & instance_extensions); +bool ggml_vk_device_is_supported(const vk::PhysicalDevice & vkdev); +bool ggml_vk_khr_cooperative_matrix_support(const vk::PhysicalDeviceProperties& props, const vk::PhysicalDeviceDriverProperties& driver_props, vk_device_architecture arch); +uint32_t ggml_vk_intel_shader_core_count(const vk::PhysicalDevice& vkdev); +bool ggml_vk_intel_windows_driver_in_range(uint32_t driver_version, uint32_t lower_major, uint32_t lower_minor, uint32_t upper_major, uint32_t upper_minor); + +// shaders +void ggml_vk_destroy_pipeline(vk::Device& device, vk_pipeline& pipeline); +vk_fa_tuning_params get_fa_tuning_params(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc); +vk_fa_pipeline_state get_fa_pipeline_state(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool aligned, bool f32acc, bool use_mask, bool use_mask_opt, bool use_logit_softcap, bool use_sparse, ggml_type k_type, ggml_type v_type); +uint32_t get_subgroup_size(const std::string &pipeline_name, const vk_device_architecture &arch); +void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested = nullptr); +bool ggml_vk_flash_attn_scalar_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type); +bool ggml_vk_flash_attn_coopmat_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type = GGML_TYPE_F16, ggml_type v_type = GGML_TYPE_F16); + +// buffers +vk_buffer ggml_vk_create_buffer_check(vk_device& device, size_t size, vk::MemoryPropertyFlags req_flags, vk::MemoryPropertyFlags fallback_flags = vk::MemoryPropertyFlags(0)); +vk_buffer ggml_vk_create_buffer_device(vk_device& device, size_t size); +void ggml_vk_destroy_buffer(vk_buffer& buf); +void * ggml_vk_host_malloc(vk_device& device, size_t size); +void ggml_vk_host_free(vk_device& device, void* ptr); +void ggml_vk_host_get(const vk_device& device, const void * ptr, vk_buffer& buf, size_t& buf_offset); +void ggml_vk_ensure_sync_staging_buffer(vk_device& device, size_t size); +void ggml_vk_ensure_sync_staging_buffer(ggml_backend_vk_context * ctx, size_t size); +bool ggml_vk_buffer_write_2d_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging = false); +bool ggml_vk_buffer_write_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t size, bool sync_staging = false); +void ggml_vk_buffer_write_2d(vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height); +void ggml_vk_buffer_write(vk_buffer& dst, size_t offset, const void * src, size_t size); +bool ggml_vk_buffer_read_2d_async(vk_context subctx, vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging = false); +void ggml_vk_buffer_read_2d(vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height); +void ggml_vk_buffer_read(vk_buffer& src, size_t offset, void * dst, size_t size); +void ggml_vk_buffer_copy_async(vk_context& ctx, vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size); +void ggml_vk_buffer_copy(vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size); +void ggml_vk_buffer_memset_async(vk_context& ctx, vk_buffer& dst, size_t offset, uint32_t c, size_t size); +void ggml_vk_buffer_memset(vk_buffer& dst, size_t offset, uint32_t c, size_t size); +vk_buffer ggml_vk_buffer_from_host_ptr(vk_device & device, void * ptr, size_t size); + +// pipelines +uint64_t vk_tensor_offset(const ggml_tensor * tensor); +uint32_t get_misalign_bytes(const ggml_backend_vk_context * ctx, const ggml_tensor * t); +void ggml_vk_wait_for_fence(ggml_backend_vk_context * ctx); +void ggml_pipeline_request_descriptor_sets(ggml_backend_vk_context *ctx, vk_pipeline& pipeline, uint32_t n); +void ggml_pipeline_allocate_descriptor_sets(ggml_backend_vk_context * ctx); +void ggml_vk_submit(vk_context& ctx, vk::Fence fence); +uint32_t ggml_vk_find_queue_family_index(std::vector& queue_family_props, const vk::QueueFlags& required, const vk::QueueFlags& avoid, int32_t compute_index, uint32_t min_num_queues); +std::unique_ptr ggml_vk_create_queue(vk_device& device, uint32_t queue_family_index, uint32_t queue_index, vk::PipelineStageFlags&& stage_flags, bool transfer_only); +std::unique_ptr ggml_vk_create_aliased_queue(vk_device& device, const std::unique_ptr& source); +vk_context ggml_vk_create_context(ggml_backend_vk_context * ctx, vk_command_pool& p); +vk_context ggml_vk_create_temporary_context(vk_command_pool& p); +void ggml_vk_command_pool_cleanup(vk_device& device, vk_command_pool& p); +void ggml_vk_queue_command_pools_cleanup(vk_device& device); +vk_subbuffer ggml_vk_subbuffer(const ggml_backend_vk_context* ctx, const vk_buffer& buf, size_t offset = 0); +void ggml_vk_sync_buffers(ggml_backend_vk_context* ctx, vk_context& subctx); +void ggml_vk_set_event(vk_context& ctx, vk::Event& event); +void ggml_vk_wait_events(vk_context& ctx, std::vector&& events); +vk_subbuffer ggml_vk_tensor_subbuffer(const ggml_backend_vk_context * ctx, const ggml_tensor * tensor, bool allow_misalign = false); +void ggml_vk_cmd_label_begin(vk::CommandBuffer buf, const char * name); +void ggml_vk_ctx_end(vk_context& ctx); +void ggml_vk_ctx_begin(vk_device& device, vk_context& subctx); +vk_context ggml_vk_get_compute_ctx(ggml_backend_vk_context * ctx); +vk_context ggml_vk_get_transfer_ctx(ggml_backend_vk_context * ctx); +bool ggml_vk_submit_transfer_ctx(ggml_backend_vk_context * ctx); +size_t ggml_vk_align_size(size_t width, size_t align); +void deferred_memcpy(void * dst, const void * src, size_t size, std::vector* memcpys = nullptr); +void deferred_memset(void * dst, uint32_t val, size_t size, std::vector* memsets = nullptr); + +// matmul +vk_pipeline ggml_vk_get_to_fp16(ggml_backend_vk_context * ctx, ggml_type type); +void ggml_vk_matmul(ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline& pipeline, vk_subbuffer&& a, vk_subbuffer&& b, vk_subbuffer&& d, vk_subbuffer&& split_k_buffer, uint32_t m, uint32_t n, uint32_t k, uint32_t stride_a, uint32_t stride_b, uint32_t stride_d, uint32_t batch_stride_a, uint32_t batch_stride_b, uint32_t batch_stride_d, uint32_t split_k, uint32_t batch, uint32_t ne02, uint32_t ne12, uint32_t broadcast2, uint32_t broadcast3, uint32_t padded_n); +bool ggml_vk_dim01_contiguous(const ggml_tensor * tensor); +vk_pipeline ggml_vk_get_cpy_pipeline(ggml_backend_vk_context * ctx, const ggml_tensor * src, const ggml_tensor * dst, ggml_type to); +vk_pipeline ggml_vk_get_quantize_pipeline(ggml_backend_vk_context * ctx, ggml_type type); +void ggml_vk_quantize_q8_1(ggml_backend_vk_context * ctx, vk_context& subctx, const vk_subbuffer & in, const vk_subbuffer & out, uint32_t ne); +void ggml_vk_dsv4_hc_comb(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * mixes, const ggml_tensor * scale, const ggml_tensor * base, ggml_tensor * dst); +void ggml_vk_dsv4_hc_pre(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * x, const ggml_tensor * weights, ggml_tensor * dst); +void ggml_vk_dsv4_hc_post(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * x, const ggml_tensor * residual, const ggml_tensor * post, const ggml_tensor * comb, ggml_tensor * dst); +void ggml_vk_mul_mat(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx); +bool ggml_vk_use_mul_mat_vec_id(const struct ggml_cgraph * cgraph, int node_idx); +void ggml_vk_mul_mat_id(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx); + +// flash-attn +void ggml_vk_flash_attn(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * q, const ggml_tensor * k, const ggml_tensor * v, const ggml_tensor * mask, const ggml_tensor * sinks, ggml_tensor * dst); + +// operators +void ggml_vk_cpy_to_contiguous(ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline pipeline, const ggml_tensor * tensor, const vk_subbuffer & in, const vk_subbuffer & out); +bool ggml_vk_can_use_fwht(const ggml_backend_vk_context * ctx, const ggml_tensor * src1, const ggml_tensor * dst); +void ggml_vk_fwht(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src, ggml_tensor * dst); +void ggml_vk_get_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_get_rows_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_acc(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_multi_add(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx); +void ggml_vk_add(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_out_prod(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_sub(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_mul(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +int ggml_vk_unary_mul_op_index(ggml_unary_op op); +void ggml_vk_unary_mul(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx); +void ggml_vk_div(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_add_id(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst); +void ggml_vk_rwkv_wkv6(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_rwkv_wkv7(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_gated_linear_attn(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_lightning_indexer(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_gated_delta_net(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_ssm_scan(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_ssm_conv(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx); +void ggml_vk_opt_step_adamw(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_opt_step_sgd(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst); +void ggml_vk_concat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_upscale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_scale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_sqr(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_sqrt(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_add1(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_arange(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_fill(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_sin(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_cos(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_log(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_diag(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_clamp(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_pad(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_pad_reflect_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_roll(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_repeat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_repeat_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_cpy(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_set_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_silu_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_group_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +uint32_t ggml_vk_rms_partials_size(ggml_backend_vk_context * ctx, const ggml_tensor *node); +void ggml_vk_rms_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx, float * op_params); +void ggml_vk_rms_norm_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_l2_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_unary(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_xielu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_glu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_diag_mask_inf(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_soft_max(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst); +void ggml_vk_soft_max_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_topk_moe(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx); +void ggml_vk_rope(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_cgraph * cgraph, int node_idx, bool backprop); +void ggml_vk_argsort(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_topk(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_topk_qsa(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_cgraph * cgraph, int node_idx); +void ggml_vk_sum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_sum_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_mean(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_cumsum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_cross_entropy_loss(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_cross_entropy_loss_back(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst); +void ggml_vk_argmax(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_count_equal(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_solve_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_im2col(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_im2col_3d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_timestep_embedding(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_conv_transpose_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_col2im_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_snake_dispatch_fused(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx); +void ggml_vk_pool_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_pool_2d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); +void ggml_vk_conv_2d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_conv_3d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_conv_2d_dw(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); +void ggml_vk_leaky_relu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst); + +// graph +void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_context subctx); +bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int node_idx, ggml_tensor *node_begin, int node_idx_begin, bool last_node, bool almost_ready, bool submit); +void ggml_vk_compute_forward(ggml_backend_vk_context* ctx, ggml_cgraph * cgraph, ggml_tensor* tensor, int tensor_idx, bool almost_ready); +void ggml_vk_graph_cleanup(ggml_backend_vk_context * ctx); +void ggml_vk_cleanup(ggml_backend_vk_context * ctx); +void ggml_vk_synchronize(ggml_backend_vk_context * ctx); +bool ggml_vk_is_empty(ggml_tensor * node); +bool ggml_vk_can_fuse(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, std::initializer_list ops); +bool ggml_vk_can_fuse_ssm_conv(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, int num_extra); +bool ggml_vk_can_fuse_topk_moe(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, topk_moe_mode mode); +bool ggml_vk_can_fuse_topk_qsa(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +bool ggml_vk_can_fuse_rope_set_rows(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +bool ggml_vk_can_fuse_rms_norm_set_rows(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +bool ggml_vk_can_fuse_snake(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +bool ggml_vk_tensors_overlap(const ggml_tensor * a, const ggml_tensor * b, bool elementwise); +bool ggml_vk_can_fuse_rms_norm_mul_rope(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +uint32_t ggml_vk_fuse_multi_add(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx); +void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * graph, struct ggml_backend_graph_optimize_params * params); + +// backend +bool ggml_backend_buffer_is_vk(ggml_backend_buffer_t buffer); +void ggml_backend_vk_buffer_free_buffer(ggml_backend_buffer_t buffer); +void * ggml_backend_vk_buffer_get_base(ggml_backend_buffer_t buffer); +enum ggml_status ggml_backend_vk_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor); +void ggml_backend_vk_buffer_memset_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, uint8_t value, size_t offset, size_t size); +void ggml_backend_vk_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size); +void ggml_backend_vk_buffer_set_tensor_2d(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size, size_t n_copies, size_t stride_tensor, size_t stride_data); +void ggml_backend_vk_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size); +void ggml_backend_vk_buffer_get_tensor_2d(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size, size_t n_copies, size_t stride_tensor, size_t stride_data); +bool ggml_backend_vk_buffer_cpy_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * src, ggml_tensor * dst); +void ggml_backend_vk_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value); +const char * ggml_backend_vk_buffer_type_name(ggml_backend_buffer_type_t buft); +ggml_backend_buffer_t ggml_backend_vk_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size); +size_t ggml_backend_vk_buffer_type_get_alignment(ggml_backend_buffer_type_t buft); +size_t ggml_backend_vk_buffer_type_get_max_size(ggml_backend_buffer_type_t buft); +size_t ggml_backend_vk_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor); +void ggml_backend_vk_free(ggml_backend_t backend); +ggml_backend_reg_t ggml_backend_vk_reg(); + +// debug +int64_t ggml_vk_get_op_batch_size(const ggml_tensor * op); + +// ggml-vulkan.cpp (residual) +bool ggml_vk_lightning_indexer_k_type_supported(ggml_type type); +void ggml_vk_print_device_fault_info(const vk_device& device); +uint64_t ggml_vk_get_node_flops(const ggml_tensor * node); +void ggml_vk_print_node_list(const ggml_cgraph * cgraph, int start, int end); +void ggml_vk_print_device_lost_info(const vk_device& device); +size_t ggml_vk_tensor_buffer_offset(const ggml_backend_vk_context * ctx, const ggml_tensor * t); +size_t ggml_vk_descriptor_offset(size_t tensor_offset, size_t alignment, size_t type_size); +uint32_t ggml_vk_concat_unit_size(ggml_type type); +bool ggml_vk_concat_supported(const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); + +template +inline void ggml_vk_dispatch_pipeline(ggml_backend_vk_context* ctx, vk_context& subctx, vk_pipeline& pipeline, std::initializer_list const& descriptor_buffer_infos, const T &push_constants, std::array elements) { + const uint32_t wg0 = CEIL_DIV(elements[0], pipeline->wg_denoms[0]); + const uint32_t wg1 = CEIL_DIV(elements[1], pipeline->wg_denoms[1]); + const uint32_t wg2 = CEIL_DIV(elements[2], pipeline->wg_denoms[2]); + VK_LOG_DEBUG("ggml_vk_dispatch_pipeline(" << pipeline->name << ", {"; + for (auto& buffer : descriptor_buffer_infos) { + std::cerr << "(" << buffer.buffer << ", " << buffer.offset << ", " << buffer.range << "), "; + } + std::cerr << "}, (" << wg0 << "," << wg1 << "," << wg2 << "))"); + GGML_ASSERT(wg0 <= ctx->device->properties.limits.maxComputeWorkGroupCount[0] && + wg1 <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && + wg2 <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + GGML_ASSERT(ctx->descriptor_set_idx < ctx->descriptor_sets.size()); + GGML_ASSERT(descriptor_buffer_infos.size() <= MAX_PARAMETER_COUNT); + GGML_ASSERT(pipeline->parameter_count == descriptor_buffer_infos.size()); + GGML_ASSERT(pipeline->push_constant_size == push_constant_size(push_constants)); + + vk::DescriptorSet& descriptor_set = ctx->descriptor_sets[ctx->descriptor_set_idx++]; + vk::WriteDescriptorSet write_descriptor_set{ descriptor_set, 0, 0, pipeline->parameter_count, vk::DescriptorType::eStorageBuffer, nullptr, descriptor_buffer_infos.begin() }; + ctx->device->device.updateDescriptorSets({ write_descriptor_set }, {}); + + subctx->s->buffer->buf.pushConstants(pipeline->layout, vk::ShaderStageFlagBits::eCompute, 0, push_constant_size(push_constants), push_constant_data(push_constants)); + subctx->s->buffer->buf.bindPipeline(vk::PipelineBindPoint::eCompute, pipeline->pipeline); + subctx->s->buffer->buf.bindDescriptorSets(vk::PipelineBindPoint::eCompute, + pipeline->layout, + 0, + { descriptor_set }, + {}); + { + ggml_vk_debug_label dbg(subctx, pipeline->name, wg0, wg1, wg2); + subctx->s->buffer->buf.dispatch(wg0, wg1, wg2); + } +} + diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp b/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp new file mode 100644 index 00000000..15abd546 --- /dev/null +++ b/ggml/src/ggml-vulkan/ggml-vulkan-debug.cpp @@ -0,0 +1,1561 @@ +#include "ggml-vulkan-common.h" + +bool vk_memory_logger_enabled = false; + +bool vk_perf_logger_enabled = false; + +bool vk_perf_logger_concurrent = false; + +bool vk_enable_sync_logger = false; + +uint32_t vk_perf_logger_frequency = 1; + +std::string vk_pipeline_stats_filter; + +void vk_memory_logger::log_allocation(vk_buffer_ref buf_ref, size_t size) { + if (!vk_memory_logger_enabled) { + return; + } + std::lock_guard guard(log_mutex); + vk_buffer buf = buf_ref.lock(); + const bool device = bool(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eDeviceLocal); + const std::string type = device ? "device" : "host"; + allocations[buf->buffer] = size; + total_device += device ? size : 0; + total_host += device ? 0 : size; + VK_LOG_MEMORY(buf->device->name << ": +" << format_size(size) << " " << type << " at " << buf->buffer << ". Total device: " << format_size(total_device) << ", total host: " << format_size(total_host)); +} + +void vk_memory_logger::log_deallocation(vk_buffer_ref buf_ref) { + if (buf_ref.expired() || buf_ref.lock()->size == 0 || !vk_memory_logger_enabled) { + return; + } + + std::lock_guard guard(log_mutex); + vk_buffer buf = buf_ref.lock(); + const bool device = bool(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eDeviceLocal); + std::string type = device ? "device" : "host"; + auto it = allocations.find(buf->buffer); + if (it != allocations.end()) { + total_device -= device ? it->second : 0; + total_host -= device ? 0 : it->second; + VK_LOG_MEMORY(buf->device->name << ": -" << format_size(it->second) << " " << type << " at " << buf->buffer << ". Total device: " << format_size(total_device) << ", total host: " << format_size(total_host)); + allocations.erase(it); + } else { + VK_LOG_MEMORY("ERROR " << buf->device->name << ": Attempted to deallocate unknown " << type << " memory at " << buf->buffer); + } +} + +#ifdef GGML_VULKAN_CHECK_RESULTS +static size_t vk_skip_checks; +static size_t vk_output_tensor; + +static void ggml_vk_print_tensor(const ggml_tensor * tensor, const char * name); +static void ggml_vk_check_results_0(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx); +static void ggml_vk_check_results_1(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx); +#endif + +#ifdef GGML_VULKAN_RUN_TESTS +static void ggml_vk_print_matrix_area(const void * data, ggml_type type, int ne0, int ne1, int i0, int i1, int i2) { + if (type != GGML_TYPE_F32 && type != GGML_TYPE_F16) { + return; + } + i0 = std::max(i0, 5); + i1 = std::max(i1, 5); + i2 = std::max(i2, 0); + fprintf(stderr, " "); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + fprintf(stderr, "%7d ", idx1); + } + fprintf(stderr, "\n"); + for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { + fprintf(stderr, "%7d: ", idx0); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + if (idx0 >= 0 && idx0 < ne0 && idx1 >= 0 && idx1 < ne1) { + float val; + if (type == GGML_TYPE_F32) { + val = *((const float *) data + i2*ne1*ne0 + idx1*ne0 + idx0); + } else if (type == GGML_TYPE_F16) { + val = ggml_fp16_to_fp32(*((const ggml_fp16_t *) data + i2*ne1*ne0 + idx1*ne0 + idx0)); + } else { + GGML_ABORT("fatal error"); + } + fprintf(stderr, "% 7.2f ", val); + } else { + fprintf(stderr, " "); + } + } + fprintf(stderr, "\n"); + } +} + +template +static void ggml_vk_test_matmul(ggml_backend_vk_context * ctx, size_t m, size_t n, size_t k, size_t batch, size_t num_it, int split_k, int shader_size) { + VK_LOG_DEBUG("ggml_vk_test_matmul(" << m << ", " << n << ", " << k << ", " << batch << ", " << num_it << ", " << split_k << ", " << shader_size << ")"); + const size_t x_ne = m * k * batch; + const size_t y_ne = k * n * batch; + const size_t d_ne = m * n * batch; + + ggml_type x_type = std::is_same() ? GGML_TYPE_F32 : GGML_TYPE_F16; + ggml_type y_type = std::is_same() ? GGML_TYPE_F32 : GGML_TYPE_F16; + vk_matmul_pipeline_key mm_test_key{x_type, y_type, false, false}; + auto mm_test_it = ctx->device->pipeline_matmul.find(mm_test_key); + GGML_ASSERT(mm_test_it != ctx->device->pipeline_matmul.end() && !mm_test_it->second.empty()); + auto& mm_test_configs = mm_test_it->second; + GGML_ASSERT(shader_size >= 0 && shader_size < (int)mm_test_configs.size()); + + std::string shname = std::string(ggml_type_name(x_type)) + "_" + std::string(ggml_type_name(y_type)) + "_ALIGNED_" + std::to_string(shader_size); + vk_pipeline p = mm_test_configs[shader_size].aligned ? mm_test_configs[shader_size].aligned : mm_test_configs[shader_size].unaligned; + + const size_t kpad = ggml_vk_align_size(k, mm_test_configs[shader_size].align); + + if (k != kpad) { + p = mm_test_configs[shader_size].unaligned; + shname = std::string(ggml_type_name(x_type)) + "_" + std::string(ggml_type_name(y_type)) + "_" + std::to_string(shader_size); + } + + if (split_k > 1) { + ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, num_it); + + if (ctx->prealloc_split_k == nullptr || ctx->prealloc_split_k->size < sizeof(float) * d_ne * split_k) { + // Resize buffer + if (ctx->prealloc_split_k != nullptr) { + ggml_vk_destroy_buffer(ctx->prealloc_split_k); + } + ctx->prealloc_split_k = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne * split_k, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + } + } + + ggml_pipeline_allocate_descriptor_sets(ctx); + + vk_buffer d_X = ggml_vk_create_buffer_check(ctx->device, sizeof(X_TYPE) * x_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer d_Y = ggml_vk_create_buffer_check(ctx->device, sizeof(Y_TYPE) * y_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer d_D = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + + X_TYPE* x = (X_TYPE *) malloc(sizeof(X_TYPE) * x_ne); + Y_TYPE* y = (Y_TYPE *) malloc(sizeof(Y_TYPE) * y_ne); + float* d = (float *) malloc(sizeof(float) * d_ne); + + for (size_t i = 0; i < x_ne; i++) { + if (std::is_same()) { + x[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; + // x[i] = 1.0f; + // x[i] = i + 1; + // x[i] = (i % k == i / k) ? 1.0f : 0.0f; + } else if (std::is_same()) { + x[i] = ggml_fp32_to_fp16((rand() / (float)RAND_MAX) * 2.0f - 1.0f); + // x[i] = ggml_fp32_to_fp16(1.0f); + // x[i] = ggml_fp32_to_fp16(i + 1); + // x[i] = ggml_fp32_to_fp16((i % k == i / k) ? 1.0f : 0.0f); + } else { + GGML_ABORT("fatal error"); + } + } + for (size_t i = 0; i < y_ne; i++) { + if (std::is_same()) { + y[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; + // y[i] = (i % k == i / k) ? 1.0f : 0.0f; + // y[i] = i + 1; + } else if (std::is_same()) { + y[i] = ggml_fp32_to_fp16((rand() / (float)RAND_MAX) * 2.0f - 1.0f); + // y[i] = ggml_fp32_to_fp16((i % k == i / k) ? 1.0f : 0.0f); + // y[i] = ggml_fp32_to_fp16(i + 1); + } else { + GGML_ABORT("fatal error"); + } + } + + ggml_vk_buffer_write(d_X, 0, x, sizeof(X_TYPE) * k * m * batch); + ggml_vk_buffer_write(d_Y, 0, y, sizeof(Y_TYPE) * k * n * batch); + + vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); + ggml_vk_ctx_begin(ctx->device, subctx); + for (size_t i = 0; i < num_it; i++) { + ggml_vk_matmul( + ctx, subctx, p, ggml_vk_subbuffer(ctx, d_X), ggml_vk_subbuffer(ctx, d_Y), ggml_vk_subbuffer(ctx, d_D), ggml_vk_subbuffer(ctx, ctx->prealloc_split_k), + m, n, k, + k, k, m, k*m, k*n, m*n, + split_k, batch, batch, batch, 1, 1, n + ); + } + ggml_vk_ctx_end(subctx); + + auto begin = std::chrono::high_resolution_clock::now(); + ggml_vk_submit(subctx, ctx->fence); + VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_matmul waitForFences", ctx->device); + ctx->device->device.resetFences({ ctx->fence }); + ggml_vk_queue_command_pools_cleanup(ctx->device); + + auto end = std::chrono::high_resolution_clock::now(); + double time = std::chrono::duration_cast(end-begin).count() / 1000.0; + + // copy dst to host + ggml_vk_buffer_read(d_D, 0, d, sizeof(float) * d_ne); + + float * d_chk = (float *) malloc(sizeof(float) * d_ne); + + ggml_init_params iparams = { + /*.mem_size =*/ 1024*1024*1024, + /*.mem_buffer =*/ NULL, + /*.no_alloc =*/ true, + }; + + ggml_context * ggml_ctx = ggml_init(iparams); + + ggml_type src0_type; + ggml_type src1_type; + + if (std::is_same()) { + src0_type = GGML_TYPE_F32; + } else if (std::is_same()) { + src0_type = GGML_TYPE_F16; + } else { + GGML_ABORT("fatal error"); + } + if (std::is_same()) { + src1_type = GGML_TYPE_F32; + } else if (std::is_same()) { + src1_type = GGML_TYPE_F16; + } else { + GGML_ABORT("fatal error"); + } + + ggml_tensor * src0_ggml = ggml_new_tensor_3d(ggml_ctx, src0_type, k, m, batch); + ggml_tensor * src1_ggml = ggml_new_tensor_3d(ggml_ctx, src1_type, k, n, batch); + ggml_tensor * tensor_ggml = ggml_mul_mat(ggml_ctx, src0_ggml, src1_ggml); + + src0_ggml->data = x; + src1_ggml->data = y; + tensor_ggml->data = d_chk; + + ggml_cgraph * cgraph = ggml_new_graph(ggml_ctx); + ggml_build_forward_expand(cgraph, tensor_ggml); + + ggml_graph_compute_with_ctx(ggml_ctx, cgraph, 1); + + ggml_free(ggml_ctx); + + double avg_err = 0.0; + int first_err_n = -1; + int first_err_m = -1; + int first_err_b = -1; + + for (size_t i = 0; i < m*n*batch; i++) { + double err = std::fabs(d[i] - d_chk[i]); + avg_err += err; + + if ((err > 0.05f || std::isnan(err)) && first_err_n == -1) { + first_err_b = i / (m * n); + first_err_n = (i % (m * n)) / m; + first_err_m = (i % (m * n)) % m; + } + } + + avg_err /= m * n; + + double tflops = 2.0*m*n*k*batch*num_it / (time / 1000.0) / (1000.0*1000.0*1000.0*1000.0); + + std::cerr << "TEST " << shname << " m=" << m << " n=" << n << " k=" << k << " batch=" << batch << " split_k=" << split_k << " matmul " << time / num_it << "ms " << tflops << " TFLOPS avg_err=" << avg_err << std::endl; + + if (avg_err > 0.1 || std::isnan(avg_err)) { + std::cerr << "m = " << first_err_m << " n = " << first_err_n << " b = " << first_err_b << std::endl; + std::cerr << "Actual result: " << std::endl << std::endl; + ggml_vk_print_matrix_area(d, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + std::cerr << "Expected result: " << std::endl << std::endl; + ggml_vk_print_matrix_area(d_chk, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + if (split_k > 1) { + float * split_k_buf = (float *) malloc(sizeof(float) * d_ne * split_k); + ggml_vk_buffer_read(ctx->prealloc_split_k, 0, split_k_buf, sizeof(float) * d_ne * split_k); + + std::cerr << "d_buf0: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf1: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf2: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + 2 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf3: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + 3 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + free(split_k_buf); + } + } + + free(d_chk); + + ggml_vk_command_pool_cleanup(ctx->device, ctx->compute_cmd_pool); + + ggml_vk_destroy_buffer(d_X); + ggml_vk_destroy_buffer(d_Y); + ggml_vk_destroy_buffer(d_D); + + free(x); + free(y); + free(d); +} + +static void ggml_vk_print_tensor_area(const ggml_tensor * tensor, int i0, int i1, int i2, int i3) { + if (tensor->type != GGML_TYPE_F32 && tensor->type != GGML_TYPE_F16) { + return; + } + i0 = std::max(i0, 5); + i1 = std::max(i1, 5); + i2 = std::max(i2, 0); + i3 = std::max(i3, 0); + fprintf(stderr, " "); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + fprintf(stderr, "%7d ", idx1); + } + fprintf(stderr, "\n"); + for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { + fprintf(stderr, "%7d: ", idx0); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + if (idx0 >= 0 && idx0 < tensor->ne[0] && idx1 >= 0 && idx1 < tensor->ne[1] && i2 >= 0 && i2 < tensor->ne[2] && i3 >= 0 && i3 < tensor->ne[3]) { + float val; + if (tensor->type == GGML_TYPE_F32) { + val = *(float *) ((char *) tensor->data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); + } else if (tensor->type == GGML_TYPE_F16) { + val = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) tensor->data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0])); + } else { + GGML_ABORT("fatal error"); + } + fprintf(stderr, "% 7.2f ", val); + } else { + fprintf(stderr, " "); + } + } + fprintf(stderr, "\n"); + } +} + +static void ggml_vk_quantize_data(const float * from, void * to, size_t ne, ggml_type quant) { + ggml_quantize_chunk(quant, from, to, 0, 1, ne, nullptr); +} + +static void ggml_vk_dequantize_data(const void * from, float * to, size_t ne, ggml_type quant) { + if (quant == GGML_TYPE_F32) { + memcpy(to, from, sizeof(float) * ne); + return; + } + + const auto * tt = ggml_get_type_traits(quant); + + ggml_to_float_t dequant_fn = tt->to_float; + + dequant_fn(from, to, ne); +} + +static void ggml_vk_test_dequant(ggml_backend_vk_context * ctx, size_t ne, ggml_type quant) { + VK_LOG_DEBUG("ggml_vk_test_dequant(" << ne << ")"); + const size_t x_sz = sizeof(float) * ne; + const size_t x_sz_f16 = sizeof(ggml_fp16_t) * ne; + const size_t qx_sz = ne * ggml_type_size(quant)/ggml_blck_size(quant); + float * x = (float *) malloc(x_sz); + void * qx = malloc(qx_sz); + vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer x_buf = ggml_vk_create_buffer_check(ctx->device, x_sz_f16, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + float * x_ref = (float *) malloc(x_sz); + ggml_fp16_t * x_chk = (ggml_fp16_t *) malloc(x_sz_f16); + + for (size_t i = 0; i < ne; i++) { + x[i] = rand() / (float)RAND_MAX; + } + + vk_pipeline p = ggml_vk_get_to_fp16(ctx, quant); + + ggml_vk_quantize_data(x, qx, ne, quant); + ggml_vk_dequantize_data(qx, x_ref, ne, quant); + + ggml_pipeline_request_descriptor_sets(ctx, p, 1); + + ggml_pipeline_allocate_descriptor_sets(ctx); + + ggml_vk_buffer_write(qx_buf, 0, qx, qx_sz); + + vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); + ggml_vk_ctx_begin(ctx->device, subctx); + const std::vector pc = { 1, (uint32_t)ne, (uint32_t)ne, (uint32_t)ne, (uint32_t)ne }; + ggml_vk_dispatch_pipeline(ctx, subctx, p, { vk_subbuffer{ qx_buf, 0, qx_sz }, vk_subbuffer{ x_buf, 0, x_sz_f16 } }, pc, { (uint32_t)ne, 1, 1}); + ggml_vk_ctx_end(subctx); + + auto begin = std::chrono::high_resolution_clock::now(); + + ggml_vk_submit(subctx, ctx->fence); + VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_dequant waitForFences", ctx->device); + ctx->device->device.resetFences({ ctx->fence }); + ggml_vk_queue_command_pools_cleanup(ctx->device); + + auto end = std::chrono::high_resolution_clock::now(); + + double ms_dequant = std::chrono::duration_cast(end-begin).count() / 1000.0; + ggml_vk_buffer_read(x_buf, 0, x_chk, x_sz_f16); + + int first_err = -1; + + double avg_err = 0.0; + for (size_t i = 0; i < ne; i++) { + double error = std::fabs(x_ref[i] - ggml_fp16_to_fp32(x_chk[i])); + avg_err += error; + + if (first_err < 0 && error > 0.05) { + first_err = i; + } + } + + avg_err /= ne; + + std::cerr << "TEST DEQUANT " << ggml_type_name(quant) << " time=" << ms_dequant << "ms avg_err=" << avg_err << std::endl; + + if (avg_err > 0.1) { + std::cerr << "first_error = " << first_err << std::endl; + std::cerr << "Actual result: " << std::endl << std::endl; + for (int i = std::max(0, first_err - 5); i < std::min((int)ne, first_err + 5); i++) { + std::cerr << ggml_fp16_to_fp32(x_chk[i]) << ", "; + } + std::cerr << std::endl << "Expected result: " << std::endl << std::endl; + for (int i = std::max(0, first_err - 5); i < std::min((int)ne, first_err + 5); i++) { + std::cerr << x_ref[i] << ", "; + } + std::cerr << std::endl; + } + + ggml_vk_destroy_buffer(x_buf); + ggml_vk_destroy_buffer(qx_buf); + + free(x); + free(qx); + free(x_ref); + free(x_chk); +} + +// This does not work without ggml q8_1 quantization support +// +// typedef uint16_t ggml_half; +// typedef uint32_t ggml_half2; +// +// #define QK8_1 32 +// typedef struct { +// union { +// struct { +// ggml_half d; // delta +// ggml_half s; // d * sum(qs[i]) +// } GGML_COMMON_AGGR_S; +// ggml_half2 ds; +// } GGML_COMMON_AGGR_U; +// int8_t qs[QK8_1]; // quants +// } block_q8_1; +// +// static void ggml_vk_test_quantize(ggml_backend_vk_context * ctx, size_t ne, ggml_type quant) { +// VK_LOG_DEBUG("ggml_vk_test_quantize(" << ne << ")"); +// GGML_ASSERT(quant == GGML_TYPE_Q8_1); +// +// const size_t x_sz = sizeof(float) * ne; +// const size_t qx_sz = ne * ggml_type_size(quant)/ggml_blck_size(quant); +// float * x = (float *) malloc(x_sz); +// block_q8_1 * qx = (block_q8_1 *)malloc(qx_sz); +// block_q8_1 * qx_res = (block_q8_1 *)malloc(qx_sz); +// vk_buffer x_buf = ggml_vk_create_buffer_check(ctx->device, x_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); +// vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); +// +// for (size_t i = 0; i < ne; i++) { +// x[i] = rand() / (float)RAND_MAX; +// } +// +// vk_pipeline p = ggml_vk_get_quantize_pipeline(ctx, quant); +// +// ggml_pipeline_request_descriptor_sets(ctx, p, 1); +// +// ggml_pipeline_allocate_descriptor_sets(ctx); +// +// ggml_vk_buffer_write(x_buf, 0, x, x_sz); +// +// vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); +// ggml_vk_ctx_begin(ctx->device, subctx); +// ggml_vk_quantize_q8_1(ctx, subctx, ggml_vk_subbuffer(ctx, x_buf), ggml_vk_subbuffer(ctx, qx_buf), ne); +// ggml_vk_ctx_end(subctx); +// +// auto begin = std::chrono::high_resolution_clock::now(); +// +// ggml_vk_submit(subctx, ctx->fence); +// VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_quantize waitForFences"); +// ctx->device->device.resetFences({ ctx->fence }); +// ggml_vk_queue_command_pools_cleanup(ctx->device); +// +// auto end = std::chrono::high_resolution_clock::now(); +// +// double ms_quant = std::chrono::duration_cast(end-begin).count() / 1000.0; +// ggml_vk_buffer_read(qx_buf, 0, qx, qx_sz); +// +// ggml_vk_quantize_data(x, qx_res, ne, quant); +// +// int first_err = -1; +// +// for (size_t i = 0; i < ne / 32; i++) { +// double error = std::fabs(ggml_fp16_to_fp32(qx_res[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) - ggml_fp16_to_fp32(qx[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d)); +// +// if (first_err < 0 && error > 0.1) { +// first_err = i; +// } +// +// error = std::fabs(ggml_fp16_to_fp32(qx_res[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) - ggml_fp16_to_fp32(qx[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s)); +// +// if (first_err < 0 && error > 0.1) { +// first_err = i; +// } +// +// for (size_t j = 0; j < 32; j++) { +// uint64_t error = std::abs(qx_res[i].qs[j] - qx[i].qs[j]); +// +// if (first_err < 0 && error > 1) { +// first_err = i; +// } +// } +// } +// +// std::cerr << "TEST QUANTIZE " << ggml_type_name(quant) << " time=" << ms_quant << "ms " << (first_err == -1 ? "CORRECT" : "INCORRECT") << std::endl; +// +// if (first_err != -1) { +// std::cerr << "first_error = " << first_err << std::endl; +// std::cerr << "Actual result: " << std::endl << std::endl; +// std::cout << "d=" << ggml_fp16_to_fp32(qx[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) << " s=" << ggml_fp16_to_fp32(qx[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) << " "; +// for (size_t j = 0; j < 32; j++) { +// std::cout << " qs" << j << "=" << (uint32_t)qx[first_err].qs[j] << " "; +// } +// std::cerr << std::endl << std::endl << "Expected result: " << std::endl << std::endl; +// std::cout << "d=" << ggml_fp16_to_fp32(qx_res[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) << " s=" << ggml_fp16_to_fp32(qx_res[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) << " "; +// for (size_t j = 0; j < 32; j++) { +// std::cout << " qs" << j << "=" << (uint32_t)qx_res[first_err].qs[j] << " "; +// } +// std::cerr << std::endl; +// } +// +// ggml_vk_destroy_buffer(x_buf); +// ggml_vk_destroy_buffer(qx_buf); +// +// free(x); +// free(qx); +// free(qx_res); +// } + +static void ggml_vk_test_dequant_matmul(ggml_backend_vk_context * ctx, size_t m, size_t n, size_t k, size_t batch, size_t num_it, size_t split_k, size_t shader_size, ggml_type quant, bool mmq = false) { + VK_LOG_DEBUG("ggml_vk_test_dequant_matmul(" << m << ", " << n << ", " << k << ", " << batch << ", " << num_it << ", " << split_k << ", " << ggml_type_name(quant) << ")"); + const size_t x_ne = m * k * batch; + const size_t y_ne = k * n * batch; + const size_t d_ne = m * n * batch; + + ggml_type b_type = mmq ? GGML_TYPE_Q8_1 : GGML_TYPE_F32; + bool f16acc = ctx->device->fp16 && !mmq; + vk_matmul_pipeline_key dq_key{quant, b_type, false, f16acc}; + auto dq_it = ctx->device->pipeline_matmul.find(dq_key); + if (dq_it == ctx->device->pipeline_matmul.end() || dq_it->second.empty()) { + if (f16acc) { + dq_key.f16acc = false; + dq_it = ctx->device->pipeline_matmul.find(dq_key); + } + } + if (dq_it == ctx->device->pipeline_matmul.end() || dq_it->second.empty()) { + std::cerr << "error: no pipeline for ggml_vk_test_dequant_matmul " << ggml_type_name(quant) << std::endl; + return; + } + auto& dq_configs = dq_it->second; + if (shader_size >= (int)dq_configs.size()) { + std::cerr << "error: shader_size " << shader_size << " >= configs.size() " << dq_configs.size() << " for " << ggml_type_name(quant) << std::endl; + return; + } + + std::string shname = std::string(ggml_type_name(quant)) + "_ALIGNED_" + std::to_string(shader_size); + vk_pipeline p = dq_configs[shader_size].aligned ? dq_configs[shader_size].aligned : dq_configs[shader_size].unaligned; + + const size_t kpad = mmq ? 0 : ggml_vk_align_size(k, dq_configs[shader_size].align); + + if (mmq || k != kpad) { + p = dq_configs[shader_size].unaligned; + shname = std::string(ggml_type_name(quant)) + "_" + std::to_string(shader_size); + } + + if (p == nullptr) { + std::cerr << "error: no pipeline for ggml_vk_test_dequant_matmul " << ggml_type_name(quant) << std::endl; + return; + } + + const size_t x_sz = sizeof(float) * x_ne; + const size_t y_sz = sizeof(float) * y_ne; + const size_t qx_sz = x_ne * ggml_type_size(quant)/ggml_blck_size(quant); + const size_t qy_sz = mmq ? y_ne * ggml_type_size(GGML_TYPE_Q8_1)/ggml_blck_size(GGML_TYPE_Q8_1) : y_sz; + const size_t d_sz = sizeof(float) * d_ne; + float * x = (float *) malloc(x_sz); + float * y = (float *) malloc(y_sz); + void * qx = malloc(qx_sz); + vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer y_buf = ggml_vk_create_buffer_check(ctx->device, y_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer qy_buf = ggml_vk_create_buffer_check(ctx->device, qy_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + vk_buffer d_buf = ggml_vk_create_buffer_check(ctx->device, d_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + float * d = (float *) malloc(d_sz); + float * d_chk = (float *) malloc(d_sz); + + for (size_t i = 0; i < x_ne; i++) { + x[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; + // x[i] = (i % k == i / k) ? 1.0f : 0.0f; + // x[i] = i % k; + } + + ggml_vk_quantize_data(x, qx, x_ne, quant); + + for (size_t i = 0; i < y_ne; i++) { + y[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; + // y[i] = (i % k == i / k) ? 1.0f : 0.0f; + // y[i] = i % k; + } + + if (split_k > 1) { + ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, num_it); + + if (ctx->prealloc_split_k == nullptr || ctx->prealloc_split_k->size < sizeof(float) * d_ne * split_k) { + // Resize buffer + if (ctx->prealloc_split_k != nullptr) { + ggml_vk_destroy_buffer(ctx->prealloc_split_k); + } + ctx->prealloc_split_k = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne * split_k, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + } + } + if (mmq) { + vk_pipeline pipeline_quantize_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); + ggml_pipeline_request_descriptor_sets(ctx, pipeline_quantize_q8_1, num_it); + } + + ggml_pipeline_allocate_descriptor_sets(ctx); + + ggml_vk_buffer_write(qx_buf, 0, qx, qx_sz); + ggml_vk_buffer_write(y_buf, 0, y, y_sz); + + vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); + ggml_vk_ctx_begin(ctx->device, subctx); + if (mmq) { + for (size_t i = 0; i < num_it; i++) { + ggml_vk_quantize_q8_1(ctx, subctx, { y_buf, 0, y_sz }, { qy_buf, 0, qy_sz }, y_ne); + ggml_vk_matmul( + ctx, subctx, p, { qx_buf, 0, qx_sz }, { qy_buf, 0, qy_sz }, { d_buf, 0, d_sz }, { ctx->prealloc_split_k, 0, ctx->prealloc_size_split_k }, + m, n, k, + k, k, m, k*m, k*n, m*n, + split_k, batch, batch, batch, 1, 1, n + ); + } + } else { + for (size_t i = 0; i < num_it; i++) { + ggml_vk_matmul( + ctx, subctx, p, { qx_buf, 0, qx_sz }, { y_buf, 0, y_sz }, { d_buf, 0, d_sz }, { ctx->prealloc_split_k, 0, ctx->prealloc_size_split_k }, + m, n, k, + k, k, m, k*m, k*n, m*n, + split_k, batch, batch, batch, 1, 1, n + ); + } + } + ggml_vk_ctx_end(subctx); + + auto begin = std::chrono::high_resolution_clock::now(); + + ggml_vk_submit(subctx, ctx->fence); + VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_dequant waitForFences", ctx->device); + ctx->device->device.resetFences({ ctx->fence }); + ggml_vk_queue_command_pools_cleanup(ctx->device); + + auto end = std::chrono::high_resolution_clock::now(); + + double time_ms = std::chrono::duration_cast(end-begin).count() / 1000.0; + ggml_vk_buffer_read(d_buf, 0, d, d_sz); + + ggml_init_params iparams = { + /*.mem_size =*/ 1024*1024*1024, + /*.mem_buffer =*/ NULL, + /*.no_alloc =*/ true, + }; + + ggml_context * ggml_ctx = ggml_init(iparams); + + ggml_tensor * src0_ggml = ggml_new_tensor_3d(ggml_ctx, quant, k, m, batch); + ggml_tensor * src1_ggml = ggml_new_tensor_3d(ggml_ctx, GGML_TYPE_F32, k, n, batch); + ggml_tensor * tensor_ggml = ggml_mul_mat(ggml_ctx, src0_ggml, src1_ggml); + + src0_ggml->data = qx; + src1_ggml->data = y; + tensor_ggml->data = d_chk; + + ggml_cgraph * cgraph = ggml_new_graph(ggml_ctx); + ggml_build_forward_expand(cgraph, tensor_ggml); + + ggml_graph_compute_with_ctx(ggml_ctx, cgraph, 1); + + ggml_free(ggml_ctx); + + double avg_err = 0.0; + int first_err_n = -1; + int first_err_m = -1; + int first_err_b = -1; + + for (size_t i = 0; i < m*n*batch; i++) { + double err = std::fabs(d[i] - d_chk[i]); + avg_err += err; + + if ((err > 0.05f || std::isnan(err)) && first_err_n == -1) { + first_err_b = i / (m * n); + first_err_n = (i % (m * n)) / m; + first_err_m = (i % (m * n)) % m; + } + } + + avg_err /= m * n; + + double tflops = 2.0*m*n*k*batch*num_it / (time_ms / 1000.0) / (1000.0*1000.0*1000.0*1000.0); + + std::cerr << "TEST dequant matmul " << shname; + if (mmq) { + std::cerr << " mmq"; + } + std::cerr << " m=" << m << " n=" << n << " k=" << k << " batch=" << batch << " split_k=" << split_k << " matmul " << time_ms / num_it << "ms " << tflops << " TFLOPS avg_err=" << avg_err << std::endl; + + if (avg_err > 0.01 || std::isnan(avg_err)) { + std::cerr << "m = " << first_err_m << " n = " << first_err_n << " b = " << first_err_b << std::endl; + std::cerr << "Actual result: " << std::endl << std::endl; + ggml_vk_print_matrix_area(d, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + std::cerr << std::endl; + std::cerr << "Expected result: " << std::endl << std::endl; + ggml_vk_print_matrix_area(d_chk, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "src0: " << std::endl << std::endl; + ggml_vk_print_matrix_area(x, GGML_TYPE_F32, k, m, first_err_m, first_err_n, first_err_b); + std::cerr << std::endl; + std::cerr << "src1: " << std::endl << std::endl; + ggml_vk_print_matrix_area(y, GGML_TYPE_F32, k, n, first_err_m, first_err_n, first_err_b); + + if (split_k > 1) { + float * split_k_buf = (float *) malloc(sizeof(float) * d_ne * split_k); + ggml_vk_buffer_read(ctx->prealloc_split_k, 0, split_k_buf, sizeof(float) * d_ne * split_k); + + std::cerr << "d_buf0: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf1: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf2: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + 2 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + std::cerr << "d_buf3: " << std::endl << std::endl; + ggml_vk_print_matrix_area(split_k_buf + 3 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + + free(split_k_buf); + } + } + + ggml_vk_destroy_buffer(qx_buf); + ggml_vk_destroy_buffer(y_buf); + ggml_vk_destroy_buffer(qy_buf); + ggml_vk_destroy_buffer(d_buf); + + free(x); + free(qx); + free(y); + free(d); + free(d_chk); +} +#endif + +int64_t ggml_vk_get_op_batch_size(const ggml_tensor * op) { + switch (op->op) { + case GGML_OP_GET_ROWS: + return 0; + case GGML_OP_MUL_MAT: + return op->ne[1]; + case GGML_OP_MUL_MAT_ID: + case GGML_OP_ROPE: + case GGML_OP_ROPE_BACK: + return op->ne[2]; + default: + return ggml_nrows(op); + } +} + +#ifdef GGML_VULKAN_CHECK_RESULTS +static void ggml_vk_print_graph_origin(const ggml_tensor * tensor, std::vector& done, int level = 0) { + if (std::find(done.begin(), done.end(), tensor) != done.end() || level > 10) { + return; + } + for (int j = 0; j < level; j++) { + std::cerr << " "; + } + std::cerr << ggml_op_name(tensor->op) << " gpu=" << (tensor->extra != nullptr) << std::endl; + + done.push_back(tensor); + + for (int i = 0; i < GGML_MAX_SRC; i++) { + if (tensor->src[i] != nullptr) { + ggml_vk_print_graph_origin(tensor->src[i], done, level + 1); + } + } +} + +static void ggml_vk_print_tensor_area(const ggml_tensor * tensor, const void * data, int i0, int i1, int i2, int i3) { + if (tensor->type != GGML_TYPE_F32 && tensor->type != GGML_TYPE_F16 && tensor->type != GGML_TYPE_I32) { + return; + } + i0 = std::max(i0, 5); + i1 = std::max(i1, 5); + i2 = std::max(i2, 0); + i3 = std::max(i3, 0); + fprintf(stderr, " "); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + fprintf(stderr, "%7d ", idx1); + } + fprintf(stderr, "\n"); + for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { + fprintf(stderr, "%7d: ", idx0); + for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { + if (idx0 >= 0 && idx0 < tensor->ne[0] && idx1 >= 0 && idx1 < tensor->ne[1] && i2 >= 0 && i2 < tensor->ne[2] && i3 >= 0 && i3 < tensor->ne[3]) { + float val; + if (tensor->type == GGML_TYPE_F32) { + val = *(const float *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); + } else if (tensor->type == GGML_TYPE_F16) { + val = ggml_fp16_to_fp32(*(const ggml_fp16_t *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0])); + } else if (tensor->type == GGML_TYPE_I32) { + val = *(const int32_t *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); + } else { + GGML_ABORT("fatal error"); + } + fprintf(stderr, "% 7.2f ", val); + } else { + fprintf(stderr, " "); + } + } + fprintf(stderr, "\n"); + } +} + +static void ggml_vk_print_tensor(const ggml_tensor * tensor, const char * name) { + void * tensor_data = tensor->data; + + const bool is_gpu = tensor->buffer != nullptr && ggml_backend_buffer_is_vk(tensor->buffer); + + if (is_gpu) { + const size_t tensor_size = ggml_nbytes(tensor); + tensor_data = malloc(tensor_size); + + ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; + + vk_buffer buffer_gpu = buf_ctx->dev_buffer; + ggml_vk_buffer_read(buffer_gpu, vk_tensor_offset(tensor) + tensor->view_offs, tensor_data, tensor_size); + } + + std::cerr << "TENSOR CHECK " << name << " (" << tensor->name << "): " << ggml_op_name(tensor->op) << std::endl; + std::cerr << "tensor=" << tensor << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << std::endl; + if (tensor->src[0] != nullptr) { + std::cerr << "tensor->src[0]=" << tensor->src[0] << " name=" << tensor->src[0]->name << " op=" << ggml_op_name(tensor->src[0]->op) << " type=" << ggml_type_name(tensor->src[0]->type) << " ne0=" << tensor->src[0]->ne[0] << " nb0=" << tensor->src[0]->nb[0] << " ne1=" << tensor->src[0]->ne[1] << " nb1=" << tensor->src[0]->nb[1] << " ne2=" << tensor->src[0]->ne[2] << " nb2=" << tensor->src[0]->nb[2] << " ne3=" << tensor->src[0]->ne[3] << " nb3=" << tensor->src[0]->nb[3] << std::endl; + } + if (tensor->src[1] != nullptr) { + std::cerr << "tensor->src[1]=" << tensor->src[1] << " name=" << tensor->src[1]->name << " op=" << ggml_op_name(tensor->src[1]->op) << " type=" << ggml_type_name(tensor->src[1]->type) << " ne0=" << tensor->src[1]->ne[0] << " nb0=" << tensor->src[1]->nb[0] << " ne1=" << tensor->src[1]->ne[1] << " nb1=" << tensor->src[1]->nb[1] << " ne2=" << tensor->src[1]->ne[2] << " nb2=" << tensor->src[1]->nb[2] << " ne3=" << tensor->src[1]->ne[3] << " nb3=" << tensor->src[1]->nb[3] << std::endl; + } + std::cerr << std::endl << "Result:" << std::endl; + ggml_vk_print_tensor_area(tensor, tensor_data, 5, 5, 0, 0); + std::cerr << std::endl; + std::vector done; + ggml_vk_print_graph_origin(tensor, done); + + if (is_gpu) { + free(tensor_data); + } +} + +void * comp_result; +size_t comp_size; +size_t comp_nb[GGML_MAX_DIMS]; +size_t check_counter = 0; +static void ggml_vk_check_results_0(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx) { + ggml_tensor * tensor = cgraph->nodes[tensor_idx + ctx->num_additional_fused_ops]; + if (tensor->op == GGML_OP_TRANSPOSE || tensor->op == GGML_OP_SET_ROWS) { + return; + } + + check_counter++; + if (!(vk_output_tensor > 0 && vk_output_tensor == check_counter) && check_counter <= vk_skip_checks) { + return; + } + + VK_LOG_DEBUG("ggml_vk_check_results_0(" << tensor->name << ")"); + + struct ggml_init_params iparams = { + /*.mem_size =*/ 2ul*1024ul*1024ul*1024ul, + /*.mem_buffer =*/ NULL, + /*.no_alloc =*/ false, + }; + + struct ggml_context * ggml_ctx = ggml_init(iparams); + + std::array src_clone = {nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr}; + const char * srci_name[GGML_MAX_SRC] = {"src0", "src1", "src2", "src3", "src4", "src5", "src6", "src7", "src8", "src9"}; + + std::map cloned_tensors; + std::vector cloned_mallocs; + + struct ggml_tensor * tensor_clone = nullptr; + + for (int f = 0; f < ctx->num_additional_fused_ops + 1; ++f) { + tensor = cgraph->nodes[tensor_idx + f]; + for (int i = 0; i < GGML_MAX_SRC; i++) { + ggml_tensor * srci = tensor->src[i]; + if (srci == nullptr) { + continue; + } + // If a src tensor has been cloned, use that one + auto it = cloned_tensors.find(srci); + if (it != cloned_tensors.end()) { + src_clone[i] = it->second; + continue; + } + ggml_tensor * srci_clone = ggml_dup_tensor(ggml_ctx, srci); + size_t srci_size = ggml_nbytes(srci); + + src_clone[i] = srci_clone; + void *src_buffer = malloc(srci_size); + cloned_mallocs.push_back(src_buffer); + + srci_clone->data = src_buffer; + if (ggml_backend_buffer_is_host(srci->buffer)) { + memcpy(srci_clone->data, srci->data, srci_size); + memcpy(srci_clone->nb, srci->nb, sizeof(size_t) * GGML_MAX_DIMS); + } else if (ggml_backend_buffer_is_vk(srci->buffer)) { + ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)srci->buffer->context; + vk_buffer& buffer_gpu = buf_ctx->dev_buffer; + uint64_t offset = vk_tensor_offset(srci) + srci->view_offs; + if (!ggml_is_contiguous(srci) && ggml_vk_dim01_contiguous(srci)) { + for (int i3 = 0; i3 < srci->ne[3]; i3++) { + for (int i2 = 0; i2 < srci->ne[2]; i2++) { + const int idx = i3*srci->ne[2] + i2; + ggml_vk_buffer_read(buffer_gpu, offset + idx * srci->nb[2], ((char *)srci_clone->data + idx * srci_clone->nb[2]), srci->ne[1] * srci->nb[1]); + } + } + + srci_clone->nb[0] = srci->nb[0]; + srci_clone->nb[1] = srci->nb[1]; + for (int i = 2; i < GGML_MAX_DIMS; i++) { + srci_clone->nb[i] = srci_clone->nb[i - 1]*srci_clone->ne[i - 1]; + } + } else { + if (offset + srci_size >= buffer_gpu->size) { + srci_size = buffer_gpu->size - offset; + } + ggml_vk_buffer_read(buffer_gpu, offset, srci_clone->data, srci_size); + memcpy(srci_clone->nb, srci->nb, sizeof(size_t) * GGML_MAX_DIMS); + } + } else { + GGML_ABORT("fatal error"); + } + + if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { + ggml_vk_print_tensor(srci, srci_name[i]); + } + } + + if (tensor->op == GGML_OP_FLASH_ATTN_EXT) { + const float * params = (const float *)tensor->op_params; + tensor_clone = ggml_flash_attn_ext(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3], params[0], params[1], params[2]); + if (src_clone[4]) { + ggml_flash_attn_ext_add_sinks(tensor_clone, src_clone[4]); + } + } else if (tensor->op == GGML_OP_MUL_MAT) { + tensor_clone = ggml_mul_mat(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_MUL_MAT_ID) { + tensor_clone = ggml_mul_mat_id(ggml_ctx, src_clone[0], src_clone[1], src_clone[2]); + } else if (tensor->op == GGML_OP_SUB) { + tensor_clone = ggml_sub(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_MUL) { + tensor_clone = ggml_mul(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_DIV) { + tensor_clone = ggml_div(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_CONCAT) { + tensor_clone = ggml_concat(ggml_ctx, src_clone[0], src_clone[1], *(int *)tensor->op_params); + } else if (tensor->op == GGML_OP_UPSCALE) { + tensor_clone = ggml_interpolate(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], (ggml_scale_mode) tensor->op_params[0]); + } else if (tensor->op == GGML_OP_SCALE) { + const float * params = (const float *)tensor->op_params; + tensor_clone = ggml_scale_bias(ggml_ctx, src_clone[0], params[0], params[1]); + } else if (tensor->op == GGML_OP_ADD1) { + tensor_clone = ggml_add1(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_ARANGE) { + const float start = ggml_get_op_params_f32(tensor, 0); + const float stop = ggml_get_op_params_f32(tensor, 1); + const float step = ggml_get_op_params_f32(tensor, 2); + tensor_clone = ggml_arange(ggml_ctx, start, stop, step); + } else if (tensor->op == GGML_OP_FILL) { + const float value = ggml_get_op_params_f32(tensor, 0); + tensor_clone = ggml_fill(ggml_ctx, src_clone[0], value); + } else if (tensor->op == GGML_OP_SQR) { + tensor_clone = ggml_sqr(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_SQRT) { + tensor_clone = ggml_sqrt(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_SIN) { + tensor_clone = ggml_sin(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_COS) { + tensor_clone = ggml_cos(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_LOG) { + tensor_clone = ggml_log(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_TRI) { + tensor_clone = ggml_tri(ggml_ctx, src_clone[0], (ggml_tri_type)ggml_get_op_params_i32(tensor, 0)); + } else if (tensor->op == GGML_OP_DIAG) { + tensor_clone = ggml_diag(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_CLAMP) { + const float * params = (const float *)tensor->op_params; + tensor_clone = ggml_clamp(ggml_ctx, src_clone[0], params[0], params[1]); + } else if (tensor->op == GGML_OP_PAD) { + tensor_clone = ggml_pad_ext(ggml_ctx, src_clone[0], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3], + tensor->op_params[4], tensor->op_params[5], tensor->op_params[6], tensor->op_params[7]); + } else if (tensor->op == GGML_OP_PAD_REFLECT_1D) { + tensor_clone = ggml_pad_reflect_1d(ggml_ctx, src_clone[0], tensor->op_params[0], tensor->op_params[1]); + } else if (tensor->op == GGML_OP_REPEAT) { + tensor_clone = ggml_repeat(ggml_ctx, src_clone[0], tensor); + } else if (tensor->op == GGML_OP_REPEAT_BACK) { + tensor_clone = ggml_repeat_back(ggml_ctx, src_clone[0], tensor); + } else if (tensor->op == GGML_OP_ADD) { + tensor_clone = ggml_add(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_ACC) { + tensor_clone = ggml_acc(ggml_ctx, src_clone[0], src_clone[1], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3]); + } else if (tensor->op == GGML_OP_SET) { + tensor_clone = ggml_set(ggml_ctx, src_clone[0], src_clone[1], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3]); + } else if (tensor->op == GGML_OP_NORM) { + tensor_clone = ggml_norm(ggml_ctx, src_clone[0], *(float *)tensor->op_params); + } else if (tensor->op == GGML_OP_GROUP_NORM) { + const float * float_params = (const float *)tensor->op_params; + tensor_clone = ggml_group_norm(ggml_ctx, src_clone[0], tensor->op_params[0], float_params[1]); + } else if (tensor->op == GGML_OP_RMS_NORM) { + tensor_clone = ggml_rms_norm(ggml_ctx, src_clone[0], *(float *)tensor->op_params); + } else if (tensor->op == GGML_OP_RMS_NORM_BACK) { + const float eps = ((float *) tensor->op_params)[0]; + tensor_clone = ggml_rms_norm_back(ggml_ctx, src_clone[0], src_clone[1], eps); + } else if (tensor->op == GGML_OP_SILU_BACK) { + tensor_clone = ggml_silu_back(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_L2_NORM) { + const float eps = ((float *) tensor->op_params)[0]; + tensor_clone = ggml_l2_norm(ggml_ctx, src_clone[0], eps); + } else if (tensor->op == GGML_OP_SOFT_MAX) { + if (tensor->src[1] != nullptr) { + const float * params = (const float *)tensor->op_params; + tensor_clone = ggml_soft_max_ext(ggml_ctx, src_clone[0], src_clone[1], params[0], params[1]); + } else { + tensor_clone = ggml_soft_max(ggml_ctx, src_clone[0]); + } + } else if (tensor->op == GGML_OP_SOFT_MAX_BACK) { + tensor_clone = ggml_soft_max_ext_back(ggml_ctx, src_clone[0], src_clone[1], ((float *)tensor->op_params)[0], ((float *)tensor->op_params)[1]); + } else if (tensor->op == GGML_OP_DIAG_MASK_INF) { + tensor_clone = ggml_diag_mask_inf(ggml_ctx, src_clone[0], tensor->op_params[0]); + } else if (tensor->op == GGML_OP_ROPE || tensor->op == GGML_OP_ROPE_BACK) { + const int n_dims = ((int32_t *) tensor->op_params)[1]; + const int mode = ((int32_t *) tensor->op_params)[2]; + //const int n_ctx_ggml = ((int32_t *) tensor->op_params)[3]; + const int n_ctx_orig_ggml = ((int32_t *) tensor->op_params)[4]; + const float freq_base = ((float *) tensor->op_params)[5]; + const float freq_scale = ((float *) tensor->op_params)[6]; + const float ext_factor = ((float *) tensor->op_params)[7]; + const float attn_factor = ((float *) tensor->op_params)[8]; + const float beta_fast = ((float *) tensor->op_params)[9]; + const float beta_slow = ((float *) tensor->op_params)[10]; + if (mode & GGML_ROPE_TYPE_MROPE) { + int32_t *sections = ((int32_t *) tensor->op_params) + 11; + if (tensor->op == GGML_OP_ROPE) { + tensor_clone = ggml_rope_multi(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, sections, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); + } else { + tensor_clone = ggml_rope_multi_back(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, sections, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); + } + } else { + if (tensor->op == GGML_OP_ROPE) { + tensor_clone = ggml_rope_ext(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); + } else { + tensor_clone = ggml_rope_ext_back(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); + } + } + const int n_offs = ((int32_t *) tensor->op_params)[15]; + if (n_offs != 0) { + tensor_clone = ggml_rope_set_offset(tensor_clone, n_offs); + } + } else if (tensor->op == GGML_OP_UNARY) { + switch (ggml_get_unary_op(tensor)) { + case GGML_UNARY_OP_EXP: + tensor_clone = ggml_exp(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_EXPM1: + tensor_clone = ggml_expm1(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_ELU: + tensor_clone = ggml_elu(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_SILU: + tensor_clone = ggml_silu(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_GELU: + tensor_clone = ggml_gelu(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_GELU_ERF: + tensor_clone = ggml_gelu_erf(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_GELU_QUICK: + tensor_clone = ggml_gelu_quick(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_RELU: + tensor_clone = ggml_relu(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_XIELU: + tensor_clone = ggml_xielu(ggml_ctx, src_clone[0], 0, 0, 0, 0); + ggml_set_op_params_f32(tensor_clone, 1, ggml_get_op_params_f32(tensor, 1)); + ggml_set_op_params_f32(tensor_clone, 2, ggml_get_op_params_f32(tensor, 2)); + ggml_set_op_params_f32(tensor_clone, 3, ggml_get_op_params_f32(tensor, 3)); + ggml_set_op_params_f32(tensor_clone, 4, ggml_get_op_params_f32(tensor, 4)); + break; + case GGML_UNARY_OP_NEG: + tensor_clone = ggml_neg(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_TANH: + tensor_clone = ggml_tanh(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_SIGMOID: + tensor_clone = ggml_sigmoid(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_HARDSIGMOID: + tensor_clone = ggml_hardsigmoid(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_HARDSWISH: + tensor_clone = ggml_hardswish(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_ABS: + tensor_clone = ggml_abs(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_SOFTPLUS: + tensor_clone = ggml_softplus(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_STEP: + tensor_clone = ggml_step(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_ROUND: + tensor_clone = ggml_round(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_CEIL: + tensor_clone = ggml_ceil(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_FLOOR: + tensor_clone = ggml_floor(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_TRUNC: + tensor_clone = ggml_trunc(ggml_ctx, src_clone[0]); + break; + case GGML_UNARY_OP_SGN: + tensor_clone = ggml_sgn(ggml_ctx, src_clone[0]); + break; + default: + std::cerr << "Missing vk_check_results OP: " << ggml_op_name(tensor->op) << std::endl; + GGML_ABORT("fatal error"); + } + } else if (tensor->op == GGML_OP_GLU) { + if (src_clone[1] == nullptr) { + tensor_clone = ggml_glu(ggml_ctx, src_clone[0], (ggml_glu_op) tensor->op_params[0], tensor->op_params[1]); + } else { + tensor_clone = ggml_glu_split(ggml_ctx, src_clone[0], src_clone[1], (ggml_glu_op) tensor->op_params[0]); + } + ggml_set_op_params_i32(tensor_clone, 2, ggml_get_op_params_i32(tensor, 2)); + ggml_set_op_params_i32(tensor_clone, 3, ggml_get_op_params_i32(tensor, 3)); + } else if (tensor->op == GGML_OP_CPY || tensor->op == GGML_OP_DUP) { + if (tensor->src[1] == nullptr) { + tensor_clone = ggml_dup(ggml_ctx, src_clone[0]); + tensor_clone->type = tensor->type; + } else { + tensor_clone = ggml_cpy(ggml_ctx, src_clone[0], src_clone[1]); + } + } else if (tensor->op == GGML_OP_CONT) { + tensor_clone = ggml_cont_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3]); + } else if (tensor->op == GGML_OP_RESHAPE) { + tensor_clone = ggml_reshape_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3]); + } else if (tensor->op == GGML_OP_VIEW) { + tensor_clone = ggml_view_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], tensor->nb[1], tensor->nb[2], tensor->nb[3], ((int32_t *) tensor->op_params)[0]); + } else if (tensor->op == GGML_OP_PERMUTE) { + int32_t * params = (int32_t *)tensor->op_params; + tensor_clone = ggml_permute(ggml_ctx, src_clone[0], params[0], params[1], params[2], params[3]); + } else if (tensor->op == GGML_OP_TRANSPOSE) { + tensor_clone = ggml_transpose(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_GET_ROWS) { + tensor_clone = ggml_get_rows(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_ARGSORT) { + tensor_clone = ggml_argsort(ggml_ctx, src_clone[0], (ggml_sort_order) *(int *)tensor->op_params); + } else if (tensor->op == GGML_OP_TOP_K) { + tensor_clone = ggml_top_k(ggml_ctx, src_clone[0], tensor->ne[0]); + } else if (tensor->op == GGML_OP_SUM) { + tensor_clone = ggml_sum(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_SUM_ROWS) { + tensor_clone = ggml_sum_rows(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_CUMSUM) { + tensor_clone = ggml_cumsum(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_DSV4_HC_COMB) { + tensor_clone = ggml_dsv4_hc_comb(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], + ggml_get_op_params_f32(tensor, 0), ggml_get_op_params_i32(tensor, 1)); + } else if (tensor->op == GGML_OP_DSV4_HC_PRE) { + if (ggml_get_op_params_i32(tensor, 1) != 0) { + tensor_clone = ggml_dsv4_hc_pre_gated(ggml_ctx, src_clone[0], src_clone[1], ggml_get_op_params_f32(tensor, 0)); + } else { + tensor_clone = ggml_dsv4_hc_pre(ggml_ctx, src_clone[0], src_clone[1]); + } + } else if (tensor->op == GGML_OP_DSV4_HC_POST) { + tensor_clone = ggml_dsv4_hc_post(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3]); + } else if (tensor->op == GGML_OP_MEAN) { + tensor_clone = ggml_mean(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_ARGMAX) { + tensor_clone = ggml_argmax(ggml_ctx, src_clone[0]); + } else if (tensor->op == GGML_OP_CROSS_ENTROPY_LOSS) { + tensor_clone = ggml_cross_entropy_loss(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_CROSS_ENTROPY_LOSS_BACK) { + tensor_clone = ggml_cross_entropy_loss_back(ggml_ctx, src_clone[0], src_clone[1], src_clone[2]); + } else if (tensor->op == GGML_OP_COUNT_EQUAL) { + tensor_clone = ggml_count_equal(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_SOLVE_TRI) { + tensor_clone = ggml_solve_tri(ggml_ctx, src_clone[0], src_clone[1], true, true, false); + } else if (tensor->op == GGML_OP_IM2COL) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t p0 = tensor->op_params[2]; + const int32_t p1 = tensor->op_params[3]; + const int32_t d0 = tensor->op_params[4]; + const int32_t d1 = tensor->op_params[5]; + + const bool is_2D = tensor->op_params[6] == 1; + tensor_clone = ggml_im2col(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1, is_2D, tensor->type); + } else if (tensor->op == GGML_OP_IM2COL_3D) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t s2 = tensor->op_params[2]; + const int32_t p0 = tensor->op_params[3]; + const int32_t p1 = tensor->op_params[4]; + const int32_t p2 = tensor->op_params[5]; + const int32_t d0 = tensor->op_params[6]; + const int32_t d1 = tensor->op_params[7]; + const int32_t d2 = tensor->op_params[8]; + const int32_t IC = tensor->op_params[9]; + + tensor_clone = ggml_im2col_3d(ggml_ctx, src_clone[0], src_clone[1], IC, s0, s1, s2, p0, p1, p2, d0, d1, d2, tensor->type); + } else if (tensor->op == GGML_OP_TIMESTEP_EMBEDDING) { + const int32_t dim = tensor->op_params[0]; + const int32_t max_period = tensor->op_params[1]; + tensor_clone = ggml_timestep_embedding(ggml_ctx, src_clone[0], dim, max_period); + } else if (tensor->op == GGML_OP_CONV_TRANSPOSE_1D){ + const int32_t s0 = tensor->op_params[0]; + const int32_t p0 = tensor->op_params[1]; + const int32_t d0 = tensor->op_params[2]; + tensor_clone = ggml_conv_transpose_1d(ggml_ctx, src_clone[0], src_clone[1], s0, p0, d0); + } else if (tensor->op == GGML_OP_COL2IM_1D) { + const int32_t stride = tensor->op_params[0]; + const int32_t oc = tensor->op_params[1]; + const int32_t p0 = tensor->op_params[2]; + tensor_clone = ggml_col2im_1d(ggml_ctx, src_clone[0], stride, oc, p0); + } else if (tensor->op == GGML_OP_POOL_1D) { + enum ggml_op_pool op = static_cast(tensor->op_params[0]); + const int32_t k0 = tensor->op_params[1]; + const int32_t s0 = tensor->op_params[2]; + const int32_t p0 = tensor->op_params[3]; + + tensor_clone = ggml_pool_1d(ggml_ctx, src_clone[0], op, k0, s0, p0); + } else if (tensor->op == GGML_OP_POOL_2D) { + enum ggml_op_pool op = static_cast(tensor->op_params[0]); + const int32_t k0 = tensor->op_params[1]; + const int32_t k1 = tensor->op_params[2]; + const int32_t s0 = tensor->op_params[3]; + const int32_t s1 = tensor->op_params[4]; + const int32_t p0 = tensor->op_params[5]; + const int32_t p1 = tensor->op_params[6]; + + tensor_clone = ggml_pool_2d(ggml_ctx, src_clone[0], op, k0, k1, s0, s1, p0, p1); + } else if (tensor->op == GGML_OP_CONV_2D) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t p0 = tensor->op_params[2]; + const int32_t p1 = tensor->op_params[3]; + const int32_t d0 = tensor->op_params[4]; + const int32_t d1 = tensor->op_params[5]; + tensor_clone = ggml_conv_2d(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1); + } else if (tensor->op == GGML_OP_CONV_3D) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t s2 = tensor->op_params[2]; + const int32_t p0 = tensor->op_params[3]; + const int32_t p1 = tensor->op_params[4]; + const int32_t p2 = tensor->op_params[5]; + const int32_t d0 = tensor->op_params[6]; + const int32_t d1 = tensor->op_params[7]; + const int32_t d2 = tensor->op_params[8]; + const int32_t IC = tensor->op_params[9]; + const int32_t N = tensor->op_params[10]; + const int32_t OC = tensor->op_params[11]; + tensor_clone = ggml_conv_3d_direct(ggml_ctx, src_clone[0], src_clone[1], s0, s1, s2, p0, p1, p2, d0, d1, d2, IC, N, OC); + } else if (tensor->op == GGML_OP_CONV_2D_DW) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t p0 = tensor->op_params[2]; + const int32_t p1 = tensor->op_params[3]; + const int32_t d0 = tensor->op_params[4]; + const int32_t d1 = tensor->op_params[5]; + tensor_clone = ggml_conv_2d_dw_direct(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1); + } else if (tensor->op == GGML_OP_CONV_TRANSPOSE_2D) { + const int32_t s = tensor->op_params[0]; + tensor_clone = ggml_conv_transpose_2d_p0(ggml_ctx, src_clone[0], src_clone[1], s); + } else if (tensor->op == GGML_OP_LEAKY_RELU) { + const float * op_params = (const float *)tensor->op_params; + tensor_clone = ggml_leaky_relu(ggml_ctx, src_clone[0], op_params[0], false); + } else if (tensor->op == GGML_OP_RWKV_WKV6) { + tensor_clone = ggml_rwkv_wkv6(ggml_ctx, src_clone[0], src_clone[1], + src_clone[2], src_clone[3], src_clone[4], src_clone[5]); + } else if (tensor->op == GGML_OP_RWKV_WKV7) { + tensor_clone = ggml_rwkv_wkv7(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3], + src_clone[4], src_clone[5], src_clone[6]); + } else if (tensor->op == GGML_OP_GATED_LINEAR_ATTN) { + const float * op_params = (const float *)tensor->op_params; + tensor_clone = ggml_gated_linear_attn(ggml_ctx, src_clone[0], src_clone[1], + src_clone[2], src_clone[3], src_clone[4], op_params[0]); + } else if (tensor->op == GGML_OP_LIGHTNING_INDEXER) { + tensor_clone = ggml_lightning_indexer(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3]); + } else if (tensor->op == GGML_OP_GATED_DELTA_NET) { + tensor_clone = ggml_gated_delta_net(ggml_ctx, src_clone[0], src_clone[1], + src_clone[2], src_clone[3], src_clone[4], src_clone[5], + ggml_get_op_params_i32(tensor, 0)); + } else if (tensor->op == GGML_OP_OPT_STEP_ADAMW) { + src_clone[0]->flags = tensor->src[0]->flags; + tensor_clone = ggml_opt_step_adamw(ggml_ctx, src_clone[0], src_clone[1], + src_clone[2], src_clone[3], src_clone[4]); + } else if (tensor->op == GGML_OP_OPT_STEP_SGD) { + src_clone[0]->flags = tensor->src[0]->flags; + tensor_clone = ggml_opt_step_sgd(ggml_ctx, src_clone[0], src_clone[1], + src_clone[2]); + } else if (tensor->op == GGML_OP_ADD_ID) { + tensor_clone = ggml_add_id(ggml_ctx, src_clone[0], src_clone[1], src_clone[2]); + } else if (tensor->op == GGML_OP_SSM_SCAN) { + const int32_t K = ggml_get_op_params_i32(tensor, 0); + tensor_clone = ggml_ssm_scan(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], + src_clone[3], src_clone[4], src_clone[5], src_clone[6], K); + } else if (tensor->op == GGML_OP_SSM_CONV) { + tensor_clone = ggml_ssm_conv(ggml_ctx, src_clone[0], src_clone[1]); + } else if (tensor->op == GGML_OP_ROLL) { + const int32_t s0 = tensor->op_params[0]; + const int32_t s1 = tensor->op_params[1]; + const int32_t s2 = tensor->op_params[2]; + const int32_t s3 = tensor->op_params[3]; + tensor_clone = ggml_roll(ggml_ctx, src_clone[0], s0, s1, s2, s3); + } + else { + std::cerr << "Missing vk_check_results OP: " << ggml_op_name(tensor->op) << std::endl; + GGML_ABORT("fatal error"); + } + cloned_tensors[tensor] = tensor_clone; + } + + ggml_cgraph * cgraph_cpu = ggml_new_graph(ggml_ctx); + ggml_build_forward_expand(cgraph_cpu, tensor_clone); + + ggml_graph_compute_with_ctx(ggml_ctx, cgraph_cpu, 8); + + if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { + ggml_vk_print_tensor(tensor_clone, "tensor_clone"); + } + + comp_size = ggml_nbytes(tensor_clone); + + comp_result = malloc(comp_size); + memcpy(comp_result, tensor_clone->data, comp_size); + memcpy(comp_nb, tensor_clone->nb, sizeof(size_t) * GGML_MAX_DIMS); + + for (auto m : cloned_mallocs) { + free(m); + } + + ggml_free(ggml_ctx); + + VK_LOG_DEBUG("END ggml_vk_check_results_0(" << tensor->name << ")"); +} + +static void ggml_vk_check_results_1(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx) { + ggml_tensor * tensor = cgraph->nodes[tensor_idx + ctx->num_additional_fused_ops]; + if (tensor->op == GGML_OP_TRANSPOSE || tensor->op == GGML_OP_SET_ROWS) { + return; + } + + if (!(vk_output_tensor > 0 && vk_output_tensor == check_counter) && check_counter <= vk_skip_checks) { + return; + } + + VK_LOG_DEBUG("ggml_vk_check_results_1(" << tensor->name << ")"); + + ggml_tensor * src0 = tensor->src[0]; + ggml_tensor * src1 = tensor->src[1]; + ggml_tensor * src2 = tensor->src[2]; + ggml_tensor * src3 = tensor->src[3]; + + void * tensor_data = tensor->data; + + if (ggml_backend_buffer_is_vk(tensor->buffer)) { + size_t tensor_size = ggml_nbytes(tensor); + tensor_data = malloc(tensor_size); + + ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; + + vk_buffer& buffer_gpu = buf_ctx->dev_buffer; + uint64_t offset = vk_tensor_offset(tensor) + tensor->view_offs; + if (offset + tensor_size >= buffer_gpu->size) { + tensor_size = buffer_gpu->size - offset; + } + + ggml_vk_buffer_read(buffer_gpu, offset, tensor_data, tensor_size); + } + + float first_error_result = -1.0f; + float first_error_correct = -1.0f; + std::array first_error = { -1, -1, -1, -1 }; + double avg_err = 0.0; + size_t counter = 0; + + for (int i3 = 0; i3 < tensor->ne[3]; i3++) { + for (int i2 = 0; i2 < tensor->ne[2]; i2++) { + for (int i1 = 0; i1 < tensor->ne[1]; i1++) { + for (int i0 = 0; i0 < tensor->ne[0]; i0++) { + const bool buffer_size_fit = i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0] < comp_size; + float correct = 0.0f; + float result = 0.0f; + + if (buffer_size_fit) { + if (tensor->type == GGML_TYPE_F32) { + correct = *(float *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); + result = *(float *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); + } else if (tensor->type == GGML_TYPE_F16) { + correct = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0])); + result = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0])); + } else if (tensor->type == GGML_TYPE_BF16) { + correct = ggml_bf16_to_fp32(*(ggml_bf16_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0])); + result = ggml_bf16_to_fp32(*(ggml_bf16_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0])); + } else if (tensor->type == GGML_TYPE_I32) { + correct = *(int32_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); + result = *(int32_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); + } else if (tensor->type == GGML_TYPE_I64) { + correct = *(int64_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); + result = *(int64_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); + } else { + std::cerr << "Results check not implemented for type " << ggml_type_name(tensor->type) << std::endl; + } + } else { + std::cerr << "Missing debug code for type " << ggml_type_name(tensor->type) << std::endl; + GGML_ABORT("fatal error"); + } + + if ((std::isnan(correct) != std::isnan(result)) || (std::isinf(correct) != std::isinf(result)) || !buffer_size_fit) { + std::cerr << "ERROR: Invalid value in " << ggml_op_name(tensor->op) << " i3=" << i3 << " i2=" << i2 << " i1=" << i1 << " i0=" << i0 << " result=" << result << " correct=" << correct << " avg_err=" << (avg_err / counter) << std::endl; + std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; + if (src0 != nullptr) { + std::cerr << "src0=" << src0 << " src0->name=" << src0->name << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; + } + if (src1 != nullptr) { + std::cerr << "src1=" << src1 << " src1->name=" << src1->name << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; + } + if (src2 != nullptr) { + std::cerr << "src2=" << src2 << " src2->name=" << src2->name << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; + } + if (src3 != nullptr) { + std::cerr << "src3=" << src3 << " src3->name=" << src3->name << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; + } + std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; + std::cerr << std::endl << "Result:" << std::endl; + ggml_vk_print_tensor_area(tensor, tensor_data, i0, i1, i2, i3); + std::cerr << std::endl << "Correct:" << std::endl; + ggml_vk_print_tensor_area(tensor, comp_result, i0, i1, i2, i3); + std::cerr << std::endl; + std::vector done; + ggml_vk_print_graph_origin(tensor, done); + GGML_ABORT("fatal error"); + } + const double denom = std::fabs(correct) > 1.0f ? (std::fabs(correct) > 1e-8 ? std::fabs(correct) : 1e-8) : 1.0f; + if (first_error[0] == -1 && std::fabs(correct - result) / denom > 0.5) { + first_error[0] = i0; + first_error[1] = i1; + first_error[2] = i2; + first_error[3] = i3; + first_error_result = result; + first_error_correct = correct; + } + + // Special case, value is infinite, avoid NaN result in avg_err + // NaN also appears in results, if both are nan error is 0 + if (!std::isinf(correct) && !std::isinf(result) && !std::isnan(correct) && !std::isnan(result)) { + avg_err += std::fabs(correct - result) / denom; + } + counter++; + } + } + } + } + + avg_err /= counter; + + if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { + std::cerr << "TENSOR CHECK: avg_err=" << avg_err << " in " << ggml_op_name(tensor->op) << " (check " << check_counter << ")" << std::endl; + std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; + if (src0 != nullptr) { + std::cerr << "src0=" << src0 << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; + } + if (src1 != nullptr) { + std::cerr << "src1=" << src1 << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; + } + if (src2 != nullptr) { + std::cerr << "src2=" << src2 << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; + } + if (src3 != nullptr) { + std::cerr << "src3=" << src3 << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; + } + std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; + std::cerr << std::endl << "Result:" << std::endl; + ggml_vk_print_tensor_area(tensor, tensor_data, 5, 5, 0, 0); + std::cerr << std::endl << "Correct:" << std::endl; + ggml_vk_print_tensor_area(tensor, comp_result, 5, 5, 0, 0); + std::cerr << std::endl; + std::vector done; + ggml_vk_print_graph_origin(tensor, done); + } + + if (avg_err > 0.01 || std::isnan(avg_err)) { + std::cerr << "ERROR: avg_err=" << avg_err << " in " << ggml_op_name(tensor->op) << " (check " << check_counter << ")" << std::endl; + std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; + if (src0 != nullptr) { + std::cerr << "src0=" << src0 << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; + } + if (src1 != nullptr) { + std::cerr << "src1=" << src1 << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; + } + if (src2 != nullptr) { + std::cerr << "src2=" << src2 << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; + } + if (src3 != nullptr) { + std::cerr << "src3=" << src3 << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; + } + std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; + std::cerr << std::endl << "Result:" << std::endl; + ggml_vk_print_tensor_area(tensor, tensor_data, first_error[0], first_error[1], first_error[2], first_error[3]); + std::cerr << std::endl << "Correct:" << std::endl; + ggml_vk_print_tensor_area(tensor, comp_result, first_error[0], first_error[1], first_error[2], first_error[3]); + std::cerr << std::endl; + std::vector done; + ggml_vk_print_graph_origin(tensor, done); + GGML_ABORT("fatal error"); + } else { + std::cerr << check_counter << " " << tensor->name << " op=" << ggml_op_name(tensor->op) << " avg_err=" << avg_err << std::endl; + } + + free(comp_result); + comp_result = nullptr; + comp_size = 0; + + if (ggml_backend_buffer_is_vk(tensor->buffer)) { + free(tensor_data); + } + + VK_LOG_DEBUG("END ggml_vk_check_results_1(" << tensor->name << ")"); +} +#endif + diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-push-constants.h b/ggml/src/ggml-vulkan/ggml-vulkan-push-constants.h new file mode 100644 index 00000000..8446e313 --- /dev/null +++ b/ggml/src/ggml-vulkan/ggml-vulkan-push-constants.h @@ -0,0 +1,1110 @@ +#pragma once +#include "ggml-vulkan-types.h" + +uint32_t get_misalign_bytes(const ggml_backend_vk_context * ctx, const ggml_tensor * t); + +uint32_t ggml_vk_concat_unit_size(ggml_type type); + +struct vk_mat_mat_push_constants { + uint32_t M; uint32_t N; uint32_t K; + uint32_t stride_a; uint32_t stride_b; uint32_t stride_d; + uint32_t batch_stride_a; uint32_t batch_stride_b; uint32_t batch_stride_d; + uint32_t base_work_group_z; uint32_t num_batches; + uint32_t k_split; + uint32_t ne02; uint32_t ne12; uint32_t broadcast2; uint32_t broadcast3; + uint32_t padded_N; +}; + +struct vk_mat_vec_push_constants { + uint32_t ncols; + uint32_t stride_a; + uint32_t stride_b; + uint32_t stride_d; + uint32_t batch_stride_a; + uint32_t batch_stride_b; + uint32_t batch_stride_d; + uint32_t fusion_flags; + uint32_t base_work_group_y; + uint32_t ne02; + uint32_t ne12; + uint32_t broadcast2; + uint32_t broadcast3; +}; + +struct vk_mat_vec_p021_push_constants { + uint32_t ncols_x; + uint32_t nrows_x; + uint32_t nchannels_x; + uint32_t nchannels_y; + uint32_t b_offset; + uint32_t d_offset; + uint32_t fusion_flags; +}; + +struct vk_mat_vec_nc_push_constants { + uint32_t ncols_x; + uint32_t nrows_x; + uint32_t row_stride_x; + uint32_t channel_stride_x; + uint32_t channel_stride_y; + uint32_t channel_x_divisor; + uint32_t ne12; + uint32_t b_offset; + uint32_t d_offset; + uint32_t nb03; + uint32_t nb13; + uint32_t nb23; + uint32_t fusion_flags; +}; + +struct vk_mat_mat_id_push_constants { + uint32_t M; uint32_t N; uint32_t K; + uint32_t stride_a; uint32_t stride_b; uint32_t stride_d; + uint32_t batch_stride_a; uint32_t batch_stride_b; uint32_t batch_stride_d; + uint32_t nei0; uint32_t nei1; uint32_t nbi1; uint32_t ne11; + uint32_t n_experts; + uint32_t hoist_row_ids; +}; + +struct vk_mat_vec_id_push_constants { + uint32_t ncols; + uint32_t stride_a; + uint32_t stride_b; + uint32_t stride_d; + uint32_t batch_stride_a; + uint32_t batch_stride_b; + uint32_t batch_stride_d; + uint32_t fusion_flags; + uint32_t nei0; + uint32_t ne11; + uint32_t expert_i1; + uint32_t nbi1; +}; + +struct vk_flash_attn_push_constants { + uint32_t N; + uint32_t KV; + + uint32_t ne1; + uint32_t ne2; + uint32_t ne3; + + uint32_t neq2; + uint32_t neq3; + uint32_t nek2; + uint32_t nek3; + uint32_t nev2; + uint32_t nev3; + uint32_t nem1; + uint32_t nem2; + uint32_t nem3; + + uint32_t nb01; + uint32_t nb02; + uint32_t nb03; + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + uint32_t nb21; + uint32_t nb22; + uint32_t nb23; + + float scale; + float max_bias; + float logit_softcap; + + uint32_t mask_n_head_log2; + float m0; + float m1; + + uint32_t gqa_ratio; + uint32_t split_kv; + uint32_t k_num; +}; + +static_assert(sizeof(vk_flash_attn_push_constants) <= 128, "sizeof(vk_flash_attn_push_constants) must be <= 128"); + +struct vk_fa_xe_opt_push_constants { + uint32_t kv_seq_len; + uint32_t activation_length; + uint32_t q_head; + uint32_t kv_head; + uint32_t qk_ratio; + uint32_t qk_sub_groups; + uint32_t flag; + uint32_t nbkv_tok; + uint32_t nbkv_head; + uint32_t batch_stride_q; + uint32_t batch_stride_k; + uint32_t batch_stride_v; + uint32_t batch_stride_m; + uint32_t batch_stride_o; + float softmax_scale; +}; + +struct vk_op_push_constants { + uint32_t KX; + uint32_t KY; + float param1; + float param2; + float param3; + float param4; +}; + +struct vk_op_fwht_push_constants { + uint32_t n_rows; + uint32_t src_offset; + uint32_t dst_offset; + float scale; +}; + +struct vk_op_dsv4_hc_comb_push_constants { + uint32_t n_tokens; + + uint32_t nbm0; uint32_t nbm1; + uint32_t nbs0; + uint32_t nbb0; + uint32_t nbd0; uint32_t nbd1; uint32_t nbd2; + + uint32_t m_offset; + uint32_t s_offset; + uint32_t b_offset; + uint32_t d_offset; + + float eps; + uint32_t n_iter; +}; + +struct vk_op_dsv4_hc_pre_push_constants { + uint32_t n_embd; + uint32_t n_tokens; + + uint32_t nbx0; uint32_t nbx1; uint32_t nbx2; + uint32_t nbw0; uint32_t nbw1; uint32_t nbw2; + uint32_t nbd0; uint32_t nbd1; + + uint32_t x_offset; + uint32_t w_offset; + uint32_t d_offset; + + float scale; +}; + +struct vk_op_dsv4_hc_post_push_constants { + uint32_t n_embd; + uint32_t n_tokens; + + uint32_t nbx0; uint32_t nbx1; + uint32_t nbr0; uint32_t nbr1; uint32_t nbr2; + uint32_t nbp0; uint32_t nbp1; + uint32_t nbc0; uint32_t nbc1; uint32_t nbc2; + uint32_t nbd0; uint32_t nbd1; uint32_t nbd2; + + uint32_t x_offset; + uint32_t r_offset; + uint32_t p_offset; + uint32_t c_offset; + uint32_t d_offset; +}; + +struct vk_op_count_experts_push_constants { + uint32_t ne00; + uint32_t ne01; + uint32_t nb00; + uint32_t nb01; + uint32_t a_offset; + uint32_t n_experts; + uint32_t hoist_row_ids; + uint32_t ne00mp; + uint32_t ne00L; +}; + +struct vk_op_glu_push_constants { + uint32_t N; + uint32_t ne00; + uint32_t ne20; + uint32_t mode; // 0: default, 1: swapped, 2: split + float alpha; // for swiglu_oai + float limit; + uint32_t nb00; + uint32_t nb01; + uint32_t nb02; + uint32_t nb03; + uint32_t nb10; + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + uint32_t nb20; + uint32_t nb21; + uint32_t nb22; + uint32_t nb23; + uint32_t ne21; + uint32_t ne22; + uint32_t misalign_offsets; + uint32_t ne2_012mp; uint32_t ne2_012L; + uint32_t ne2_01mp; uint32_t ne2_01L; + uint32_t ne2_0mp; uint32_t ne2_0L; +}; + +static_assert(sizeof(vk_op_glu_push_constants) <= 128, "sizeof(vk_op_glu_push_constants) must be <= 128"); + +struct vk_op_unary_push_constants { + uint32_t ne; + uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; + uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; + uint32_t misalign_offsets; + float param1; float param2; float param3; float param4; + uint32_t ne0_012mp; uint32_t ne0_01mp; uint32_t ne0_0mp; uint32_t ne0_Ls; + uint32_t ne1_012mp; uint32_t ne1_01mp; uint32_t ne1_0mp; uint32_t ne1_Ls; +}; + +static_assert(sizeof(vk_op_unary_push_constants) <= 128, "sizeof(vk_op_unary_push_constants) must be <= 128"); + +static vk_op_unary_push_constants vk_op_unary_push_constants_init(const ggml_tensor * src0, const ggml_tensor * dst, int64_t ne = 0) { + GGML_ASSERT(ne != 0 || (ggml_nelements(src0) == ggml_nelements(dst))); + ne = ne != 0 ? ne : ggml_nelements(dst); + GGML_ASSERT(ne <= (int64_t)std::numeric_limits::max()); + + vk_op_unary_push_constants p{}; + p.ne = (uint32_t)ne; + + size_t src0_tsize = ggml_type_size(src0->type); + p.ne00 = (uint32_t)src0->ne[0]; + p.ne01 = (uint32_t)src0->ne[1]; + p.ne02 = (uint32_t)src0->ne[2]; + p.ne03 = (uint32_t)src0->ne[3]; + p.nb00 = (uint32_t)(src0->nb[0] / src0_tsize); + p.nb01 = (uint32_t)(src0->nb[1] / src0_tsize); + p.nb02 = (uint32_t)(src0->nb[2] / src0_tsize); + p.nb03 = (uint32_t)(src0->nb[3] / src0_tsize); + + size_t dst_tsize = ggml_type_size(dst->type); + p.ne10 = (uint32_t)dst->ne[0]; + p.ne11 = (uint32_t)dst->ne[1]; + p.ne12 = (uint32_t)dst->ne[2]; + p.ne13 = (uint32_t)dst->ne[3]; + p.nb10 = (uint32_t)(dst->nb[0] / dst_tsize); + p.nb11 = (uint32_t)(dst->nb[1] / dst_tsize); + p.nb12 = (uint32_t)(dst->nb[2] / dst_tsize); + p.nb13 = (uint32_t)(dst->nb[3] / dst_tsize); + + return p; // offsets are initialized later in ggml_vk_op +} + +struct vk_op_pad_push_constants { + uint32_t ne; + uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; + uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; + uint32_t misalign_offsets; + uint32_t circular; + + uint32_t lp0; uint32_t rp0; + uint32_t lp1; uint32_t rp1; + uint32_t lp2; uint32_t rp2; + uint32_t lp3; uint32_t rp3; +}; + +static vk_op_pad_push_constants vk_op_pad_push_constants_init(const ggml_tensor * src0, const ggml_tensor * dst) { + int64_t ne = ggml_nelements(dst); + GGML_ASSERT(ne <= (int64_t)std::numeric_limits::max()); + + vk_op_pad_push_constants p{}; + p.ne = (uint32_t)ne; + + size_t src0_tsize = ggml_type_size(src0->type); + p.ne00 = (uint32_t)src0->ne[0]; + p.ne01 = (uint32_t)src0->ne[1]; + p.ne02 = (uint32_t)src0->ne[2]; + p.ne03 = (uint32_t)src0->ne[3]; + p.nb00 = (uint32_t)(src0->nb[0] / src0_tsize); + p.nb01 = (uint32_t)(src0->nb[1] / src0_tsize); + p.nb02 = (uint32_t)(src0->nb[2] / src0_tsize); + p.nb03 = (uint32_t)(src0->nb[3] / src0_tsize); + + size_t dst_tsize = ggml_type_size(dst->type); + p.ne10 = (uint32_t)dst->ne[0]; + p.ne11 = (uint32_t)dst->ne[1]; + p.ne12 = (uint32_t)dst->ne[2]; + p.ne13 = (uint32_t)dst->ne[3]; + p.nb10 = (uint32_t)(dst->nb[0] / dst_tsize); + p.nb11 = (uint32_t)(dst->nb[1] / dst_tsize); + p.nb12 = (uint32_t)(dst->nb[2] / dst_tsize); + p.nb13 = (uint32_t)(dst->nb[3] / dst_tsize); + + p.lp0 = dst->op_params[0]; + p.rp0 = dst->op_params[1]; + p.lp1 = dst->op_params[2]; + p.rp1 = dst->op_params[3]; + p.lp2 = dst->op_params[4]; + p.rp2 = dst->op_params[5]; + p.lp3 = dst->op_params[6]; + p.rp3 = dst->op_params[7]; + p.circular = dst->op_params[8]; + + return p; // fastdiv values and offsets are initialized later in ggml_vk_op +} + +static void init_fastdiv_values(uint32_t d, uint32_t &mp, uint32_t &L) +{ + // compute L = ceil(log2(d)); + L = 0; + while (L < 32 && (uint32_t{1} << L) < d) { + L++; + } + + mp = (uint32_t)((uint64_t{1} << 32) * ((uint64_t{1} << L) - d) / d + 1); +} + +static uint32_t pack_fastdiv_L(uint32_t L0, uint32_t L1, uint32_t L2) { + return L0 | (L1 << 8) | (L2 << 16); +} + +template void init_pushconst_fastdiv(T &p) { + GGML_UNUSED(p); + static_assert(!std::is_const::value, "unexpected type"); +} + +template <> inline void init_pushconst_fastdiv(vk_op_unary_push_constants &p) { + // Compute magic values to divide by these six numbers. + uint32_t ne0_012L; + uint32_t ne0_01L; + uint32_t ne0_0L; + uint32_t ne1_012L; + uint32_t ne1_01L; + uint32_t ne1_0L; + + init_fastdiv_values(p.ne02*p.ne01*p.ne00, p.ne0_012mp, ne0_012L); + init_fastdiv_values(p.ne01*p.ne00, p.ne0_01mp, ne0_01L); + init_fastdiv_values(p.ne00, p.ne0_0mp, ne0_0L); + init_fastdiv_values(p.ne12*p.ne11*p.ne10, p.ne1_012mp, ne1_012L); + init_fastdiv_values(p.ne11*p.ne10, p.ne1_01mp, ne1_01L); + init_fastdiv_values(p.ne10, p.ne1_0mp, ne1_0L); + + p.ne0_Ls = pack_fastdiv_L(ne0_012L, ne0_01L, ne0_0L); + p.ne1_Ls = pack_fastdiv_L(ne1_012L, ne1_01L, ne1_0L); +} + +template <> inline void init_pushconst_fastdiv(vk_op_glu_push_constants &p) { + // GLU linearizes over dst, then uses dst coordinates for src0/src1. + init_fastdiv_values(p.ne22*p.ne21*p.ne20, p.ne2_012mp, p.ne2_012L); + init_fastdiv_values(p.ne21*p.ne20, p.ne2_01mp, p.ne2_01L); + init_fastdiv_values(p.ne20, p.ne2_0mp, p.ne2_0L); +} + +template <> inline void init_pushconst_fastdiv(vk_op_count_experts_push_constants &p) { + init_fastdiv_values(p.ne00, p.ne00mp, p.ne00L); +} + +struct vk_op_binary_push_constants { + uint32_t ne; + uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; + uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; + uint32_t ne20; uint32_t ne21; uint32_t ne22; uint32_t ne23; uint32_t nb20; uint32_t nb21; uint32_t nb22; uint32_t nb23; + uint32_t misalign_offsets; + float param1; float param2; int32_t param3; +}; + +struct vk_op_concat_push_constants : vk_op_binary_push_constants {}; + +static_assert(sizeof(vk_op_concat_push_constants) == sizeof(vk_op_binary_push_constants)); + +static_assert(std::is_standard_layout_v); + +struct vk_op_multi_add_push_constants { + // shape for dst + uint32_t ne20; uint32_t ne21; uint32_t ne22; uint32_t ne23; + + // strides for srcs+dst + uint32_t nb[MAX_PARAMETER_COUNT][4]; + + uint32_t rms_partials; +}; + +static_assert(MAX_PARAMETER_COUNT == 12); + +static_assert(sizeof(vk_op_multi_add_push_constants) <= 256); + +struct vk_op_topk_moe_push_constants { + uint32_t n_rows; + uint32_t n_experts_push; + uint32_t n_expert_used; + float clamp_min; + float clamp_max; + uint32_t gating_func; + uint32_t has_bias; + uint32_t with_norm; + float output_scale; + float output_bias; +}; + +struct vk_op_add_id_push_constants { + uint32_t ne0; + uint32_t ne1; + uint32_t s01; + uint32_t s02; + uint32_t s11; + uint32_t s21; +}; + +struct vk_op_diag_mask_push_constants { + uint32_t ncols; + uint32_t rows_per_channel; + int32_t n_past; +}; + +struct vk_op_rope_push_constants { + uint32_t rope_mode; + uint32_t nrows; + uint32_t n_dims; + uint32_t n_offs; + float freq_scale; + float freq_base; + float ext_factor; + float attn_factor; + float corr_dims[2]; + float theta_scale; + uint32_t has_ff; + int32_t sections[4]; + uint32_t is_imrope; + uint32_t is_back; + uint32_t set_rows_stride; + uint32_t ne00; + uint32_t ne01; + uint32_t ne02; + uint32_t nb01; + uint32_t nb02; + uint32_t nb03; + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + uint32_t a_offset; + uint32_t d_offset; +}; + +static_assert(sizeof(vk_op_rope_push_constants) <= 128, "sizeof(vk_op_rope_push_constants) must be <= 128"); + +struct vk_op_rms_norm_mul_rope_push_constants { + vk_op_binary_push_constants bin; + vk_op_rope_push_constants rope; +}; + +struct vk_op_soft_max_push_constants { + uint32_t KX; + uint32_t KY; + uint32_t ne00; + uint32_t ne01; + uint32_t ne02; + uint32_t ne12; + uint32_t ne13; + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + float scale; + float max_bias; + float m0; + float m1; + uint32_t n_head_log2; + uint32_t nrows_x; + uint32_t has_sinks; +}; + +struct vk_op_argsort_push_constants { + uint32_t ncols; + uint32_t ncols_padded; + uint32_t ncols_padded_log2; + uint32_t nrows; + uint32_t order; + uint32_t outer_start; + uint32_t outer_end; + uint32_t inner_start; + uint32_t inner_end; +}; + +struct vk_op_topk_push_constants { + uint32_t orig_ncols; + uint32_t ncols_input; + uint32_t ncols_output; + uint32_t k; + uint32_t nrows; + uint32_t first_pass; + uint32_t last_pass; +}; + +struct vk_op_topk_radix_push_constants { + uint32_t ncols; + uint32_t k; + uint32_t nrows; + uint32_t n_tps; // QSA only + uint32_t n_blocks; // QSA only + uint32_t n_stream; // QSA only +}; + +struct vk_op_im2col_push_constants { + uint64_t dst_addr; + uint32_t batch_offset; uint32_t offset_delta; + uint32_t IC; + uint32_t IW; uint32_t IH; + uint32_t OW; uint32_t OH; + uint32_t KW; uint32_t KH; + uint32_t OH_batch; + uint32_t CHW; + int32_t s0; int32_t s1; + int32_t p0; int32_t p1; + int32_t d0; int32_t d1; + uint32_t batch_IC; +}; + +struct vk_op_im2col_3d_push_constants { + uint64_t dst_addr; + uint32_t nb10; + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + uint32_t s0; + uint32_t s1; + uint32_t s2; + uint32_t p0; + uint32_t p1; + uint32_t p2; + uint32_t d0; + uint32_t d1; + uint32_t d2; + uint32_t IW; + uint32_t IH; + uint32_t ID; + uint32_t IC; + uint32_t KW; + uint32_t OH; + uint32_t KD_KH_KW; + uint32_t KH_KW; + uint32_t IC_KD_KH_KW; + uint32_t N_OD_OH; + uint32_t OD_OH; + uint32_t OD_OH_OW_IC_KD_KH_KW; + uint32_t OH_OW_IC_KD_KH_KW; + uint32_t OW_IC_KD_KH_KW; + uint32_t misalign_offsets; +}; + +struct vk_op_timestep_embedding_push_constants { + uint32_t nb1; + uint32_t dim; + uint32_t max_period; +}; + +struct vk_op_col2im_1d_push_constants { + uint32_t T_out; + uint32_t OC; + uint32_t K_OC; + uint32_t T_in; + uint32_t K; + int32_t stride; + int32_t p0; +}; + +struct vk_op_conv_transpose_1d_push_constants { + uint32_t Cout; + uint32_t Cin; + uint32_t K; + uint32_t L; + uint32_t KL; + + uint32_t nb01; + uint32_t nb02; + uint32_t nb11; + uint32_t nb1; + + int32_t s0; +}; + +struct vk_op_snake_push_constants { + uint32_t ne0; + uint32_t ne1; +}; + +struct vk_op_pool1d_push_constants { + uint32_t IL; + uint32_t OL; + uint32_t OC; + uint32_t pelements; + uint32_t op; + int32_t k0; + int32_t s0; + int32_t p0; +}; + +struct vk_op_pool2d_push_constants { + uint32_t IW; uint32_t IH; + uint32_t OW; uint32_t OH; + uint32_t OC; + uint32_t pelements; + uint32_t op; + int32_t k0; int32_t k1; + int32_t s0; int32_t s1; + int32_t p0; int32_t p1; +}; + +struct vk_op_rwkv_wkv6_push_constants { + uint32_t B; + uint32_t T; + uint32_t C; + uint32_t H; +}; + +struct vk_op_rwkv_wkv7_push_constants { + uint32_t B; + uint32_t T; + uint32_t C; + uint32_t H; +}; + +struct vk_op_gated_linear_attn_push_constants { + uint32_t B; + uint32_t T; + uint32_t C; + uint32_t H; + float scale; +}; + +struct vk_op_lightning_indexer_push_constants { + uint32_t n_kv; + uint32_t n_heads; + uint32_t n_tokens; + uint32_t n_streams; + uint32_t n_masks; + uint32_t dispatch_x; + uint32_t q_nb1; + uint32_t q_nb2; + uint32_t q_nb3; + uint32_t k_nb2; + uint32_t k_nb3; + uint32_t w_nb1; + uint32_t w_nb3; + uint32_t m_nb1; + uint32_t m_nb3; + uint32_t d_nb1; + uint32_t d_nb3; +}; + +static_assert(sizeof(vk_op_lightning_indexer_push_constants) <= 128); + +struct vk_op_gated_delta_net_push_constants { + uint32_t H; + uint32_t n_tokens; + uint32_t n_seqs; + uint32_t s_off; + uint32_t sq1, sq2, sq3; + uint32_t sv1, sv2, sv3; + uint32_t sb1, sb2, sb3; + uint32_t neq1, rq3; + float scale; + uint32_t K; +}; + +struct vk_op_ssm_scan_push_constants { + uint32_t nb02, nb03, nb12, nb13; + uint32_t nb21, nb22, nb31; + uint32_t nb42, nb43, nb52, nb53; + uint32_t s_off; + uint32_t n_head, d_head, n_group, n_tok; + uint32_t n_seq, K; +}; + +struct vk_op_ssm_conv_push_constants { + uint32_t nb01, nb02; + uint32_t nb11; + uint32_t dst_nb0, dst_nb1, dst_nb2; + uint32_t nc, ncs, nr, n_t, n_s; +}; + +struct vk_op_conv2d_push_constants { + uint32_t Cout; + uint32_t Cin; + uint32_t N; + + uint32_t W; + uint32_t H; + uint32_t OW; + uint32_t OH; + + uint32_t nb01; + uint32_t nb02; + uint32_t nb03; + + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + + uint32_t nb1; + uint32_t nb2; + uint32_t nb3; + + // init_fastdiv_values constants for dividing by OW, OW*OH + uint32_t OWmp; uint32_t OWL; + uint32_t OWOHmp; uint32_t OWOHL; +}; + +template <> inline void init_pushconst_fastdiv(vk_op_conv2d_push_constants &p) { + // Compute magic values to divide by OW, OW*OH + init_fastdiv_values(p.OW, p.OWmp, p.OWL); + init_fastdiv_values(p.OW*p.OH, p.OWOHmp, p.OWOHL); +} + +struct vk_op_conv3d_push_constants { + uint32_t OC; + uint32_t IC; + uint32_t N; + + uint32_t IW; + uint32_t IH; + uint32_t ID; + uint32_t OW; + uint32_t OH; + uint32_t OD; + + uint32_t nb01; + uint32_t nb02; + uint32_t nb03; + + uint32_t nb11; + uint32_t nb12; + uint32_t nb13; + + uint32_t nb1; + uint32_t nb2; + uint32_t nb3; + + uint32_t OWmp; uint32_t OWL; + uint32_t OWOHmp; uint32_t OWOHL; + uint32_t OWOHODmp; uint32_t OWOHODL; +}; + +template <> inline void init_pushconst_fastdiv(vk_op_conv3d_push_constants &p) { + init_fastdiv_values(p.OW, p.OWmp, p.OWL); + init_fastdiv_values(p.OW*p.OH, p.OWOHmp, p.OWOHL); + init_fastdiv_values(p.OW*p.OH*p.OD, p.OWOHODmp, p.OWOHODL); +} + +struct vk_op_conv2d_dw_push_constants { + uint32_t ne; + uint32_t batches; + uint32_t channels; + uint32_t dst_w; + uint32_t dst_h; + uint32_t src_w; + uint32_t src_h; + uint32_t knl_w; + uint32_t knl_h; + int32_t stride_x; + int32_t stride_y; + int32_t pad_x; + int32_t pad_y; + int32_t dilation_x; + int32_t dilation_y; +}; + +struct vk_op_upscale_push_constants { + uint32_t ne; uint32_t a_offset; uint32_t d_offset; + uint32_t ne00; uint32_t ne01; + uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; + uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; + float sf0; float sf1; float sf2; float sf3; + float pixel_offset; +}; + +struct vk_op_sum_rows_push_constants +{ + uint32_t n_cols; + uint32_t ne01, ne02; + uint32_t nb01, nb02, nb03; + uint32_t nb11, nb12, nb13; + float weight; + uint32_t misalign_offsets; + uint32_t ne0_12mp, ne0_12L; + uint32_t ne0_1mp, ne0_1L; +}; + +static vk_op_sum_rows_push_constants vk_op_sum_rows_push_constants_init(const ggml_tensor * src, const ggml_tensor * dst, int64_t n_cols) { + uint32_t type_size = (uint32_t)ggml_type_size(src->type); + vk_op_sum_rows_push_constants p = {}; + p.n_cols = (uint32_t)n_cols; + p.ne01 = (uint32_t)src->ne[1]; + p.ne02 = (uint32_t)src->ne[2]; + p.nb01 = (uint32_t)src->nb[1] / type_size; + p.nb02 = (uint32_t)src->nb[2] / type_size; + p.nb03 = (uint32_t)src->nb[3] / type_size; + p.nb11 = (uint32_t)dst->nb[1] / type_size; + p.nb12 = (uint32_t)dst->nb[2] / type_size; + p.nb13 = (uint32_t)dst->nb[3] / type_size; + p.weight = 1.0f; + return p; +} + +template <> inline void init_pushconst_fastdiv(vk_op_sum_rows_push_constants &p) { + init_fastdiv_values(p.ne01*p.ne02, p.ne0_12mp, p.ne0_12L); + init_fastdiv_values(p.ne01, p.ne0_1mp, p.ne0_1L); +} + +struct vk_quantize_q8_1_push_constants { + uint32_t ne; + uint32_t num_blocks; +}; + +struct vk_op_flash_attn_split_k_reduce_push_constants { + uint32_t D; + uint32_t ne1; + uint32_t ne2; + uint32_t ne3; + uint32_t k_num; + uint32_t sinks; +}; + +struct vk_op_flash_attn_mask_opt_push_constants { + uint32_t nem0; + uint32_t nem1; + uint32_t nem2; + uint32_t nbm1; + uint32_t nbm2; + uint32_t nbm3; + uint32_t nbd1; + uint32_t nbd2; + uint32_t nbd3; +}; + +struct vk_op_flash_attn_sparse_compact_push_constants { + uint32_t KV; + uint32_t nem1; + uint32_t nem2; + uint32_t nbm1; + uint32_t nbm2; + uint32_t nbm3; + uint32_t n_kv_max; +}; + +template void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, T &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + GGML_UNUSED(p); + GGML_UNUSED(src0); + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); + GGML_UNUSED(dst); + static_assert(!std::is_const::value, "unexpected type"); + GGML_ASSERT(!src0 || get_misalign_bytes(ctx, src0) == 0); + GGML_ASSERT(!src1 || get_misalign_bytes(ctx, src1) == 0); + GGML_ASSERT(!src2 || get_misalign_bytes(ctx, src2) == 0); + GGML_ASSERT(!src3 || get_misalign_bytes(ctx, src3) == 0); + GGML_ASSERT(!dst || get_misalign_bytes(ctx, dst) == 0); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_mat_vec_p021_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.b_offset = b_offset; + p.d_offset = d_offset; + + GGML_UNUSED(src0); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_mat_vec_nc_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.b_offset = b_offset; + p.d_offset = d_offset; + + GGML_UNUSED(src0); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_fwht_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + p.src_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + p.dst_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_dsv4_hc_comb_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + p.m_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + p.s_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + p.b_offset = get_misalign_bytes(ctx, src2) / ggml_type_size(src2->type); + p.d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_dsv4_hc_pre_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + p.x_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + p.w_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + p.d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_dsv4_hc_post_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + p.x_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + p.r_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + p.p_offset = get_misalign_bytes(ctx, src2) / ggml_type_size(src2->type); + p.c_offset = src3 ? get_misalign_bytes(ctx, src3) / ggml_type_size(src3->type) : 0; + p.d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); +} + +template size_t push_constant_size(const T &t) { + static_assert(std::is_class::value, "T must be a struct/class"); + GGML_UNUSED(t); + return sizeof(T); +} + +template size_t push_constant_size(const std::vector &t) { + GGML_UNUSED(t); + return sizeof(T) * t.size(); +} + +template size_t push_constant_size(const std::array &t) { + GGML_UNUSED(t); + return sizeof(T) * N; +} + +template const T *push_constant_data(const T &t) { + static_assert(std::is_class::value, "T must be a struct/class"); + return &t; +} + +template const T *push_constant_data(const std::vector &t) { + return t.data(); +} + +template const T *push_constant_data(const std::array &t) { + return t.data(); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_unary_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.misalign_offsets = (a_offset << 16) | d_offset; + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_glu_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t b_offset = src1 ? get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type) : a_offset; + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_ASSERT(a_offset < (1u << 8)); + GGML_ASSERT(b_offset < (1u << 8)); + GGML_ASSERT(d_offset < (1u << 8)); + + p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; + + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_sum_rows_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.misalign_offsets = (a_offset << 16) | d_offset; + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_pad_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.misalign_offsets = (a_offset << 16) | d_offset; + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_im2col_3d_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.misalign_offsets = (a_offset << 16) | d_offset; + + GGML_UNUSED(src0); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_binary_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_ASSERT(a_offset <= 0xFFFF); + GGML_ASSERT(b_offset <= 0xFF); + GGML_ASSERT(d_offset <= 0xFF); + + p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; + + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_concat_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t unit_size = ggml_vk_concat_unit_size(dst->type); + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / unit_size; + const uint32_t b_offset = get_misalign_bytes(ctx, src1) / unit_size; + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / unit_size; + + p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; + + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_upscale_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + p.a_offset = a_offset; + p.d_offset = d_offset; + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +template <> inline void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_rope_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { + p.a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); + p.d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + + GGML_UNUSED(src1); + GGML_UNUSED(src2); + GGML_UNUSED(src3); +} + +static vk_op_binary_push_constants ggml_vk_rms_norm_push_constants( + const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst, + float eps, uint32_t num_partials) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); + + return { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + eps, 0.0f, (int32_t)num_partials, + }; +} + diff --git a/ggml/src/ggml-vulkan/ggml-vulkan-types.h b/ggml/src/ggml-vulkan/ggml-vulkan-types.h new file mode 100644 index 00000000..5df1c390 --- /dev/null +++ b/ggml/src/ggml-vulkan/ggml-vulkan-types.h @@ -0,0 +1,1433 @@ +#pragma once + +#include "ggml-vulkan.h" + +#include + +#if defined(GGML_VULKAN_RUN_TESTS) || defined(GGML_VULKAN_CHECK_RESULTS) +#include +#include "ggml-cpu.h" +#endif + +#define VULKAN_HPP_DISPATCH_LOADER_DYNAMIC 1 + +#if VK_HEADER_VERSION >= 301 +namespace vk::detail { class DispatchLoaderDynamic; } +using vk::detail::DispatchLoaderDynamic; +#else +namespace vk { class DispatchLoaderDynamic; } +using vk::DispatchLoaderDynamic; +#endif + +DispatchLoaderDynamic & ggml_vk_default_dispatcher(); + +#define VULKAN_HPP_DEFAULT_DISPATCHER ggml_vk_default_dispatcher() + +#include + +#ifndef VK_NV_cooperative_matrix_decode_vector +#define VK_NV_cooperative_matrix_decode_vector 1 +#define VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME "VK_NV_cooperative_matrix_decode_vector" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV ((VkStructureType)1000689000) +typedef struct VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV { + VkStructureType sType; + void* pNext; + VkBool32 cooperativeMatrixDecodeVector; +} VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV; +#endif + +#if __has_include() +# include +#elif __has_include() +# include +#elif __has_include() +# include +#else + // Fallback to let the compiler throw a standard "file not found" error +# include +#endif + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#include + +#if defined(_MSC_VER) +# define NOMINMAX 1 +# include +# define YIELD() YieldProcessor() +#elif defined(__clang__) || defined(__GNUC__) +# if defined(__x86_64__) ||defined(__i386__) +# include +# define YIELD() _mm_pause() +# elif defined(__arm__) || defined(__aarch64__) +# if defined(__clang__) +# include +# define YIELD() __yield() +# else +# define YIELD() asm volatile("yield") +# endif +# endif +#endif + +#if !defined(YIELD) +#define YIELD() +#endif + +#include "ggml-impl.h" + +#include "ggml-backend-impl.h" + +#include "ggml-vulkan-shaders.hpp" + +#if !defined(VK_KHR_shader_bfloat16) + +#define VK_KHR_shader_bfloat16 1 +#define VK_KHR_SHADER_BFLOAT16_SPEC_VERSION 1 +#define VK_KHR_SHADER_BFLOAT16_EXTENSION_NAME "VK_KHR_shader_bfloat16" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR ((VkStructureType)1000141000) +#define VK_COMPONENT_TYPE_BFLOAT16_KHR ((VkComponentTypeKHR)1000141000) + +typedef struct VkPhysicalDeviceShaderBfloat16FeaturesKHR { + VkStructureType sType; + void* pNext; + VkBool32 shaderBFloat16Type; + VkBool32 shaderBFloat16DotProduct; + VkBool32 shaderBFloat16CooperativeMatrix; +} VkPhysicalDeviceShaderBfloat16FeaturesKHR; +#endif + +#if !defined(VK_VALVE_shader_mixed_float_dot_product) +#define VK_VALVE_shader_mixed_float_dot_product 1 +#define VK_VALVE_SHADER_MIXED_FLOAT_DOT_PRODUCT_SPEC_VERSION 1 +#define VK_VALVE_SHADER_MIXED_FLOAT_DOT_PRODUCT_EXTENSION_NAME "VK_VALVE_shader_mixed_float_dot_product" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE ((VkStructureType)1000673000) +typedef struct VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE { + VkStructureType sType; + void* pNext; + VkBool32 shaderMixedFloatDotProductFloat16AccFloat32; + VkBool32 shaderMixedFloatDotProductFloat16AccFloat16; + VkBool32 shaderMixedFloatDotProductBFloat16Acc; + VkBool32 shaderMixedFloatDotProductFloat8AccFloat32; +} VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE; +#endif + +#if !defined(VK_EXT_shader_ocp_microscaling_types) +#define VK_EXT_shader_ocp_microscaling_types 1 +#define VK_EXT_SHADER_OCP_MICROSCALING_TYPES_SPEC_VERSION 1 +#define VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME "VK_EXT_shader_ocp_microscaling_types" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT ((VkStructureType)1000672000) +typedef struct VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT { + VkStructureType sType; + void* pNext; + VkBool32 shaderFloat4; + VkBool32 shaderFloat6; + VkBool32 shaderFloat8UnsignedE8M0; + VkBool32 shaderMXInt8; +} VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT; +#endif + +#if !defined(VK_EXT_shader_float8) +#define VK_EXT_shader_float8 1 +#define VK_EXT_SHADER_FLOAT8_SPEC_VERSION 1 +#define VK_EXT_SHADER_FLOAT8_EXTENSION_NAME "VK_EXT_shader_float8" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT ((VkStructureType)1000567000) +typedef struct VkPhysicalDeviceShaderFloat8FeaturesEXT { + VkStructureType sType; + void* pNext; + VkBool32 shaderFloat8; + VkBool32 shaderFloat8CooperativeMatrix; +} VkPhysicalDeviceShaderFloat8FeaturesEXT; +#endif + +#ifndef VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME +#define VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME "VK_KHR_internally_synchronized_queues" +#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INTERNALLY_SYNCHRONIZED_QUEUES_FEATURES_KHR ((VkStructureType)1000504000) +#define VK_DEVICE_QUEUE_CREATE_INTERNALLY_SYNCHRONIZED_BIT_KHR ((VkDeviceQueueCreateFlagBits)0x00000004) + +// Compile-time constant guaranteed; no runtime initialization overhead +static constexpr vk::DeviceQueueCreateFlagBits eInternallySynchronizedKHR = + static_cast(0x00000004); + +typedef struct VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR { + VkStructureType sType; + void* pNext; + VkBool32 internallySynchronizedQueues; +} VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR; +#else +static constexpr vk::DeviceQueueCreateFlagBits eInternallySynchronizedKHR = vk::DeviceQueueCreateFlagBits::eInternallySynchronizedKHR; +#endif + +#define ROUNDUP_POW2(M, N) (((M) + (N) - 1) & ~((N) - 1)) + +#define CEIL_DIV(M, N) (((M) / (N)) + (((M) % (N)) != 0)) + +static bool is_pow2(uint32_t x) { return x > 1 && (x & (x-1)) == 0; } + +#define VK_VENDOR_ID_AMD 0x1002 + +#define VK_VENDOR_ID_APPLE 0x106b + +#define VK_VENDOR_ID_INTEL 0x8086 + +#define VK_VENDOR_ID_NVIDIA 0x10de + +#define VK_VENDOR_ID_QUALCOMM 0x5143 + +#define VK_DEVICE_DESCRIPTOR_POOL_SIZE 256 + +#define VK_CHECK(err, msg, dev) \ + do { \ + vk::Result err_; \ + try { \ + err_ = (err); \ + } catch (vk::DeviceLostError &) { \ + ggml_vk_print_device_lost_info(dev); \ + GGML_LOG_ERROR("ggml_vulkan: %s at %s:%d\n", \ + #err, __FILE__, __LINE__); \ + throw; \ + } \ + if (err_ != vk::Result::eSuccess) { \ + GGML_LOG_ERROR("ggml_vulkan: %s error %s at %s:%d\n", \ + #err, to_string(err_).c_str(), __FILE__, __LINE__); \ + throw vk::SystemError(vk::make_error_code(err_), \ + "ggml_vulkan: " msg); \ + } \ + } while (0) + +#ifdef GGML_VULKAN_DEBUG +#define VK_LOG_DEBUG(msg) std::cerr << msg << std::endl +#else +#define VK_LOG_DEBUG(msg) ((void) 0) +#endif // GGML_VULKAN_DEBUG + +#define MAX_PARAMETER_COUNT 12 + +#define MAX_FUSED_ADDS (MAX_PARAMETER_COUNT - 3) + +struct vk_pipeline_struct; + +typedef std::shared_ptr vk_pipeline; + +struct vk_pipeline_struct { + std::string name; + vk::ShaderModule shader_module; + vk::PipelineLayout layout; + vk::Pipeline pipeline; + uint32_t push_constant_size; + uint32_t parameter_count; + std::array wg_denoms; + uint32_t align; + // true if fields have been set by ggml_vk_create_pipeline + bool initialized {}; + // true while a compile is in flight, used to dedupe concurrent claims. + // Protected by device->compile_mutex. + bool compile_pending {}; + // set to true when the shader has been compiled + std::atomic compiled {}; + // number of registers used, extracted from pipeline executable properties + uint32_t register_count {}; + +#if defined(VK_EXT_shader_64bit_indexing) + bool is_64b_indexing {}; +#endif + // linked list of pipelines for multiple compilation variants. + // currently only used to compile a 64-bit indexing variant. + vk_pipeline next; +}; + +typedef std::weak_ptr vk_pipeline_ref; + +struct vk_matmul_pipeline_key { + ggml_type type_a; + ggml_type type_b; + bool mul_mat_id; + bool f16acc; + + bool operator<(const vk_matmul_pipeline_key & o) const { + return std::tie(type_a, type_b, mul_mat_id, f16acc) + < std::tie(o.type_a, o.type_b, o.mul_mat_id, o.f16acc); + } +}; + +struct vk_matmul_pipeline_pair { + vk_pipeline unaligned; + vk_pipeline aligned; + uint32_t align; +}; + +struct vk_tile_config { + std::vector warptile; + std::array wg_denoms; + uint32_t align; +}; + +using matmul_tile_selector_t = std::function& configs)>; + +struct vk_device_struct; + +typedef std::shared_ptr vk_device; + +typedef std::weak_ptr vk_device_ref; + +struct vk_buffer_struct; + +typedef std::shared_ptr vk_buffer; + +typedef std::weak_ptr vk_buffer_ref; + +struct ggml_backend_vk_buffer_type_context { + std::string name; + vk_device device; +}; + +struct vk_command_buffer { + vk::CommandBuffer buf; + uint64_t use_counter = 0; + bool in_use = false; +}; + +struct vk_queue; + +struct vk_command_pool { + void init(vk_device& device, vk_queue *q_); + void destroy(vk::Device& device); + + vk::CommandPool pool; + // Using deque so the pointers to command buffers + // remain valid even if we add more + std::deque cmd_buffers; + + vk_queue *q; + + size_t buffers_in_use() const { + return std::count_if(cmd_buffers.begin(), cmd_buffers.end(), + [](const auto& cb) { return cb.in_use; }); + } +}; + +struct vk_queue_handle { + vk::Queue queue; + vk_device_ref device; + std::mutex * device_submit_mutex = nullptr; + virtual void submit(vk::ArrayProxy submits, vk::Fence fence) = 0; + virtual void lock() {} // no-op by default (internally synchronized case) + virtual void unlock() {} + virtual ~vk_queue_handle() = default; +}; + +struct vk_queue_handle_synchronized : vk_queue_handle { + std::mutex mutex; + void submit(vk::ArrayProxy submits, vk::Fence fence) override; + + void lock() override { mutex.lock(); } + void unlock() override { mutex.unlock(); } +}; + +struct vk_queue_handle_unsynchronized : vk_queue_handle { + void submit(vk::ArrayProxy submits, vk::Fence fence) override; + + // lock()/unlock() inherited no-ops +}; + +struct vk_queue { + uint32_t queue_family_index; + std::shared_ptr handle; + + vk_command_pool cmd_pool; + + vk::PipelineStageFlags stage_flags; + + bool transfer_only; +}; + +static constexpr uint32_t mul_mat_vec_max_cols = 8; + +static constexpr uint32_t p021_max_gqa_ratio = 8; + +enum vk_device_architecture { + OTHER, + AMD_GCN, + AMD_RDNA1, + AMD_RDNA2, + AMD_RDNA3, + INTEL_XE1, + INTEL_XE2, + NVIDIA_PRE_TURING, + NVIDIA_TURING, +}; + +enum vk_conv_shapes { + CONV_SHAPE_128x128, + CONV_SHAPE_64x32, + CONV_SHAPE_32x256, + CONV_SHAPE_64x128, + CONV_SHAPE_COUNT, +}; + +struct vk_conv_block_size { + uint32_t K; + uint32_t NPQ; + uint32_t CRS; +}; + +inline vk_conv_block_size vk_conv_block_sizes[CONV_SHAPE_COUNT] = { + // K NPQ CRS + { 128, 128, 16 }, // CONV_SHAPE_128x128 + { 64, 32, 32 }, // CONV_SHAPE_64x32 + { 32, 256, 16 }, // CONV_SHAPE_32x256 + { 64, 128, 16 }, // CONV_SHAPE_64x128 +}; + +enum dmmv_wg_sizes { + DMMV_WG_SIZE_SUBGROUP, + DMMV_WG_SIZE_LARGE, + DMMV_WG_SIZE_COUNT, +}; + +enum FaCodePath { + FA_SCALAR, + FA_COOPMAT1, + FA_COOPMAT2, +}; + +struct vk_fa_pipeline_state { + uint32_t HSK, HSV; + uint32_t Br, Bc; + uint32_t D_split, row_split; + bool shmem_staging; + FaCodePath path; + uint32_t workgroup_size, subgroup_size; + bool aligned; + bool f32acc; + uint32_t flags; + uint32_t limit_occupancy_shmem; + ggml_type k_type; + ggml_type v_type; + + bool operator<(const vk_fa_pipeline_state &b) const { + return std::tie(HSK, HSV, Br, Bc, D_split, row_split, shmem_staging, path, workgroup_size, subgroup_size, aligned, f32acc, flags, limit_occupancy_shmem, k_type, v_type) < + std::tie(b.HSK, b.HSV, b.Br, b.Bc, b.D_split, b.row_split, b.shmem_staging, b.path, b.workgroup_size, b.subgroup_size, b.aligned, b.f32acc, b.flags, b.limit_occupancy_shmem, b.k_type, b.v_type); + } +}; + +struct vk_conv2d_pipeline_state { + vk_conv2d_pipeline_state(uint32_t s0, uint32_t s1, uint32_t p0, uint32_t p1, uint32_t d0, uint32_t d1, uint32_t KW, uint32_t KH, uint32_t aligned) + : s0(s0), s1(s1), p0(p0), p1(p1), d0(d0), d1(d1), KW(KW), KH(KH), aligned(aligned) {} + + uint32_t s0, s1, p0, p1, d0, d1, KW, KH; + // when set, shader can skip K/CRS/NPQ bounds checks and address clamps + uint32_t aligned; + + bool operator<(const vk_conv2d_pipeline_state &b) const { + return std::tie(s0, s1, p0, p1, d0, d1, KW, KH, aligned) < + std::tie(b.s0, b.s1, b.p0, b.p1, b.d0, b.d1, b.KW, b.KH, b.aligned); + } +}; + +struct vk_conv3d_pipeline_state { + vk_conv3d_pipeline_state(uint32_t s0, uint32_t s1, uint32_t s2, uint32_t p0, uint32_t p1, uint32_t p2, + uint32_t d0, uint32_t d1, uint32_t d2, uint32_t KW, uint32_t KH, uint32_t KD, uint32_t aligned) + : s0(s0), s1(s1), s2(s2), p0(p0), p1(p1), p2(p2), d0(d0), d1(d1), d2(d2), KW(KW), KH(KH), KD(KD), aligned(aligned) {} + + uint32_t s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD; + uint32_t aligned; + + bool operator<(const vk_conv3d_pipeline_state &b) const { + return std::tie(s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD, aligned) < + std::tie(b.s0, b.s1, b.s2, b.p0, b.p1, b.p2, b.d0, b.d1, b.d2, b.KW, b.KH, b.KD, b.aligned); + } +}; + +struct vk_solve_tri_pipeline_state { + vk_solve_tri_pipeline_state(uint32_t N, uint32_t K) + : N(N), K(K) {} + + uint32_t N, K; + + bool operator<(const vk_solve_tri_pipeline_state &b) const { + return std::tie(N, K) < + std::tie(b.N, b.K); + } +}; + +enum shader_reduction_mode { + SHADER_REDUCTION_MODE_SHMEM, + SHADER_REDUCTION_MODE_HYBRID, + SHADER_REDUCTION_MODE_SUBGROUP, + SHADER_REDUCTION_MODE_COUNT, +}; + +static constexpr uint32_t num_argsort_pipelines = 11; + +static constexpr uint32_t num_topk_moe_pipelines = 10; + +static constexpr uint32_t num_topk_pipelines = 11; + +static constexpr std::initializer_list topk_moe_early_softmax_norm{ GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, + GGML_OP_VIEW, GGML_OP_GET_ROWS, GGML_OP_RESHAPE, + GGML_OP_SUM_ROWS, GGML_OP_CLAMP, GGML_OP_DIV, + GGML_OP_RESHAPE }; + +static constexpr std::initializer_list topk_moe_sigmoid_norm_bias{ GGML_OP_UNARY, GGML_OP_RESHAPE, GGML_OP_ADD, + GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS, + GGML_OP_RESHAPE, GGML_OP_SUM_ROWS, GGML_OP_CLAMP, + GGML_OP_DIV, GGML_OP_RESHAPE }; + +static constexpr std::initializer_list topk_moe_sqrt_softplus_norm_bias{ GGML_OP_UNARY, GGML_OP_SQRT, + GGML_OP_RESHAPE, GGML_OP_ADD, + GGML_OP_ARGSORT, GGML_OP_VIEW, + GGML_OP_GET_ROWS, GGML_OP_RESHAPE, + GGML_OP_SUM_ROWS, GGML_OP_CLAMP, + GGML_OP_DIV, GGML_OP_RESHAPE }; + +static constexpr std::initializer_list topk_moe_early_softmax { GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, + GGML_OP_VIEW, GGML_OP_GET_ROWS }; + +static constexpr std::initializer_list topk_moe_late_softmax { GGML_OP_ARGSORT, GGML_OP_VIEW, + GGML_OP_GET_ROWS, GGML_OP_RESHAPE, + GGML_OP_SOFT_MAX, GGML_OP_RESHAPE }; + +static constexpr std::initializer_list snake_pattern { GGML_OP_MUL, GGML_OP_SIN, + GGML_OP_SQR, GGML_OP_MUL, + GGML_OP_ADD }; + +static constexpr std::initializer_list topk_qsa_pattern { GGML_OP_GET_ROWS, GGML_OP_PERMUTE, + GGML_OP_CONT, GGML_OP_CPY, + GGML_OP_RESHAPE, GGML_OP_ADD, + GGML_OP_TOP_K }; + +static constexpr std::initializer_list> topk_qsa_edges { + { 1, 0, 0 }, // permute->src[0] == get_rows + { 2, 0, 1 }, // cont->src[0] == permute + { 4, 0, 3 }, // reshape->src[0] == cpy (mask cast) + { 5, 0, 2 }, // add->src[0] == cont + { 5, 1, 4 }, // add->src[1] == reshape + { 6, 0, 5 }, // top_k->src[0] == add +}; + +static constexpr std::initializer_list rms_norm_mul_add_mul_pattern { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD, GGML_OP_MUL }; + +static constexpr std::initializer_list rms_norm_mul_add_pattern { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }; + +static constexpr std::initializer_list rms_norm_mul_rope_view_set_rows_pattern { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ROPE, GGML_OP_VIEW, GGML_OP_SET_ROWS }; + +static constexpr std::initializer_list rms_norm_view_set_rows_pattern { GGML_OP_RMS_NORM, GGML_OP_VIEW, GGML_OP_SET_ROWS }; + +static constexpr std::initializer_list rope_view_set_rows_pattern { GGML_OP_ROPE, GGML_OP_VIEW, GGML_OP_SET_ROWS }; + +static constexpr std::initializer_list> topk_moe_early_softmax_norm_edges { + { 1, 0, 0 }, // reshape->src[0] == softmax + { 2, 0, 0 }, // argsort->src[0] == softmax + { 3, 0, 2 }, // view->src[0] == argsort + { 4, 0, 1 }, // get_rows->src[0] == reshape + { 4, 1, 3 }, // get_rows->src[1] == view + { 5, 0, 4 }, // reshape->src[0] == get_rows + { 6, 0, 5 }, // sum_rows->src[0] == reshape + { 7, 0, 6 }, // clamp->src[0] == sum_rows + { 8, 0, 5 }, // div->src[0] == reshape + { 8, 1, 7 }, // div->src[1] == clamp + { 9, 0, 8 }, // reshape->src[0] == div +}; + +static constexpr std::initializer_list> topk_moe_sigmoid_norm_bias_edges { + { 1, 0, 0 }, // reshape->src[0] == sigmoid + { 2, 0, 0 }, // add->src[0] == sigmoid + { 3, 0, 2 }, // argsort->src[0] == add + { 4, 0, 3 }, // view->src[0] == argsort + { 5, 0, 1 }, // get_rows->src[0] == reshape + { 5, 1, 4 }, // get_rows->src[1] == view + { 6, 0, 5 }, // reshape->src[0] == get_rows + { 7, 0, 6 }, // sum_rows->src[0] == reshape + { 8, 0, 7 }, // clamp->src[0] == sum_rows + { 9, 0, 6 }, // div->src[0] == reshape + { 9, 1, 8 }, // div->src[1] == clamp + {10, 0, 9 }, // reshape->src[0] == div +}; + +static constexpr std::initializer_list> topk_moe_sqrt_softplus_norm_bias_edges { + { 1, 0, 0 }, // sqrt->src[0] == softplus + { 2, 0, 1 }, // reshape->src[0] == sqrt + { 3, 0, 1 }, // add->src[0] == sqrt + { 4, 0, 3 }, // argsort->src[0] == add + { 5, 0, 4 }, // view->src[0] == argsort + { 6, 0, 2 }, // get_rows->src[0] == reshape + { 6, 1, 5 }, // get_rows->src[1] == view + { 7, 0, 6 }, // reshape->src[0] == get_rows + { 8, 0, 7 }, // sum_rows->src[0] == reshape + { 9, 0, 8 }, // clamp->src[0] == sum_rows + {10, 0, 7 }, // div->src[0] == reshape + {10, 1, 9 }, // div->src[1] == clamp + {11, 0,10 }, // reshape->src[0] == div +}; + +static constexpr std::initializer_list> topk_moe_early_softmax_edges { + { 1, 0, 0 }, // reshape->src[0] == softmax + { 2, 0, 0 }, // argsort->src[0] == softmax + { 3, 0, 2 }, // view->src[0] == argsort + { 4, 0, 1 }, // get_rows->src[0] == reshape + { 4, 1, 3 }, // get_rows->src[1] == view +}; + +static constexpr std::initializer_list> topk_moe_late_softmax_edges { + { 1, 0, 0 }, // view->src[0] == argsort + { 2, 1, 1 }, // get_rows->src[1] == view + { 3, 0, 2 }, // reshape->src[0] == get_rows + { 4, 0, 3 }, // soft_max->src[0] == reshape + { 5, 0, 4 }, // reshape->src[0] == soft_max +}; + +enum topk_moe_mode { + TOPK_MOE_EARLY_SOFTMAX, + TOPK_MOE_EARLY_SOFTMAX_NORM, + TOPK_MOE_LATE_SOFTMAX, + TOPK_MOE_SIGMOID_NORM_BIAS, + TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS, + TOPK_MOE_COUNT, +}; + +enum rms_norm_mode { + RMS_NORM_MUL, + RMS_NORM_MUL_ADD, + RMS_NORM_MUL_ADD_MUL, + RMS_NORM_MUL_ROPE, + RMS_NORM_MUL_ROPE_VIEW_SET_ROWS, + RMS_NORM_VIEW_SET_ROWS, + RMS_NORM_COUNT, +}; + +static constexpr std::initializer_list> rope_view_set_rows_edges { + { 1, 0, 0 }, // view->src[0] == rope + { 2, 0, 1 }, // set_rows->src[0] == view +}; + +static constexpr std::initializer_list> rms_norm_mul_rope_view_set_rows_edges { + { 1, 0, 0 }, // mul->src[0] == rms + { 2, 0, 1 }, // rope->src[0] == mul + { 3, 0, 2 }, // view->src[0] == rope + { 4, 0, 3 }, // set_rows->src[0] == view +}; + +static constexpr std::initializer_list> rms_norm_view_set_rows_edges { + { 1, 0, 0 }, // view->src[0] == rms_norm + { 2, 0, 1 }, // set_rows->src[0] == view +}; + +static constexpr std::array lightning_indexer_k_types = { + GGML_TYPE_F32, + GGML_TYPE_F16, + GGML_TYPE_BF16, + GGML_TYPE_Q8_0, + GGML_TYPE_Q5_1, + GGML_TYPE_Q5_0, + GGML_TYPE_Q4_1, + GGML_TYPE_Q4_0, + GGML_TYPE_IQ4_NL, +}; + +class vk_memory_logger; + +struct vk_device_struct { + std::recursive_mutex mutex; + std::mutex queue_submit_mutex; + mutable std::shared_mutex pinned_memory_mutex; + + // Guards compile_pending, all_pipelines, and the dynamic pipeline maps + // (flash_attn, fa_mask_opt, solve_tri, conv2d, etc). The actual compile + // runs with no lock held, so different pipelines can compile in parallel. + // Lock order is device->mutex -> compile_mutex, never the reverse. + std::mutex compile_mutex; + std::condition_variable compile_cv; + + uint32_t debug_cmdbuf_idx {}; + + vk::PhysicalDevice physical_device; + vk::PhysicalDeviceProperties properties; + std::string name; + uint64_t max_memory_allocation_size; + uint64_t max_buffer_size; + uint64_t suballocation_block_size; + uint64_t min_imported_host_pointer_alignment; + bool external_memory_host {}; + bool fp16; + bool bf16; + bool pipeline_robustness; + bool memory_priority; + vk::Device device; + uint32_t vendor_id; + vk::DriverId driver_id; + vk_device_architecture architecture; + std::unique_ptr compute_queue; + std::unique_ptr transfer_queue; + bool single_queue; + bool support_async; + bool async_use_transfer_queue; + bool has_internally_synchronized_queues = false; + uint32_t subgroup_size; + uint32_t subgroup_size_log2; + uint32_t shader_core_count; + bool uma; + bool prefer_host_memory; + bool float_controls_rte_fp16; + bool float_controls_denorm_preserve_fp16; + bool subgroup_basic; + bool subgroup_arithmetic; + bool subgroup_shuffle; + bool subgroup_ballot; + bool subgroup_clustered; + bool subgroup_vote; + bool multi_add; + bool shader_int64; + bool buffer_device_address; + bool vulkan_memory_model; + + bool add_rms_fusion; + uint32_t partials_binding_alignment; + uint32_t max_nodes_per_submit; + + bool shader_64b_indexing; + + bool integer_dot_product; + // 0: default, 1: force mmvq, -1: disable mmvq + int32_t mmvq_mode; + + bool subgroup_size_control; + uint32_t subgroup_min_size; + uint32_t subgroup_max_size; + bool subgroup_require_full_support; + + // floor(log2(maxComputeWorkGroupInvocations)) + uint32_t max_workgroup_size_log2 {}; + + bool coopmat_support; + bool coopmat_acc_f32_support {}; + bool coopmat_acc_f16_support {}; + bool coopmat_bf16_support {}; + bool coopmat_support_16x16x16_f16acc {}; + bool coopmat_support_16x16x16_f32acc {}; + bool coopmat1_fa_support {}; + uint32_t coopmat_m; + uint32_t coopmat_n; + uint32_t coopmat_k; + + bool coopmat_int_support; + uint32_t coopmat_int_m; + uint32_t coopmat_int_n; + uint32_t coopmat_int_k; + + bool coopmat2; + bool coopmat2_bf16_support {}; + bool coopmat2_decode_vector; + + bool dot2_f16 {}; + bool ocp_fp4 {}; + + bool pipeline_executable_properties_support {}; + + bool device_fault {}; + PFN_vkGetDeviceFaultInfoEXT pfn_vkGetDeviceFaultInfoEXT {}; + + bool serialize_submissions {}; + + const ggml_cgraph * diag_cgraph {}; + int diag_prev_start = -1; + int diag_prev_end = -1; + + size_t idx; + + bool mul_mat_l[GGML_TYPE_COUNT]; + bool mul_mat_m[GGML_TYPE_COUNT]; + bool mul_mat_s[GGML_TYPE_COUNT]; + bool mul_mat_id_l[GGML_TYPE_COUNT]; + bool mul_mat_id_m[GGML_TYPE_COUNT]; + bool mul_mat_id_s[GGML_TYPE_COUNT]; + + // Separate flags for the q8_1 (integer dot) mmq path, whose shader uses + // a different shared-memory layout than the float matmul shaders. + bool mul_mat_l_int[GGML_TYPE_COUNT]; + bool mul_mat_m_int[GGML_TYPE_COUNT]; + bool mul_mat_s_int[GGML_TYPE_COUNT]; + bool mul_mat_id_l_int[GGML_TYPE_COUNT]; + bool mul_mat_id_m_int[GGML_TYPE_COUNT]; + bool mul_mat_id_s_int[GGML_TYPE_COUNT]; + + vk::DescriptorSetLayout dsl; + + std::map> pipeline_matmul; + matmul_tile_selector_t matmul_tile_selector; + matmul_tile_selector_t matmul_id_tile_selector; + + vk_pipeline pipeline_matmul_split_k_reduce; + vk_pipeline pipeline_quantize_q8_1_x4; + + vk_pipeline pipeline_dequant[GGML_TYPE_COUNT]; + vk_pipeline pipeline_dequant_transpose[GGML_TYPE_COUNT]; // fused dequant+transpose for FA quant-KV + vk_pipeline pipeline_dequant_mul_mat_vec_f32_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; + vk_pipeline pipeline_dequant_mul_mat_vec_f16_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; + vk_pipeline pipeline_dequant_mul_mat_vec_id_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT]; + + vk_pipeline pipeline_dequant_mul_mat_vec_q8_1_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; + vk_pipeline pipeline_dequant_mul_mat_vec_id_q8_1_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT]; + + vk_pipeline pipeline_mul_mat_vec_p021_f16_f32[p021_max_gqa_ratio]; + vk_pipeline pipeline_mul_mat_vec_nc_f16_f32; + vk_pipeline pipeline_get_rows[GGML_TYPE_COUNT]; + vk_pipeline pipeline_get_rows_f32[GGML_TYPE_COUNT]; + vk_pipeline pipeline_get_rows_back_f32; + vk_pipeline pipeline_acc_f32; + vk_pipeline pipeline_set_f32; + + // [src0 0=fp32,1=fp16][src1 0=fp32,1=fp16][dst 0=fp32,1=fp16] + vk_pipeline pipeline_add[2][2][2]; + vk_pipeline pipeline_add_norepeat[2][2][2]; + vk_pipeline pipeline_sub[2][2][2]; + vk_pipeline pipeline_sub_norepeat[2][2][2]; + vk_pipeline pipeline_mul[2][2][2]; + vk_pipeline pipeline_mul_norepeat[2][2][2]; + vk_pipeline pipeline_div[2][2][2]; + vk_pipeline pipeline_div_norepeat[2][2][2]; + vk_pipeline pipeline_add_rms[2][2][2]; + vk_pipeline pipeline_add_rms_norepeat[2][2][2]; + + // indexed by num_additional_fused_ops == num_adds - 1 + vk_pipeline pipeline_multi_add[MAX_FUSED_ADDS]; + vk_pipeline pipeline_multi_add_rms[MAX_FUSED_ADDS]; + + vk_pipeline pipeline_add_id_f32; + + vk_pipeline pipeline_concat_i8, pipeline_concat_i16, pipeline_concat_i32, pipeline_concat_i64; + vk_pipeline pipeline_upscale_nearest_f32, pipeline_upscale_bilinear_f32, pipeline_upscale_bicubic_f32, pipeline_upscale_bilinear_antialias_f32; + vk_pipeline pipeline_scale_f32; + vk_pipeline pipeline_log[2]; + vk_pipeline pipeline_tri[2]; + vk_pipeline pipeline_diag[2]; + vk_pipeline pipeline_clamp[2]; + vk_pipeline pipeline_pad_f32; + vk_pipeline pipeline_pad_reflect_1d_f32; + vk_pipeline pipeline_roll_f32; + vk_pipeline pipeline_repeat_i32, pipeline_repeat_back_f32; + vk_pipeline pipeline_repeat_i16; + vk_pipeline pipeline_cpy_f32_f32, pipeline_cpy_f32_f16, pipeline_cpy_f16_f16, pipeline_cpy_f16_f32, pipeline_cpy_f32_bf16, pipeline_cpy_bf16_f32, pipeline_cpy_f32_i32, pipeline_cpy_i32_f32; + vk_pipeline pipeline_contig_cpy_f32_f32, pipeline_contig_cpy_f32_f16, pipeline_contig_cpy_f16_f16, pipeline_contig_cpy_f16_f32, pipeline_contig_cpy_f32_bf16, pipeline_contig_cpy_bf16_f32, pipeline_contig_cpy_f32_i32, pipeline_contig_cpy_i32_f32; + vk_pipeline pipeline_cpy_f32_quant[GGML_TYPE_COUNT]; + vk_pipeline pipeline_cpy_quant_f32[GGML_TYPE_COUNT]; + vk_pipeline pipeline_cpy_transpose_16, pipeline_cpy_transpose_32; + vk_pipeline pipeline_cpy_transpose_02_16, pipeline_cpy_transpose_02_32; + // [src0 0=fp32,1=fp16][dst] + vk_pipeline pipeline_set_rows_i32[2][GGML_TYPE_COUNT]; + vk_pipeline pipeline_set_rows_i64[2][GGML_TYPE_COUNT]; + vk_pipeline pipeline_norm_f32; + vk_pipeline pipeline_group_norm_f32; + vk_pipeline pipeline_rms_norm_f32; + vk_pipeline pipeline_rms_norm_mul_f32; + vk_pipeline pipeline_rms_norm_mul_add_f32; + vk_pipeline pipeline_rms_norm_mul_add_mul_f32; + vk_pipeline pipeline_rms_norm_mul_add_partials_f32; + vk_pipeline pipeline_rms_norm_mul_add_mul_partials_f32; + vk_pipeline pipeline_rms_norm_set_rows_f32_f32; + vk_pipeline pipeline_rms_norm_set_rows_f32_f16; + vk_pipeline pipeline_rms_norm_partials_f32; + vk_pipeline pipeline_rms_norm_mul_partials_f32; + vk_pipeline pipeline_rms_norm_mul_rope_f32_f32; + vk_pipeline pipeline_rms_norm_mul_rope_f32_f16; + vk_pipeline pipeline_rms_norm_back_f32; + vk_pipeline pipeline_l2_norm_f32; + + // [src/dst 0=fp32,1=fp16] + vk_pipeline pipeline_exp[2]; + vk_pipeline pipeline_expm1[2]; + vk_pipeline pipeline_elu[2]; + vk_pipeline pipeline_gelu[2]; + vk_pipeline pipeline_gelu_erf[2]; + vk_pipeline pipeline_gelu_quick[2]; + vk_pipeline pipeline_silu[2]; + vk_pipeline pipeline_relu[2]; + vk_pipeline pipeline_sqr[2]; + vk_pipeline pipeline_sqrt[2]; + vk_pipeline pipeline_sin[2]; + vk_pipeline pipeline_cos[2]; + vk_pipeline pipeline_xielu[2]; + vk_pipeline pipeline_neg[2]; + vk_pipeline pipeline_tanh[2]; + vk_pipeline pipeline_sigmoid[2]; + vk_pipeline pipeline_hardsigmoid[2]; + vk_pipeline pipeline_hardswish[2]; + vk_pipeline pipeline_abs[2]; + vk_pipeline pipeline_softplus[2]; + vk_pipeline pipeline_step[2]; + vk_pipeline pipeline_round[2]; + vk_pipeline pipeline_ceil[2]; + vk_pipeline pipeline_floor[2]; + vk_pipeline pipeline_trunc[2]; + vk_pipeline pipeline_sgn[2]; + + // fused UNARY+MUL pipelines: [op][f16][norepeat][op_on_b] + vk_pipeline pipeline_unary_mul[4][2][2][2]; + + vk_pipeline pipeline_add1_f16_f16; + vk_pipeline pipeline_add1_f16_f32; + vk_pipeline pipeline_add1_f32_f32; + + vk_pipeline pipeline_arange_f32; + + vk_pipeline pipeline_fill_f32; + vk_pipeline pipeline_fill_f16; + + vk_pipeline pipeline_geglu[2]; + vk_pipeline pipeline_reglu[2]; + vk_pipeline pipeline_swiglu[2]; + vk_pipeline pipeline_swiglu_oai[2]; + vk_pipeline pipeline_swiglu_clamp[2]; + vk_pipeline pipeline_geglu_erf[2]; + vk_pipeline pipeline_geglu_quick[2]; + + vk_pipeline pipeline_leaky_relu[2]; + vk_pipeline pipeline_silu_back_f32; + vk_pipeline pipeline_diag_mask_inf_f32; + vk_pipeline pipeline_soft_max_f32, pipeline_soft_max_f32_f16; + vk_pipeline pipeline_soft_max_f32_wg512, pipeline_soft_max_f32_f16_wg512; + vk_pipeline pipeline_soft_max_back_f32; + + vk_pipeline pipeline_soft_max_large1_f32, pipeline_soft_max_large1_f32_f16; + vk_pipeline pipeline_soft_max_large2_f32, pipeline_soft_max_large2_f32_f16; + vk_pipeline pipeline_soft_max_large3_f32, pipeline_soft_max_large3_f32_f16; + + vk_pipeline pipeline_rope_norm_f32, pipeline_rope_norm_f16, pipeline_rope_norm_f32_f16; + vk_pipeline pipeline_rope_neox_f32, pipeline_rope_neox_f16, pipeline_rope_neox_f32_f16; + vk_pipeline pipeline_rope_multi_f32, pipeline_rope_multi_f16, pipeline_rope_multi_f32_f16; + vk_pipeline pipeline_rope_vision_f32, pipeline_rope_vision_f16; + vk_pipeline pipeline_argsort_f32[num_argsort_pipelines]; + vk_pipeline pipeline_argsort_large_f32[num_argsort_pipelines]; + vk_pipeline pipeline_topk_f32[num_topk_pipelines]; + vk_pipeline pipeline_topk_radix_f32; + vk_pipeline pipeline_topk_radix_qsa; // qwen4 QSA indexer fusion (f16 mask) + vk_pipeline pipeline_sum_rows_f32; + vk_pipeline pipeline_cross_entropy_loss_f32, pipeline_cross_entropy_loss_f32_wg512; + vk_pipeline pipeline_cross_entropy_loss_back_f32, pipeline_cross_entropy_loss_back_f32_wg512; + vk_pipeline pipeline_fwht_f32[4]; + vk_pipeline pipeline_cumsum_f32; + vk_pipeline pipeline_cumsum_small_f32; + vk_pipeline pipeline_cumsum_multipass1_f32; + vk_pipeline pipeline_cumsum_multipass2_f32; + vk_pipeline pipeline_argmax_f32; + vk_pipeline pipeline_count_equal_i32; + vk_pipeline pipeline_dsv4_hc_comb_f32; + vk_pipeline pipeline_dsv4_hc_pre_f32; + vk_pipeline pipeline_dsv4_hc_pre_gated_f32; + vk_pipeline pipeline_dsv4_hc_post_f32; + vk_pipeline pipeline_dsv4_hc_post_nocomb_f32; + std::map pipeline_solve_tri_f32; + vk_pipeline pipeline_im2col_f32, pipeline_im2col_f32_f16; + vk_pipeline pipeline_im2col_3d_f32, pipeline_im2col_3d_f32_f16; + vk_pipeline pipeline_timestep_embedding_f32; + vk_pipeline pipeline_conv_transpose_1d_f32; + vk_pipeline pipeline_col2im_1d_f32; + vk_pipeline pipeline_col2im_1d_f16; + vk_pipeline pipeline_col2im_1d_bf16; + vk_pipeline pipeline_out_prod_f32; + vk_pipeline pipeline_snake_f32; + vk_pipeline pipeline_snake_f16; + vk_pipeline pipeline_snake_bf16; + vk_pipeline pipeline_pool1d_f32; + vk_pipeline pipeline_pool2d_f32; + vk_pipeline pipeline_rwkv_wkv6_f32; + vk_pipeline pipeline_rwkv_wkv7_f32; + vk_pipeline pipeline_gated_linear_attn_f32; + vk_pipeline pipeline_lightning_indexer_f32[GGML_TYPE_COUNT]; + // [size_idx][kda] where size_idx: 0=d16, 1=d32, 2=d64, 3=d128 + vk_pipeline pipeline_gated_delta_net[4][2]; + vk_pipeline pipeline_ssm_scan_f32_d128; + vk_pipeline pipeline_ssm_scan_f32_d256; + vk_pipeline pipeline_ssm_conv_f32; + vk_pipeline pipeline_ssm_conv_silu_f32; + vk_pipeline pipeline_ssm_conv_bias_silu_f32; + vk_pipeline pipeline_opt_step_adamw_f32; + vk_pipeline pipeline_opt_step_sgd_f32; + std::map pipeline_conv2d_f32[CONV_SHAPE_COUNT]; + std::map pipeline_conv2d_f16_f32[CONV_SHAPE_COUNT]; + std::map pipeline_conv_transpose_2d_f32[CONV_SHAPE_COUNT]; + std::map pipeline_conv_transpose_2d_f16_f32[CONV_SHAPE_COUNT]; + std::map pipeline_conv3d_f32[CONV_SHAPE_COUNT]; + std::map pipeline_conv3d_f16_f32[CONV_SHAPE_COUNT]; + vk_pipeline pipeline_conv2d_dw_whcn_f32, pipeline_conv2d_dw_whcn_f16_f32; + vk_pipeline pipeline_conv2d_dw_cwhn_f32, pipeline_conv2d_dw_cwhn_f16_f32; + + std::map pipeline_flash_attn_f32_f16; + + std::map, vk_pipeline> pipeline_fa_mask_opt; + + vk_pipeline pipeline_fa_sparse_compact; + vk_pipeline pipeline_fa_sparse_compact_subgroup; + bool fa_sparse_compact_use_subgroups; + + vk_pipeline pipeline_flash_attn_split_k_reduce; + std::map, std::pair> pipeline_xe_fa_decode_dual_phases; + vk_pipeline pipeline_count_experts; + + // [2] is for whether to take n_experts from spec constant (0) or push constant (1) + vk_pipeline pipeline_topk_moe[num_topk_moe_pipelines][2]; + + std::vector all_pipelines; + + std::vector> pinned_memory; + + vk::Fence fence; + vk_buffer sync_staging; + + ggml_backend_buffer_type buffer_type; + + bool disable_fusion; + bool disable_host_visible_vidmem; + bool allow_sysmem_fallback; + bool disable_graph_optimize; + + std::unique_ptr memory_logger; + + ~vk_device_struct(); + +}; + +inline void vk_command_pool::init(vk_device& device, vk_queue *q_) { + cmd_buffers.clear(); + q = q_; + + vk::CommandPoolCreateInfo command_pool_create_info( + vk::CommandPoolCreateFlags(VK_COMMAND_POOL_CREATE_TRANSIENT_BIT | VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT), + q->queue_family_index); + pool = device->device.createCommandPool(command_pool_create_info); +} + +inline void vk_command_pool::destroy(vk::Device& device) { + device.destroyCommandPool(pool); + pool = nullptr; + cmd_buffers.clear(); +} + +struct vk_buffer_struct { + vk::Buffer buffer = VK_NULL_HANDLE; + vk::DeviceMemory device_memory = VK_NULL_HANDLE; + vk::MemoryPropertyFlags memory_property_flags; + void * ptr; + size_t size = 0; + vk::DeviceAddress bda_addr {}; + + vk_device device; + + ~vk_buffer_struct() { + if (size == 0) { + return; + } + VK_LOG_DEBUG("~vk_buffer_struct(" << buffer << ", " << size << ")"); + + device->device.freeMemory(device_memory); + device->device.destroyBuffer(buffer); + } +}; + +struct vk_subbuffer { + vk_buffer buffer; + uint64_t offset; + uint64_t size; + + operator vk::DescriptorBufferInfo() const { + return { buffer->buffer, offset, size }; + } +}; + +struct vk_semaphore { + vk::Semaphore s; + uint64_t value; +}; + +struct vk_event { + std::vector events_free; // Events available for reuse + std::vector events_submitted; // Events that are fully submitted and can be reused on next synchronize + vk::Event event; + bool has_event; + + vk_semaphore tl_semaphore; + vk_command_buffer* cmd_buffer = nullptr; + uint64_t cmd_buffer_use_counter = 0; +}; + +struct vk_submission { + vk_command_buffer* buffer = nullptr; + std::vector wait_semaphores; + std::vector signal_semaphores; +}; + +typedef std::vector vk_sequence; + +#define MAT_VEC_FUSION_FLAGS_BIAS0 0x1 + +#define MAT_VEC_FUSION_FLAGS_BIAS1 0x2 + +#define MAT_VEC_FUSION_FLAGS_SCALE0 0x4 + +#define MAT_VEC_FUSION_FLAGS_SCALE1 0x8 + +struct vk_staging_memcpy { + vk_staging_memcpy(void * _dst, const void * _src, size_t _n) : dst(_dst), src(_src), n(_n) {} + + void * dst; + const void * src; + size_t n; +}; + +struct vk_staging_memset { + vk_staging_memset(void * _dst, uint32_t _val, size_t _n) : dst(_dst), val(_val), n(_n) {} + + void * dst; + uint32_t val; + size_t n; +}; + +struct vk_context_struct { + vk_submission * s; + std::vector seqs; + + int exit_tensor_idx; + + std::vector in_memcpys; + std::vector out_memcpys; + std::vector memsets; + + std::vector debug_labels; + + vk_command_pool * p {}; +}; + +typedef std::shared_ptr vk_context; + +typedef std::weak_ptr vk_context_ref; + +struct ggml_vk_garbage_collector { + std::vector tl_semaphores; + std::vector semaphores; + std::vector events; + std::vector contexts; +}; + +#define VK_LOG_MEMORY(msg) if (vk_memory_logger_enabled) { std::cerr << "ggml_vulkan memory: " << msg << std::endl; } + +static std::string format_size(size_t size) { + const size_t kib = 1024; + const size_t mib = kib * 1024; + const size_t gib = mib * 1024; + + std::ostringstream oss; + oss << std::fixed << std::setprecision(2); + + if (size >= gib) { + oss << static_cast(size) / gib << " GiB"; + } else if (size >= mib) { + oss << static_cast(size) / mib << " MiB"; + } else if (size >= kib) { + oss << static_cast(size) / kib << " KiB"; + } else { + oss << size << " B"; + } + + return oss.str(); +} + +class vk_memory_logger { +public: + vk_memory_logger(): total_device(0), total_host(0) {} + void log_allocation(vk_buffer_ref buf_ref, size_t size); + void log_deallocation(vk_buffer_ref buf_ref); + +private: + std::map allocations; // Track allocations + size_t total_device; + size_t total_host; + static std::mutex log_mutex; +}; + +inline std::mutex vk_memory_logger::log_mutex; + +class vk_perf_logger { + public: + void print_timings(bool force = false); + + + std::string get_node_fusion_name(const ggml_tensor * node, const char *fusion_name, uint64_t *n_flops); + + + void log_timing(const ggml_tensor * node, const char *fusion_name, uint64_t time) { + uint64_t n_flops; + std::string name = get_node_fusion_name(node, fusion_name, &n_flops); + if (n_flops) { + flops[name].push_back(n_flops); + } + timings[name].push_back(time); + } + + void log_timing(const std::vector &nodes, const std::vector &names, uint64_t time) { + uint64_t total_flops = 0; + std::string name; + for (size_t n = 0; n < nodes.size(); ++n) { + uint64_t n_flops = 0; + name += get_node_fusion_name(nodes[n], names[n], &n_flops); + total_flops += n_flops; + + if (n != nodes.size() - 1) { + name += ", "; + } + } + if (total_flops) { + flops[name].push_back(total_flops); + } + timings[name].push_back(time); + } + + private: + std::map> timings; + std::map> flops; + uint32_t print_count {}; +}; + +struct ggml_backend_vk_context { + std::string name; + + vk_device device; + + size_t semaphore_idx, event_idx; + ggml_vk_garbage_collector gc; + size_t prealloc_size_x, prealloc_size_y, prealloc_size_split_k, prealloc_size_add_rms_partials, prealloc_size_add_rms_partials_offset; + vk_buffer prealloc_x, prealloc_y, prealloc_split_k, prealloc_add_rms_partials, sync_staging; + vk::Fence fence, almost_ready_fence; + bool submit_pending {}; + bool almost_ready_fence_pending {}; + // Set before op_add and unset after op_rms_norm to indicate that the add should + // write partial sums to accumulate the square of the vector components + bool do_add_rms_partials_offset_calculation; + bool do_add_rms_partials; + + uint64_t last_total_flops {UINT64_MAX}; + + // Cache most recent tensor that was converted into prealloc_y, and what pipeline it used to convert. + vk_pipeline_struct * prealloc_y_last_pipeline_used {}; + const ggml_tensor * prealloc_y_last_tensor_used {}; + // True when the K dimension in prealloc_y is padded. + bool prealloc_y_last_k_padded {}; + + // Track which nodes have been used since the last sync, and whether they were written to + std::vector unsynced_nodes_written; + std::vector unsynced_nodes_read; + // Track which prealloc buffers have pending reads that need to be synchronized. + // These are checked before writing to the buffer (and call ggml_vk_sync_buffers if set), + // and set to true after the buffer contents are consumed. + bool prealloc_x_need_sync, prealloc_y_need_sync, prealloc_split_k_need_sync; + + vk_context_ref compute_ctx; + + vk_context_ref transfer_ctx; + vk_semaphore transfer_semaphore; + uint64_t transfer_semaphore_last_submitted {}; + + std::vector tensor_ctxs; + + std::vector descriptor_pools; + std::vector descriptor_sets; + uint32_t descriptor_set_idx {}; + uint32_t pipeline_descriptor_set_requirements {}; + + vk_command_pool compute_cmd_pool; + vk_command_pool transfer_cmd_pool; + + // number of additional consecutive nodes that are being fused with the + // node currently being processed + int num_additional_fused_ops {}; + // Bitmask of which fused ops need to write an intermediate value to memory. + // Bit 'i' means nodes[start_of_fusion + i] writes to memory. + // If there's no fusion, bit 0 is still set. + int fused_ops_write_mask {}; + topk_moe_mode fused_topk_moe_mode {}; + bool fused_topk_moe_scale {}; + // QSA indexer gather+add+top_k fused into one radix-select + bool fused_topk_qsa {}; + rms_norm_mode fused_rms_norm_mode {RMS_NORM_COUNT}; + + // for GGML_VK_PERF_LOGGER + std::unique_ptr perf_logger; + vk::QueryPool query_pool; + std::vector query_fusion_names; + std::vector query_fusion_node_count; + std::vector query_nodes; + std::vector query_node_idx; + int32_t num_queries {}; + int32_t query_idx {}; +}; + +struct ggml_backend_vk_buffer_context { + vk_device_ref device; + vk_buffer dev_buffer; + std::string name; + + ggml_backend_vk_buffer_context(vk_device_ref device, vk_buffer&& dev_buffer, std::string& name) : + device(device), + dev_buffer(dev_buffer), + name(name) { + } + + ~ggml_backend_vk_buffer_context(); + +}; + +struct vk_instance_t { + vk::Instance instance; + + bool debug_utils_support = false; // VK_EXT_debug_utils enabled + PFN_vkSetDebugUtilsObjectNameEXT pfn_vkSetDebugUtilsObjectNameEXT = {}; + PFN_vkQueueBeginDebugUtilsLabelEXT pfn_vkQueueBeginDebugUtilsLabelEXT = {}; + PFN_vkQueueEndDebugUtilsLabelEXT pfn_vkQueueEndDebugUtilsLabelEXT = {}; + PFN_vkCmdBeginDebugUtilsLabelEXT pfn_vkCmdBeginDebugUtilsLabelEXT = {}; + PFN_vkCmdEndDebugUtilsLabelEXT pfn_vkCmdEndDebugUtilsLabelEXT = {}; + PFN_vkCmdInsertDebugUtilsLabelEXT pfn_vkCmdInsertDebugUtilsLabelEXT = {}; + + std::vector device_indices; + std::vector device_supports_membudget; + vk_device devices[GGML_VK_MAX_DEVICES]; +}; + +typedef void (*ggml_vk_func_t)(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); + +static constexpr uint32_t kSpvOpCooperativeMatrixLoadTensorNV = 5367; + +static constexpr uint32_t kSpvCapabilityCooperativeMatrixDecodeVectorNV = 5447; + +static constexpr uint32_t kSpvTensorAddressingDecodeVectorFuncBit = 0x4; + +struct vk_fa_tuning_params { + FaCodePath path; + uint32_t workgroup_size; + uint32_t subgroup_size; + uint32_t block_rows; + uint32_t block_cols; + uint32_t d_split; + uint32_t row_split; + bool shmem_staging; + bool disable_subgroups; + uint32_t limit_occupancy_shmem; + + void print() const { + std::cerr << "path=" << path << " workgroup_size=" << workgroup_size << " subgroup_size=" << subgroup_size << + " block_rows=" << block_rows << " block_cols=" << block_cols << " d_split=" << d_split << + " row_split=" << row_split << " shmem_staging=" << shmem_staging << " disable_subgroups=" << disable_subgroups << + " limit_occupancy_shmem=" << limit_occupancy_shmem << std::endl; + } +}; + +struct GpuPipelineConfig { + // GPU architecture identifier. + // Example: vk_device_architecture::AMD_GCN + vk_device_architecture arch; + + // Mapping of pipeline names to their specific subgroup sizes. + // Example: {"soft_max_f32", 64} + std::unordered_map pipelines; + + // Default subgroup size for this GPU. + // Defaults to 0 if not explicitly provided. + uint32_t default_subgroup_size = 0; +}; + +static constexpr uint32_t RDNA_DEFAULT_SUBGROUP_SIZE = 32; + +struct CompileTask { + vk_pipeline pipeline; + size_t spv_size; + const void * spv_data; + std::string entrypoint; + uint32_t parameter_count; + std::array wg_denoms; + std::vector specialization_constants; + bool disable_robustness; + bool require_full_subgroups; + uint32_t required_subgroup_size; +}; + +struct ggml_vk_debug_label { + // at most one of these is set, depending on the scope the label was opened in + vk_context_struct * subctx {}; + vk_queue_handle * qhandle {}; + + // one region per dispatch, e.g. "matmul_q4_k_f32_f16acc_aligned_m (192,8,1)". + // RGP cannot recover the pipeline name on its own, it only has the hash + ggml_vk_debug_label(vk_context & ctx, const std::string & pipeline_name, uint32_t wg0, uint32_t wg1, uint32_t wg2); + + + // one region per graph node + // fused nodes are joined with '+', e.g. "RMS_NORM+MUL+ROPE Qcur-19" + ggml_vk_debug_label(vk_context & ctx, const ggml_cgraph * cgraph, int node_idx, int n_fused); + + + // one region per graph evaluation, opened on the queue instead of a command buffer + // so it spans every submit the evaluation makes + ggml_vk_debug_label(vk_queue_handle * handle, const char * name); + + + // call before the command buffer can end, the destructor covers the rest + void close(); + + + ~ggml_vk_debug_label() { + close(); + } + + ggml_vk_debug_label(const ggml_vk_debug_label &) = delete; + ggml_vk_debug_label & operator=(const ggml_vk_debug_label &) = delete; + +private: + // the constructors check this too, so the name is not built when markers are off + void begin(vk_context & ctx, const std::string & name); + +}; + +#define UNUSED GGML_UNUSED + +struct ggml_backend_vk_device_context { + size_t device; + std::string name; + std::string description; + bool is_integrated_gpu; + std::string pci_bus_id; + int op_offload_min_batch_size; +}; + diff --git a/ggml/src/ggml-vulkan/ggml-vulkan.cpp b/ggml/src/ggml-vulkan/ggml-vulkan.cpp index 585e10d4..1720657c 100644 --- a/ggml/src/ggml-vulkan/ggml-vulkan.cpp +++ b/ggml/src/ggml-vulkan/ggml-vulkan.cpp @@ -1,406 +1,10 @@ -#include "ggml-vulkan.h" -#include -#if defined(GGML_VULKAN_RUN_TESTS) || defined(GGML_VULKAN_CHECK_RESULTS) -#include -#include "ggml-cpu.h" -#endif - -// See https://github.com/KhronosGroup/Vulkan-Hpp?tab=readme-ov-file#extensions--per-device-function-pointers- -#define VULKAN_HPP_DISPATCH_LOADER_DYNAMIC 1 -// We use VULKAN_HPP_DEFAULT_DISPATCHER, but not VULKAN_HPP_DEFAULT_DISPATCH_LOADER_DYNAMIC_STORAGE -// to avoid conflicts with applications or other libraries who might use it. -#if VK_HEADER_VERSION >= 301 -namespace vk::detail { class DispatchLoaderDynamic; } -using vk::detail::DispatchLoaderDynamic; -#else -namespace vk { class DispatchLoaderDynamic; } -using vk::DispatchLoaderDynamic; -#endif -DispatchLoaderDynamic & ggml_vk_default_dispatcher(); -#define VULKAN_HPP_DEFAULT_DISPATCHER ggml_vk_default_dispatcher() - -#include - -// Fallback definitions for VK_NV_cooperative_matrix_decode_vector in case the -// installed Vulkan headers predate the extension. -#ifndef VK_NV_cooperative_matrix_decode_vector -#define VK_NV_cooperative_matrix_decode_vector 1 -#define VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME "VK_NV_cooperative_matrix_decode_vector" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV ((VkStructureType)1000689000) -typedef struct VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV { - VkStructureType sType; - void* pNext; - VkBool32 cooperativeMatrixDecodeVector; -} VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV; -#endif - -// SPIR-V Headers: different SDK installations expose different include paths. -// LunarG Vulkan SDK on Windows typically provides . -// Linux packages, MSYS2 and MinGW often use the Khronos layout . -#if __has_include() -# include -#elif __has_include() -# include -#elif __has_include() -# include -#else - // Fallback to let the compiler throw a standard "file not found" error -# include -#endif - -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include -#include - -#if defined(_MSC_VER) -# define NOMINMAX 1 -# include -# define YIELD() YieldProcessor() -#elif defined(__clang__) || defined(__GNUC__) -# if defined(__x86_64__) ||defined(__i386__) -# include -# define YIELD() _mm_pause() -# elif defined(__arm__) || defined(__aarch64__) -# if defined(__clang__) -# include -# define YIELD() __yield() -# else -# define YIELD() asm volatile("yield") -# endif -# endif -#endif - -#if !defined(YIELD) -#define YIELD() -#endif - -#include "ggml-impl.h" -#include "ggml-backend-impl.h" - -#include "ggml-vulkan-shaders.hpp" - -// remove this once it's more widely available in the SDK -#if !defined(VK_KHR_shader_bfloat16) - -#define VK_KHR_shader_bfloat16 1 -#define VK_KHR_SHADER_BFLOAT16_SPEC_VERSION 1 -#define VK_KHR_SHADER_BFLOAT16_EXTENSION_NAME "VK_KHR_shader_bfloat16" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR ((VkStructureType)1000141000) -#define VK_COMPONENT_TYPE_BFLOAT16_KHR ((VkComponentTypeKHR)1000141000) - -typedef struct VkPhysicalDeviceShaderBfloat16FeaturesKHR { - VkStructureType sType; - void* pNext; - VkBool32 shaderBFloat16Type; - VkBool32 shaderBFloat16DotProduct; - VkBool32 shaderBFloat16CooperativeMatrix; -} VkPhysicalDeviceShaderBfloat16FeaturesKHR; -#endif - -#if !defined(VK_VALVE_shader_mixed_float_dot_product) -#define VK_VALVE_shader_mixed_float_dot_product 1 -#define VK_VALVE_SHADER_MIXED_FLOAT_DOT_PRODUCT_SPEC_VERSION 1 -#define VK_VALVE_SHADER_MIXED_FLOAT_DOT_PRODUCT_EXTENSION_NAME "VK_VALVE_shader_mixed_float_dot_product" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE ((VkStructureType)1000673000) -typedef struct VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE { - VkStructureType sType; - void* pNext; - VkBool32 shaderMixedFloatDotProductFloat16AccFloat32; - VkBool32 shaderMixedFloatDotProductFloat16AccFloat16; - VkBool32 shaderMixedFloatDotProductBFloat16Acc; - VkBool32 shaderMixedFloatDotProductFloat8AccFloat32; -} VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE; -#endif - -#if !defined(VK_EXT_shader_ocp_microscaling_types) -#define VK_EXT_shader_ocp_microscaling_types 1 -#define VK_EXT_SHADER_OCP_MICROSCALING_TYPES_SPEC_VERSION 1 -#define VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME "VK_EXT_shader_ocp_microscaling_types" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT ((VkStructureType)1000672000) -typedef struct VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT { - VkStructureType sType; - void* pNext; - VkBool32 shaderFloat4; - VkBool32 shaderFloat6; - VkBool32 shaderFloat8UnsignedE8M0; - VkBool32 shaderMXInt8; -} VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT; -#endif - -#if !defined(VK_EXT_shader_float8) -#define VK_EXT_shader_float8 1 -#define VK_EXT_SHADER_FLOAT8_SPEC_VERSION 1 -#define VK_EXT_SHADER_FLOAT8_EXTENSION_NAME "VK_EXT_shader_float8" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT ((VkStructureType)1000567000) -typedef struct VkPhysicalDeviceShaderFloat8FeaturesEXT { - VkStructureType sType; - void* pNext; - VkBool32 shaderFloat8; - VkBool32 shaderFloat8CooperativeMatrix; -} VkPhysicalDeviceShaderFloat8FeaturesEXT; -#endif - -#ifndef VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME -#define VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME "VK_KHR_internally_synchronized_queues" -#define VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INTERNALLY_SYNCHRONIZED_QUEUES_FEATURES_KHR ((VkStructureType)1000504000) -#define VK_DEVICE_QUEUE_CREATE_INTERNALLY_SYNCHRONIZED_BIT_KHR ((VkDeviceQueueCreateFlagBits)0x00000004) - -// Compile-time constant guaranteed; no runtime initialization overhead -static constexpr vk::DeviceQueueCreateFlagBits eInternallySynchronizedKHR = - static_cast(0x00000004); - -typedef struct VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR { - VkStructureType sType; - void* pNext; - VkBool32 internallySynchronizedQueues; -} VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR; -#else -static constexpr vk::DeviceQueueCreateFlagBits eInternallySynchronizedKHR = vk::DeviceQueueCreateFlagBits::eInternallySynchronizedKHR; -#endif - -#define ROUNDUP_POW2(M, N) (((M) + (N) - 1) & ~((N) - 1)) -#define CEIL_DIV(M, N) (((M) / (N)) + (((M) % (N)) != 0)) -static bool is_pow2(uint32_t x) { return x > 1 && (x & (x-1)) == 0; } - -#define VK_VENDOR_ID_AMD 0x1002 -#define VK_VENDOR_ID_APPLE 0x106b -#define VK_VENDOR_ID_INTEL 0x8086 -#define VK_VENDOR_ID_NVIDIA 0x10de -#define VK_VENDOR_ID_QUALCOMM 0x5143 - -#define VK_DEVICE_DESCRIPTOR_POOL_SIZE 256 - -#define VK_CHECK(err, msg, dev) \ - do { \ - vk::Result err_; \ - try { \ - err_ = (err); \ - } catch (vk::DeviceLostError &) { \ - ggml_vk_print_device_lost_info(dev); \ - GGML_LOG_ERROR("ggml_vulkan: %s at %s:%d\n", \ - #err, __FILE__, __LINE__); \ - throw; \ - } \ - if (err_ != vk::Result::eSuccess) { \ - GGML_LOG_ERROR("ggml_vulkan: %s error %s at %s:%d\n", \ - #err, to_string(err_).c_str(), __FILE__, __LINE__); \ - throw vk::SystemError(vk::make_error_code(err_), \ - "ggml_vulkan: " msg); \ - } \ - } while (0) - -#ifdef GGML_VULKAN_DEBUG -#define VK_LOG_DEBUG(msg) std::cerr << msg << std::endl -#else -#define VK_LOG_DEBUG(msg) ((void) 0) -#endif // GGML_VULKAN_DEBUG - -struct ggml_backend_vk_context; - -#define MAX_PARAMETER_COUNT 12 -// Max number of adds that can be fused without exceeding MAX_PARAMETER_COUNT. -#define MAX_FUSED_ADDS (MAX_PARAMETER_COUNT - 3) - -typedef std::shared_ptr vk_pipeline; - -struct vk_pipeline_struct { - std::string name; - vk::ShaderModule shader_module; - vk::PipelineLayout layout; - vk::Pipeline pipeline; - uint32_t push_constant_size; - uint32_t parameter_count; - std::array wg_denoms; - uint32_t align; - // true if fields have been set by ggml_vk_create_pipeline - bool initialized {}; - // true while a compile is in flight, used to dedupe concurrent claims. - // Protected by device->compile_mutex. - bool compile_pending {}; - // set to true when the shader has been compiled - std::atomic compiled {}; - // number of registers used, extracted from pipeline executable properties - uint32_t register_count {}; - -#if defined(VK_EXT_shader_64bit_indexing) - bool is_64b_indexing {}; -#endif - // linked list of pipelines for multiple compilation variants. - // currently only used to compile a 64-bit indexing variant. - vk_pipeline next; -}; - -typedef std::weak_ptr vk_pipeline_ref; - -static void ggml_vk_destroy_pipeline(vk::Device& device, vk_pipeline& pipeline); - -struct vk_matmul_pipeline_struct { - vk_pipeline l, m, s; - vk_pipeline a_l, a_m, a_s; - // Returns true when all unaligned pipelines are null. - // We only check for unaligned variants since one of the unaligned pipelines must exist - // while aligned pipelines are optional - bool is_empty() const { - return l == nullptr && m == nullptr && s == nullptr; - } -}; -typedef std::shared_ptr vk_matmul_pipeline; - -struct vk_matmul_pipeline2 { - vk_matmul_pipeline2() { - f16acc = std::make_shared(); - f32acc = std::make_shared(); - } - vk_matmul_pipeline f32acc; - vk_matmul_pipeline f16acc; -}; - -struct vk_device_struct; -typedef std::shared_ptr vk_device; -typedef std::weak_ptr vk_device_ref; - -struct vk_buffer_struct; -typedef std::shared_ptr vk_buffer; -typedef std::weak_ptr vk_buffer_ref; - -struct ggml_backend_vk_buffer_type_context { - std::string name; - vk_device device; -}; - -struct vk_queue; - -struct vk_command_buffer { - vk::CommandBuffer buf; - uint64_t use_counter = 0; - bool in_use = false; -}; - -// Stores command pool/buffers. There's an instance of this -// for each (context,queue) pair and for each (device,queue) pair. -struct vk_command_pool { - void init(vk_device& device, vk_queue *q_); - void destroy(vk::Device& device); - - vk::CommandPool pool; - // Using deque so the pointers to command buffers - // remain valid even if we add more - std::deque cmd_buffers; - - vk_queue *q; - - size_t buffers_in_use() const { - return std::count_if(cmd_buffers.begin(), cmd_buffers.end(), - [](const auto& cb) { return cb.in_use; }); - } -}; - -static void ggml_vk_print_device_fault_info(const vk_device& device); -static void ggml_vk_print_device_lost_info(const vk_device& device); - -// Prevent simultaneous submissions to the same queue. -struct vk_queue_handle { - vk::Queue queue; - vk_device_ref device; - virtual void submit(vk::ArrayProxy submits, vk::Fence fence) = 0; - virtual void lock() {} // no-op by default (internally synchronized case) - virtual void unlock() {} - virtual ~vk_queue_handle() = default; -}; - -struct vk_queue_handle_synchronized : vk_queue_handle { - std::mutex mutex; - void submit(vk::ArrayProxy submits, vk::Fence fence) override { - std::lock_guard guard(mutex); - try { - queue.submit(submits, fence); - } catch (vk::DeviceLostError &) { - if (auto dev = device.lock()) { - ggml_vk_print_device_lost_info(dev); - } - throw; - } - } - void lock() override { mutex.lock(); } - void unlock() override { mutex.unlock(); } -}; - -struct vk_queue_handle_unsynchronized : vk_queue_handle { - void submit(vk::ArrayProxy submits, vk::Fence fence) override { - // Driver guarantees internal synchronization via VK_KHR_internally_synchronized_queues - try { - queue.submit(submits, fence); - } catch (vk::DeviceLostError &) { - if (auto dev = device.lock()) { - ggml_vk_print_device_lost_info(dev); - } - throw; - } - } - // lock()/unlock() inherited no-ops -}; - -struct vk_queue { - uint32_t queue_family_index; - std::shared_ptr handle; - - vk_command_pool cmd_pool; - - vk::PipelineStageFlags stage_flags; - - bool transfer_only; -}; - -static const char * ggml_backend_vk_buffer_type_name(ggml_backend_buffer_type_t buft); -static ggml_backend_buffer_t ggml_backend_vk_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size); -static size_t ggml_backend_vk_buffer_type_get_alignment(ggml_backend_buffer_type_t buft); -static size_t ggml_backend_vk_buffer_type_get_max_size(ggml_backend_buffer_type_t buft); -static size_t ggml_backend_vk_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor); -static ggml_backend_buffer_type_i ggml_backend_vk_buffer_type_interface = { - /* .get_name = */ ggml_backend_vk_buffer_type_name, - /* .alloc_buffer = */ ggml_backend_vk_buffer_type_alloc_buffer, - /* .get_alignment = */ ggml_backend_vk_buffer_type_get_alignment, - /* .get_max_size = */ ggml_backend_vk_buffer_type_get_max_size, - /* .get_alloc_size = */ ggml_backend_vk_buffer_type_get_alloc_size, - /* .is_host = */ NULL, -}; - -class vk_memory_logger; -class vk_perf_logger; -static void ggml_vk_destroy_buffer(vk_buffer& buf); -static void ggml_vk_synchronize(ggml_backend_vk_context * ctx); - -static constexpr uint32_t mul_mat_vec_max_cols = 8; -static constexpr uint32_t p021_max_gqa_ratio = 8; - -enum vk_device_architecture { - OTHER, - AMD_GCN, - AMD_RDNA1, - AMD_RDNA2, - AMD_RDNA3, - INTEL_XE1, - INTEL_XE2, - NVIDIA_PRE_TURING, - NVIDIA_TURING, -}; +#include "ggml-vulkan-common.h" +namespace { +inline std::ostream & operator<<(std::ostream & os, vk::Buffer buffer) { + return os << static_cast(buffer); +} +} static vk_device_architecture get_device_architecture(const vk::PhysicalDevice& device) { vk::PhysicalDeviceProperties props = device.getProperties(); @@ -521,640 +125,10 @@ static vk_device_architecture get_device_architecture(const vk::PhysicalDevice& return vk_device_architecture::OTHER; } -enum vk_conv_shapes { - CONV_SHAPE_128x128, - CONV_SHAPE_64x32, - CONV_SHAPE_32x256, - CONV_SHAPE_64x128, - CONV_SHAPE_COUNT, -}; - -struct vk_conv_block_size { - uint32_t K; - uint32_t NPQ; - uint32_t CRS; -}; - -vk_conv_block_size vk_conv_block_sizes[CONV_SHAPE_COUNT] = { - // K NPQ CRS - { 128, 128, 16 }, // CONV_SHAPE_128x128 - { 64, 32, 32 }, // CONV_SHAPE_64x32 - { 32, 256, 16 }, // CONV_SHAPE_32x256 - { 64, 128, 16 }, // CONV_SHAPE_64x128 -}; - -enum dmmv_wg_sizes { - DMMV_WG_SIZE_SUBGROUP, - DMMV_WG_SIZE_LARGE, - DMMV_WG_SIZE_COUNT, -}; - -enum FaCodePath { - FA_SCALAR, - FA_COOPMAT1, - FA_COOPMAT2, -}; - -struct vk_fa_pipeline_state { - uint32_t HSK, HSV; - uint32_t Br, Bc; - uint32_t D_split, row_split; - bool shmem_staging; - FaCodePath path; - uint32_t workgroup_size, subgroup_size; - bool aligned; - bool f32acc; - uint32_t flags; - uint32_t limit_occupancy_shmem; - ggml_type k_type; - ggml_type v_type; - - bool operator<(const vk_fa_pipeline_state &b) const { - return std::tie(HSK, HSV, Br, Bc, D_split, row_split, shmem_staging, path, workgroup_size, subgroup_size, aligned, f32acc, flags, limit_occupancy_shmem, k_type, v_type) < - std::tie(b.HSK, b.HSV, b.Br, b.Bc, b.D_split, b.row_split, b.shmem_staging, b.path, b.workgroup_size, b.subgroup_size, b.aligned, b.f32acc, b.flags, b.limit_occupancy_shmem, b.k_type, b.v_type); - } -}; - -struct vk_conv2d_pipeline_state { - vk_conv2d_pipeline_state(uint32_t s0, uint32_t s1, uint32_t p0, uint32_t p1, uint32_t d0, uint32_t d1, uint32_t KW, uint32_t KH, uint32_t aligned) - : s0(s0), s1(s1), p0(p0), p1(p1), d0(d0), d1(d1), KW(KW), KH(KH), aligned(aligned) {} - - uint32_t s0, s1, p0, p1, d0, d1, KW, KH; - // when set, shader can skip K/CRS/NPQ bounds checks and address clamps - uint32_t aligned; - - bool operator<(const vk_conv2d_pipeline_state &b) const { - return std::tie(s0, s1, p0, p1, d0, d1, KW, KH, aligned) < - std::tie(b.s0, b.s1, b.p0, b.p1, b.d0, b.d1, b.KW, b.KH, b.aligned); - } -}; - -struct vk_conv3d_pipeline_state { - vk_conv3d_pipeline_state(uint32_t s0, uint32_t s1, uint32_t s2, uint32_t p0, uint32_t p1, uint32_t p2, - uint32_t d0, uint32_t d1, uint32_t d2, uint32_t KW, uint32_t KH, uint32_t KD, uint32_t aligned) - : s0(s0), s1(s1), s2(s2), p0(p0), p1(p1), p2(p2), d0(d0), d1(d1), d2(d2), KW(KW), KH(KH), KD(KD), aligned(aligned) {} - - uint32_t s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD; - uint32_t aligned; - - bool operator<(const vk_conv3d_pipeline_state &b) const { - return std::tie(s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD, aligned) < - std::tie(b.s0, b.s1, b.s2, b.p0, b.p1, b.p2, b.d0, b.d1, b.d2, b.KW, b.KH, b.KD, b.aligned); - } -}; - -struct vk_solve_tri_pipeline_state { - vk_solve_tri_pipeline_state(uint32_t N, uint32_t K) - : N(N), K(K) {} - - uint32_t N, K; - - bool operator<(const vk_solve_tri_pipeline_state &b) const { - return std::tie(N, K) < - std::tie(b.N, b.K); - } -}; - -enum shader_reduction_mode { - SHADER_REDUCTION_MODE_SHMEM, - SHADER_REDUCTION_MODE_HYBRID, - SHADER_REDUCTION_MODE_SUBGROUP, - SHADER_REDUCTION_MODE_COUNT, -}; - -// argsort pipelines for up to 1<<10 invocations per workgroup -static constexpr uint32_t num_argsort_pipelines = 11; -static constexpr uint32_t num_topk_moe_pipelines = 10; -static constexpr uint32_t num_topk_pipelines = 11; - -static constexpr std::initializer_list topk_moe_early_softmax_norm{ GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, - GGML_OP_VIEW, GGML_OP_GET_ROWS, GGML_OP_RESHAPE, - GGML_OP_SUM_ROWS, GGML_OP_CLAMP, GGML_OP_DIV, - GGML_OP_RESHAPE }; - -static constexpr std::initializer_list topk_moe_sigmoid_norm_bias{ GGML_OP_UNARY, GGML_OP_RESHAPE, GGML_OP_ADD, - GGML_OP_ARGSORT, GGML_OP_VIEW, GGML_OP_GET_ROWS, - GGML_OP_RESHAPE, GGML_OP_SUM_ROWS, GGML_OP_CLAMP, - GGML_OP_DIV, GGML_OP_RESHAPE }; - -static constexpr std::initializer_list topk_moe_sqrt_softplus_norm_bias{ GGML_OP_UNARY, GGML_OP_SQRT, - GGML_OP_RESHAPE, GGML_OP_ADD, - GGML_OP_ARGSORT, GGML_OP_VIEW, - GGML_OP_GET_ROWS, GGML_OP_RESHAPE, - GGML_OP_SUM_ROWS, GGML_OP_CLAMP, - GGML_OP_DIV, GGML_OP_RESHAPE }; - -static constexpr std::initializer_list topk_moe_early_softmax { GGML_OP_SOFT_MAX, GGML_OP_RESHAPE, GGML_OP_ARGSORT, - GGML_OP_VIEW, GGML_OP_GET_ROWS }; - -static constexpr std::initializer_list topk_moe_late_softmax { GGML_OP_ARGSORT, GGML_OP_VIEW, - GGML_OP_GET_ROWS, GGML_OP_RESHAPE, - GGML_OP_SOFT_MAX, GGML_OP_RESHAPE }; - -// Snake activation: y = x + sin(a*x)^2 * inv_b. Used by the optimize_graph reorder -// pass so it keeps the chain contiguous and by the dispatcher to detect the fusion. -static constexpr std::initializer_list snake_pattern { GGML_OP_MUL, GGML_OP_SIN, - GGML_OP_SQR, GGML_OP_MUL, - GGML_OP_ADD }; - -//node #978 ( SOFT_MAX): ffn_moe_probs-15 ( 0K) [Vulka ] use=2: ffn_moe_logits-15 ( 0K) [Vulka ] -//node #979 ( RESHAPE): ffn_moe_probs-15 (re ( 0K) [Vulka ] use=1: ffn_moe_probs-15 ( 0K) [Vulka ] -//node #980 ( ARGSORT): ffn_moe_argsort-15 ( 0K) [Vulka ] use=1: ffn_moe_probs-15 ( 0K) [Vulka ] -//node #981 ( VIEW): ffn_moe_topk-15 ( 0K) [Vulka ] use=4: ffn_moe_argsort-15 ( 0K) [Vulka ] -//node #982 ( GET_ROWS): ffn_moe_weights-15 ( 0K) [Vulka ] use=1: ffn_moe_probs-15 (re ( 0K) [Vulka ] ffn_moe_topk-15 ( 0K) [Vulka ] -//node #983 ( RESHAPE): ffn_moe_weights-15 ( ( 0K) [Vulka ] use=2: ffn_moe_weights-15 ( 0K) [Vulka ] -//node #984 ( SUM_ROWS): ffn_moe_weights_sum- ( 0K) [Vulka ] use=1: ffn_moe_weights-15 ( ( 0K) [Vulka ] -//node #985 ( CLAMP): ffn_moe_weights_sum_ ( 0K) [Vulka ] use=1: ffn_moe_weights_sum- ( 0K) [Vulka ] -//node #986 ( DIV): ffn_moe_weights_norm ( 0K) [Vulka ] use=1: ffn_moe_weights-15 ( ( 0K) [Vulka ] ffn_moe_weights_sum_ ( 0K) [Vulka ] -//node #987 ( RESHAPE): ffn_moe_weights_norm ( 0K) [Vulka ] use=1: ffn_moe_weights_norm ( 0K) [Vulka ] -static constexpr std::initializer_list> topk_moe_early_softmax_norm_edges { - { 1, 0, 0 }, // reshape->src[0] == softmax - { 2, 0, 0 }, // argsort->src[0] == softmax - { 3, 0, 2 }, // view->src[0] == argsort - { 4, 0, 1 }, // get_rows->src[0] == reshape - { 4, 1, 3 }, // get_rows->src[1] == view - { 5, 0, 4 }, // reshape->src[0] == get_rows - { 6, 0, 5 }, // sum_rows->src[0] == reshape - { 7, 0, 6 }, // clamp->src[0] == sum_rows - { 8, 0, 5 }, // div->src[0] == reshape - { 8, 1, 7 }, // div->src[1] == clamp - { 9, 0, 8 }, // reshape->src[0] == div -}; - -//node #436 ( UNARY): ffn_moe_probs-10 ( 256K) [Vulka ] use=2: ffn_moe_logits-10 ( 256K) [Vulka ] -//node #437 ( RESHAPE): ffn_moe_probs-10 (re ( 256K) [Vulka ] use=1: ffn_moe_probs-10 ( 256K) [Vulka ] -//node #438 ( ADD): ffn_moe_probs_biased ( 256K) [Vulka ] use=1: ffn_moe_probs-10 ( 256K) [Vulka ] blk.10.exp_probs_b.b ( 0K) [Vulka ] -//node #439 ( ARGSORT): ffn_moe_argsort-10 ( 256K) [Vulka ] use=1: ffn_moe_probs_biased ( 256K) [Vulka ] -//node #440 ( VIEW): ffn_moe_topk-10 ( 255K) [Vulka ] use=3: ffn_moe_argsort-10 ( 256K) [Vulka ] -//node #441 ( GET_ROWS): ffn_moe_weights-10 ( 12K) [Vulka ] use=1: ffn_moe_probs-10 (re ( 256K) [Vulka ] ffn_moe_topk-10 ( 255K) [Vulka ] -//node #442 ( RESHAPE): ffn_moe_weights-10 ( ( 12K) [Vulka ] use=2: ffn_moe_weights-10 ( 12K) [Vulka ] -//node #443 ( SUM_ROWS): ffn_moe_weights_sum- ( 2K) [Vulka ] use=1: ffn_moe_weights-10 ( ( 12K) [Vulka ] -//node #444 ( CLAMP): ffn_moe_weights_sum_ ( 2K) [Vulka ] use=1: ffn_moe_weights_sum- ( 2K) [Vulka ] -//node #445 ( DIV): ffn_moe_weights_norm ( 12K) [Vulka ] use=1: ffn_moe_weights-10 ( ( 12K) [Vulka ] ffn_moe_weights_sum_ ( 2K) [Vulka ] -//node #446 ( RESHAPE): ffn_moe_weights_norm ( 12K) [Vulka ] use=1: ffn_moe_weights_norm ( 12K) [Vulka ] -static constexpr std::initializer_list> topk_moe_sigmoid_norm_bias_edges { - { 1, 0, 0 }, // reshape->src[0] == sigmoid - { 2, 0, 0 }, // add->src[0] == sigmoid - { 3, 0, 2 }, // argsort->src[0] == add - { 4, 0, 3 }, // view->src[0] == argsort - { 5, 0, 1 }, // get_rows->src[0] == reshape - { 5, 1, 4 }, // get_rows->src[1] == view - { 6, 0, 5 }, // reshape->src[0] == get_rows - { 7, 0, 6 }, // sum_rows->src[0] == reshape - { 8, 0, 7 }, // clamp->src[0] == sum_rows - { 9, 0, 6 }, // div->src[0] == reshape - { 9, 1, 8 }, // div->src[1] == clamp - {10, 0, 9 }, // reshape->src[0] == div -}; - -static constexpr std::initializer_list> topk_moe_sqrt_softplus_norm_bias_edges { - { 1, 0, 0 }, // sqrt->src[0] == softplus - { 2, 0, 1 }, // reshape->src[0] == sqrt - { 3, 0, 1 }, // add->src[0] == sqrt - { 4, 0, 3 }, // argsort->src[0] == add - { 5, 0, 4 }, // view->src[0] == argsort - { 6, 0, 2 }, // get_rows->src[0] == reshape - { 6, 1, 5 }, // get_rows->src[1] == view - { 7, 0, 6 }, // reshape->src[0] == get_rows - { 8, 0, 7 }, // sum_rows->src[0] == reshape - { 9, 0, 8 }, // clamp->src[0] == sum_rows - {10, 0, 7 }, // div->src[0] == reshape - {10, 1, 9 }, // div->src[1] == clamp - {11, 0,10 }, // reshape->src[0] == div -}; - -// same as early_softmax_norm but ending after the get_rows -static constexpr std::initializer_list> topk_moe_early_softmax_edges { - { 1, 0, 0 }, // reshape->src[0] == softmax - { 2, 0, 0 }, // argsort->src[0] == softmax - { 3, 0, 2 }, // view->src[0] == argsort - { 4, 0, 1 }, // get_rows->src[0] == reshape - { 4, 1, 3 }, // get_rows->src[1] == view -}; - -//node #652 ( ARGSORT): ffn_moe_argsort-11 ( 0K) [Vulka ] use=1: ffn_moe_probs-11 ( 0K) [Vulka ] -//node #653 ( VIEW): ffn_moe_topk-11 ( 0K) [Vulka ] use=7: ffn_moe_argsort-11 ( 0K) [Vulka ] -//node #654 ( GET_ROWS): ffn_moe_weights-11 ( 0K) [Vulka ] use=1: ffn_moe_probs-11 (re ( 0K) [Vulka ] ffn_moe_topk-11 ( 0K) [Vulka ] -//node #655 ( RESHAPE): ffn_moe_weights-11 ( ( 0K) [Vulka ] use=1: ffn_moe_weights-11 ( 0K) [Vulka ] -//node #656 ( SOFT_MAX): node_656 ( 0K) [Vulka ] use=1: ffn_moe_weights-11 ( ( 0K) [Vulka ] -//node #657 ( RESHAPE): ffn_moe_weights_soft ( 0K) [Vulka ] use=1: node_656 ( 0K) [Vulka ] -static constexpr std::initializer_list> topk_moe_late_softmax_edges { - { 1, 0, 0 }, // view->src[0] == argsort - { 2, 1, 1 }, // get_rows->src[1] == view - { 3, 0, 2 }, // reshape->src[0] == get_rows - { 4, 0, 3 }, // soft_max->src[0] == reshape - { 5, 0, 4 }, // reshape->src[0] == soft_max -}; - -enum topk_moe_mode { - TOPK_MOE_EARLY_SOFTMAX, - TOPK_MOE_EARLY_SOFTMAX_NORM, - TOPK_MOE_LATE_SOFTMAX, - TOPK_MOE_SIGMOID_NORM_BIAS, - TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS, - TOPK_MOE_COUNT, -}; - -static constexpr std::initializer_list> rope_view_set_rows_edges { - { 1, 0, 0 }, // view->src[0] == rope - { 2, 0, 1 }, // set_rows->src[0] == view -}; - -static constexpr std::initializer_list> rms_norm_mul_rope_view_set_rows_edges { - { 1, 0, 0 }, // mul->src[0] == rms - { 2, 0, 1 }, // rope->src[0] == mul - { 3, 0, 2 }, // view->src[0] == rope - { 4, 0, 3 }, // set_rows->src[0] == view -}; - - -struct vk_device_struct { - std::recursive_mutex mutex; - mutable std::shared_mutex pinned_memory_mutex; - - // Guards compile_pending, all_pipelines, and the dynamic pipeline maps - // (flash_attn, fa_mask_opt, solve_tri, conv2d, etc). The actual compile - // runs with no lock held, so different pipelines can compile in parallel. - // Lock order is device->mutex -> compile_mutex, never the reverse. - std::mutex compile_mutex; - std::condition_variable compile_cv; - - vk::PhysicalDevice physical_device; - vk::PhysicalDeviceProperties properties; - std::string name; - uint64_t max_memory_allocation_size; - uint64_t max_buffer_size; - uint64_t suballocation_block_size; - uint64_t min_imported_host_pointer_alignment; - bool external_memory_host {}; - bool fp16; - bool bf16; - bool pipeline_robustness; - bool memory_priority; - vk::Device device; - uint32_t vendor_id; - vk::DriverId driver_id; - vk_device_architecture architecture; - std::unique_ptr compute_queue; - std::unique_ptr transfer_queue; - bool single_queue; - bool support_async; - bool async_use_transfer_queue; - bool has_internally_synchronized_queues = false; - uint32_t subgroup_size; - uint32_t subgroup_size_log2; - uint32_t shader_core_count; - bool uma; - bool prefer_host_memory; - bool float_controls_rte_fp16; - bool float_controls_denorm_preserve_fp16; - bool subgroup_basic; - bool subgroup_arithmetic; - bool subgroup_shuffle; - bool subgroup_ballot; - bool subgroup_clustered; - bool subgroup_vote; - bool multi_add; - bool shader_int64; - bool buffer_device_address; - bool vulkan_memory_model; - - bool add_rms_fusion; - uint32_t partials_binding_alignment; - uint32_t max_nodes_per_submit; - - bool shader_64b_indexing; - - bool integer_dot_product; - // 0: default, 1: force mmvq, -1: disable mmvq - int32_t mmvq_mode; - - bool subgroup_size_control; - uint32_t subgroup_min_size; - uint32_t subgroup_max_size; - bool subgroup_require_full_support; - - // floor(log2(maxComputeWorkGroupInvocations)) - uint32_t max_workgroup_size_log2 {}; - - bool coopmat_support; - bool coopmat_acc_f32_support {}; - bool coopmat_acc_f16_support {}; - bool coopmat_bf16_support {}; - bool coopmat_support_16x16x16_f16acc {}; - bool coopmat_support_16x16x16_f32acc {}; - bool coopmat1_fa_support {}; - uint32_t coopmat_m; - uint32_t coopmat_n; - uint32_t coopmat_k; - - bool coopmat_int_support; - uint32_t coopmat_int_m; - uint32_t coopmat_int_n; - uint32_t coopmat_int_k; - - bool coopmat2; - bool coopmat2_bf16_support {}; - bool coopmat2_decode_vector; - - bool dot2_f16 {}; - bool ocp_fp4 {}; - - bool pipeline_executable_properties_support {}; - - bool device_fault {}; - PFN_vkGetDeviceFaultInfoEXT pfn_vkGetDeviceFaultInfoEXT {}; - - bool serialize_submissions {}; - - const ggml_cgraph * diag_cgraph {}; - int diag_prev_start = -1; - int diag_prev_end = -1; - - size_t idx; - - bool mul_mat_l[GGML_TYPE_COUNT]; - bool mul_mat_m[GGML_TYPE_COUNT]; - bool mul_mat_s[GGML_TYPE_COUNT]; - bool mul_mat_id_l[GGML_TYPE_COUNT]; - bool mul_mat_id_m[GGML_TYPE_COUNT]; - bool mul_mat_id_s[GGML_TYPE_COUNT]; - - // Separate flags for the q8_1 (integer dot) mmq path, whose shader uses - // a different shared-memory layout than the float matmul shaders. - bool mul_mat_l_int[GGML_TYPE_COUNT]; - bool mul_mat_m_int[GGML_TYPE_COUNT]; - bool mul_mat_s_int[GGML_TYPE_COUNT]; - bool mul_mat_id_l_int[GGML_TYPE_COUNT]; - bool mul_mat_id_m_int[GGML_TYPE_COUNT]; - bool mul_mat_id_s_int[GGML_TYPE_COUNT]; - - vk::DescriptorSetLayout dsl; - - vk_matmul_pipeline pipeline_matmul_f32 {}; - vk_matmul_pipeline pipeline_matmul_f32_f16 {}; - vk_matmul_pipeline pipeline_matmul_bf16 {}; - vk_matmul_pipeline2 pipeline_matmul_f16; - vk_matmul_pipeline2 pipeline_matmul_f16_f32; - - vk_matmul_pipeline2 pipeline_dequant_mul_mat_mat[GGML_TYPE_COUNT]; - vk_matmul_pipeline2 pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_COUNT]; - vk_matmul_pipeline2 pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_COUNT]; - - vk_matmul_pipeline pipeline_matmul_id_f32 {}; - vk_matmul_pipeline pipeline_matmul_id_bf16 {}; - vk_matmul_pipeline2 pipeline_matmul_id_f16; - vk_matmul_pipeline2 pipeline_matmul_id_f16_f32; - - vk_matmul_pipeline2 pipeline_dequant_mul_mat_mat_id[GGML_TYPE_COUNT]; - vk_matmul_pipeline2 pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_COUNT]; - - vk_pipeline pipeline_matmul_split_k_reduce; - vk_pipeline pipeline_quantize_q8_1_x4; - - vk_pipeline pipeline_dequant[GGML_TYPE_COUNT]; - vk_pipeline pipeline_dequant_mul_mat_vec_f32_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; - vk_pipeline pipeline_dequant_mul_mat_vec_f16_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; - vk_pipeline pipeline_dequant_mul_mat_vec_id_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT]; - - vk_pipeline pipeline_dequant_mul_mat_vec_q8_1_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT][mul_mat_vec_max_cols]; - vk_pipeline pipeline_dequant_mul_mat_vec_id_q8_1_f32[DMMV_WG_SIZE_COUNT][GGML_TYPE_COUNT]; - - vk_pipeline pipeline_mul_mat_vec_p021_f16_f32[p021_max_gqa_ratio]; - vk_pipeline pipeline_mul_mat_vec_nc_f16_f32; - vk_pipeline pipeline_get_rows[GGML_TYPE_COUNT]; - vk_pipeline pipeline_get_rows_f32[GGML_TYPE_COUNT]; - vk_pipeline pipeline_get_rows_back_f32; - vk_pipeline pipeline_acc_f32; - vk_pipeline pipeline_set_f32; - - // [src0 0=fp32,1=fp16][src1 0=fp32,1=fp16][dst 0=fp32,1=fp16] - vk_pipeline pipeline_add[2][2][2]; - vk_pipeline pipeline_add_norepeat[2][2][2]; - vk_pipeline pipeline_sub[2][2][2]; - vk_pipeline pipeline_sub_norepeat[2][2][2]; - vk_pipeline pipeline_mul[2][2][2]; - vk_pipeline pipeline_mul_norepeat[2][2][2]; - vk_pipeline pipeline_div[2][2][2]; - vk_pipeline pipeline_div_norepeat[2][2][2]; - vk_pipeline pipeline_add_rms[2][2][2]; - vk_pipeline pipeline_add_rms_norepeat[2][2][2]; - - // indexed by num_additional_fused_ops == num_adds - 1 - vk_pipeline pipeline_multi_add[MAX_FUSED_ADDS]; - vk_pipeline pipeline_multi_add_rms[MAX_FUSED_ADDS]; - - vk_pipeline pipeline_add_id_f32; - - vk_pipeline pipeline_concat_i8, pipeline_concat_i16, pipeline_concat_i32, pipeline_concat_i64; - vk_pipeline pipeline_upscale_nearest_f32, pipeline_upscale_bilinear_f32, pipeline_upscale_bicubic_f32, pipeline_upscale_bilinear_antialias_f32; - vk_pipeline pipeline_scale_f32; - vk_pipeline pipeline_log[2]; - vk_pipeline pipeline_tri[2]; - vk_pipeline pipeline_diag[2]; - vk_pipeline pipeline_clamp[2]; - vk_pipeline pipeline_pad_f32; - vk_pipeline pipeline_roll_f32; - vk_pipeline pipeline_repeat_i32, pipeline_repeat_back_f32; - vk_pipeline pipeline_repeat_i16; - vk_pipeline pipeline_cpy_f32_f32, pipeline_cpy_f32_f16, pipeline_cpy_f16_f16, pipeline_cpy_f16_f32, pipeline_cpy_f32_bf16, pipeline_cpy_bf16_f32, pipeline_cpy_f32_i32, pipeline_cpy_i32_f32; - vk_pipeline pipeline_contig_cpy_f32_f32, pipeline_contig_cpy_f32_f16, pipeline_contig_cpy_f16_f16, pipeline_contig_cpy_f16_f32, pipeline_contig_cpy_f32_bf16, pipeline_contig_cpy_bf16_f32, pipeline_contig_cpy_f32_i32, pipeline_contig_cpy_i32_f32; - vk_pipeline pipeline_cpy_f32_quant[GGML_TYPE_COUNT]; - vk_pipeline pipeline_cpy_quant_f32[GGML_TYPE_COUNT]; - vk_pipeline pipeline_cpy_transpose_16, pipeline_cpy_transpose_32; - // [src0 0=fp32,1=fp16][dst] - vk_pipeline pipeline_set_rows_i32[2][GGML_TYPE_COUNT]; - vk_pipeline pipeline_set_rows_i64[2][GGML_TYPE_COUNT]; - vk_pipeline pipeline_norm_f32; - vk_pipeline pipeline_group_norm_f32; - vk_pipeline pipeline_rms_norm_f32; - vk_pipeline pipeline_rms_norm_mul_f32; - vk_pipeline pipeline_rms_norm_partials_f32; - vk_pipeline pipeline_rms_norm_mul_partials_f32; - vk_pipeline pipeline_rms_norm_mul_rope_f32_f32; - vk_pipeline pipeline_rms_norm_mul_rope_f32_f16; - vk_pipeline pipeline_rms_norm_back_f32; - vk_pipeline pipeline_l2_norm_f32; - - // [src/dst 0=fp32,1=fp16] - vk_pipeline pipeline_exp[2]; - vk_pipeline pipeline_expm1[2]; - vk_pipeline pipeline_elu[2]; - vk_pipeline pipeline_gelu[2]; - vk_pipeline pipeline_gelu_erf[2]; - vk_pipeline pipeline_gelu_quick[2]; - vk_pipeline pipeline_silu[2]; - vk_pipeline pipeline_relu[2]; - vk_pipeline pipeline_sqr[2]; - vk_pipeline pipeline_sqrt[2]; - vk_pipeline pipeline_sin[2]; - vk_pipeline pipeline_cos[2]; - vk_pipeline pipeline_xielu[2]; - vk_pipeline pipeline_neg[2]; - vk_pipeline pipeline_tanh[2]; - vk_pipeline pipeline_sigmoid[2]; - vk_pipeline pipeline_hardsigmoid[2]; - vk_pipeline pipeline_hardswish[2]; - vk_pipeline pipeline_abs[2]; - vk_pipeline pipeline_softplus[2]; - vk_pipeline pipeline_step[2]; - vk_pipeline pipeline_round[2]; - vk_pipeline pipeline_ceil[2]; - vk_pipeline pipeline_floor[2]; - vk_pipeline pipeline_trunc[2]; - vk_pipeline pipeline_sgn[2]; - - vk_pipeline pipeline_add1_f16_f16; - vk_pipeline pipeline_add1_f16_f32; - vk_pipeline pipeline_add1_f32_f32; - - vk_pipeline pipeline_arange_f32; - - vk_pipeline pipeline_fill_f32; - vk_pipeline pipeline_fill_f16; - - vk_pipeline pipeline_geglu[2]; - vk_pipeline pipeline_reglu[2]; - vk_pipeline pipeline_swiglu[2]; - vk_pipeline pipeline_swiglu_oai[2]; - vk_pipeline pipeline_geglu_erf[2]; - vk_pipeline pipeline_geglu_quick[2]; - - vk_pipeline pipeline_leaky_relu[2]; - vk_pipeline pipeline_silu_back_f32; - vk_pipeline pipeline_diag_mask_inf_f32; - vk_pipeline pipeline_soft_max_f32, pipeline_soft_max_f32_f16; - vk_pipeline pipeline_soft_max_f32_wg512, pipeline_soft_max_f32_f16_wg512; - vk_pipeline pipeline_soft_max_back_f32; - - vk_pipeline pipeline_soft_max_large1_f32, pipeline_soft_max_large1_f32_f16; - vk_pipeline pipeline_soft_max_large2_f32, pipeline_soft_max_large2_f32_f16; - vk_pipeline pipeline_soft_max_large3_f32, pipeline_soft_max_large3_f32_f16; - - vk_pipeline pipeline_rope_norm_f32, pipeline_rope_norm_f16, pipeline_rope_norm_f32_f16; - vk_pipeline pipeline_rope_neox_f32, pipeline_rope_neox_f16, pipeline_rope_neox_f32_f16; - vk_pipeline pipeline_rope_multi_f32, pipeline_rope_multi_f16, pipeline_rope_multi_f32_f16; - vk_pipeline pipeline_rope_vision_f32, pipeline_rope_vision_f16; - vk_pipeline pipeline_argsort_f32[num_argsort_pipelines]; - vk_pipeline pipeline_argsort_large_f32[num_argsort_pipelines]; - vk_pipeline pipeline_topk_f32[num_topk_pipelines]; - vk_pipeline pipeline_sum_rows_f32; - vk_pipeline pipeline_fwht_f32[4]; - vk_pipeline pipeline_cumsum_f32; - vk_pipeline pipeline_cumsum_small_f32; - vk_pipeline pipeline_cumsum_multipass1_f32; - vk_pipeline pipeline_cumsum_multipass2_f32; - vk_pipeline pipeline_argmax_f32; - vk_pipeline pipeline_count_equal_i32; - std::map pipeline_solve_tri_f32; - vk_pipeline pipeline_im2col_f32, pipeline_im2col_f32_f16; - vk_pipeline pipeline_im2col_3d_f32, pipeline_im2col_3d_f32_f16; - vk_pipeline pipeline_timestep_embedding_f32; - vk_pipeline pipeline_conv_transpose_1d_f32; - vk_pipeline pipeline_col2im_1d_f32; - vk_pipeline pipeline_col2im_1d_f16; - vk_pipeline pipeline_col2im_1d_bf16; - vk_pipeline pipeline_out_prod_f32; - vk_pipeline pipeline_snake_f32; - vk_pipeline pipeline_snake_f16; - vk_pipeline pipeline_snake_bf16; - vk_pipeline pipeline_pool1d_f32; - vk_pipeline pipeline_pool2d_f32; - vk_pipeline pipeline_rwkv_wkv6_f32; - vk_pipeline pipeline_rwkv_wkv7_f32; - vk_pipeline pipeline_gated_linear_attn_f32; - // [size_idx][kda] where size_idx: 0=d16, 1=d32, 2=d64, 3=d128 - vk_pipeline pipeline_gated_delta_net[4][2]; - vk_pipeline pipeline_ssm_scan_f32_d128; - vk_pipeline pipeline_ssm_scan_f32_d256; - vk_pipeline pipeline_ssm_conv_f32; - vk_pipeline pipeline_ssm_conv_silu_f32; - vk_pipeline pipeline_ssm_conv_bias_silu_f32; - vk_pipeline pipeline_opt_step_adamw_f32; - vk_pipeline pipeline_opt_step_sgd_f32; - std::map pipeline_conv2d_f32[CONV_SHAPE_COUNT]; - std::map pipeline_conv2d_f16_f32[CONV_SHAPE_COUNT]; - std::map pipeline_conv_transpose_2d_f32[CONV_SHAPE_COUNT]; - std::map pipeline_conv_transpose_2d_f16_f32[CONV_SHAPE_COUNT]; - std::map pipeline_conv3d_f32[CONV_SHAPE_COUNT]; - std::map pipeline_conv3d_f16_f32[CONV_SHAPE_COUNT]; - vk_pipeline pipeline_conv2d_dw_whcn_f32, pipeline_conv2d_dw_whcn_f16_f32; - vk_pipeline pipeline_conv2d_dw_cwhn_f32, pipeline_conv2d_dw_cwhn_f16_f32; - - std::map pipeline_flash_attn_f32_f16; - - std::map, vk_pipeline> pipeline_fa_mask_opt; - - vk_pipeline pipeline_flash_attn_split_k_reduce; - vk_pipeline pipeline_count_experts; - - // [2] is for whether to take n_experts from spec constant (0) or push constant (1) - vk_pipeline pipeline_topk_moe[num_topk_moe_pipelines][2]; - - std::vector all_pipelines; - - std::vector> pinned_memory; - - vk::Fence fence; - vk_buffer sync_staging; - - ggml_backend_buffer_type buffer_type; - - bool disable_fusion; - bool disable_host_visible_vidmem; - bool allow_sysmem_fallback; - bool disable_graph_optimize; - - std::unique_ptr memory_logger; - - ~vk_device_struct() { - VK_LOG_DEBUG("destroy device " << name); - - device.destroyFence(fence); - - ggml_vk_destroy_buffer(sync_staging); - - if (compute_queue) compute_queue->cmd_pool.destroy(device); - if (transfer_queue) transfer_queue->cmd_pool.destroy(device); - - // Explicitly clear to ensure queues drop their shared_ptrs to handles - // before the Vulkan logical device instance is destroyed - compute_queue.reset(); - transfer_queue.reset(); - - for (auto& pipeline : all_pipelines) { - if (pipeline.expired()) { - continue; - } - - vk_pipeline pl = pipeline.lock(); - ggml_vk_destroy_pipeline(device, pl); - } - all_pipelines.clear(); - - device.destroyDescriptorSetLayout(dsl); - - device.destroy(); - } -}; - -void vk_command_pool::init(vk_device& device, vk_queue *q_) { - cmd_buffers.clear(); - q = q_; - - vk::CommandPoolCreateInfo command_pool_create_info( - vk::CommandPoolCreateFlags(VK_COMMAND_POOL_CREATE_TRANSIENT_BIT | VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT), - q->queue_family_index); - pool = device->device.createCommandPool(command_pool_create_info); -} - -void vk_command_pool::destroy(vk::Device& device) { - device.destroyCommandPool(pool); - pool = nullptr; - cmd_buffers.clear(); +bool ggml_vk_lightning_indexer_k_type_supported(ggml_type type) { + return std::find(lightning_indexer_k_types.begin(), lightning_indexer_k_types.end(), type) != lightning_indexer_k_types.end(); } - -static void ggml_vk_print_device_fault_info(const vk_device& device) { +void ggml_vk_print_device_fault_info(const vk_device& device) { if (!device->device_fault || !device->pfn_vkGetDeviceFaultInfoEXT) { return; } @@ -1204,6516 +178,5723 @@ static void ggml_vk_print_device_fault_info(const vk_device& device) { (unsigned long long)info.vendorFaultData); } } - -struct vk_buffer_struct { - vk::Buffer buffer = VK_NULL_HANDLE; - vk::DeviceMemory device_memory = VK_NULL_HANDLE; - vk::MemoryPropertyFlags memory_property_flags; - void * ptr; - size_t size = 0; - vk::DeviceAddress bda_addr {}; - - vk_device device; - - ~vk_buffer_struct() { - if (size == 0) { - return; +uint64_t ggml_vk_get_node_flops(const ggml_tensor * node) { + if (node->op == GGML_OP_MUL_MAT || node->op == GGML_OP_MUL_MAT_ID) { + const uint64_t m = node->ne[0]; + const uint64_t n = node->ne[1]; + const uint64_t k = node->src[1]->ne[0]; + const uint64_t batch = node->ne[2] * node->ne[3]; + return m * n * (k + (k - 1)) * batch; + } + if (node->op == GGML_OP_CONV_2D || node->op == GGML_OP_CONV_TRANSPOSE_2D) { + const ggml_tensor * knl = node->src[0]; + const uint64_t Cout = node->ne[2]; + const uint64_t size_K = node->src[1]->ne[2] * knl->ne[0] * knl->ne[1]; + const uint64_t size_N = node->ne[3] * node->ne[0] * node->ne[1]; + return Cout * size_N * (size_K + (size_K - 1)); + } + if (node->op == GGML_OP_CONV_3D) { + const ggml_tensor * knl = node->src[0]; + const uint64_t OC = ggml_get_op_params_i32(node, 11); + const uint64_t IC = ggml_get_op_params_i32(node, 9); + const uint64_t size_K = IC * knl->ne[0] * knl->ne[1] * knl->ne[2]; + const uint64_t size_N = node->ne[3] / OC * node->ne[0] * node->ne[1] * node->ne[2]; + return OC * size_N * (size_K + (size_K - 1)); + } + if (node->op == GGML_OP_FLASH_ATTN_EXT) { + const ggml_tensor * q = node->src[0]; + const ggml_tensor * k = node->src[1]; + const ggml_tensor * v = node->src[2]; + return 2ull * q->ne[1] * q->ne[2] * (k->ne[0] + v->ne[0]) * k->ne[1] * q->ne[3]; + } + return 0; +} +void ggml_vk_print_node_list(const ggml_cgraph * cgraph, int start, int end) { + uint64_t total_flops = 0; + int n_ops = 0; + for (int j = start; j <= end && j < cgraph->n_nodes; j++) { + uint64_t flops = ggml_vk_get_node_flops(cgraph->nodes[j]); + total_flops += flops; + n_ops++; + if (flops > 0) { + GGML_LOG_CONT(" node %d: %s (%s) [%.2f GFLOP]\n", + j, cgraph->nodes[j]->name, ggml_op_name(cgraph->nodes[j]->op), + flops / 1e9); + } else { + GGML_LOG_CONT(" node %d: %s (%s)\n", + j, cgraph->nodes[j]->name, ggml_op_name(cgraph->nodes[j]->op)); } - VK_LOG_DEBUG("~vk_buffer_struct(" << buffer << ", " << size << ")"); - - device->device.freeMemory(device_memory); - device->device.destroyBuffer(buffer); } -}; + GGML_LOG_CONT(" total: %d ops, %.2f GFLOP\n", n_ops, total_flops / 1e9); +} +void ggml_vk_print_device_lost_info(const vk_device& device) { + ggml_vk_print_device_fault_info(device); + if (device->serialize_submissions && device->diag_cgraph != nullptr && device->diag_prev_start >= 0) { + GGML_LOG_ERROR("ggml_vulkan: device lost on %s, likely caused by previous submission (nodes %d to %d):\n", + device->name.c_str(), device->diag_prev_start, device->diag_prev_end); + ggml_vk_print_node_list(device->diag_cgraph, device->diag_prev_start, device->diag_prev_end); + } else { + GGML_LOG_ERROR("ggml_vulkan: device lost on %s\n", device->name.c_str()); + } +} +void * const vk_ptr_base = (void *)(uintptr_t) 0x1000; // NOLINT -struct vk_subbuffer { - vk_buffer buffer; - uint64_t offset; - uint64_t size; +uint64_t vk_tensor_offset(const ggml_tensor * tensor) { + if (tensor->view_src) { + return (uint8_t *) tensor->view_src->data - (uint8_t *) vk_ptr_base; + } + return (uint8_t *) tensor->data - (uint8_t *) vk_ptr_base; +} - operator vk::DescriptorBufferInfo() const { - return { buffer->buffer, offset, size }; +size_t ggml_vk_tensor_buffer_offset(const ggml_backend_vk_context * ctx, const ggml_tensor * t) { + // vk_tensor_offset() is relative to vk_ptr_base, but mapped host tensors need an offset relative to their Vulkan buffer. + if (ctx->device->uma) { + vk_buffer buf = nullptr; + size_t off = 0; + ggml_vk_host_get(ctx->device, t->data, buf, off); + if (buf) { + return off; + } + } + return (size_t)(vk_tensor_offset(t) + t->view_offs); +} +size_t ggml_vk_descriptor_offset(size_t tensor_offset, size_t alignment, size_t type_size) { + // Move the descriptor back until its distance to the tensor is divisible by the tensor type size. + size_t descriptor_offset = tensor_offset & ~(alignment - 1); + while ((tensor_offset - descriptor_offset) % type_size != 0) { + GGML_ASSERT(descriptor_offset >= alignment); + descriptor_offset -= alignment; } -}; -struct vk_semaphore { - vk::Semaphore s; - uint64_t value; -}; + return descriptor_offset; +} +uint32_t get_misalign_bytes(const ggml_backend_vk_context * ctx, const ggml_tensor * t) { + const size_t tensor_offset = ggml_vk_tensor_buffer_offset(ctx, t); + const size_t descriptor_offset = ggml_vk_descriptor_offset( + tensor_offset, ctx->device->properties.limits.minStorageBufferOffsetAlignment, ggml_type_size(t->type)); + GGML_ASSERT(tensor_offset - descriptor_offset <= UINT32_MAX); + return tensor_offset - descriptor_offset; +} -// vk_event is used for the event-related backend interfaces. It uses vk::Events for -// event_wait and a timeline semaphore for event_synchronize. Polling on an event for -// event_synchronize wouldn't be sufficient to wait for command buffers to complete, -// and would lead to validation errors. -struct vk_event { - std::vector events_free; // Events available for reuse - std::vector events_submitted; // Events that are fully submitted and can be reused on next synchronize - vk::Event event; - bool has_event; - - vk_semaphore tl_semaphore; - vk_command_buffer* cmd_buffer = nullptr; - uint64_t cmd_buffer_use_counter = 0; -}; +uint32_t ggml_vk_concat_unit_size(ggml_type type) { + const uint32_t type_size = ggml_type_size(type); -struct vk_submission { - vk_command_buffer* buffer = nullptr; - std::vector wait_semaphores; - std::vector signal_semaphores; -}; + if (!ggml_is_quantized(type)) { + return type_size; + } -typedef std::vector vk_sequence; + // Use the widest existing concat shader that evenly divides a quant block. + if (type_size % 8 == 0) { + return 8; + } + if (type_size % 4 == 0) { + return 4; + } + if (type_size % 2 == 0) { + return 2; + } + return 1; +} +bool ggml_vk_concat_supported(const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { + if (src0->type != src1->type || src0->type != dst->type) { + return false; + } -struct vk_mat_mat_push_constants { - uint32_t M; uint32_t N; uint32_t K; - uint32_t stride_a; uint32_t stride_b; uint32_t stride_d; - uint32_t batch_stride_a; uint32_t batch_stride_b; uint32_t batch_stride_d; - uint32_t base_work_group_z; uint32_t num_batches; - uint32_t k_split; - uint32_t ne02; uint32_t ne12; uint32_t broadcast2; uint32_t broadcast3; - uint32_t padded_N; -}; + if (!ggml_is_quantized(src0->type)) { + const size_t type_size = ggml_type_size(src0->type); + return type_size == 1 || type_size == 2 || type_size == 4 || type_size == 8; + } -#define MAT_VEC_FUSION_FLAGS_BIAS0 0x1 -#define MAT_VEC_FUSION_FLAGS_BIAS1 0x2 -#define MAT_VEC_FUSION_FLAGS_SCALE0 0x4 -#define MAT_VEC_FUSION_FLAGS_SCALE1 0x8 - -struct vk_mat_vec_push_constants { - uint32_t ncols; - uint32_t stride_a; - uint32_t stride_b; - uint32_t stride_d; - uint32_t batch_stride_a; - uint32_t batch_stride_b; - uint32_t batch_stride_d; - uint32_t fusion_flags; - uint32_t base_work_group_y; - uint32_t ne02; - uint32_t ne12; - uint32_t broadcast2; - uint32_t broadcast3; -}; + // Quantized tensor rows are block-aligned when created. + return ggml_is_contiguous_rows(src0) && ggml_is_contiguous_rows(src1) && ggml_is_contiguous_rows(dst); +} +static bool vk_instance_initialized = false; -struct vk_mat_vec_p021_push_constants { - uint32_t ncols_x; - uint32_t nrows_x; - uint32_t nchannels_x; - uint32_t nchannels_y; - uint32_t b_offset; - uint32_t d_offset; - uint32_t fusion_flags; -}; +vk_instance_t vk_instance; -struct vk_mat_vec_nc_push_constants { - uint32_t ncols_x; - uint32_t nrows_x; - uint32_t row_stride_x; - uint32_t channel_stride_x; - uint32_t channel_stride_y; - uint32_t channel_x_divisor; - uint32_t ne12; - uint32_t b_offset; - uint32_t d_offset; - uint32_t nb03; - uint32_t nb13; - uint32_t nb23; - uint32_t fusion_flags; -}; +static VkDeviceSize ggml_vk_get_max_buffer_range(const ggml_backend_vk_context * ctx, const vk_buffer &buf, const VkDeviceSize offset) { + const VkDeviceSize range = std::min(VkDeviceSize{buf->size - offset}, + VkDeviceSize{ctx->device->properties.limits.maxStorageBufferRange}); + return range; +} -struct vk_mat_mat_id_push_constants { - uint32_t M; uint32_t N; uint32_t K; - uint32_t stride_a; uint32_t stride_b; uint32_t stride_d; - uint32_t batch_stride_a; uint32_t batch_stride_b; uint32_t batch_stride_d; - uint32_t nei0; uint32_t nei1; uint32_t nbi1; uint32_t ne11; - uint32_t padded_N; -}; -struct vk_mat_vec_id_push_constants { - uint32_t ncols; - uint32_t stride_a; - uint32_t stride_b; - uint32_t stride_d; - uint32_t batch_stride_a; - uint32_t batch_stride_b; - uint32_t batch_stride_d; - uint32_t fusion_flags; - uint32_t nei0; - uint32_t ne11; - uint32_t expert_i1; - uint32_t nbi1; -}; - -struct vk_flash_attn_push_constants { - uint32_t N; - uint32_t KV; - - uint32_t ne1; - uint32_t ne2; - uint32_t ne3; - - uint32_t neq2; - uint32_t neq3; - uint32_t nek2; - uint32_t nek3; - uint32_t nev2; - uint32_t nev3; - uint32_t nem1; - uint32_t nem2; - uint32_t nem3; - - uint32_t nb01; - uint32_t nb02; - uint32_t nb03; - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; - uint32_t nb21; - uint32_t nb22; - uint32_t nb23; - - float scale; - float max_bias; - float logit_softcap; - - uint32_t mask_n_head_log2; - float m0; - float m1; - - uint32_t gqa_ratio; - uint32_t split_kv; - uint32_t k_num; -}; -static_assert(sizeof(vk_flash_attn_push_constants) <= 128, "sizeof(vk_flash_attn_push_constants) must be <= 128"); - -struct vk_op_push_constants { - uint32_t KX; - uint32_t KY; - float param1; - float param2; - float param3; - float param4; -}; - -struct vk_op_fwht_push_constants { - uint32_t n_rows; - uint32_t src_offset; - uint32_t dst_offset; - float scale; -}; - -struct vk_op_count_experts_push_constants { - uint32_t ne00; - uint32_t ne01; - uint32_t nb00; - uint32_t nb01; - uint32_t a_offset; -}; - -struct vk_op_glu_push_constants { - uint32_t N; - uint32_t ne00; - uint32_t ne20; - uint32_t mode; // 0: default, 1: swapped, 2: split - float alpha; // for swiglu_oai - float limit; - uint32_t nb00; - uint32_t nb01; - uint32_t nb02; - uint32_t nb03; - uint32_t nb10; - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; - uint32_t nb20; - uint32_t nb21; - uint32_t nb22; - uint32_t nb23; - uint32_t ne21; - uint32_t ne22; - uint32_t misalign_offsets; - uint32_t ne2_012mp; uint32_t ne2_012L; - uint32_t ne2_01mp; uint32_t ne2_01L; - uint32_t ne2_0mp; uint32_t ne2_0L; -}; -static_assert(sizeof(vk_op_glu_push_constants) <= 128, "sizeof(vk_op_glu_push_constants) must be <= 128"); - -struct vk_op_unary_push_constants { - uint32_t ne; - uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; - uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; - uint32_t misalign_offsets; - float param1; float param2; float param3; float param4; - uint32_t ne0_012mp; uint32_t ne0_01mp; uint32_t ne0_0mp; uint32_t ne0_Ls; - uint32_t ne1_012mp; uint32_t ne1_01mp; uint32_t ne1_0mp; uint32_t ne1_Ls; -}; -static_assert(sizeof(vk_op_unary_push_constants) <= 128, "sizeof(vk_op_unary_push_constants) must be <= 128"); - -static vk_op_unary_push_constants vk_op_unary_push_constants_init(const ggml_tensor * src0, const ggml_tensor * dst, int64_t ne = 0) { - GGML_ASSERT(ne != 0 || (ggml_nelements(src0) == ggml_nelements(dst))); - ne = ne != 0 ? ne : ggml_nelements(dst); - GGML_ASSERT(ne <= (int64_t)std::numeric_limits::max()); - - vk_op_unary_push_constants p{}; - p.ne = (uint32_t)ne; - - size_t src0_tsize = ggml_type_size(src0->type); - p.ne00 = (uint32_t)src0->ne[0]; - p.ne01 = (uint32_t)src0->ne[1]; - p.ne02 = (uint32_t)src0->ne[2]; - p.ne03 = (uint32_t)src0->ne[3]; - p.nb00 = (uint32_t)(src0->nb[0] / src0_tsize); - p.nb01 = (uint32_t)(src0->nb[1] / src0_tsize); - p.nb02 = (uint32_t)(src0->nb[2] / src0_tsize); - p.nb03 = (uint32_t)(src0->nb[3] / src0_tsize); - - size_t dst_tsize = ggml_type_size(dst->type); - p.ne10 = (uint32_t)dst->ne[0]; - p.ne11 = (uint32_t)dst->ne[1]; - p.ne12 = (uint32_t)dst->ne[2]; - p.ne13 = (uint32_t)dst->ne[3]; - p.nb10 = (uint32_t)(dst->nb[0] / dst_tsize); - p.nb11 = (uint32_t)(dst->nb[1] / dst_tsize); - p.nb12 = (uint32_t)(dst->nb[2] / dst_tsize); - p.nb13 = (uint32_t)(dst->nb[3] / dst_tsize); - - return p; // offsets are initialized later in ggml_vk_op -} - -struct vk_op_pad_push_constants { - uint32_t ne; - uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; - uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; - uint32_t misalign_offsets; - uint32_t circular; - - uint32_t lp0; uint32_t rp0; - uint32_t lp1; uint32_t rp1; - uint32_t lp2; uint32_t rp2; - uint32_t lp3; uint32_t rp3; -}; - -static vk_op_pad_push_constants vk_op_pad_push_constants_init(const ggml_tensor * src0, const ggml_tensor * dst) { - int64_t ne = ggml_nelements(dst); - GGML_ASSERT(ne <= (int64_t)std::numeric_limits::max()); - - vk_op_pad_push_constants p{}; - p.ne = (uint32_t)ne; - - size_t src0_tsize = ggml_type_size(src0->type); - p.ne00 = (uint32_t)src0->ne[0]; - p.ne01 = (uint32_t)src0->ne[1]; - p.ne02 = (uint32_t)src0->ne[2]; - p.ne03 = (uint32_t)src0->ne[3]; - p.nb00 = (uint32_t)(src0->nb[0] / src0_tsize); - p.nb01 = (uint32_t)(src0->nb[1] / src0_tsize); - p.nb02 = (uint32_t)(src0->nb[2] / src0_tsize); - p.nb03 = (uint32_t)(src0->nb[3] / src0_tsize); - - size_t dst_tsize = ggml_type_size(dst->type); - p.ne10 = (uint32_t)dst->ne[0]; - p.ne11 = (uint32_t)dst->ne[1]; - p.ne12 = (uint32_t)dst->ne[2]; - p.ne13 = (uint32_t)dst->ne[3]; - p.nb10 = (uint32_t)(dst->nb[0] / dst_tsize); - p.nb11 = (uint32_t)(dst->nb[1] / dst_tsize); - p.nb12 = (uint32_t)(dst->nb[2] / dst_tsize); - p.nb13 = (uint32_t)(dst->nb[3] / dst_tsize); - - p.lp0 = dst->op_params[0]; - p.rp0 = dst->op_params[1]; - p.lp1 = dst->op_params[2]; - p.rp1 = dst->op_params[3]; - p.lp2 = dst->op_params[4]; - p.rp2 = dst->op_params[5]; - p.lp3 = dst->op_params[6]; - p.rp3 = dst->op_params[7]; - p.circular = dst->op_params[8]; - - return p; // fastdiv values and offsets are initialized later in ggml_vk_op -} - -// See https://gmplib.org/~tege/divcnst-pldi94.pdf figure 4.1. -// Precompute mp (m' in the paper) and L such that division -// can be computed using a multiply (high 32b of 64b result) -// and a shift: -// -// n/d = (mulhi(n, mp) + n) >> L; -static void init_fastdiv_values(uint32_t d, uint32_t &mp, uint32_t &L) -{ - // compute L = ceil(log2(d)); - L = 0; - while (L < 32 && (uint32_t{1} << L) < d) { - L++; +void ggml_vk_wait_for_fence(ggml_backend_vk_context * ctx) { + // Use waitForFences while most of the graph executes. Hopefully the CPU can sleep + // during this wait. + if (ctx->almost_ready_fence_pending) { + VK_CHECK(ctx->device->device.waitForFences({ ctx->almost_ready_fence }, true, UINT64_MAX), "almost_ready_fence", ctx->device); + ctx->device->device.resetFences({ ctx->almost_ready_fence }); + ctx->almost_ready_fence_pending = false; } - mp = (uint32_t)((uint64_t{1} << 32) * ((uint64_t{1} << L) - d) / d + 1); -} - -static uint32_t pack_fastdiv_L(uint32_t L0, uint32_t L1, uint32_t L2) { - return L0 | (L1 << 8) | (L2 << 16); -} - -template void init_pushconst_fastdiv(T &p) { - GGML_UNUSED(p); - static_assert(!std::is_const::value, "unexpected type"); -} - -template <> void init_pushconst_fastdiv(vk_op_unary_push_constants &p) { - // Compute magic values to divide by these six numbers. - uint32_t ne0_012L; - uint32_t ne0_01L; - uint32_t ne0_0L; - uint32_t ne1_012L; - uint32_t ne1_01L; - uint32_t ne1_0L; - - init_fastdiv_values(p.ne02*p.ne01*p.ne00, p.ne0_012mp, ne0_012L); - init_fastdiv_values(p.ne01*p.ne00, p.ne0_01mp, ne0_01L); - init_fastdiv_values(p.ne00, p.ne0_0mp, ne0_0L); - init_fastdiv_values(p.ne12*p.ne11*p.ne10, p.ne1_012mp, ne1_012L); - init_fastdiv_values(p.ne11*p.ne10, p.ne1_01mp, ne1_01L); - init_fastdiv_values(p.ne10, p.ne1_0mp, ne1_0L); - - p.ne0_Ls = pack_fastdiv_L(ne0_012L, ne0_01L, ne0_0L); - p.ne1_Ls = pack_fastdiv_L(ne1_012L, ne1_01L, ne1_0L); -} - -template <> void init_pushconst_fastdiv(vk_op_glu_push_constants &p) { - // GLU linearizes over dst, then uses dst coordinates for src0/src1. - init_fastdiv_values(p.ne22*p.ne21*p.ne20, p.ne2_012mp, p.ne2_012L); - init_fastdiv_values(p.ne21*p.ne20, p.ne2_01mp, p.ne2_01L); - init_fastdiv_values(p.ne20, p.ne2_0mp, p.ne2_0L); + // Spin (w/pause) waiting for the graph to finish executing. + vk::Result result; + for (;;) { + try { + result = ctx->device->device.getFenceStatus(ctx->fence); + } catch (vk::DeviceLostError &) { + ggml_vk_print_device_lost_info(ctx->device); + GGML_LOG_ERROR("ggml_vulkan: getFenceStatus at %s:%d\n", __FILE__, __LINE__); + throw; + } + if (result == vk::Result::eSuccess) { + break; + } + if (result != vk::Result::eNotReady) { + GGML_LOG_ERROR("ggml_vulkan: error %s at %s:%d\n", to_string(result).c_str(), __FILE__, __LINE__); + throw vk::SystemError(vk::make_error_code(result), "ggml_vulkan: getFenceStatus"); + } + for (uint32_t i = 0; i < 100; ++i) { + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + YIELD(); + } + } + ctx->device->device.resetFences({ ctx->fence }); } -struct vk_op_binary_push_constants { - uint32_t ne; - uint32_t ne00; uint32_t ne01; uint32_t ne02; uint32_t ne03; uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; - uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; uint32_t nb10; uint32_t nb11; uint32_t nb12; uint32_t nb13; - uint32_t ne20; uint32_t ne21; uint32_t ne22; uint32_t ne23; uint32_t nb20; uint32_t nb21; uint32_t nb22; uint32_t nb23; - uint32_t misalign_offsets; - float param1; float param2; int32_t param3; -}; +static bool ggml_vk_strip_decode_vector(const uint32_t * code, size_t word_count, std::vector & out) { + static const char kDecodeVectorExt[] = "SPV_NV_cooperative_matrix_decode_vector"; -// Distinct type with the same layout so concat can overload tensor offset initialization. -struct vk_op_concat_push_constants : vk_op_binary_push_constants {}; -static_assert(sizeof(vk_op_concat_push_constants) == sizeof(vk_op_binary_push_constants)); -static_assert(std::is_standard_layout_v); + if (word_count < 5) { + return false; + } -struct vk_op_multi_add_push_constants { - // shape for dst - uint32_t ne20; uint32_t ne21; uint32_t ne22; uint32_t ne23; + bool uses_decode_vector = false; + for (size_t pos = 5; pos < word_count; ) { + uint32_t word = code[pos]; + uint32_t wc = word >> spv::WordCountShift; + uint32_t op = word & spv::OpCodeMask; + GGML_ASSERT(wc > 0 && pos + wc <= word_count); + if (op == spv::OpExtension && wc >= 2) { + const char * s = reinterpret_cast(&code[pos + 1]); + if (strcmp(s, kDecodeVectorExt) == 0) { + uses_decode_vector = true; + break; + } + } + pos += wc; + } - // strides for srcs+dst - uint32_t nb[MAX_PARAMETER_COUNT][4]; + if (!uses_decode_vector) { + return false; + } - uint32_t rms_partials; -}; -// update multi_add.comp if this changes -static_assert(MAX_PARAMETER_COUNT == 12); -static_assert(sizeof(vk_op_multi_add_push_constants) <= 256); - -struct vk_op_topk_moe_push_constants { - uint32_t n_rows; - uint32_t n_experts_push; - uint32_t n_expert_used; - float clamp_min; - float clamp_max; - uint32_t gating_func; - uint32_t has_bias; - uint32_t with_norm; - float output_scale; - float output_bias; -}; + VK_LOG_DEBUG("ggml_vk_strip_decode_vector: stripping SPV_NV_cooperative_matrix_decode_vector"); -struct vk_op_add_id_push_constants { - uint32_t ne0; - uint32_t ne1; - uint32_t s01; - uint32_t s02; - uint32_t s11; - uint32_t s21; -}; + // Bulk-copy unchanged runs and only break the run when an instruction needs to + // be dropped or patched. Use reserve + insert/push_back so the destination buffer + // is touched exactly once (no zero-initialization pass from resize()). + out.clear(); + out.reserve(word_count); -struct vk_op_diag_mask_push_constants { - uint32_t ncols; - uint32_t rows_per_channel; - int32_t n_past; -}; + size_t run_start = 0; + auto flush_run = [&](size_t up_to) { + if (up_to > run_start) { + out.insert(out.end(), code + run_start, code + up_to); + } + }; -struct vk_op_rope_push_constants { - uint32_t rope_mode; - uint32_t nrows; - uint32_t n_dims; - float freq_scale; - float freq_base; - float ext_factor; - float attn_factor; - float corr_dims[2]; - float theta_scale; - uint32_t has_ff; - int32_t sections[4]; - uint32_t is_imrope; - uint32_t is_back; - uint32_t set_rows_stride; - uint32_t ne00; - uint32_t ne01; - uint32_t ne02; - uint32_t nb01; - uint32_t nb02; - uint32_t nb03; - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; - uint32_t a_offset; - uint32_t d_offset; -}; -static_assert(sizeof(vk_op_rope_push_constants) <= 128, "sizeof(vk_op_rope_push_constants) must be <= 128"); + for (size_t pos = 5; pos < word_count; ) { + uint32_t word = code[pos]; + uint32_t wc = word >> spv::WordCountShift; + uint32_t op = word & spv::OpCodeMask; + GGML_ASSERT(wc > 0 && pos + wc <= word_count); -// For fused rms_norm+mul+rope(+view+set_rows) -struct vk_op_rms_norm_mul_rope_push_constants { - vk_op_binary_push_constants bin; - vk_op_rope_push_constants rope; -}; + if (op == spv::OpExtension && wc >= 2) { + const char * s = reinterpret_cast(&code[pos + 1]); + if (strcmp(s, kDecodeVectorExt) == 0) { + flush_run(pos); + pos += wc; + run_start = pos; + continue; + } + } -struct vk_op_soft_max_push_constants { - uint32_t KX; - uint32_t KY; - uint32_t ne00; - uint32_t ne01; - uint32_t ne02; - uint32_t ne12; - uint32_t ne13; - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; - float scale; - float max_bias; - float m0; - float m1; - uint32_t n_head_log2; - uint32_t nrows_x; - uint32_t has_sinks; -}; + if (op == spv::OpCapability && wc == 2 && code[pos + 1] == kSpvCapabilityCooperativeMatrixDecodeVectorNV) { + flush_run(pos); + pos += wc; + run_start = pos; + continue; + } -struct vk_op_argsort_push_constants { - uint32_t ncols; - uint32_t ncols_padded; - uint32_t ncols_padded_log2; - uint32_t nrows; - uint32_t order; - uint32_t outer_start; - uint32_t outer_end; - uint32_t inner_start; - uint32_t inner_end; -}; + if (op == kSpvOpCooperativeMatrixLoadTensorNV) { + // [opcode/wc][ResultType][Result][Pointer][Object][TensorLayout][MemOperand mask][mem extras...][TA mask][ta extras...] + GGML_ASSERT(wc >= 8); -struct vk_op_topk_push_constants { - uint32_t orig_ncols; - uint32_t ncols_input; - uint32_t ncols_output; - uint32_t k; - uint32_t nrows; - uint32_t first_pass; - uint32_t last_pass; -}; + uint32_t mem_mask = code[pos + 6]; + size_t cur = pos + 7; + // Each of these MemoryAccess bits (when set) carries one trailing operand. + cur += (mem_mask & 0x2) ? 1 : 0; // Aligned + cur += (mem_mask & 0x8) ? 1 : 0; // MakePointerAvailable + cur += (mem_mask & 0x10) ? 1 : 0; // MakePointerVisible + cur += (mem_mask & 0x10000) ? 1 : 0; // AliasScopeINTELMask + cur += (mem_mask & 0x20000) ? 1 : 0; // NoAliasINTELMask + GGML_ASSERT(cur < pos + wc); -struct vk_op_im2col_push_constants { - uint64_t dst_addr; - uint32_t batch_offset; uint32_t offset_delta; - uint32_t IC; - uint32_t IW; uint32_t IH; - uint32_t OW; uint32_t OH; - uint32_t KW; uint32_t KH; - uint32_t OH_batch; - uint32_t CHW; - int32_t s0; int32_t s1; - int32_t p0; int32_t p1; - int32_t d0; int32_t d1; - uint32_t batch_IC; -}; + uint32_t ta_mask = code[cur]; + if ((ta_mask & kSpvTensorAddressingDecodeVectorFuncBit) == 0) { + pos += wc; + continue; // leave instruction inside the current unchanged run + } -struct vk_op_im2col_3d_push_constants { - uint64_t dst_addr; - uint32_t nb10; - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; - uint32_t s0; - uint32_t s1; - uint32_t s2; - uint32_t p0; - uint32_t p1; - uint32_t p2; - uint32_t d0; - uint32_t d1; - uint32_t d2; - uint32_t IW; - uint32_t IH; - uint32_t ID; - uint32_t IC; - uint32_t KW; - uint32_t OH; - uint32_t KD_KH_KW; - uint32_t KH_KW; - uint32_t IC_KD_KH_KW; - uint32_t N_OD_OH; - uint32_t OD_OH; - uint32_t OD_OH_OW_IC_KD_KH_KW; - uint32_t OH_OW_IC_KD_KH_KW; - uint32_t OW_IC_KD_KH_KW; - uint32_t misalign_offsets; -}; + flush_run(pos); -struct vk_op_timestep_embedding_push_constants { - uint32_t nb1; - uint32_t dim; - uint32_t max_period; -}; + // Append unchanged prefix of the instruction (header through the mem-extras). + size_t inst_start = out.size(); + size_t pre_n = cur - pos; + out.insert(out.end(), code + pos, code + pos + pre_n); -struct vk_op_col2im_1d_push_constants { - uint32_t T_out; - uint32_t OC; - uint32_t K_OC; - uint32_t T_in; - uint32_t K; - int32_t stride; - int32_t p0; -}; + // Emit TA mask with the DecodeVectorFunc bit cleared. + out.push_back(ta_mask & ~kSpvTensorAddressingDecodeVectorFuncBit); -struct vk_op_conv_transpose_1d_push_constants { - uint32_t Cout; - uint32_t Cin; - uint32_t K; - uint32_t L; - uint32_t KL; + // TA extras: TensorView (0x1) and DecodeFunc (0x2) are kept verbatim; + // DecodeVectorFunc (0x4) is dropped along with its trailing id operand. + size_t keep_ta_extras = ((ta_mask & 0x1) ? 1 : 0) + ((ta_mask & 0x2) ? 1 : 0); + if (keep_ta_extras) { + out.insert(out.end(), code + cur + 1, code + cur + 1 + keep_ta_extras); + } - uint32_t nb01; - uint32_t nb02; - uint32_t nb11; - uint32_t nb1; + GGML_ASSERT(wc == pre_n + 1 + keep_ta_extras + 1); - int32_t s0; -}; + // Patch the instruction header with the new (one-shorter) word count. + uint32_t new_wc = wc - 1; + out[inst_start] = (new_wc << spv::WordCountShift) | op; -struct vk_op_snake_push_constants { - uint32_t ne0; - uint32_t ne1; -}; + pos += wc; + run_start = pos; + continue; + } -struct vk_op_pool1d_push_constants { - uint32_t IL; - uint32_t OL; - uint32_t OC; - uint32_t pelements; - uint32_t op; - int32_t k0; - int32_t s0; - int32_t p0; -}; + pos += wc; + } -struct vk_op_pool2d_push_constants { - uint32_t IW; uint32_t IH; - uint32_t OW; uint32_t OH; - uint32_t OC; - uint32_t pelements; - uint32_t op; - int32_t k0; int32_t k1; - int32_t s0; int32_t s1; - int32_t p0; int32_t p1; -}; + flush_run(word_count); + return true; +} -struct vk_op_rwkv_wkv6_push_constants { - uint32_t B; - uint32_t T; - uint32_t C; - uint32_t H; -}; +static bool ggml_vk_roll_bk_loop(const uint32_t * code, size_t word_count, std::vector & out) { + if (word_count < 5) { + return false; + } -struct vk_op_rwkv_wkv7_push_constants { - uint32_t B; - uint32_t T; - uint32_t C; - uint32_t H; -}; -struct vk_op_gated_linear_attn_push_constants { - uint32_t B; - uint32_t T; - uint32_t C; - uint32_t H; - float scale; -}; -struct vk_op_gated_delta_net_push_constants { - uint32_t H; - uint32_t n_tokens; - uint32_t n_seqs; - uint32_t s_off; - uint32_t sq1, sq2, sq3; - uint32_t sv1, sv2, sv3; - uint32_t sb1, sb2, sb3; - uint32_t neq1, rq3; - float scale; - uint32_t K; -}; + struct vk_spv_loop { + size_t header; + size_t end; + uint32_t control; + }; -struct vk_op_ssm_scan_push_constants { - uint32_t nb02, nb03, nb12, nb13; - uint32_t nb21, nb22, nb31; - uint32_t nb42, nb43, nb52, nb53; - uint32_t s_off; - uint32_t n_head, d_head, n_group, n_tok; - uint32_t n_seq, K; -}; -struct vk_op_ssm_conv_push_constants { - uint32_t nb01, nb02; - uint32_t nb11; - uint32_t dst_nb0, dst_nb1, dst_nb2; - uint32_t nc, ncs, nr, n_t, n_s; -}; + std::vector loops; -struct vk_op_conv2d_push_constants { - uint32_t Cout; - uint32_t Cin; - uint32_t N; + // Collect a list of all loops in the module. + for (size_t pos = 5; pos < word_count; ) { + const uint32_t wc = code[pos] >> spv::WordCountShift; + const uint32_t op = code[pos] & spv::OpCodeMask; + if (wc == 0 || pos + wc > word_count) { + return false; + } - uint32_t W; - uint32_t H; - uint32_t OW; - uint32_t OH; + if (op == spv::OpLoopMerge && wc >= 4) { loops.push_back({ pos, 0, code[pos + 3] }); } - uint32_t nb01; - uint32_t nb02; - uint32_t nb03; + if (op == spv::OpLabel && wc >= 2) { + for (auto & l : loops) { + if (l.end == 0 && code[l.header + 1] == code[pos + 1]) { l.end = pos; } + } + } - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; + pos += wc; + } - uint32_t nb1; - uint32_t nb2; - uint32_t nb3; + auto encloses = [](const vk_spv_loop & a, const vk_spv_loop & b) { + return a.header < b.header && b.header < a.end; + }; - // init_fastdiv_values constants for dividing by OW, OW*OH - uint32_t OWmp; uint32_t OWL; - uint32_t OWOHmp; uint32_t OWOHL; -}; + // Find the BK loop. + const vk_spv_loop * bk = nullptr; + for (const auto & h : loops) { + if (h.control != spv::LoopControlUnrollMask) { + continue; + } + const vk_spv_loop * parent = nullptr; + bool has_child = false; + for (const auto & g : loops) { + if (encloses(g, h) && (!parent || g.header > parent->header)) { + parent = &g; + } + if (encloses(h, g)) { + has_child = true; + } + } + // BK loop should be the last loop nested inside the loop with no hint + // and have at least one child loop. + if (parent && + parent->control == spv::LoopControlMaskNone && + has_child && + (!bk || h.header > bk->header)) { + bk = &h; + } + } + if (!bk) { + return false; + } -template <> void init_pushconst_fastdiv(vk_op_conv2d_push_constants &p) { - // Compute magic values to divide by OW, OW*OH - init_fastdiv_values(p.OW, p.OWmp, p.OWL); - init_fastdiv_values(p.OW*p.OH, p.OWOHmp, p.OWOHL); + // set DontUnroll instead of Unroll + out.assign(code, code + word_count); + out[bk->header + 3] = spv::LoopControlDontUnrollMask; + return true; } -struct vk_op_conv3d_push_constants { - uint32_t OC; - uint32_t IC; - uint32_t N; - - uint32_t IW; - uint32_t IH; - uint32_t ID; - uint32_t OW; - uint32_t OH; - uint32_t OD; - - uint32_t nb01; - uint32_t nb02; - uint32_t nb03; +static void ggml_vk_create_pipeline_func(vk_device& device, vk_pipeline& pipeline, size_t spv_size, const void* spv_data, const std::string entrypoint, + uint32_t parameter_count, std::array wg_denoms, std::vector specialization_constants, + bool disable_robustness, bool require_full_subgroups, uint32_t required_subgroup_size) { + VK_LOG_DEBUG("ggml_vk_create_pipeline(" << device->name << ", " << pipeline->name << ", " << entrypoint << ", " << parameter_count << + ", (" << wg_denoms[0] << "," << wg_denoms[1] << "," << wg_denoms[2] << "), specialization_constants, " << + disable_robustness << ", " << require_full_subgroups << ", " << required_subgroup_size << ")"); + GGML_ASSERT(parameter_count > 0); + GGML_ASSERT(parameter_count <= MAX_PARAMETER_COUNT); + GGML_ASSERT(wg_denoms[0] > 0 && wg_denoms[1] > 0 && wg_denoms[2] > 0); // NOLINT - uint32_t nb11; - uint32_t nb12; - uint32_t nb13; + vk::ShaderModuleCreateInfo shader_module_create_info({}, spv_size, reinterpret_cast(spv_data)); - uint32_t nb1; - uint32_t nb2; - uint32_t nb3; + // Patch SPIR-V to enable supported FP16 float controls, avoiding the need + // for separate shader variants. + std::vector spirv; + if (device->float_controls_rte_fp16 || device->float_controls_denorm_preserve_fp16) { + const uint32_t* spv_words = reinterpret_cast(spv_data); + size_t word_count = spv_size / sizeof(uint32_t); + spirv.assign(spv_words, spv_words + word_count); - uint32_t OWmp; uint32_t OWL; - uint32_t OWOHmp; uint32_t OWOHL; - uint32_t OWOHODmp; uint32_t OWOHODL; -}; + // Find insertion points respecting SPIR-V layout order: + // Header(5) -> OpCapability -> OpExtension -> ... -> OpEntryPoint -> OpExecutionMode -> ... + size_t pos = 5; // skip header + size_t cap_insert_pos = pos; + size_t ext_insert_pos = pos; + size_t exec_insert_pos = pos; + uint32_t entry_point_id = 0; -template <> void init_pushconst_fastdiv(vk_op_conv3d_push_constants &p) { - init_fastdiv_values(p.OW, p.OWmp, p.OWL); - init_fastdiv_values(p.OW*p.OH, p.OWOHmp, p.OWOHL); - init_fastdiv_values(p.OW*p.OH*p.OD, p.OWOHODmp, p.OWOHODL); -} - -struct vk_op_conv2d_dw_push_constants { - uint32_t ne; - uint32_t batches; - uint32_t channels; - uint32_t dst_w; - uint32_t dst_h; - uint32_t src_w; - uint32_t src_h; - uint32_t knl_w; - uint32_t knl_h; - int32_t stride_x; - int32_t stride_y; - int32_t pad_x; - int32_t pad_y; - int32_t dilation_x; - int32_t dilation_y; -}; + while (pos < spirv.size()) { + uint32_t opcode = spirv[pos] & spv::OpCodeMask; + uint32_t len = spirv[pos] >> spv::WordCountShift; + if (len == 0) break; -struct vk_op_upscale_push_constants { - uint32_t ne; uint32_t a_offset; uint32_t d_offset; - uint32_t ne00; uint32_t ne01; - uint32_t nb00; uint32_t nb01; uint32_t nb02; uint32_t nb03; - uint32_t ne10; uint32_t ne11; uint32_t ne12; uint32_t ne13; - float sf0; float sf1; float sf2; float sf3; - float pixel_offset; -}; + if (opcode == spv::OpCapability) { + cap_insert_pos = pos + len; + ext_insert_pos = pos + len; + } else if (opcode == spv::OpExtension) { + ext_insert_pos = pos + len; + } else if (opcode == spv::OpEntryPoint) { + entry_point_id = spirv[pos + 2]; + exec_insert_pos = pos + len; + } else if (opcode == spv::OpExecutionMode || opcode == spv::OpExecutionModeId) { + exec_insert_pos = pos + len; + } else if (entry_point_id != 0) { + break; + } -struct vk_op_sum_rows_push_constants -{ - uint32_t n_cols; - uint32_t ne01, ne02; - uint32_t nb01, nb02, nb03; - uint32_t nb11, nb12, nb13; - float weight; - uint32_t misalign_offsets; - uint32_t ne0_12mp, ne0_12L; - uint32_t ne0_1mp, ne0_1L; -}; + pos += len; + } -static vk_op_sum_rows_push_constants vk_op_sum_rows_push_constants_init(const ggml_tensor * src, const ggml_tensor * dst, int64_t n_cols) { - uint32_t type_size = (uint32_t)ggml_type_size(src->type); - vk_op_sum_rows_push_constants p = {}; - p.n_cols = (uint32_t)n_cols; - p.ne01 = (uint32_t)src->ne[1]; - p.ne02 = (uint32_t)src->ne[2]; - p.nb01 = (uint32_t)src->nb[1] / type_size; - p.nb02 = (uint32_t)src->nb[2] / type_size; - p.nb03 = (uint32_t)src->nb[3] / type_size; - p.nb11 = (uint32_t)dst->nb[1] / type_size; - p.nb12 = (uint32_t)dst->nb[2] / type_size; - p.nb13 = (uint32_t)dst->nb[3] / type_size; - p.weight = 1.0f; - return p; -} - -template <> void init_pushconst_fastdiv(vk_op_sum_rows_push_constants &p) { - init_fastdiv_values(p.ne01*p.ne02, p.ne0_12mp, p.ne0_12L); - init_fastdiv_values(p.ne01, p.ne0_1mp, p.ne0_1L); -} - -struct vk_quantize_q8_1_push_constants { - uint32_t ne; - uint32_t num_blocks; -}; + // Insert from latest position first so earlier indices stay valid. -struct vk_op_flash_attn_split_k_reduce_push_constants { - uint32_t D; - uint32_t ne1; - uint32_t ne2; - uint32_t ne3; - uint32_t k_num; - uint32_t sinks; -}; + if (device->float_controls_rte_fp16) { + // OpExecutionMode %entrypoint RoundingModeRTE 16 + uint32_t exec_mode[] = { (4u << spv::WordCountShift) | spv::OpExecutionMode, entry_point_id, spv::ExecutionModeRoundingModeRTE, 16 }; + spirv.insert(spirv.begin() + exec_insert_pos, std::begin(exec_mode), std::end(exec_mode)); + } -struct vk_op_flash_attn_mask_opt_push_constants { - uint32_t nem0; - uint32_t nem1; - uint32_t nem2; - uint32_t nbm1; - uint32_t nbm2; - uint32_t nbm3; - uint32_t nbd1; - uint32_t nbd2; - uint32_t nbd3; -}; + if (device->float_controls_denorm_preserve_fp16) { + // OpExecutionMode %entrypoint DenormPreserve 16 + uint32_t exec_mode[] = { (4u << spv::WordCountShift) | spv::OpExecutionMode, entry_point_id, spv::ExecutionModeDenormPreserve, 16 }; + spirv.insert(spirv.begin() + exec_insert_pos, std::begin(exec_mode), std::end(exec_mode)); + } -// Allow pre-recording command buffers -struct vk_staging_memcpy { - vk_staging_memcpy(void * _dst, const void * _src, size_t _n) : dst(_dst), src(_src), n(_n) {} + // OpExtension "SPV_KHR_float_controls" + const char ext_str[] = "SPV_KHR_float_controls"; + size_t ext_str_words = CEIL_DIV(sizeof(ext_str), sizeof(uint32_t)); + std::vector extension(1 + ext_str_words, 0); + extension[0] = (uint32_t)((1 + ext_str_words) << spv::WordCountShift) | spv::OpExtension; + memcpy(&extension[1], ext_str, sizeof(ext_str)); + spirv.insert(spirv.begin() + ext_insert_pos, extension.begin(), extension.end()); - void * dst; - const void * src; - size_t n; -}; + if (device->float_controls_rte_fp16) { + // OpCapability RoundingModeRTE + uint32_t capability[] = { (2u << spv::WordCountShift) | spv::OpCapability, spv::CapabilityRoundingModeRTE }; + spirv.insert(spirv.begin() + cap_insert_pos, std::begin(capability), std::end(capability)); + } -struct vk_staging_memset { - vk_staging_memset(void * _dst, uint32_t _val, size_t _n) : dst(_dst), val(_val), n(_n) {} + if (device->float_controls_denorm_preserve_fp16) { + // OpCapability DenormPreserve + uint32_t capability[] = { (2u << spv::WordCountShift) | spv::OpCapability, spv::CapabilityDenormPreserve }; + spirv.insert(spirv.begin() + cap_insert_pos, std::begin(capability), std::end(capability)); + } - void * dst; - uint32_t val; - size_t n; -}; + shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); + } -struct vk_context_struct { - vk_submission * s; - std::vector seqs; +#if defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) + if (device->coopmat2 && !device->coopmat2_decode_vector) { + const uint32_t * src = spirv.empty() ? reinterpret_cast(spv_data) : spirv.data(); + size_t src_n = spirv.empty() ? spv_size / sizeof(uint32_t) : spirv.size(); + std::vector stripped; + if (ggml_vk_strip_decode_vector(src, src_n, stripped)) { + spirv = std::move(stripped); + shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); + } + } +#endif - int exit_tensor_idx; +#if VK_HEADER_VERSION >= 287 + // Roll the mul_mm BK loop on Asahi Linux. Skip bf16 and the mul_mmq pipelines. + if (device->driver_id == vk::DriverId::eMesaHoneykrisp && + pipeline->name.rfind("matmul", 0) == 0 && + pipeline->name.find("bf16") == std::string::npos && + pipeline->name.find("q8_1") == std::string::npos) { + const uint32_t * src = spirv.empty() ? reinterpret_cast(spv_data) : spirv.data(); + size_t src_n = spirv.empty() ? spv_size / sizeof(uint32_t) : spirv.size(); + std::vector rolled; + if (ggml_vk_roll_bk_loop(src, src_n, rolled)) { + spirv = std::move(rolled); + shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); + } + } +#endif - std::vector in_memcpys; - std::vector out_memcpys; - std::vector memsets; + pipeline->shader_module = device->device.createShaderModule(shader_module_create_info); - vk_command_pool * p {}; -}; -typedef std::shared_ptr vk_context; -typedef std::weak_ptr vk_context_ref; - -struct ggml_vk_garbage_collector { - std::vector tl_semaphores; - std::vector semaphores; - std::vector events; - std::vector contexts; -}; + vk::PushConstantRange pcr( + vk::ShaderStageFlagBits::eCompute, + 0, + pipeline->push_constant_size + ); -static void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_context subctx); -static void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested = nullptr); -static void ggml_pipeline_allocate_descriptor_sets(ggml_backend_vk_context * ctx); -static bool ggml_vk_intel_windows_driver_in_range(uint32_t driver_version, uint32_t lower_major, uint32_t lower_minor, uint32_t upper_major, uint32_t upper_minor); + vk::PipelineLayoutCreateInfo pipeline_layout_create_info(vk::PipelineLayoutCreateFlags(), device->dsl, pcr); + pipeline->layout = device->device.createPipelineLayout(pipeline_layout_create_info); -static bool vk_memory_logger_enabled = false; + std::vector specialization_entries(specialization_constants.size()); -#define VK_LOG_MEMORY(msg) if (vk_memory_logger_enabled) { std::cerr << "ggml_vulkan memory: " << msg << std::endl; } + for (size_t i = 0; i < specialization_constants.size(); i++) { + specialization_entries[i].constantID = i; + specialization_entries[i].offset = i * sizeof(uint32_t); + specialization_entries[i].size = sizeof(uint32_t); + } -static std::string format_size(size_t size) { - const size_t kib = 1024; - const size_t mib = kib * 1024; - const size_t gib = mib * 1024; + vk::SpecializationInfo specialization_info( + specialization_entries.size(), + specialization_entries.data(), + specialization_constants.size() * sizeof(uint32_t), + specialization_constants.data() + ); - std::ostringstream oss; - oss << std::fixed << std::setprecision(2); + vk::PipelineShaderStageCreateFlags pipeline_shader_stage_create_flags{}; - if (size >= gib) { - oss << static_cast(size) / gib << " GiB"; - } else if (size >= mib) { - oss << static_cast(size) / mib << " MiB"; - } else if (size >= kib) { - oss << static_cast(size) / kib << " KiB"; - } else { - oss << size << " B"; + if (device->subgroup_require_full_support && require_full_subgroups) { + pipeline_shader_stage_create_flags |= vk::PipelineShaderStageCreateFlagBits::eRequireFullSubgroupsEXT; } - return oss.str(); -} - -class vk_memory_logger { -public: - vk_memory_logger(): total_device(0), total_host(0) {} - void log_allocation(vk_buffer_ref buf_ref, size_t size); - void log_deallocation(vk_buffer_ref buf_ref); + vk::PipelineShaderStageCreateInfo pipeline_shader_create_info( + pipeline_shader_stage_create_flags, + vk::ShaderStageFlagBits::eCompute, + pipeline->shader_module, + entrypoint.c_str(), + &specialization_info); -private: - std::map allocations; // Track allocations - size_t total_device; - size_t total_host; - static std::mutex log_mutex; -}; + vk::PipelineShaderStageRequiredSubgroupSizeCreateInfoEXT pipeline_shader_stage_required_subgroup_size_create_info; + pipeline_shader_stage_required_subgroup_size_create_info.requiredSubgroupSize = required_subgroup_size; + if (device->subgroup_size_control && required_subgroup_size > 0) { + GGML_ASSERT(device->subgroup_min_size <= required_subgroup_size && required_subgroup_size <= device->subgroup_max_size); + pipeline_shader_create_info.setPNext(&pipeline_shader_stage_required_subgroup_size_create_info); + } -std::mutex vk_memory_logger::log_mutex; + vk::ComputePipelineCreateInfo compute_pipeline_create_info( + device->pipeline_executable_properties_support ? + vk::PipelineCreateFlagBits::eCaptureStatisticsKHR : + vk::PipelineCreateFlags{}, + pipeline_shader_create_info, + pipeline->layout); -static bool vk_perf_logger_enabled = false; -static bool vk_perf_logger_concurrent = false; -static bool vk_enable_sync_logger = false; -// number of calls between perf logger prints -static uint32_t vk_perf_logger_frequency = 1; -static std::string vk_pipeline_stats_filter; + vk::PipelineRobustnessCreateInfoEXT rci; -static uint64_t ggml_vk_get_node_flops(const ggml_tensor * node) { - if (node->op == GGML_OP_MUL_MAT || node->op == GGML_OP_MUL_MAT_ID) { - const uint64_t m = node->ne[0]; - const uint64_t n = node->ne[1]; - const uint64_t k = node->src[1]->ne[0]; - const uint64_t batch = node->ne[2] * node->ne[3]; - return m * n * (k + (k - 1)) * batch; - } - if (node->op == GGML_OP_CONV_2D || node->op == GGML_OP_CONV_TRANSPOSE_2D) { - const ggml_tensor * knl = node->src[0]; - const uint64_t Cout = node->ne[2]; - const uint64_t size_K = node->src[1]->ne[2] * knl->ne[0] * knl->ne[1]; - const uint64_t size_N = node->ne[3] * node->ne[0] * node->ne[1]; - return Cout * size_N * (size_K + (size_K - 1)); - } - if (node->op == GGML_OP_CONV_3D) { - const ggml_tensor * knl = node->src[0]; - const uint64_t OC = ggml_get_op_params_i32(node, 11); - const uint64_t IC = ggml_get_op_params_i32(node, 9); - const uint64_t size_K = IC * knl->ne[0] * knl->ne[1] * knl->ne[2]; - const uint64_t size_N = node->ne[3] / OC * node->ne[0] * node->ne[1] * node->ne[2]; - return OC * size_N * (size_K + (size_K - 1)); - } - if (node->op == GGML_OP_FLASH_ATTN_EXT) { - const ggml_tensor * q = node->src[0]; - const ggml_tensor * k = node->src[1]; - const ggml_tensor * v = node->src[2]; - return 2ull * q->ne[1] * q->ne[2] * (k->ne[0] + v->ne[0]) * k->ne[1] * q->ne[3]; + if (device->pipeline_robustness && disable_robustness) { + rci.storageBuffers = vk::PipelineRobustnessBufferBehaviorEXT::eDisabled; + rci.uniformBuffers = vk::PipelineRobustnessBufferBehaviorEXT::eDisabled; + compute_pipeline_create_info.setPNext(&rci); } - return 0; -} -static void ggml_vk_print_node_list(const ggml_cgraph * cgraph, int start, int end) { - uint64_t total_flops = 0; - int n_ops = 0; - for (int j = start; j <= end && j < cgraph->n_nodes; j++) { - uint64_t flops = ggml_vk_get_node_flops(cgraph->nodes[j]); - total_flops += flops; - n_ops++; - if (flops > 0) { - GGML_LOG_CONT(" node %d: %s (%s) [%.2f GFLOP]\n", - j, cgraph->nodes[j]->name, ggml_op_name(cgraph->nodes[j]->op), - flops / 1e9); - } else { - GGML_LOG_CONT(" node %d: %s (%s)\n", - j, cgraph->nodes[j]->name, ggml_op_name(cgraph->nodes[j]->op)); +#if defined(VK_EXT_shader_64bit_indexing) + vk::PipelineCreateFlags2CreateInfo pipelineFlags2CreateInfo; + if (pipeline->is_64b_indexing) + { + pipelineFlags2CreateInfo.flags = vk::PipelineCreateFlagBits2::e64BitIndexingEXT; + if (device->pipeline_executable_properties_support) { + pipelineFlags2CreateInfo.flags |= vk::PipelineCreateFlagBits2::eCaptureStatisticsKHR; } + pipelineFlags2CreateInfo.setPNext(compute_pipeline_create_info.pNext); + compute_pipeline_create_info.setPNext(&pipelineFlags2CreateInfo); } - GGML_LOG_CONT(" total: %d ops, %.2f GFLOP\n", n_ops, total_flops / 1e9); -} +#endif -static void ggml_vk_print_device_lost_info(const vk_device& device) { - ggml_vk_print_device_fault_info(device); - if (device->serialize_submissions && device->diag_cgraph != nullptr && device->diag_prev_start >= 0) { - GGML_LOG_ERROR("ggml_vulkan: device lost on %s, likely caused by previous submission (nodes %d to %d):\n", - device->name.c_str(), device->diag_prev_start, device->diag_prev_end); - ggml_vk_print_node_list(device->diag_cgraph, device->diag_prev_start, device->diag_prev_end); - } else { - GGML_LOG_ERROR("ggml_vulkan: device lost on %s\n", device->name.c_str()); + try { + pipeline->pipeline = device->device.createComputePipeline(VK_NULL_HANDLE, compute_pipeline_create_info).value; + } catch (const vk::SystemError& e) { + std::cerr << "ggml_vulkan: Compute pipeline creation failed for " << pipeline->name << std::endl; + std::cerr << "ggml_vulkan: " << e.what() << std::endl; + throw e; } -} -class vk_perf_logger { - public: - void print_timings(bool force = false) { - if (timings.empty()) { - return; - } - print_count++; - if ((print_count % vk_perf_logger_frequency) != 0 && !force) { - return; - } - print_count = 0; - uint64_t total_all_op_times = 0; - std::cerr << "----------------\nVulkan Timings:" << std::endl; - for (const auto & t : timings) { - uint64_t total_op_times = 0; - for (const auto & time : t.second) { - total_op_times += time; - } - std::cerr << t.first << ": " << t.second.size() << " x " << (total_op_times / t.second.size() / 1000.0) - << " us = " << (total_op_times / 1000.0) << " us"; - - // If we have as many flops entries as timing entries for the op, then compute and log the flops/S. - auto it = flops.find(t.first); - if (it != flops.end() && (it->second).size() == t.second.size()) { - uint64_t total_op_flops = 0; - for (const auto & elem : it->second) { - total_op_flops += elem; - } - std::cerr << " (" - << (double(total_op_flops) / (1000.0 * 1000.0 * 1000.0)) / - (double(total_op_times) / (1000.0 * 1000.0 * 1000.0)) - << " GFLOPS/s)"; - } + if (vk_instance.debug_utils_support) { + vk::DebugUtilsObjectNameInfoEXT duoni; + duoni.objectType = vk::ObjectType::ePipeline; + duoni.pObjectName = pipeline->name.c_str(); + duoni.objectHandle = /*reinterpret_cast*/(uint64_t)(static_cast(pipeline->pipeline)); + vk_instance.pfn_vkSetDebugUtilsObjectNameEXT(device->device, &static_cast(duoni)); + } - total_all_op_times += total_op_times; + if (device->pipeline_executable_properties_support) { + vk::PipelineExecutableInfoKHR executableInfo; + executableInfo.pipeline = pipeline->pipeline; - std::cerr << std::endl; - } + auto statistics = device->device.getPipelineExecutableStatisticsKHR(executableInfo); - if (timings.size() > 0) { - std::cerr << "Total time: " << total_all_op_times / 1000.0 << " us." << std::endl; + bool print_stats = !vk_pipeline_stats_filter.empty() && + pipeline->name.find(vk_pipeline_stats_filter) != std::string::npos; + if (print_stats) { + std::cerr << "ggml_vulkan: pipeline stats for " << pipeline->name << ":" << std::endl; } - timings.clear(); - flops.clear(); - } - - std::string get_node_fusion_name(const ggml_tensor * node, const char *fusion_name, uint64_t *n_flops) { - *n_flops = ggml_vk_get_node_flops(node); - std::string fusion_str; - if (fusion_name) { - fusion_str = fusion_name + std::string(" "); - } - if (node->op == GGML_OP_UNARY) { - return fusion_str + ggml_unary_op_name(ggml_get_unary_op(node)); - } - if (node->op == GGML_OP_MUL_MAT || node->op == GGML_OP_MUL_MAT_ID) { - const uint64_t m = node->ne[0]; - const uint64_t n = node->ne[1]; - const uint64_t k = node->src[1]->ne[0]; - const uint64_t batch = node->ne[2] * node->ne[3]; - std::string name = ggml_op_name(node->op); - if ((node->op == GGML_OP_MUL_MAT && n <= mul_mat_vec_max_cols) || - (node->op == GGML_OP_MUL_MAT_ID && node->src[2]->ne[1] == 1)) { - name += "_VEC"; - } - name += " "; - name += ggml_type_name(node->src[0]->type); - name += " m=" + std::to_string(m) + " n=" + std::to_string(n) + " k=" + std::to_string(k); - if (node->op == GGML_OP_MUL_MAT_ID) { - name += " n_expert=" + std::to_string(node->src[0]->ne[2]); - } - if (batch > 1) { - name += " batch=" + std::to_string(batch); + for (auto & s : statistics) { + if (print_stats) { + std::cerr << "ggml_vulkan: " << s.name.data() << ": "; + switch (s.format) { + case vk::PipelineExecutableStatisticFormatKHR::eBool32: + std::cerr << (s.value.b32 ? "true" : "false"); + break; + case vk::PipelineExecutableStatisticFormatKHR::eInt64: + std::cerr << s.value.i64; + break; + case vk::PipelineExecutableStatisticFormatKHR::eUint64: + std::cerr << s.value.u64; + break; + case vk::PipelineExecutableStatisticFormatKHR::eFloat64: + std::cerr << s.value.f64; + break; + } + std::cerr << std::endl; } - return fusion_str + name; - } - if (node->op == GGML_OP_CONV_2D || node->op == GGML_OP_CONV_TRANSPOSE_2D) { - std::string name = ggml_op_name(node->op); - const ggml_tensor * knl = node->src[0]; - uint64_t Cout = node->ne[2]; - uint64_t size_K = node->src[1]->ne[2] * knl->ne[0] * knl->ne[1]; - uint64_t size_N = node->ne[3] * node->ne[0] * node->ne[1]; - name += " M=Cout=" + std::to_string(Cout) + ", K=Cin*KW*KH=" + std::to_string(size_K) + - ", N=N*OW*OH=" + std::to_string(size_N); - return fusion_str + name; - } - if (node->op == GGML_OP_RMS_NORM) { - std::string name = ggml_op_name(node->op); - name += "(" + std::to_string(node->ne[0]) + "," + std::to_string(node->ne[1]) + "," + std::to_string(node->ne[2]) + "," + std::to_string(node->ne[3]) + ")"; - return fusion_str + name; - } - if (node->op == GGML_OP_FLASH_ATTN_EXT) { - const ggml_tensor * dst = node; - const ggml_tensor * q = node->src[0]; - const ggml_tensor * k = node->src[1]; - const ggml_tensor * v = node->src[2]; - const ggml_tensor * m = node->src[3]; - std::stringstream name; - name << fusion_str; - name << ggml_op_name(node->op) << - " dst(" << dst->ne[0] << "," << dst->ne[1] << "," << dst->ne[2] << "," << dst->ne[3] << "), " << - " q(" << q->ne[0] << "," << q->ne[1] << "," << q->ne[2] << "," << q->ne[3] << "), " << - " k(" << k->ne[0] << "," << k->ne[1] << "," << k->ne[2] << "," << k->ne[3] << "), " << - " v(" << v->ne[0] << "," << v->ne[1] << "," << v->ne[2] << "," << v->ne[3] << "), " << - " m(" << (m?m->ne[0]:0) << "," << (m?m->ne[1]:0) << "," << (m?m->ne[2]:0) << "," << (m?m->ne[3]:0) << ")"; - return name.str(); - } - if (node->op == GGML_OP_TOP_K) { - std::stringstream name; - name << fusion_str; - name << ggml_op_name(node->op) << - " K=" << node->ne[0] << - " (" << node->src[0]->ne[0] << "," << node->src[0]->ne[1] << "," << node->src[0]->ne[2] << "," << node->src[0]->ne[3] << ")"; - return name.str(); - } - return fusion_str + ggml_op_name(node->op); - } - - void log_timing(const ggml_tensor * node, const char *fusion_name, uint64_t time) { - uint64_t n_flops; - std::string name = get_node_fusion_name(node, fusion_name, &n_flops); - if (n_flops) { - flops[name].push_back(n_flops); - } - timings[name].push_back(time); - } - - void log_timing(const std::vector &nodes, const std::vector &names, uint64_t time) { - uint64_t total_flops = 0; - std::string name; - for (size_t n = 0; n < nodes.size(); ++n) { - uint64_t n_flops = 0; - name += get_node_fusion_name(nodes[n], names[n], &n_flops); - total_flops += n_flops; - - if (n != nodes.size() - 1) { - name += ", "; + // "Register Count" is reported by NVIDIA drivers. + if (strcmp(s.name, "Register Count") == 0) { + VK_LOG_DEBUG(pipeline->name << " " << s.name << ": " << s.value.u64 << " registers"); + pipeline->register_count = (uint32_t)s.value.u64; } } - if (total_flops) { - flops[name].push_back(total_flops); - } - timings[name].push_back(time); } - private: - std::map> timings; - std::map> flops; - uint32_t print_count {}; -}; + { + std::lock_guard guard(device->compile_mutex); + device->all_pipelines.push_back(pipeline); + pipeline->compiled = true; + pipeline->compile_pending = false; + } + device->compile_cv.notify_all(); +} -struct ggml_backend_vk_context { - std::string name; - - vk_device device; - - size_t semaphore_idx, event_idx; - ggml_vk_garbage_collector gc; - size_t prealloc_size_x, prealloc_size_y, prealloc_size_split_k, prealloc_size_add_rms_partials, prealloc_size_add_rms_partials_offset; - vk_buffer prealloc_x, prealloc_y, prealloc_split_k, prealloc_add_rms_partials, sync_staging; - vk::Fence fence, almost_ready_fence; - bool submit_pending {}; - bool almost_ready_fence_pending {}; - // Set before op_add and unset after op_rms_norm to indicate that the add should - // write partial sums to accumulate the square of the vector components - bool do_add_rms_partials_offset_calculation; - bool do_add_rms_partials; - - uint64_t last_total_flops {UINT64_MAX}; - - // Cache most recent tensor that was converted into prealloc_y, and what pipeline it used to convert. - vk_pipeline_struct * prealloc_y_last_pipeline_used {}; - const ggml_tensor * prealloc_y_last_tensor_used {}; - // True when prealloc_y holds the padded fp16 layout used by the coopmat2 B decode-vector callback. - // If false, then it's contiguous. - bool prealloc_y_last_decode_vector_staging {}; - - // Track which nodes have been used since the last sync, and whether they were written to - std::vector unsynced_nodes_written; - std::vector unsynced_nodes_read; - // Track which prealloc buffers have pending reads that need to be synchronized. - // These are checked before writing to the buffer (and call ggml_vk_sync_buffers if set), - // and set to true after the buffer contents are consumed. - bool prealloc_x_need_sync, prealloc_y_need_sync, prealloc_split_k_need_sync; - - vk_context_ref compute_ctx; - - vk_context_ref transfer_ctx; - vk_semaphore transfer_semaphore; - uint64_t transfer_semaphore_last_submitted {}; - - std::vector tensor_ctxs; - - std::vector descriptor_pools; - std::vector descriptor_sets; - uint32_t descriptor_set_idx {}; - uint32_t pipeline_descriptor_set_requirements {}; - - vk_command_pool compute_cmd_pool; - vk_command_pool transfer_cmd_pool; - - // number of additional consecutive nodes that are being fused with the - // node currently being processed - int num_additional_fused_ops {}; - // Bitmask of which fused ops need to write an intermediate value to memory. - // Bit 'i' means nodes[start_of_fusion + i] writes to memory. - // If there's no fusion, bit 0 is still set. - int fused_ops_write_mask {}; - topk_moe_mode fused_topk_moe_mode {}; - bool fused_topk_moe_scale {}; - - // for GGML_VK_PERF_LOGGER - std::unique_ptr perf_logger; - vk::QueryPool query_pool; - std::vector query_fusion_names; - std::vector query_fusion_node_count; - std::vector query_nodes; - std::vector query_node_idx; - int32_t num_queries {}; - int32_t query_idx {}; -}; +void ggml_vk_destroy_pipeline(vk::Device& device, vk_pipeline& pipeline) { + VK_LOG_DEBUG("ggml_pipeline_destroy_pipeline(" << pipeline->name << ")"); + device.destroyPipelineLayout(pipeline->layout); -static void * const vk_ptr_base = (void *)(uintptr_t) 0x1000; // NOLINT + device.destroyShaderModule(pipeline->shader_module); -static uint64_t vk_tensor_offset(const ggml_tensor * tensor) { - if (tensor->view_src) { - return (uint8_t *) tensor->view_src->data - (uint8_t *) vk_ptr_base; - } - return (uint8_t *) tensor->data - (uint8_t *) vk_ptr_base; + device.destroyPipeline(pipeline->pipeline); } -static uint32_t get_misalign_bytes(const ggml_backend_vk_context * ctx, const ggml_tensor * t) -{ - return ((vk_tensor_offset(t) + t->view_offs) & (ctx->device->properties.limits.minStorageBufferOffsetAlignment - 1));; +void ggml_pipeline_request_descriptor_sets(ggml_backend_vk_context *ctx, vk_pipeline& pipeline, uint32_t n) { + VK_LOG_DEBUG("ggml_pipeline_request_descriptor_sets(" << pipeline->name << ", " << n << ")"); + ctx->pipeline_descriptor_set_requirements += n; + if (!pipeline->compiled) { + ggml_vk_load_shaders(ctx->device, pipeline); + } + ggml_pipeline_allocate_descriptor_sets(ctx); } -static uint32_t ggml_vk_concat_unit_size(ggml_type type) { - const uint32_t type_size = ggml_type_size(type); +void ggml_pipeline_allocate_descriptor_sets(ggml_backend_vk_context * ctx) { - if (!ggml_is_quantized(type)) { - return type_size; + if (ctx->descriptor_sets.size() >= ctx->pipeline_descriptor_set_requirements) { + // Enough descriptors are available + return; } - // Use the widest existing concat shader that evenly divides a quant block. - if (type_size % 8 == 0) { - return 8; - } - if (type_size % 4 == 0) { - return 4; - } - if (type_size % 2 == 0) { - return 2; - } - return 1; -} + vk_device& device = ctx->device; -static bool ggml_vk_concat_supported(const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { - if (src0->type != src1->type || src0->type != dst->type) { - return false; - } + // Grow by 50% to avoid frequent allocations + uint32_t needed = std::max(3 * ctx->descriptor_sets.size() / 2, size_t{ctx->pipeline_descriptor_set_requirements}); + uint32_t to_alloc = needed - ctx->descriptor_sets.size(); + uint32_t pool_remaining = VK_DEVICE_DESCRIPTOR_POOL_SIZE - ctx->descriptor_sets.size() % VK_DEVICE_DESCRIPTOR_POOL_SIZE; + uint32_t pool_idx = ctx->descriptor_sets.size() / VK_DEVICE_DESCRIPTOR_POOL_SIZE; - if (!ggml_is_quantized(src0->type)) { - const size_t type_size = ggml_type_size(src0->type); - return type_size == 1 || type_size == 2 || type_size == 4 || type_size == 8; - } + while (to_alloc > 0) { + const uint32_t alloc_count = std::min(pool_remaining, to_alloc); + to_alloc -= alloc_count; + pool_remaining = VK_DEVICE_DESCRIPTOR_POOL_SIZE; - // Quantized tensor rows are block-aligned when created. - return ggml_is_contiguous_rows(src0) && ggml_is_contiguous_rows(src1) && ggml_is_contiguous_rows(dst); + if (pool_idx >= ctx->descriptor_pools.size()) { + vk::DescriptorPoolSize descriptor_pool_size(vk::DescriptorType::eStorageBuffer, MAX_PARAMETER_COUNT * VK_DEVICE_DESCRIPTOR_POOL_SIZE); + vk::DescriptorPoolCreateInfo descriptor_pool_create_info({}, VK_DEVICE_DESCRIPTOR_POOL_SIZE, descriptor_pool_size); + ctx->descriptor_pools.push_back(device->device.createDescriptorPool(descriptor_pool_create_info)); + } + + std::vector layouts(alloc_count); + for (uint32_t i = 0; i < alloc_count; i++) { + layouts[i] = device->dsl; + } + vk::DescriptorSetAllocateInfo descriptor_set_alloc_info(ctx->descriptor_pools[pool_idx], alloc_count, layouts.data()); + std::vector sets = device->device.allocateDescriptorSets(descriptor_set_alloc_info); + ctx->descriptor_sets.insert(ctx->descriptor_sets.end(), sets.begin(), sets.end()); + + pool_idx++; + } } -template void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, T &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - GGML_UNUSED(p); - GGML_UNUSED(src0); - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); - GGML_UNUSED(dst); - static_assert(!std::is_const::value, "unexpected type"); - GGML_ASSERT(!src0 || get_misalign_bytes(ctx, src0) == 0); - GGML_ASSERT(!src1 || get_misalign_bytes(ctx, src1) == 0); - GGML_ASSERT(!src2 || get_misalign_bytes(ctx, src2) == 0); - GGML_ASSERT(!src3 || get_misalign_bytes(ctx, src3) == 0); - GGML_ASSERT(!dst || get_misalign_bytes(ctx, dst) == 0); +static vk_command_buffer* ggml_vk_create_cmd_buffer(vk_device& device, vk_command_pool& p) { + VK_LOG_DEBUG("ggml_vk_create_cmd_buffer()"); + vk::CommandBufferAllocateInfo command_buffer_alloc_info( + p.pool, + vk::CommandBufferLevel::ePrimary, + 1); + const std::vector cmd_buffers = device->device.allocateCommandBuffers(command_buffer_alloc_info); + p.cmd_buffers.push_back({ cmd_buffers.front(), 0, true }); + return &p.cmd_buffers[p.cmd_buffers.size()-1]; } -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_mat_vec_p021_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); +void ggml_vk_submit(vk_context& ctx, vk::Fence fence) { + if (ctx->seqs.empty()) { + if (fence) { + ctx->p->q->handle->submit({}, fence); + } + return; + } + VK_LOG_DEBUG("ggml_vk_submit(" << ctx << ", " << fence << ")"); - p.b_offset = b_offset; - p.d_offset = d_offset; + std::vector> tl_wait_vals; + std::vector> tl_signal_vals; + std::vector> tl_wait_semaphores; + std::vector> tl_signal_semaphores; + std::vector tl_submit_infos; + std::vector submit_infos; + int idx = -1; + std::vector> stage_flags; - GGML_UNUSED(src0); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} + size_t reserve = 0; -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_mat_vec_nc_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + for (const auto& sequence : ctx->seqs) { + reserve += sequence.size(); + } - p.b_offset = b_offset; - p.d_offset = d_offset; + // Pre-reserve vectors to prevent reallocation, which invalidates pointers + tl_wait_semaphores.reserve(reserve); + tl_wait_vals.reserve(reserve); + tl_signal_semaphores.reserve(reserve); + tl_signal_vals.reserve(reserve); + tl_submit_infos.reserve(reserve); + submit_infos.reserve(reserve); + stage_flags.reserve(reserve); - GGML_UNUSED(src0); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} + for (const auto& sequence : ctx->seqs) { + for (const auto& submission : sequence) { + stage_flags.push_back({}); + idx++; + tl_wait_vals.push_back({}); + tl_wait_semaphores.push_back({}); + tl_signal_vals.push_back({}); + tl_signal_semaphores.push_back({}); + for (size_t i = 0; i < submission.wait_semaphores.size(); i++) { + stage_flags[idx].push_back(ctx->p->q->stage_flags); + tl_wait_vals[idx].push_back(submission.wait_semaphores[i].value); + tl_wait_semaphores[idx].push_back(submission.wait_semaphores[i].s); + } + for (size_t i = 0; i < submission.signal_semaphores.size(); i++) { + tl_signal_vals[idx].push_back(submission.signal_semaphores[i].value); + tl_signal_semaphores[idx].push_back(submission.signal_semaphores[i].s); + } + tl_submit_infos.push_back({ + (uint32_t) submission.wait_semaphores.size(), + tl_wait_vals[idx].data(), + (uint32_t) submission.signal_semaphores.size(), + tl_signal_vals[idx].data(), + }); + tl_submit_infos[idx].sType = vk::StructureType::eTimelineSemaphoreSubmitInfo; + tl_submit_infos[idx].pNext = nullptr; + vk::SubmitInfo si{ + (uint32_t) submission.wait_semaphores.size(), + tl_wait_semaphores[idx].data(), + stage_flags[idx].data(), + 1, + &submission.buffer->buf, + (uint32_t) submission.signal_semaphores.size(), + tl_signal_semaphores[idx].data(), + }; + si.setPNext(&tl_submit_infos[idx]); + submit_infos.push_back(si); + } + } -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_fwht_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - p.src_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - p.dst_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); + ctx->p->q->handle->submit(submit_infos, fence); - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); + ctx->seqs.clear(); } -struct ggml_backend_vk_buffer_context { - vk_device_ref device; - vk_buffer dev_buffer; - std::string name; +uint32_t ggml_vk_find_queue_family_index(std::vector& queue_family_props, const vk::QueueFlags& required, const vk::QueueFlags& avoid, int32_t compute_index, uint32_t min_num_queues) { + VK_LOG_DEBUG("ggml_vk_find_queue_family_index()"); + const uint32_t qfsize = queue_family_props.size(); - ggml_backend_vk_buffer_context(vk_device_ref device, vk_buffer&& dev_buffer, std::string& name) : - device(device), - dev_buffer(dev_buffer), - name(name) { + // Try with avoid preferences first + for (uint32_t i = 0; i < qfsize; i++) { + if (queue_family_props[i].queueCount >= min_num_queues && (compute_index < 0 || i != (uint32_t) compute_index) && queue_family_props[i].queueFlags & required && !(queue_family_props[i].queueFlags & avoid)) { + return i; + } } - ~ggml_backend_vk_buffer_context() { - ggml_vk_destroy_buffer(dev_buffer); + // Fall back to only required + for (size_t i = 0; i < qfsize; i++) { + if (queue_family_props[i].queueCount >= min_num_queues && (compute_index < 0 || i != (uint32_t) compute_index) && queue_family_props[i].queueFlags & required) { + return i; + } } -}; -void vk_memory_logger::log_allocation(vk_buffer_ref buf_ref, size_t size) { - if (!vk_memory_logger_enabled) { - return; + // Fall back to reusing compute queue + for (size_t i = 0; i < qfsize; i++) { + if (queue_family_props[i].queueCount >= min_num_queues && queue_family_props[i].queueFlags & required) { + return i; + } } - std::lock_guard guard(log_mutex); - vk_buffer buf = buf_ref.lock(); - const bool device = bool(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eDeviceLocal); - const std::string type = device ? "device" : "host"; - allocations[buf->buffer] = size; - total_device += device ? size : 0; - total_host += device ? 0 : size; - VK_LOG_MEMORY(buf->device->name << ": +" << format_size(size) << " " << type << " at " << buf->buffer << ". Total device: " << format_size(total_device) << ", total host: " << format_size(total_host)); -} -void vk_memory_logger::log_deallocation(vk_buffer_ref buf_ref) { - if (buf_ref.expired() || buf_ref.lock()->size == 0 || !vk_memory_logger_enabled) { - return; + // Fall back to ignoring min_num_queries + for (size_t i = 0; i < qfsize; i++) { + if (queue_family_props[i].queueFlags & required) { + return i; + } } - std::lock_guard guard(log_mutex); - vk_buffer buf = buf_ref.lock(); - const bool device = bool(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eDeviceLocal); - std::string type = device ? "device" : "host"; - auto it = allocations.find(buf->buffer); - if (it != allocations.end()) { - total_device -= device ? it->second : 0; - total_host -= device ? 0 : it->second; - VK_LOG_MEMORY(buf->device->name << ": -" << format_size(it->second) << " " << type << " at " << buf->buffer << ". Total device: " << format_size(total_device) << ", total host: " << format_size(total_host)); - allocations.erase(it); - } else { - VK_LOG_MEMORY("ERROR " << buf->device->name << ": Attempted to deallocate unknown " << type << " memory at " << buf->buffer); + // All commands that are allowed on a queue that supports transfer operations are also allowed on a queue that supports either graphics or compute operations. + // Thus, if the capabilities of a queue family include VK_QUEUE_GRAPHICS_BIT or VK_QUEUE_COMPUTE_BIT, then reporting the VK_QUEUE_TRANSFER_BIT capability separately for that queue family is optional. + if (compute_index >= 0) { + return compute_index; } -} -struct vk_instance_t { - vk::Instance instance; + std::cerr << "ggml_vulkan: No suitable queue family index found." << std::endl; - bool debug_utils_support = false; // VK_EXT_debug_utils enabled - PFN_vkSetDebugUtilsObjectNameEXT pfn_vkSetDebugUtilsObjectNameEXT = {}; - PFN_vkQueueBeginDebugUtilsLabelEXT pfn_vkQueueBeginDebugUtilsLabelEXT = {}; - PFN_vkQueueEndDebugUtilsLabelEXT pfn_vkQueueEndDebugUtilsLabelEXT = {}; - PFN_vkCmdBeginDebugUtilsLabelEXT pfn_vkCmdBeginDebugUtilsLabelEXT = {}; - PFN_vkCmdEndDebugUtilsLabelEXT pfn_vkCmdEndDebugUtilsLabelEXT = {}; - PFN_vkCmdInsertDebugUtilsLabelEXT pfn_vkCmdInsertDebugUtilsLabelEXT = {}; + for(auto &q_family : queue_family_props) { + std::cerr << "Queue number: " + std::to_string(q_family.queueCount) << " flags: " + to_string(q_family.queueFlags) << std::endl; + } + abort(); +} - std::vector device_indices; - std::vector device_supports_membudget; - vk_device devices[GGML_VK_MAX_DEVICES]; -}; +std::unique_ptr ggml_vk_create_queue(vk_device& device, uint32_t queue_family_index, uint32_t queue_index, vk::PipelineStageFlags&& stage_flags, bool transfer_only) { + VK_LOG_DEBUG("ggml_vk_create_queue()"); + std::lock_guard guard(device->mutex); -static bool vk_instance_initialized = false; -static vk_instance_t vk_instance; + auto q = std::make_unique(); + q->queue_family_index = queue_family_index; + q->transfer_only = transfer_only; -#ifdef GGML_VULKAN_CHECK_RESULTS -static size_t vk_skip_checks; -static size_t vk_output_tensor; + std::shared_ptr h; + vk::DeviceQueueInfo2 queue_info2{}; + queue_info2.queueFamilyIndex = queue_family_index; + queue_info2.queueIndex = queue_index; -static void ggml_vk_print_tensor(const ggml_tensor * tensor, const char * name); -static void ggml_vk_check_results_0(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx); -static void ggml_vk_check_results_1(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx); -#endif + if (device->has_internally_synchronized_queues) { + h = std::make_shared(); + queue_info2.flags = eInternallySynchronizedKHR; + } else { + h = std::make_shared(); + } -typedef void (*ggml_vk_func_t)(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst); + h->queue = device->device.getQueue2(queue_info2); + h->device = device; + // Avoid concurrent submissions on NVIDIA due to driver bug. + if (device->vendor_id == VK_VENDOR_ID_NVIDIA) { + h->device_submit_mutex = &device->queue_submit_mutex; + } + q->handle = h; -static void ggml_backend_vk_free(ggml_backend_t backend); + q->cmd_pool.init(device, q.get()); -static VkDeviceSize ggml_vk_get_max_buffer_range(const ggml_backend_vk_context * ctx, const vk_buffer &buf, const VkDeviceSize offset) { - const VkDeviceSize range = std::min(VkDeviceSize{buf->size - offset}, - VkDeviceSize{ctx->device->properties.limits.maxStorageBufferRange}); - return range; + q->stage_flags = stage_flags; + return q; } -// Wait for ctx->fence to be signaled. -static void ggml_vk_wait_for_fence(ggml_backend_vk_context * ctx) { - // Use waitForFences while most of the graph executes. Hopefully the CPU can sleep - // during this wait. - if (ctx->almost_ready_fence_pending) { - VK_CHECK(ctx->device->device.waitForFences({ ctx->almost_ready_fence }, true, UINT64_MAX), "almost_ready_fence", ctx->device); - ctx->device->device.resetFences({ ctx->almost_ready_fence }); - ctx->almost_ready_fence_pending = false; - } +std::unique_ptr ggml_vk_create_aliased_queue(vk_device& device, const std::unique_ptr& source) { + std::lock_guard guard(device->mutex); + auto q = std::make_unique(); + q->handle = source->handle; + q->queue_family_index = source->queue_family_index; + q->stage_flags = source->stage_flags; + q->transfer_only = source->transfer_only; + q->cmd_pool.init(device, q.get()); + return q; +} - // Spin (w/pause) waiting for the graph to finish executing. - vk::Result result; - for (;;) { - try { - result = ctx->device->device.getFenceStatus(ctx->fence); - } catch (vk::DeviceLostError &) { - ggml_vk_print_device_lost_info(ctx->device); - GGML_LOG_ERROR("ggml_vulkan: getFenceStatus at %s:%d\n", __FILE__, __LINE__); - throw; - } - if (result == vk::Result::eSuccess) { - break; - } - if (result != vk::Result::eNotReady) { - GGML_LOG_ERROR("ggml_vulkan: error %s at %s:%d\n", to_string(result).c_str(), __FILE__, __LINE__); - throw vk::SystemError(vk::make_error_code(result), "ggml_vulkan: getFenceStatus"); - } - for (uint32_t i = 0; i < 100; ++i) { - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - YIELD(); - } - } - ctx->device->device.resetFences({ ctx->fence }); +vk_context ggml_vk_create_context(ggml_backend_vk_context * ctx, vk_command_pool& p) { + vk_context result = std::make_shared(); + VK_LOG_DEBUG("ggml_vk_create_context(" << result << ")"); + ctx->gc.contexts.emplace_back(result); + result->p = &p; + return result; } -static constexpr uint32_t kSpvOpCooperativeMatrixLoadTensorNV = 5367; -static constexpr uint32_t kSpvCapabilityCooperativeMatrixDecodeVectorNV = 5447; -static constexpr uint32_t kSpvTensorAddressingDecodeVectorFuncBit = 0x4; +vk_context ggml_vk_create_temporary_context(vk_command_pool& p) { + vk_context result = std::make_shared(); + VK_LOG_DEBUG("ggml_vk_create_temporary_context(" << result << ")"); + result->p = &p; + return result; +} -// Remove SPV_NV_cooperative_matrix_decode_vector usage from a SPIR-V module so it -// can be loaded on drivers that only support SPV_NV_cooperative_matrix2. Drops the -// OpExtension declaration, the CooperativeMatrixDecodeVectorNV OpCapability, and the -// DecodeVectorFunc operand from any OpCooperativeMatrixLoadTensorNV instruction. -// Returns true when the input used the extension (and `out` was populated with a -// stripped copy); returns false otherwise without touching `out`. -static bool ggml_vk_strip_decode_vector(const uint32_t * code, size_t word_count, std::vector & out) { - static const char kDecodeVectorExt[] = "SPV_NV_cooperative_matrix_decode_vector"; +static vk_semaphore * ggml_vk_create_binary_semaphore(ggml_backend_vk_context * ctx) { + VK_LOG_DEBUG("ggml_vk_create_timeline_semaphore()"); + vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eBinary, 0 }; + vk::SemaphoreCreateInfo ci{}; + ci.setPNext(&tci); + vk::Semaphore semaphore = ctx->device->device.createSemaphore(ci); + ctx->gc.semaphores.push_back({ semaphore, 0 }); + return &ctx->gc.semaphores[ctx->gc.semaphores.size() - 1]; +} - if (word_count < 5) { - return false; +static vk_semaphore * ggml_vk_create_timeline_semaphore(ggml_backend_vk_context * ctx) { + VK_LOG_DEBUG("ggml_vk_create_timeline_semaphore()"); + if (ctx->semaphore_idx >= ctx->gc.tl_semaphores.size()) { + vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eTimeline, 0 }; + vk::SemaphoreCreateInfo ci{}; + ci.setPNext(&tci); + vk::Semaphore semaphore = ctx->device->device.createSemaphore(ci); + ctx->gc.tl_semaphores.push_back({ semaphore, 0 }); } + return &ctx->gc.tl_semaphores[ctx->semaphore_idx++]; +} - bool uses_decode_vector = false; - for (size_t pos = 5; pos < word_count; ) { - uint32_t word = code[pos]; - uint32_t wc = word >> spv::WordCountShift; - uint32_t op = word & spv::OpCodeMask; - GGML_ASSERT(wc > 0 && pos + wc <= word_count); - if (op == spv::OpExtension && wc >= 2) { - const char * s = reinterpret_cast(&code[pos + 1]); - if (strcmp(s, kDecodeVectorExt) == 0) { - uses_decode_vector = true; - break; - } - } - pos += wc; +static vk::Event ggml_vk_create_event(ggml_backend_vk_context * ctx) { + if (ctx->event_idx >= ctx->gc.events.size()) { + ctx->gc.events.push_back(ctx->device->device.createEvent({})); } + return ctx->gc.events[ctx->event_idx++]; +} - if (!uses_decode_vector) { - return false; +void ggml_vk_command_pool_cleanup(vk_device& device, vk_command_pool& p) { + VK_LOG_DEBUG("ggml_vk_command_pool_cleanup()"); + + // Requires command buffers to be done + device->device.resetCommandPool(p.pool); + // Don't clear the command buffers and mark them as not in use. + // This allows us to reuse them + for (auto& cmd_buffer : p.cmd_buffers) { + cmd_buffer.in_use = false; } +} - VK_LOG_DEBUG("ggml_vk_strip_decode_vector: stripping SPV_NV_cooperative_matrix_decode_vector"); +void ggml_vk_queue_command_pools_cleanup(vk_device& device) { + VK_LOG_DEBUG("ggml_vk_queue_command_pools_cleanup()"); - // Bulk-copy unchanged runs and only break the run when an instruction needs to - // be dropped or patched. Use reserve + insert/push_back so the destination buffer - // is touched exactly once (no zero-initialization pass from resize()). - out.clear(); - out.reserve(word_count); + // Arbitrary frequency to cleanup/reuse command buffers + static constexpr uint32_t cleanup_frequency = 10; - size_t run_start = 0; - auto flush_run = [&](size_t up_to) { - if (up_to > run_start) { - out.insert(out.end(), code + run_start, code + up_to); - } - }; + if (device->compute_queue && device->compute_queue->cmd_pool.buffers_in_use() >= cleanup_frequency) { + ggml_vk_command_pool_cleanup(device, device->compute_queue->cmd_pool); + } + if (device->transfer_queue && device->transfer_queue->cmd_pool.buffers_in_use() >= cleanup_frequency) { + ggml_vk_command_pool_cleanup(device, device->transfer_queue->cmd_pool); + } +} - for (size_t pos = 5; pos < word_count; ) { - uint32_t word = code[pos]; - uint32_t wc = word >> spv::WordCountShift; - uint32_t op = word & spv::OpCodeMask; - GGML_ASSERT(wc > 0 && pos + wc <= word_count); +vk_subbuffer ggml_vk_subbuffer(const ggml_backend_vk_context* ctx, const vk_buffer& buf, size_t offset) { + return { buf, offset, ggml_vk_get_max_buffer_range(ctx, buf, offset) }; +} - if (op == spv::OpExtension && wc >= 2) { - const char * s = reinterpret_cast(&code[pos + 1]); - if (strcmp(s, kDecodeVectorExt) == 0) { - flush_run(pos); - pos += wc; - run_start = pos; - continue; - } - } +void ggml_vk_sync_buffers(ggml_backend_vk_context* ctx, vk_context& subctx) { + VK_LOG_DEBUG("ggml_vk_sync_buffers()"); - if (op == spv::OpCapability && wc == 2 && code[pos + 1] == kSpvCapabilityCooperativeMatrixDecodeVectorNV) { - flush_run(pos); - pos += wc; - run_start = pos; - continue; - } + const bool transfer_queue = subctx->p->q->transfer_only; - if (op == kSpvOpCooperativeMatrixLoadTensorNV) { - // [opcode/wc][ResultType][Result][Pointer][Object][TensorLayout][MemOperand mask][mem extras...][TA mask][ta extras...] - GGML_ASSERT(wc >= 8); + if (ctx) { + ctx->prealloc_x_need_sync = ctx->prealloc_y_need_sync = ctx->prealloc_split_k_need_sync = false; + } - uint32_t mem_mask = code[pos + 6]; - size_t cur = pos + 7; - // Each of these MemoryAccess bits (when set) carries one trailing operand. - cur += (mem_mask & 0x2) ? 1 : 0; // Aligned - cur += (mem_mask & 0x8) ? 1 : 0; // MakePointerAvailable - cur += (mem_mask & 0x10) ? 1 : 0; // MakePointerVisible - cur += (mem_mask & 0x10000) ? 1 : 0; // AliasScopeINTELMask - cur += (mem_mask & 0x20000) ? 1 : 0; // NoAliasINTELMask - GGML_ASSERT(cur < pos + wc); + subctx->s->buffer->buf.pipelineBarrier( + subctx->p->q->stage_flags, + subctx->p->q->stage_flags, + {}, + { { + { !transfer_queue ? (vk::AccessFlagBits::eShaderRead | vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) : (vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) }, + { !transfer_queue ? (vk::AccessFlagBits::eShaderRead | vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) : (vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) } + } }, + {}, + {} + ); +} - uint32_t ta_mask = code[cur]; - if ((ta_mask & kSpvTensorAddressingDecodeVectorFuncBit) == 0) { - pos += wc; - continue; // leave instruction inside the current unchanged run - } +static void ggml_vk_reset_event(vk_context& ctx, vk::Event& event) { + VK_LOG_DEBUG("ggml_vk_set_event()"); - flush_run(pos); + ctx->s->buffer->buf.resetEvent( + event, + ctx->p->q->stage_flags + ); +} - // Append unchanged prefix of the instruction (header through the mem-extras). - size_t inst_start = out.size(); - size_t pre_n = cur - pos; - out.insert(out.end(), code + pos, code + pos + pre_n); +void ggml_vk_set_event(vk_context& ctx, vk::Event& event) { + VK_LOG_DEBUG("ggml_vk_set_event()"); - // Emit TA mask with the DecodeVectorFunc bit cleared. - out.push_back(ta_mask & ~kSpvTensorAddressingDecodeVectorFuncBit); + ctx->s->buffer->buf.setEvent( + event, + ctx->p->q->stage_flags + ); +} - // TA extras: TensorView (0x1) and DecodeFunc (0x2) are kept verbatim; - // DecodeVectorFunc (0x4) is dropped along with its trailing id operand. - size_t keep_ta_extras = ((ta_mask & 0x1) ? 1 : 0) + ((ta_mask & 0x2) ? 1 : 0); - if (keep_ta_extras) { - out.insert(out.end(), code + cur + 1, code + cur + 1 + keep_ta_extras); - } +void ggml_vk_wait_events(vk_context& ctx, std::vector&& events) { + VK_LOG_DEBUG("ggml_vk_wait_events()"); + if (events.empty()) { + return; + } - GGML_ASSERT(wc == pre_n + 1 + keep_ta_extras + 1); + ctx->s->buffer->buf.waitEvents( + events, + ctx->p->q->stage_flags, + ctx->p->q->stage_flags, + {}, + {}, + {} + ); +} - // Patch the instruction header with the new (one-shorter) word count. - uint32_t new_wc = wc - 1; - out[inst_start] = (new_wc << spv::WordCountShift) | op; +static vk_fa_tuning_params get_fa_tuning_params_scalar(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { - pos += wc; - run_start = pos; - continue; - } + vk_fa_tuning_params result{}; + result.path = FA_SCALAR; - pos += wc; + if (device->vendor_id == VK_VENDOR_ID_INTEL) { + // Disable subgroup use due to performance issues when enforcing subgroup sizes + result.subgroup_size = 32; + result.disable_subgroups = true; + } else if (device->vendor_id == VK_VENDOR_ID_AMD && device->architecture != AMD_GCN) { + result.subgroup_size = n_rows < 4 ? 32 : device->subgroup_size; + } else { + result.subgroup_size = device->subgroup_size; } - flush_run(word_count); - return true; -} - -// Remove the loop unrolling hint of the matmul shader's BK loop -// and replace it with the dont_unroll hint for better performance on -// hardware like Apple M1/M2. -// Assumes 1. code comes from mul_mm.comp 2. the K-tile loop has no loop -// control hint and 3. the BK loop is the last loop nested directly inside -// the K-tile loop. -// Returns true when the input was modified; returns false otherwise -// without touching `out`. -static bool ggml_vk_roll_bk_loop(const uint32_t * code, size_t word_count, std::vector & out) { - if (word_count < 5) { - return false; + // Row split splits the workgroup so that synchronization only has to happen within subgroups, which avoids barriers + uint32_t row_split_max_hsk = 64; + if (device->vendor_id == VK_VENDOR_ID_AMD && device->architecture != AMD_GCN && !device->uma) { + row_split_max_hsk = n_rows <= 8 ? 64 : 128; } + result.row_split = (n_rows < 4 || hsk <= row_split_max_hsk) ? 1 : 4; - struct vk_spv_loop { - size_t header; - size_t end; - uint32_t control; - }; - - std::vector loops; + if (result.subgroup_size > 32 && (n_rows < 4 || hsk < (result.row_split == 1 ? 128 : 64))) { + result.workgroup_size = result.subgroup_size * 2; + } else { + result.workgroup_size = result.subgroup_size * 4; + } - // Collect a list of all loops in the module. - for (size_t pos = 5; pos < word_count; ) { - const uint32_t wc = code[pos] >> spv::WordCountShift; - const uint32_t op = code[pos] & spv::OpCodeMask; - if (wc == 0 || pos + wc > word_count) { - return false; - } + const uint32_t D = hsk | hsv; - if (op == spv::OpLoopMerge && wc >= 4) { loops.push_back({ pos, 0, code[pos + 3] }); } + const bool reduce_block_rows = D & 8 || n_kv < 1024 || device->vendor_id == VK_VENDOR_ID_INTEL; - if (op == spv::OpLabel && wc >= 2) { - for (auto & l : loops) { - if (l.end == 0 && code[l.header + 1] == code[pos + 1]) { l.end = pos; } - } + if (n_rows == 1) { + result.block_rows = 1; + result.block_cols = 64; + } else { + // row_split 1 means higher register use per row, so block size has to be adjusted + if (result.row_split == 1) { + result.block_rows = n_rows == 2 ? 2 : ((n_rows <= 4 || reduce_block_rows) ? 4 : 8); + } else { + result.block_rows = n_rows <= 4 ? 4 : ((n_rows <= 8 || reduce_block_rows) ? 8 : 16); } - pos += wc; + result.block_cols = (D & 8) ? 64 : 32; } - auto encloses = [](const vk_spv_loop & a, const vk_spv_loop & b) { - return a.header < b.header && b.header < a.end; - }; + const uint32_t D_lsb = D ^ (D & (D-1)); // extract lowest set bit - // Find the BK loop. - const vk_spv_loop * bk = nullptr; - for (const auto & h : loops) { - if (h.control != spv::LoopControlUnrollMask) { - continue; - } - const vk_spv_loop * parent = nullptr; - bool has_child = false; - for (const auto & g : loops) { - if (encloses(g, h) && (!parent || g.header > parent->header)) { - parent = &g; - } - if (encloses(h, g)) { - has_child = true; - } - } - // BK loop should be the last loop nested inside the loop with no hint - // and have at least one child loop. - if (parent && - parent->control == spv::LoopControlMaskNone && - has_child && - (!bk || h.header > bk->header)) { - bk = &h; - } + result.d_split = std::min(std::min(result.subgroup_size, 8u), D_lsb / 4); + + result.shmem_staging = (device->vendor_id == VK_VENDOR_ID_NVIDIA && hsk < 256 && hsv < 256) ? 1 : 0; + + if (!reduce_block_rows && !ggml_vk_flash_attn_scalar_shmem_support(device, result, hsk, hsv, f32acc, k_type, v_type)) { + result.block_rows /= 2; } - if (!bk) { - return false; + + // On AMD RDNA, for small head sizes and big batch size the shader uses few registers, so too many subgroups get scheduled + // at once and end up thrashing the cache. Fix this by setting a large (unused) shmem buffer that reduces occupancy. + // This targets an occupancy of 4 subgroups per SIMD. + if (device->vendor_id == VK_VENDOR_ID_AMD && device->properties.limits.maxComputeSharedMemorySize == 65536) { + if (device->architecture != AMD_GCN && n_rows >= 64 && hsk <= 128) { + // 30kb target for hsk > 64, 26kb for <= 64 due to smaller workgroup size + // Values are guessed, tested on RDNA2 + result.limit_occupancy_shmem = (hsk <= 64 ? 26 : 30) * 1024 / 4 / 4; + } else if (device->architecture == AMD_GCN && n_rows <= 8 && hsk >= 256) { + // Same thing for GCN, with an occupancy target of 2 subgroups per SIMD. + // Here low-batch FA with large head size is affected. + // n_rows < 4 switch because workgroup size switches from 128 to 256 there. + result.limit_occupancy_shmem = (n_rows < 4 ? 14 : 26) * 1024 / 4 / 4; + } } - // set DontUnroll instead of Unroll - out.assign(code, code + word_count); - out[bk->header + 3] = spv::LoopControlDontUnrollMask; - return true; + return result; } -static void ggml_vk_create_pipeline_func(vk_device& device, vk_pipeline& pipeline, size_t spv_size, const void* spv_data, const std::string entrypoint, - uint32_t parameter_count, std::array wg_denoms, std::vector specialization_constants, - bool disable_robustness, bool require_full_subgroups, uint32_t required_subgroup_size) { - VK_LOG_DEBUG("ggml_vk_create_pipeline(" << device->name << ", " << pipeline->name << ", " << entrypoint << ", " << parameter_count << - ", (" << wg_denoms[0] << "," << wg_denoms[1] << "," << wg_denoms[2] << "), specialization_constants, " << - disable_robustness << ", " << require_full_subgroups << ", " << required_subgroup_size << ")"); - GGML_ASSERT(parameter_count > 0); - GGML_ASSERT(parameter_count <= MAX_PARAMETER_COUNT); - GGML_ASSERT(wg_denoms[0] > 0 && wg_denoms[1] > 0 && wg_denoms[2] > 0); // NOLINT - - vk::ShaderModuleCreateInfo shader_module_create_info({}, spv_size, reinterpret_cast(spv_data)); - - // Patch SPIR-V to enable supported FP16 float controls, avoiding the need - // for separate shader variants. - std::vector spirv; - if (device->float_controls_rte_fp16 || device->float_controls_denorm_preserve_fp16) { - const uint32_t* spv_words = reinterpret_cast(spv_data); - size_t word_count = spv_size / sizeof(uint32_t); - spirv.assign(spv_words, spv_words + word_count); +static vk_fa_tuning_params get_fa_tuning_params_coopmat1(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { + GGML_UNUSED(n_rows); + GGML_UNUSED(n_kv); + GGML_UNUSED(k_type); + GGML_UNUSED(v_type); + GGML_UNUSED(f32acc); - // Find insertion points respecting SPIR-V layout order: - // Header(5) -> OpCapability -> OpExtension -> ... -> OpEntryPoint -> OpExecutionMode -> ... - size_t pos = 5; // skip header - size_t cap_insert_pos = pos; - size_t ext_insert_pos = pos; - size_t exec_insert_pos = pos; - uint32_t entry_point_id = 0; + vk_fa_tuning_params result{}; + result.path = FA_COOPMAT1; - while (pos < spirv.size()) { - uint32_t opcode = spirv[pos] & spv::OpCodeMask; - uint32_t len = spirv[pos] >> spv::WordCountShift; - if (len == 0) break; + const uint32_t D = hsk | hsv; - if (opcode == spv::OpCapability) { - cap_insert_pos = pos + len; - ext_insert_pos = pos + len; - } else if (opcode == spv::OpExtension) { - ext_insert_pos = pos + len; - } else if (opcode == spv::OpEntryPoint) { - entry_point_id = spirv[pos + 2]; - exec_insert_pos = pos + len; - } else if (opcode == spv::OpExecutionMode || opcode == spv::OpExecutionModeId) { - exec_insert_pos = pos + len; - } else if (entry_point_id != 0) { - break; - } + const uint32_t coopmat_block_rows = 16; + const uint32_t coopmat_block_cols = 16; - pos += len; - } + const uint32_t num_subgroups = 4; - // Insert from latest position first so earlier indices stay valid. + result.block_rows = coopmat_block_rows; + result.block_cols = coopmat_block_cols * num_subgroups; + result.row_split = num_subgroups; + result.subgroup_size = device->subgroup_size; + result.workgroup_size = num_subgroups * result.subgroup_size; - if (device->float_controls_rte_fp16) { - // OpExecutionMode %entrypoint RoundingModeRTE 16 - uint32_t exec_mode[] = { (4u << spv::WordCountShift) | spv::OpExecutionMode, entry_point_id, spv::ExecutionModeRoundingModeRTE, 16 }; - spirv.insert(spirv.begin() + exec_insert_pos, std::begin(exec_mode), std::end(exec_mode)); - } + const uint32_t D_lsb = D ^ (D & (D-1)); // extract lowest set bit + result.d_split = std::min(std::min(result.subgroup_size, 8u), D_lsb / 4); - if (device->float_controls_denorm_preserve_fp16) { - // OpExecutionMode %entrypoint DenormPreserve 16 - uint32_t exec_mode[] = { (4u << spv::WordCountShift) | spv::OpExecutionMode, entry_point_id, spv::ExecutionModeDenormPreserve, 16 }; - spirv.insert(spirv.begin() + exec_insert_pos, std::begin(exec_mode), std::end(exec_mode)); - } + result.shmem_staging = (device->vendor_id == VK_VENDOR_ID_NVIDIA && hsk < 256 && hsv < 256) ? 1 : 0; - // OpExtension "SPV_KHR_float_controls" - const char ext_str[] = "SPV_KHR_float_controls"; - size_t ext_str_words = CEIL_DIV(sizeof(ext_str), sizeof(uint32_t)); - std::vector extension(1 + ext_str_words, 0); - extension[0] = (uint32_t)((1 + ext_str_words) << spv::WordCountShift) | spv::OpExtension; - memcpy(&extension[1], ext_str, sizeof(ext_str)); - spirv.insert(spirv.begin() + ext_insert_pos, extension.begin(), extension.end()); + return result; +} - if (device->float_controls_rte_fp16) { - // OpCapability RoundingModeRTE - uint32_t capability[] = { (2u << spv::WordCountShift) | spv::OpCapability, spv::CapabilityRoundingModeRTE }; - spirv.insert(spirv.begin() + cap_insert_pos, std::begin(capability), std::end(capability)); - } +static vk_fa_tuning_params get_fa_tuning_params_coopmat2(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { + GGML_UNUSED(n_kv); + GGML_UNUSED(f32acc); - if (device->float_controls_denorm_preserve_fp16) { - // OpCapability DenormPreserve - uint32_t capability[] = { (2u << spv::WordCountShift) | spv::OpCapability, spv::CapabilityDenormPreserve }; - spirv.insert(spirv.begin() + cap_insert_pos, std::begin(capability), std::end(capability)); - } + vk_fa_tuning_params result{}; + result.path = FA_COOPMAT2; - shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); - } + const uint32_t D = hsk | hsv; -#if defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) - if (device->coopmat2 && !device->coopmat2_decode_vector) { - const uint32_t * src = spirv.empty() ? reinterpret_cast(spv_data) : spirv.data(); - size_t src_n = spirv.empty() ? spv_size / sizeof(uint32_t) : spirv.size(); - std::vector stripped; - if (ggml_vk_strip_decode_vector(src, src_n, stripped)) { - spirv = std::move(stripped); - shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); - } - } -#endif + const bool small_rows = n_rows < 32; -#if VK_HEADER_VERSION >= 287 - // Roll the mul_mm BK loop on Asahi Linux. Skip bf16 and the mul_mmq pipelines. - if (device->driver_id == vk::DriverId::eMesaHoneykrisp && - pipeline->name.rfind("matmul", 0) == 0 && - pipeline->name.find("bf16") == std::string::npos && - pipeline->name.find("q8_1") == std::string::npos) { - const uint32_t * src = spirv.empty() ? reinterpret_cast(spv_data) : spirv.data(); - size_t src_n = spirv.empty() ? spv_size / sizeof(uint32_t) : spirv.size(); - std::vector rolled; - if (ggml_vk_roll_bk_loop(src, src_n, rolled)) { - spirv = std::move(rolled); - shader_module_create_info = vk::ShaderModuleCreateInfo({}, spirv.size() * sizeof(uint32_t), spirv.data()); - } + if (small_rows) { + result.block_rows = 32; + result.block_cols = 32; + } else if (ggml_is_quantized(k_type) || ggml_is_quantized(v_type) || hsk >= 256 || hsv >= 256) { + result.block_rows = (hsk >= 512 || hsv >= 512) ? 32 : 64; + result.block_cols = 32; + } else { + result.block_rows = 64; + result.block_cols = 64; } -#endif - - pipeline->shader_module = device->device.createShaderModule(shader_module_create_info); - vk::PushConstantRange pcr( - vk::ShaderStageFlagBits::eCompute, - 0, - pipeline->push_constant_size - ); + result.subgroup_size = device->subgroup_size; + result.workgroup_size = (small_rows && (D % 32) == 0) ? 256 : 128; - vk::PipelineLayoutCreateInfo pipeline_layout_create_info(vk::PipelineLayoutCreateFlags(), device->dsl, pcr); - pipeline->layout = device->device.createPipelineLayout(pipeline_layout_create_info); + return result; +} - std::vector specialization_entries(specialization_constants.size()); +vk_fa_tuning_params get_fa_tuning_params(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { + FaCodePath path = device->coopmat2 ? FA_COOPMAT2 : + device->coopmat1_fa_support ? FA_COOPMAT1 : FA_SCALAR; - for (size_t i = 0; i < specialization_constants.size(); i++) { - specialization_entries[i].constantID = i; - specialization_entries[i].offset = i * sizeof(uint32_t); - specialization_entries[i].size = sizeof(uint32_t); + if (path == FA_COOPMAT2 && k_type == GGML_TYPE_BF16 && !device->coopmat2_bf16_support) { + path = FA_COOPMAT1; + } + if (path == FA_COOPMAT1 && k_type == GGML_TYPE_BF16 && !device->coopmat_bf16_support) { + path = FA_SCALAR; } - vk::SpecializationInfo specialization_info( - specialization_entries.size(), - specialization_entries.data(), - specialization_constants.size() * sizeof(uint32_t), - specialization_constants.data() - ); - - vk::PipelineShaderStageCreateFlags pipeline_shader_stage_create_flags{}; - - if (device->subgroup_require_full_support && require_full_subgroups) { - pipeline_shader_stage_create_flags |= vk::PipelineShaderStageCreateFlagBits::eRequireFullSubgroupsEXT; - } - - vk::PipelineShaderStageCreateInfo pipeline_shader_create_info( - pipeline_shader_stage_create_flags, - vk::ShaderStageFlagBits::eCompute, - pipeline->shader_module, - entrypoint.c_str(), - &specialization_info); - - vk::PipelineShaderStageRequiredSubgroupSizeCreateInfoEXT pipeline_shader_stage_required_subgroup_size_create_info; - pipeline_shader_stage_required_subgroup_size_create_info.requiredSubgroupSize = required_subgroup_size; - if (device->subgroup_size_control && required_subgroup_size > 0) { - GGML_ASSERT(device->subgroup_min_size <= required_subgroup_size && required_subgroup_size <= device->subgroup_max_size); - pipeline_shader_create_info.setPNext(&pipeline_shader_stage_required_subgroup_size_create_info); + if (path == FA_COOPMAT1 && device->architecture == vk_device_architecture::NVIDIA_TURING) { + // Nvidia compiler bug, see https://github.com/ggml-org/llama.cpp/pull/19075#issuecomment-3820716090 + path = FA_SCALAR; } - vk::ComputePipelineCreateInfo compute_pipeline_create_info( - device->pipeline_executable_properties_support ? - vk::PipelineCreateFlagBits::eCaptureStatisticsKHR : - vk::PipelineCreateFlags{}, - pipeline_shader_create_info, - pipeline->layout); - - vk::PipelineRobustnessCreateInfoEXT rci; - - if (device->pipeline_robustness && disable_robustness) { - rci.storageBuffers = vk::PipelineRobustnessBufferBehaviorEXT::eDisabled; - rci.uniformBuffers = vk::PipelineRobustnessBufferBehaviorEXT::eDisabled; - compute_pipeline_create_info.setPNext(&rci); - } + if (path == FA_COOPMAT1) { + bool shape_ok = (f32acc && device->coopmat_support_16x16x16_f32acc) || + (!f32acc && device->coopmat_support_16x16x16_f16acc); + const vk_fa_tuning_params params = get_fa_tuning_params_coopmat1(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); + bool shmem_ok = ggml_vk_flash_attn_coopmat_shmem_support(device, params, hsk, hsv, f32acc, k_type, v_type); -#if defined(VK_EXT_shader_64bit_indexing) - vk::PipelineCreateFlags2CreateInfo pipelineFlags2CreateInfo; - if (pipeline->is_64b_indexing) - { - pipelineFlags2CreateInfo.flags = vk::PipelineCreateFlagBits2::e64BitIndexingEXT; - if (device->pipeline_executable_properties_support) { - pipelineFlags2CreateInfo.flags |= vk::PipelineCreateFlagBits2::eCaptureStatisticsKHR; + if (!shape_ok || !shmem_ok) { + path = FA_SCALAR; } - pipelineFlags2CreateInfo.setPNext(compute_pipeline_create_info.pNext); - compute_pipeline_create_info.setPNext(&pipelineFlags2CreateInfo); } -#endif - try { - pipeline->pipeline = device->device.createComputePipeline(VK_NULL_HANDLE, compute_pipeline_create_info).value; - } catch (const vk::SystemError& e) { - std::cerr << "ggml_vulkan: Compute pipeline creation failed for " << pipeline->name << std::endl; - std::cerr << "ggml_vulkan: " << e.what() << std::endl; - throw e; + // scalar is faster than coopmat when N==1 + if (n_rows == 1 && (path == FA_COOPMAT1 || path == FA_COOPMAT2)) { + path = FA_SCALAR; } - if (vk_instance.debug_utils_support) { - vk::DebugUtilsObjectNameInfoEXT duoni; - duoni.objectType = vk::ObjectType::ePipeline; - duoni.pObjectName = pipeline->name.c_str(); - duoni.objectHandle = /*reinterpret_cast*/(uint64_t)(static_cast(pipeline->pipeline)); - vk_instance.pfn_vkSetDebugUtilsObjectNameEXT(device->device, &static_cast(duoni)); + switch (path) { + case FA_SCALAR: + return get_fa_tuning_params_scalar(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); + case FA_COOPMAT1: + return get_fa_tuning_params_coopmat1(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); + case FA_COOPMAT2: + return get_fa_tuning_params_coopmat2(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); + default: + throw std::runtime_error("unsupported FaCodePath"); } +} - if (device->pipeline_executable_properties_support) { - vk::PipelineExecutableInfoKHR executableInfo; - executableInfo.pipeline = pipeline->pipeline; - - auto statistics = device->device.getPipelineExecutableStatisticsKHR(executableInfo); +vk_fa_pipeline_state get_fa_pipeline_state(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool aligned, bool f32acc, + bool use_mask, bool use_mask_opt, bool use_logit_softcap, bool use_sparse, ggml_type k_type, ggml_type v_type) { + const bool old_amd_windows = device->vendor_id == VK_VENDOR_ID_AMD && device->driver_id == vk::DriverId::eAmdProprietary && + (device->architecture == AMD_GCN || device->architecture == AMD_RDNA1 || device->architecture == AMD_RDNA2); - bool print_stats = !vk_pipeline_stats_filter.empty() && - pipeline->name.find(vk_pipeline_stats_filter) != std::string::npos; - if (print_stats) { - std::cerr << "ggml_vulkan: pipeline stats for " << pipeline->name << ":" << std::endl; - } + uint32_t flags = (use_mask_opt ? 1 : 0) | + (use_mask ? 2 : 0) | + (use_logit_softcap ? 4 : 0) | + (old_amd_windows ? 8 : 0) | + (use_sparse ? 16 : 0); - for (auto & s : statistics) { - if (print_stats) { - std::cerr << "ggml_vulkan: " << s.name.data() << ": "; - switch (s.format) { - case vk::PipelineExecutableStatisticFormatKHR::eBool32: - std::cerr << (s.value.b32 ? "true" : "false"); - break; - case vk::PipelineExecutableStatisticFormatKHR::eInt64: - std::cerr << s.value.i64; - break; - case vk::PipelineExecutableStatisticFormatKHR::eUint64: - std::cerr << s.value.u64; - break; - case vk::PipelineExecutableStatisticFormatKHR::eFloat64: - std::cerr << s.value.f64; - break; - } - std::cerr << std::endl; - } - // "Register Count" is reported by NVIDIA drivers. - if (strcmp(s.name, "Register Count") == 0) { - VK_LOG_DEBUG(pipeline->name << " " << s.name << ": " << s.value.u64 << " registers"); - pipeline->register_count = (uint32_t)s.value.u64; - } - } - } + const uint32_t subgroup_size = params.disable_subgroups ? 0 : params.subgroup_size; - { - std::lock_guard guard(device->compile_mutex); - device->all_pipelines.push_back(pipeline); - pipeline->compiled = true; - pipeline->compile_pending = false; - } - device->compile_cv.notify_all(); + return vk_fa_pipeline_state{hsk, hsv, params.block_rows, params.block_cols, params.d_split, params.row_split, params.shmem_staging, params.path, params.workgroup_size, subgroup_size, aligned, f32acc, flags, params.limit_occupancy_shmem, k_type, v_type}; } -static void ggml_vk_destroy_pipeline(vk::Device& device, vk_pipeline& pipeline) { - VK_LOG_DEBUG("ggml_pipeline_destroy_pipeline(" << pipeline->name << ")"); - device.destroyPipelineLayout(pipeline->layout); - - device.destroyShaderModule(pipeline->shader_module); - - device.destroyPipeline(pipeline->pipeline); +static uint32_t fa_block_bytes(ggml_type t) { + if (t == GGML_TYPE_F32) { + return 16u; + } + return (uint32_t) ggml_type_size(t); } -static void ggml_pipeline_request_descriptor_sets(ggml_backend_vk_context *ctx, vk_pipeline& pipeline, uint32_t n) { - VK_LOG_DEBUG("ggml_pipeline_request_descriptor_sets(" << pipeline->name << ", " << n << ")"); - ctx->pipeline_descriptor_set_requirements += n; - if (!pipeline->compiled) { - ggml_vk_load_shaders(ctx->device, pipeline); - } - ggml_pipeline_allocate_descriptor_sets(ctx); +static std::vector get_fa_spec_constants(const vk_fa_pipeline_state& state) { + return { + /* 0 WorkGroupSize */ state.workgroup_size, + /* 1 Br */ state.Br, + /* 2 Bc */ state.Bc, + /* 3 HSK */ state.HSK, + /* 4 HSV */ state.HSV, + /* 5 Clamp */ static_cast(!state.aligned), + /* 6 D_split */ state.D_split, + /* 7 row_split */ state.row_split, + /* 8 SubGroupSize */ state.subgroup_size, + /* 9 SHMEM_STAGING */ state.shmem_staging ? 1u : 0u, + /*10 Flags */ state.flags, + /*11 LIMIT_OCCUPANCY_SHMEM */ state.limit_occupancy_shmem, + /*12 FaTypeK */ static_cast(state.k_type), + /*13 FaTypeV */ static_cast(state.v_type), + /*14 FaBlockBytesK */ fa_block_bytes(state.k_type), + /*15 FaBlockBytesV */ fa_block_bytes(state.v_type), + }; } -static void ggml_pipeline_allocate_descriptor_sets(ggml_backend_vk_context * ctx) { +static bool ggml_vk_matmul_shmem_support(const vk_device& device, const std::vector& warptile, bool mul_mat_id, ggml_type src0_type) { - if (ctx->descriptor_sets.size() >= ctx->pipeline_descriptor_set_requirements) { - // Enough descriptors are available - return; + uint32_t lut_size = 0; + switch (src0_type) { + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + // Regular matmul uses the compact uint16_t IQ1 grid; the expanded + // uint32_t grid is only enabled for the q8_1/int-dot vector path. + lut_size = 2*2048; + break; + case GGML_TYPE_IQ2_XXS: + lut_size = 8*256; + break; + case GGML_TYPE_IQ2_XS: + lut_size = 8*512; + break; + case GGML_TYPE_IQ2_S: + lut_size = 8*1024; + break; + case GGML_TYPE_IQ3_XXS: + lut_size = 4*256; + break; + case GGML_TYPE_IQ3_S: + lut_size = 4*512; + break; + case GGML_TYPE_IQ4_NL: + case GGML_TYPE_IQ4_XS: + case GGML_TYPE_MXFP4: + lut_size = 4*16; + break; + case GGML_TYPE_NVFP4: + // Same kvalues budget as MXFP4 plus ue4m3_fp32_lut[128] (types.glsl, DATA_A_NVFP4). + lut_size = 4*16 + 128u * (uint32_t)sizeof(float); + break; + default: + break; } - vk_device& device = ctx->device; - - // Grow by 50% to avoid frequent allocations - uint32_t needed = std::max(3 * ctx->descriptor_sets.size() / 2, size_t{ctx->pipeline_descriptor_set_requirements}); - uint32_t to_alloc = needed - ctx->descriptor_sets.size(); - uint32_t pool_remaining = VK_DEVICE_DESCRIPTOR_POOL_SIZE - ctx->descriptor_sets.size() % VK_DEVICE_DESCRIPTOR_POOL_SIZE; - uint32_t pool_idx = ctx->descriptor_sets.size() / VK_DEVICE_DESCRIPTOR_POOL_SIZE; + // Needs to be kept up to date on shader changes + // Needs to stay aligned with ggml_vk_mul_mm_spec. + const bool intel_shmem_stride_pad_zero = device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support && + device->driver_id == vk::DriverId::eIntelProprietaryWindows; + const uint32_t bank_conflict_offset = intel_shmem_stride_pad_zero ? 0 : (device->coopmat_support ? 8 : 1); + const uint32_t type_size = device->fp16 ? sizeof(ggml_fp16_t) : sizeof(float); + const uint32_t warps = warptile[0] / warptile[10]; - while (to_alloc > 0) { - const uint32_t alloc_count = std::min(pool_remaining, to_alloc); - to_alloc -= alloc_count; - pool_remaining = VK_DEVICE_DESCRIPTOR_POOL_SIZE; + const uint32_t load_bufs = (warptile[1] + warptile[2]) * (warptile[3] + bank_conflict_offset) * type_size; + const uint32_t mmid_row_ids = mul_mat_id ? (warptile[2] * 2 * sizeof(uint16_t)) : 0; + const uint32_t coopmat_stage = device->coopmat_support ? warptile[7] * warptile[8] / warps * sizeof(float) : 0; + const uint32_t ballots_sh = mul_mat_id ? (warps * 4 * sizeof(uint32_t)) : 0; - if (pool_idx >= ctx->descriptor_pools.size()) { - vk::DescriptorPoolSize descriptor_pool_size(vk::DescriptorType::eStorageBuffer, MAX_PARAMETER_COUNT * VK_DEVICE_DESCRIPTOR_POOL_SIZE); - vk::DescriptorPoolCreateInfo descriptor_pool_create_info({}, VK_DEVICE_DESCRIPTOR_POOL_SIZE, descriptor_pool_size); - ctx->descriptor_pools.push_back(device->device.createDescriptorPool(descriptor_pool_create_info)); - } + const uint32_t total_size = load_bufs + mmid_row_ids + coopmat_stage + lut_size + ballots_sh; + const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - std::vector layouts(alloc_count); - for (uint32_t i = 0; i < alloc_count; i++) { - layouts[i] = device->dsl; - } - vk::DescriptorSetAllocateInfo descriptor_set_alloc_info(ctx->descriptor_pools[pool_idx], alloc_count, layouts.data()); - std::vector sets = device->device.allocateDescriptorSets(descriptor_set_alloc_info); - ctx->descriptor_sets.insert(ctx->descriptor_sets.end(), sets.begin(), sets.end()); + VK_LOG_DEBUG("ggml_vk_matmul_shmem_support(warptile=(" << warptile[0] << "," << warptile[1] << "," << warptile[2] << "), " + "mul_mat_id=" << mul_mat_id << ", src0_type=" << ggml_type_name(src0_type) << ", supported=" << supported); - pool_idx++; - } + return supported; } -static vk_command_buffer* ggml_vk_create_cmd_buffer(vk_device& device, vk_command_pool& p) { - VK_LOG_DEBUG("ggml_vk_create_cmd_buffer()"); - vk::CommandBufferAllocateInfo command_buffer_alloc_info( - p.pool, - vk::CommandBufferLevel::ePrimary, - 1); - const std::vector cmd_buffers = device->device.allocateCommandBuffers(command_buffer_alloc_info); - p.cmd_buffers.push_back({ cmd_buffers.front(), 0, true }); - return &p.cmd_buffers[p.cmd_buffers.size()-1]; -} +static bool ggml_vk_matmul_int_shmem_support(const vk_device& device, const std::vector& warptile, bool mul_mat_id, ggml_type src0_type) { -static void ggml_vk_submit(vk_context& ctx, vk::Fence fence) { - if (ctx->seqs.empty()) { - if (fence) { - ctx->p->q->handle->submit({}, fence); + // FLOAT_TYPE in the shader is float16_t with fp16 support, otherwise float. + const uint32_t fp_size = device->fp16 ? 2u : 4u; + const uint32_t fp_align = fp_size; + const uint32_t fp2_size = 2u * fp_size; + const uint32_t fp2_align = device->fp16 ? 4u : 8u; + + struct member { uint32_t size, align; }; + auto std430_size = [](std::initializer_list members) { + uint32_t off = 0, struct_align = 1; + for (const auto &m : members) { + off = (off + m.align - 1) & ~(m.align - 1); + off += m.size; + struct_align = std::max(struct_align, m.align); } - return; + return (off + struct_align - 1) & ~(struct_align - 1); + }; + + uint32_t block_a_size = 0; + switch (src0_type) { + case GGML_TYPE_Q2_0: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + dm + case GGML_TYPE_Q4_0: block_a_size = std430_size({{16, 4}, {fp_size, fp_align}}); break; // qs[16/4] + dm + case GGML_TYPE_Q4_1: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[16/4] + dm(vec2) + case GGML_TYPE_Q5_0: block_a_size = std430_size({{16, 4}, {4, 4}, {fp_size, fp_align}}); break; // qs[16/4] + qh + dm + case GGML_TYPE_Q5_1: block_a_size = std430_size({{16, 4}, {4, 4}, {fp2_size, fp2_align}}); break; // qs[16/4] + qh + dm(vec2) + case GGML_TYPE_Q8_0: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + dm + case GGML_TYPE_IQ4_XS: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + d + case GGML_TYPE_MXFP4: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + d + case GGML_TYPE_Q2_K: block_a_size = std430_size({{ 8, 4}, {2, 2}, {fp2_size, fp2_align}}); break; // qs[2] + scales(u8vec2) + dm(vec2) + case GGML_TYPE_Q3_K: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[4] + d_scales(vec2) + case GGML_TYPE_Q4_K: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[4] + dm(vec2) + case GGML_TYPE_Q5_K: block_a_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); break; // qs[8] + dm(vec2) + case GGML_TYPE_Q6_K: block_a_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); break; // qs[8] + d_scales(vec2) + case GGML_TYPE_IQ3_S: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + d + default: + return false; } - VK_LOG_DEBUG("ggml_vk_submit(" << ctx << ", " << fence << ")"); - std::vector> tl_wait_vals; - std::vector> tl_signal_vals; - std::vector> tl_wait_semaphores; - std::vector> tl_signal_semaphores; - std::vector tl_submit_infos; - std::vector submit_infos; - int idx = -1; - std::vector> stage_flags; + // IQ3_S also copies its 512-entry grid into shared memory (types.glsl, init_iq_shmem) + const uint32_t lut_size = (src0_type == GGML_TYPE_IQ3_S) ? 4*512 : 0; - size_t reserve = 0; + // block_b_cache: { int32_t qs[8]; FLOAT_TYPEV2 ds; } + const uint32_t block_b_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); - for (const auto& sequence : ctx->seqs) { - reserve += sequence.size(); - } + const uint32_t BM = warptile[1]; + const uint32_t BN = warptile[2]; + // mul_mmq.comp: BK_STEP=1 for MUL_MAT_ID, 4 otherwise. + const uint32_t BK_STEP = mul_mat_id ? 1u : 4u; - // Pre-reserve vectors to prevent reallocation, which invalidates pointers - tl_wait_semaphores.reserve(reserve); - tl_wait_vals.reserve(reserve); - tl_signal_semaphores.reserve(reserve); - tl_signal_vals.reserve(reserve); - tl_submit_infos.reserve(reserve); - submit_infos.reserve(reserve); - stage_flags.reserve(reserve); + const uint32_t buf_a_size = BM * BK_STEP * block_a_size; + const uint32_t buf_b_size = BN * BK_STEP * block_b_size; + const uint32_t mmid_row_ids = mul_mat_id ? (BN * 2u * (uint32_t)sizeof(uint16_t)) : 0u; - for (const auto& sequence : ctx->seqs) { - for (const auto& submission : sequence) { - stage_flags.push_back({}); - idx++; - tl_wait_vals.push_back({}); - tl_wait_semaphores.push_back({}); - tl_signal_vals.push_back({}); - tl_signal_semaphores.push_back({}); - for (size_t i = 0; i < submission.wait_semaphores.size(); i++) { - stage_flags[idx].push_back(ctx->p->q->stage_flags); - tl_wait_vals[idx].push_back(submission.wait_semaphores[i].value); - tl_wait_semaphores[idx].push_back(submission.wait_semaphores[i].s); - } - for (size_t i = 0; i < submission.signal_semaphores.size(); i++) { - tl_signal_vals[idx].push_back(submission.signal_semaphores[i].value); - tl_signal_semaphores[idx].push_back(submission.signal_semaphores[i].s); - } - tl_submit_infos.push_back({ - (uint32_t) submission.wait_semaphores.size(), - tl_wait_vals[idx].data(), - (uint32_t) submission.signal_semaphores.size(), - tl_signal_vals[idx].data(), - }); - tl_submit_infos[idx].sType = vk::StructureType::eTimelineSemaphoreSubmitInfo; - tl_submit_infos[idx].pNext = nullptr; - vk::SubmitInfo si{ - (uint32_t) submission.wait_semaphores.size(), - tl_wait_semaphores[idx].data(), - stage_flags[idx].data(), - 1, - &submission.buffer->buf, - (uint32_t) submission.signal_semaphores.size(), - tl_signal_semaphores[idx].data(), - }; - si.setPNext(&tl_submit_infos[idx]); - submit_infos.push_back(si); - } - } + const uint32_t warps = warptile[0] / warptile[10]; + const uint32_t ballots_sh = mul_mat_id ? (warps * 4u * (uint32_t)sizeof(uint32_t)) : 0u; - ctx->p->q->handle->submit(submit_infos, fence); + const uint32_t total_size = buf_a_size + buf_b_size + mmid_row_ids + ballots_sh + lut_size; + const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - ctx->seqs.clear(); -} + VK_LOG_DEBUG("ggml_vk_matmul_int_shmem_support(warptile=(" << warptile[0] << "," << warptile[1] << "," << warptile[2] << "), " + "mul_mat_id=" << mul_mat_id << ", src0_type=" << ggml_type_name(src0_type) << ", total=" << total_size << ", supported=" << supported); -static uint32_t ggml_vk_find_queue_family_index(std::vector& queue_family_props, const vk::QueueFlags& required, const vk::QueueFlags& avoid, int32_t compute_index, uint32_t min_num_queues) { - VK_LOG_DEBUG("ggml_vk_find_queue_family_index()"); - const uint32_t qfsize = queue_family_props.size(); + return supported; +} - // Try with avoid preferences first - for (uint32_t i = 0; i < qfsize; i++) { - if (queue_family_props[i].queueCount >= min_num_queues && (compute_index < 0 || i != (uint32_t) compute_index) && queue_family_props[i].queueFlags & required && !(queue_family_props[i].queueFlags & avoid)) { - return i; - } - } +static const std::unordered_map rdna1_pipelines = { + {"soft_max", 64}, {"im2col", 64}, + {"argmax", 64}, {"mul_mat_vec", 64}, + {"mul_mat_vec_f16", 32}, {"mul_mat_vec_f32_f16", 32} +}; - // Fall back to only required - for (size_t i = 0; i < qfsize; i++) { - if (queue_family_props[i].queueCount >= min_num_queues && (compute_index < 0 || i != (uint32_t) compute_index) && queue_family_props[i].queueFlags & required) { - return i; - } - } +static const std::unordered_map rdna2_pipelines = { + {"soft_max", 64}, {"im2col", 64}, +}; - // Fall back to reusing compute queue - for (size_t i = 0; i < qfsize; i++) { - if (queue_family_props[i].queueCount >= min_num_queues && queue_family_props[i].queueFlags & required) { - return i; - } - } +static std::vector gpu_pipeline_configs = { + { + vk_device_architecture::AMD_RDNA1, + { + rdna1_pipelines, + }, + RDNA_DEFAULT_SUBGROUP_SIZE + }, + { + vk_device_architecture::AMD_RDNA2, + { + rdna2_pipelines, + }, + RDNA_DEFAULT_SUBGROUP_SIZE + }, +}; - // Fall back to ignoring min_num_queries - for (size_t i = 0; i < qfsize; i++) { - if (queue_family_props[i].queueFlags & required) { - return i; +uint32_t get_subgroup_size(const std::string &pipeline_name, const vk_device_architecture &arch) { + for (const auto &config : gpu_pipeline_configs) { + if (config.arch == arch) { + auto pipIt = config.pipelines.find(pipeline_name); + if (pipIt != config.pipelines.end()) { + return pipIt->second; + } + std::vector> sorted_pipelines(config.pipelines.begin(), config.pipelines.end()); + std::sort(sorted_pipelines.begin(), sorted_pipelines.end(), + [](const auto &a, const auto &b) { return a.first.size() > b.first.size(); }); + for (const auto &entry : sorted_pipelines) { + if (pipeline_name.find(entry.first) != std::string::npos) { + return entry.second; + } + } + return config.default_subgroup_size; } } + return 0; // If no matching configuration is found +} - // All commands that are allowed on a queue that supports transfer operations are also allowed on a queue that supports either graphics or compute operations. - // Thus, if the capabilities of a queue family include VK_QUEUE_GRAPHICS_BIT or VK_QUEUE_COMPUTE_BIT, then reporting the VK_QUEUE_TRANSFER_BIT capability separately for that queue family is optional. - if (compute_index >= 0) { - return compute_index; +static bool ggml_vk_fa_type_needs_shmem(ggml_type type) { + switch (type) { + case GGML_TYPE_IQ4_NL: + return true; + default: + return false; } +} - std::cerr << "ggml_vulkan: No suitable queue family index found." << std::endl; - - for(auto &q_family : queue_family_props) { - std::cerr << "Queue number: " + std::to_string(q_family.queueCount) << " flags: " + to_string(q_family.queueFlags) << std::endl; - } - abort(); +static bool ggml_vk_fa_scalar_uses_mmq(const vk_device& device, ggml_type k_type, ggml_type v_type) { +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + return device->integer_dot_product && device->subgroup_clustered && + !ggml_vk_fa_type_needs_shmem(v_type) && + (k_type == GGML_TYPE_Q4_0 || k_type == GGML_TYPE_Q4_1 || + k_type == GGML_TYPE_Q5_0 || k_type == GGML_TYPE_Q5_1 || + k_type == GGML_TYPE_Q8_0); +#else + GGML_UNUSED(device); + GGML_UNUSED(k_type); + GGML_UNUSED(v_type); + return false; +#endif } -static std::unique_ptr ggml_vk_create_queue(vk_device& device, uint32_t queue_family_index, uint32_t queue_index, vk::PipelineStageFlags&& stage_flags, bool transfer_only) { - VK_LOG_DEBUG("ggml_vk_create_queue()"); - std::lock_guard guard(device->mutex); +void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested) { + VK_LOG_DEBUG("ggml_vk_load_shaders(" << device->name << ")"); - auto q = std::make_unique(); - q->queue_family_index = queue_family_index; - q->transfer_only = transfer_only; + // some shaders have a minimum subgroup size + const uint32_t subgroup_size_8 = std::max(device->subgroup_size, 8u); + const uint32_t subgroup_size_16 = std::max(device->subgroup_size, 16u); + const uint32_t subgroup_size_32 = std::max(device->subgroup_size, 32u); - std::shared_ptr h; - vk::DeviceQueueInfo2 queue_info2{}; - queue_info2.queueFamilyIndex = queue_family_index; - queue_info2.queueIndex = queue_index; + // clamp WARP for l_/m_ warptiles so WM <= BM (breaks on subgroupSize > 64) + const uint32_t mm_warp_8 = std::min(subgroup_size_8, 64u); + const uint32_t mm_warp_16 = std::min(subgroup_size_16, 64u); - if (device->has_internally_synchronized_queues) { - h = std::make_shared(); - queue_info2.flags = eInternallySynchronizedKHR; - } else { - h = std::make_shared(); - } + const uint32_t mul_mat_subgroup_size = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; + const uint32_t mul_mat_subgroup_size_8 = std::max(mul_mat_subgroup_size, 8u); + const uint32_t mul_mat_subgroup_size_16 = std::max(mul_mat_subgroup_size, 16u); + const uint32_t mul_mat_subgroup_size_32 = std::max(mul_mat_subgroup_size, 32u); + const uint32_t mul_mat_mm_warp_8 = std::min(mul_mat_subgroup_size_8, 64u); + const uint32_t mul_mat_mm_warp_16 = std::min(mul_mat_subgroup_size_16, 64u); - h->queue = device->device.getQueue2(queue_info2); - h->device = device; - q->handle = h; + const bool subgroup_min_size_16 = (!device->subgroup_size_control && device->subgroup_size >= 16) || + (device->subgroup_size_control && device->subgroup_max_size >= 16); - q->cmd_pool.init(device, q.get()); + // mulmat + // Warptile layout (indices match mul_mm.comp constantIDs): + // [0..9] : BLOCK_SIZE, BM, BN, BK, WM, WN, WMITER, TM, TN, TK + // [10] : WARP / required_subgroup_size (read via WARP_SIZE_IDX) + static constexpr size_t WARP_SIZE_IDX = 10; + std::vector l_warptile, m_warptile, s_warptile, + l_warptile_id, m_warptile_id, s_warptile_id, + l_warptile_mmq, m_warptile_mmq, s_warptile_mmq, + l_warptile_mmq_int, m_warptile_mmq_int, s_warptile_mmq_int, + l_warptile_mmq_int_k, m_warptile_mmq_int_k, s_warptile_mmq_int_k, + l_warptile_mmq_k, m_warptile_mmq_k, s_warptile_mmq_k, + l_warptile_mmqid, m_warptile_mmqid, s_warptile_mmqid, + l_warptile_mmqid_int, m_warptile_mmqid_int, s_warptile_mmqid_int, + l_warptile_mmqid_int_k, m_warptile_mmqid_int_k, s_warptile_mmqid_int_k; + std::array l_wg_denoms, m_wg_denoms, s_wg_denoms, + l_mmq_wg_denoms, m_mmq_wg_denoms, s_mmq_wg_denoms, + l_mmq_wg_denoms_k, m_mmq_wg_denoms_k, s_mmq_wg_denoms_k, + l_mmqid_wg_denoms, m_mmqid_wg_denoms, s_mmqid_wg_denoms; - q->stage_flags = stage_flags; - return q; -} + uint32_t l_align, m_align, s_align; -static std::unique_ptr ggml_vk_create_aliased_queue(vk_device& device, const std::unique_ptr& source) { - std::lock_guard guard(device->mutex); - auto q = std::make_unique(); - q->handle = source->handle; - q->queue_family_index = source->queue_family_index; - q->stage_flags = source->stage_flags; - q->transfer_only = source->transfer_only; - q->cmd_pool.init(device, q.get()); - return q; -} - -static vk_context ggml_vk_create_context(ggml_backend_vk_context * ctx, vk_command_pool& p) { - vk_context result = std::make_shared(); - VK_LOG_DEBUG("ggml_vk_create_context(" << result << ")"); - ctx->gc.contexts.emplace_back(result); - result->p = &p; - return result; -} - -static vk_context ggml_vk_create_temporary_context(vk_command_pool& p) { - vk_context result = std::make_shared(); - VK_LOG_DEBUG("ggml_vk_create_temporary_context(" << result << ")"); - result->p = &p; - return result; -} - -static vk_semaphore * ggml_vk_create_binary_semaphore(ggml_backend_vk_context * ctx) { - VK_LOG_DEBUG("ggml_vk_create_timeline_semaphore()"); - vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eBinary, 0 }; - vk::SemaphoreCreateInfo ci{}; - ci.setPNext(&tci); - vk::Semaphore semaphore = ctx->device->device.createSemaphore(ci); - ctx->gc.semaphores.push_back({ semaphore, 0 }); - return &ctx->gc.semaphores[ctx->gc.semaphores.size() - 1]; -} - -static vk_semaphore * ggml_vk_create_timeline_semaphore(ggml_backend_vk_context * ctx) { - VK_LOG_DEBUG("ggml_vk_create_timeline_semaphore()"); - if (ctx->semaphore_idx >= ctx->gc.tl_semaphores.size()) { - vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eTimeline, 0 }; - vk::SemaphoreCreateInfo ci{}; - ci.setPNext(&tci); - vk::Semaphore semaphore = ctx->device->device.createSemaphore(ci); - ctx->gc.tl_semaphores.push_back({ semaphore, 0 }); - } - return &ctx->gc.tl_semaphores[ctx->semaphore_idx++]; -} - -static vk::Event ggml_vk_create_event(ggml_backend_vk_context * ctx) { - if (ctx->event_idx >= ctx->gc.events.size()) { - ctx->gc.events.push_back(ctx->device->device.createEvent({})); - } - return ctx->gc.events[ctx->event_idx++]; -} - -static void ggml_vk_command_pool_cleanup(vk_device& device, vk_command_pool& p) { - VK_LOG_DEBUG("ggml_vk_command_pool_cleanup()"); - - // Requires command buffers to be done - device->device.resetCommandPool(p.pool); - // Don't clear the command buffers and mark them as not in use. - // This allows us to reuse them - for (auto& cmd_buffer : p.cmd_buffers) { - cmd_buffer.in_use = false; - } -} - -static void ggml_vk_queue_command_pools_cleanup(vk_device& device) { - VK_LOG_DEBUG("ggml_vk_queue_command_pools_cleanup()"); + vk_pipeline wait_pipeline; + CompileTask claimed_task {}; + bool has_claimed_task = false; - // Arbitrary frequency to cleanup/reuse command buffers - static constexpr uint32_t cleanup_frequency = 10; + // The rest of the walk reads and writes shared device state, so hold the + // lock until we're done deciding what to compile. + std::unique_lock compile_lock(device->compile_mutex); - if (device->compute_queue->cmd_pool.buffers_in_use() >= cleanup_frequency) { - ggml_vk_command_pool_cleanup(device, device->compute_queue->cmd_pool); - } - if (device->transfer_queue->cmd_pool.buffers_in_use() >= cleanup_frequency) { - ggml_vk_command_pool_cleanup(device, device->transfer_queue->cmd_pool); - } -} + if (device->coopmat2) { + // spec constants and tile sizes for non-quant matmul/matmul_id + l_warptile = { 256, 128, 256, 64, 1 }; + m_warptile = { 256, 128, 128, 64, 0 }; + s_warptile = { 128, 64, 64, 64, 0 }; + l_wg_denoms = {128, 256, 1 }; + m_wg_denoms = {128, 128, 1 }; + s_wg_denoms = { 64, 64, 1 }; -static std::vector ggml_vk_find_memory_properties(const vk::PhysicalDeviceMemoryProperties* mem_props, vk::MemoryRequirements* mem_req, vk::MemoryPropertyFlags flags) { - std::vector indices; + // spec constants and tile sizes for quant matmul (non-Qi_K) + l_warptile_mmq = { 256, 128, 256, 64, 1 }; + m_warptile_mmq = { 256, 128, 128, 64, 1 }; + s_warptile_mmq = { 256, 32, 64, 128, 0 }; + l_mmq_wg_denoms = { 128, 256, 1 }; + m_mmq_wg_denoms = { 128, 128, 1 }; + s_mmq_wg_denoms = { 32, 64, 1 }; - for (uint32_t i = 0; i < mem_props->memoryTypeCount; ++i) { - vk::MemoryType memory_type = mem_props->memoryTypes[i]; - if ((mem_req->memoryTypeBits & ((uint64_t)1 << i)) && - (flags & memory_type.propertyFlags) == flags && - mem_props->memoryHeaps[memory_type.heapIndex].size >= mem_req->size) { - indices.push_back(i); - } - } - return indices; -} + // spec constants and tile sizes for quant matmul (Qi_K) + l_warptile_mmq_k = { 256, 128, 256, 64, 1 }; + m_warptile_mmq_k = { 256, 128, 128, 64, 1 }; + s_warptile_mmq_k = { 256, 32, 64, 128, 0 }; + l_mmq_wg_denoms_k = { 128, 256, 1 }; + m_mmq_wg_denoms_k = { 128, 128, 1 }; + s_mmq_wg_denoms_k = { 32, 64, 1 }; -static vk_buffer ggml_vk_create_buffer(vk_device& device, size_t size, const std::initializer_list & req_flags_list, - void *import_ptr = nullptr) { - VK_LOG_DEBUG("ggml_vk_create_buffer(" << device->name << ", " << size << ", " << to_string(req_flags_list.begin()[0]) << ", " << to_string(req_flags_list.begin()[req_flags_list.size()-1]) << ")"); - if (size > device->max_buffer_size) { - throw vk::OutOfDeviceMemoryError("Requested buffer size exceeds device buffer size limit"); - } + // spec constants and tile sizes for quant matmul_id + const uint32_t mmqid_bk = device->coopmat2_decode_vector ? 64u : 32u; + l_warptile_mmqid = { 256, 128, 128, mmqid_bk, 1 }; + m_warptile_mmqid = { 256, 128, 64, mmqid_bk, 0 }; + s_warptile_mmqid = { 256, 128, 64, mmqid_bk, 0 }; + l_mmqid_wg_denoms = { 128, 128, 1 }; + m_mmqid_wg_denoms = { 128, 64, 1 }; + s_mmqid_wg_denoms = { 128, 64, 1 }; - vk_buffer buf = std::make_shared(); + l_align = 128; + m_align = 64; + s_align = 32; + } else { + // Matrix cores require different warp group sizes + const uint32_t tm_l = device->coopmat_support ? device->coopmat_m : 4; + const uint32_t tm_m = device->coopmat_support ? device->coopmat_m : 4; + const uint32_t tm_s = device->coopmat_support ? device->coopmat_m : 2; + const uint32_t tn_l = device->coopmat_support ? device->coopmat_n : 4; + const uint32_t tn_m = device->coopmat_support ? device->coopmat_n : 2; + const uint32_t tn_s = device->coopmat_support ? device->coopmat_n : 2; + const uint32_t tk_l = device->coopmat_support ? device->coopmat_k : 1; + const uint32_t tk_m = device->coopmat_support ? device->coopmat_k : 1; + const uint32_t tk_s = device->coopmat_support ? device->coopmat_k : 1; - if (size == 0) { - buf->size = 0; - return buf; - } + const uint32_t s_warptile_wm = device->subgroup_size == 8 ? 8 : 32; - vk::BufferUsageFlags usage_flags = vk::BufferUsageFlagBits::eStorageBuffer | vk::BufferUsageFlagBits::eTransferSrc | vk::BufferUsageFlagBits::eTransferDst; - vk::MemoryAllocateFlags mem_flags {}; - if (device->buffer_device_address) { - usage_flags |= vk::BufferUsageFlagBits::eShaderDeviceAddress; - mem_flags |= vk::MemoryAllocateFlagBits::eDeviceAddress; - } + l_warptile = { 128, 128, 128, 16, mm_warp_8 * 2, 64, 2, tm_l, tn_l, tk_l, mm_warp_8 }; + m_warptile = { 128, 64, 64, 16, mm_warp_8, 32, 2, tm_m, tn_m, tk_m, mm_warp_8 }; + s_warptile = { subgroup_size_32, 32, 32, 16, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, subgroup_size_8 }; - vk::BufferCreateInfo buffer_create_info{ - vk::BufferCreateFlags(), - size, - usage_flags, - vk::SharingMode::eExclusive, - 0, - nullptr, - }; + l_warptile_mmq = { 128, 128, 128, 32, mm_warp_8 * 2, 64, 2, tm_l, tn_l, tk_l, mm_warp_8 }; + m_warptile_mmq = { 128, 64, 64, 32, mm_warp_8, 32, 2, tm_m, tn_m, tk_m, mm_warp_8 }; + s_warptile_mmq = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, subgroup_size_8 }; - vk::ExternalMemoryBufferCreateInfo external_memory_bci; - if (import_ptr) { - external_memory_bci.handleTypes = vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT; - buffer_create_info.setPNext(&external_memory_bci); - } + // Integer MMQ has a smaller shared memory profile, but heavier register use + l_warptile_mmq_int = { 128, 128, 128, 32, mm_warp_8 * 2, 64, 2, 4, 4, 1, mm_warp_8 }; + m_warptile_mmq_int = { 128, 64, 64, 32, mm_warp_8, 32, 2, 2, 2, 1, mm_warp_8 }; + s_warptile_mmq_int = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, 2, 1, 1, subgroup_size_8 }; - buf->buffer = device->device.createBuffer(buffer_create_info); + // K-quants use even more registers, mitigate by setting WMITER to 1 + l_warptile_mmq_int_k = { 128, 128, 128, 32, mm_warp_8 * 2, 64, 1, 4, 4, 1, mm_warp_8 }; + m_warptile_mmq_int_k = { 128, 64, 64, 32, mm_warp_8, 32, 1, 2, 2, 1, mm_warp_8 }; + s_warptile_mmq_int_k = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 1, 2, 1, 1, subgroup_size_8 }; - vk::MemoryRequirements mem_req = device->device.getBufferMemoryRequirements(buf->buffer); + l_warptile_id = { 128, 128, 128, 16, mul_mat_mm_warp_16 * 2, 64, 2, tm_l, tn_l, tk_l, mul_mat_mm_warp_16 }; + m_warptile_id = { 128, 64, 64, 16, mul_mat_mm_warp_16, 32, 2, tm_m, tn_m, tk_m, mul_mat_mm_warp_16 }; + s_warptile_id = { mul_mat_subgroup_size_16, 32, 32, 16, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, mul_mat_subgroup_size_16 }; - vk::PhysicalDeviceMemoryProperties mem_props = device->physical_device.getMemoryProperties(); + l_warptile_mmqid = { 128, 128, 128, 32, mul_mat_mm_warp_8 * 2, 64, 2, tm_l, tn_l, tk_l, mul_mat_mm_warp_8 }; + m_warptile_mmqid = { 128, 64, 64, 32, mul_mat_mm_warp_8, 32, 2, tm_m, tn_m, tk_m, mul_mat_mm_warp_8 }; + s_warptile_mmqid = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, mul_mat_subgroup_size_8 }; - const vk::MemoryPriorityAllocateInfoEXT mem_priority_info { 1.0f }; + l_warptile_mmqid_int = { 128, 128, 128, 32, mul_mat_mm_warp_8 * 2, 64, 2, 4, 4, 1, mul_mat_mm_warp_8 }; + m_warptile_mmqid_int = { 128, 64, 64, 32, mul_mat_mm_warp_8, 32, 2, 2, 2, 1, mul_mat_mm_warp_8 }; + s_warptile_mmqid_int = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, 2, 1, 1, mul_mat_subgroup_size_8 }; - vk::MemoryAllocateFlagsInfo mem_flags_info { mem_flags }; + l_warptile_mmqid_int_k = { 128, 128, 128, 32, mul_mat_mm_warp_16 * 2, 64, 1, 4, 4, 1, mul_mat_mm_warp_16 }; + m_warptile_mmqid_int_k = { 128, 64, 64, 32, mul_mat_mm_warp_16, 32, 1, 2, 2, 1, mul_mat_mm_warp_16 }; + s_warptile_mmqid_int_k = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 1, 2, 1, 1, mul_mat_subgroup_size_16 }; - if (device->memory_priority) { - mem_flags_info.setPNext(&mem_priority_info); - } + // chip specific tuning + if ((device->architecture == AMD_GCN) && (device->driver_id != vk::DriverId::eAmdProprietary)) { + m_warptile_mmq = m_warptile_mmq_int = { 256, 64, 64, 32, 16, 16, 2, 2, 2, 1, 16 }; + m_warptile_mmqid = m_warptile_mmqid_int = { 256, 64, 64, 32, 16, 16, 2, 2, 2, 1, 16 }; + } else if (device->vendor_id == VK_VENDOR_ID_AMD && device->coopmat_support && device->driver_id != vk::DriverId::eAmdProprietary) { + // This is intentionally using tx_m values, slight performance increase + l_warptile = { 256, 128, 128, 16, mm_warp_8, 64, 2, tm_m, tn_m, tk_m, mm_warp_8 }; + l_warptile_mmq = l_warptile_mmq_int = { 256, 128, 128, 32, mm_warp_8, 64, 2, tm_m, tn_m, tk_m, mm_warp_8 }; + l_warptile_mmq_int_k = { 256, 128, 128, 32, mm_warp_16, 64, 1, 4, 2, 1, mm_warp_16 }; + } - if (import_ptr) { - vk::MemoryHostPointerPropertiesEXT host_pointer_props; - try { - host_pointer_props = device->device.getMemoryHostPointerPropertiesEXT(vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT, import_ptr); - } catch (vk::SystemError& e) { - GGML_LOG_WARN("ggml_vulkan: Failed getMemoryHostPointerPropertiesEXT (%s)\n", e.what()); - device->device.destroyBuffer(buf->buffer); - return {}; - } - vk::PhysicalDeviceMemoryProperties mem_props = device->physical_device.getMemoryProperties(); - - uint32_t memory_type_idx; - vk::MemoryPropertyFlags property_flags = *req_flags_list.begin(); - for (memory_type_idx = 0; memory_type_idx < 32; ++memory_type_idx) { - if (!(host_pointer_props.memoryTypeBits & (1u << memory_type_idx))) { - continue; - } - if (!(mem_req.memoryTypeBits & (1u << memory_type_idx))) { - continue; - } + l_mmq_wg_denoms = l_wg_denoms = {128, 128, 1 }; + m_mmq_wg_denoms = m_wg_denoms = { 64, 64, 1 }; + s_mmq_wg_denoms = s_wg_denoms = { 32, 32, 1 }; + l_align = 128; + m_align = 64; + s_align = 32; - vk::MemoryType memory_type = mem_props.memoryTypes[memory_type_idx]; - // check for visible+coherent+cached. Other flags (e.g. devicelocal) are allowed - if ((memory_type.propertyFlags & property_flags) == property_flags) { - property_flags = memory_type.propertyFlags; - break; + if (device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support) { + // Xe1/Xe2/Xe3 with coopmat enabled - warptile performance tuning + l_warptile = { 512, 128, 128, 16, mm_warp_8, 32, 2, tm_l, tn_l, tk_l, mm_warp_8 }; + if (device->architecture == INTEL_XE1) { + l_warptile_mmq = { 512, 256, 128, 32, 32, 32, 2, tm_l, tn_l, tk_l, 16 }; + l_mmq_wg_denoms = { 256, 128, 1 }; + l_align = 32; //set as BK + } else { + l_warptile_mmq = { 512, 128, 256, 32, 32, 32, 2, tm_l, tn_l, tk_l, 16 }; + l_mmq_wg_denoms = { 128, 256, 1 }; + l_align = 32; //set as BK } } - if (memory_type_idx == 32) { - GGML_LOG_WARN("ggml_vulkan: Memory type for host allocation not found\n"); - device->device.destroyBuffer(buf->buffer); - return {}; - } - - buf->memory_property_flags = mem_props.memoryTypes[memory_type_idx].propertyFlags; - try { - vk::ImportMemoryHostPointerInfoEXT import_info; - import_info.handleType = vk::ExternalMemoryHandleTypeFlagBits::eHostAllocationEXT; - import_info.pHostPointer = import_ptr; - import_info.setPNext(&mem_flags_info); - buf->device_memory = device->device.allocateMemory({ size, memory_type_idx, &import_info }); - } catch (const vk::SystemError& e) { - } - } else { - for (auto it = req_flags_list.begin(); it != req_flags_list.end(); it++) { - const auto & req_flags = *it; - const std::vector memory_type_indices = ggml_vk_find_memory_properties(&mem_props, &mem_req, req_flags); + for (uint32_t i = 0; i < GGML_TYPE_COUNT; ++i) { + ggml_type t = (ggml_type)i; + // Disable medium and large matrix multiplication if not enough shared memory is available + // Check mmq warptiles as the largest configuration + // Throw an error if not enough for any matrix multiplication is available + if (!ggml_vk_matmul_shmem_support(device, s_warptile_mmq, false, t)) { + std::cerr << "ggml_vulkan: Error: Shared memory size too small for matrix multiplication." << std::endl; + throw std::runtime_error("Shared memory size too small for matrix multiplication."); + } else if (!ggml_vk_matmul_shmem_support(device, m_warptile_mmq, false, t)) { + device->mul_mat_m[i] = false; + device->mul_mat_l[i] = false; + } else if (!ggml_vk_matmul_shmem_support(device, l_warptile_mmq, false, t)) { + device->mul_mat_l[i] = false; + } - if (memory_type_indices.empty()) { - continue; + // Disable mul_mat_id if not enough shared memory is available + if (!ggml_vk_matmul_shmem_support(device, s_warptile_mmqid, true, t)) { + device->mul_mat_id_s[i] = false; + device->mul_mat_id_m[i] = false; + device->mul_mat_id_l[i] = false; + } else if (!ggml_vk_matmul_shmem_support(device, m_warptile_mmqid, true, t)) { + device->mul_mat_id_m[i] = false; + device->mul_mat_id_l[i] = false; + } else if (!ggml_vk_matmul_shmem_support(device, l_warptile_mmqid, true, t)) { + device->mul_mat_id_l[i] = false; } - bool done = false; + // The q8_1 mmq path has its own (larger) shmem layout, check it separately. + // K-quants and IQ3_S use the _int_k warptiles, others use _int. + const bool is_k_quant = (t == GGML_TYPE_Q2_K || t == GGML_TYPE_Q3_K || + t == GGML_TYPE_Q4_K || t == GGML_TYPE_Q5_K || + t == GGML_TYPE_Q6_K || t == GGML_TYPE_IQ3_S); + const auto & s_int = is_k_quant ? s_warptile_mmq_int_k : s_warptile_mmq_int; + const auto & m_int = is_k_quant ? m_warptile_mmq_int_k : m_warptile_mmq_int; + const auto & l_int = is_k_quant ? l_warptile_mmq_int_k : l_warptile_mmq_int; + const auto & s_intid = is_k_quant ? s_warptile_mmqid_int_k : s_warptile_mmqid_int; + const auto & m_intid = is_k_quant ? m_warptile_mmqid_int_k : m_warptile_mmqid_int; + const auto & l_intid = is_k_quant ? l_warptile_mmqid_int_k : l_warptile_mmqid_int; - for (auto mtype_it = memory_type_indices.begin(); mtype_it != memory_type_indices.end(); mtype_it++) { - try { - buf->device_memory = device->device.allocateMemory({ mem_req.size, *mtype_it, &mem_flags_info }); - buf->memory_property_flags = mem_props.memoryTypes[*mtype_it].propertyFlags; - done = true; - break; - } catch (const vk::SystemError& e) { - // loop and retry - // during last attempt throw the exception - if (it + 1 == req_flags_list.end() && mtype_it + 1 == memory_type_indices.end()) { - device->device.destroyBuffer(buf->buffer); - throw e; - } - } + if (!ggml_vk_matmul_int_shmem_support(device, s_int, false, t)) { + device->mul_mat_s_int[i] = false; + device->mul_mat_m_int[i] = false; + device->mul_mat_l_int[i] = false; + } else if (!ggml_vk_matmul_int_shmem_support(device, m_int, false, t)) { + device->mul_mat_m_int[i] = false; + device->mul_mat_l_int[i] = false; + } else if (!ggml_vk_matmul_int_shmem_support(device, l_int, false, t)) { + device->mul_mat_l_int[i] = false; } - if (done) { - break; + if (!ggml_vk_matmul_int_shmem_support(device, s_intid, true, t)) { + device->mul_mat_id_s_int[i] = false; + device->mul_mat_id_m_int[i] = false; + device->mul_mat_id_l_int[i] = false; + } else if (!ggml_vk_matmul_int_shmem_support(device, m_intid, true, t)) { + device->mul_mat_id_m_int[i] = false; + device->mul_mat_id_l_int[i] = false; + } else if (!ggml_vk_matmul_int_shmem_support(device, l_intid, true, t)) { + device->mul_mat_id_l_int[i] = false; } } } - if (!buf->device_memory) { - device->device.destroyBuffer(buf->buffer); - throw vk::OutOfDeviceMemoryError("No suitable memory type found"); - } - - buf->ptr = nullptr; + auto const &ggml_vk_create_pipeline = [&](vk_device& device, vk_pipeline& base_pipeline, const char *name, size_t spv_size, const void* spv_data, const char *entrypoint, + uint32_t parameter_count, uint32_t push_constant_size, std::array wg_denoms, const std::vector& specialization_constants, + uint32_t align, bool disable_robustness = false, bool require_full_subgroups = false, uint32_t required_subgroup_size = 0) { - if (import_ptr) { - buf->ptr = import_ptr; - } else { - if (buf->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { - buf->ptr = device->device.mapMemory(buf->device_memory, 0, VK_WHOLE_SIZE); + if (!require_full_subgroups && required_subgroup_size == 0) { + required_subgroup_size = get_subgroup_size(name, device->architecture); } - } - device->device.bindBufferMemory(buf->buffer, buf->device_memory, 0); + vk_pipeline *ptr = &base_pipeline; - buf->device = device; - buf->size = size; + int num_pipelines = 1; +#if defined(VK_EXT_shader_64bit_indexing) + if (device->shader_64b_indexing) { + num_pipelines = 2; + } +#endif + for (int i = 0; i < num_pipelines; ++i, ptr = &(*ptr)->next) { + vk_pipeline &pipeline = *ptr; + if (!pipeline) { + pipeline = std::make_shared(); + } + if (!pipeline->initialized) { + pipeline->name = name; + pipeline->parameter_count = parameter_count; + pipeline->push_constant_size = push_constant_size; + pipeline->wg_denoms = wg_denoms; + pipeline->align = align; + pipeline->initialized = true; +#if defined(VK_EXT_shader_64bit_indexing) + pipeline->is_64b_indexing = (i == 1); +#endif + } - if (device->buffer_device_address) { - const vk::BufferDeviceAddressInfo addressInfo(buf->buffer); - buf->bda_addr = device->device.getBufferAddress(addressInfo); - } + // We only care about the pipeline this call asked for; the rest + // (including the 64-bit indexing variant) are handled by their + // own request_descriptor_sets / load_shaders calls. + if (pipeline.get() != requested.get()) { + continue; + } - device->memory_logger->log_allocation(buf, size); + if (pipeline->compiled) { + continue; + } - return buf; -} + wait_pipeline = pipeline; -static vk_buffer ggml_vk_create_buffer_check(vk_device& device, size_t size, vk::MemoryPropertyFlags req_flags, vk::MemoryPropertyFlags fallback_flags = vk::MemoryPropertyFlags(0)) { - try { - return ggml_vk_create_buffer(device, size, {req_flags, fallback_flags}); - } catch (const vk::SystemError& e) { - std::cerr << "ggml_vulkan: Memory allocation of size " << size << " failed." << std::endl; - std::cerr << "ggml_vulkan: " << e.what() << std::endl; - throw e; - } -} + if (!pipeline->compile_pending) { + pipeline->compile_pending = true; + claimed_task.pipeline = pipeline; + claimed_task.spv_size = spv_size; + claimed_task.spv_data = spv_data; + claimed_task.entrypoint = entrypoint; + claimed_task.parameter_count = parameter_count; + claimed_task.wg_denoms = wg_denoms; + claimed_task.specialization_constants = specialization_constants; + claimed_task.disable_robustness = disable_robustness; + claimed_task.require_full_subgroups = require_full_subgroups; + claimed_task.required_subgroup_size = required_subgroup_size; + has_claimed_task = true; + } + } + }; -static vk_buffer ggml_vk_create_buffer_device(vk_device& device, size_t size) { - vk_buffer buf; - try { - if (device->prefer_host_memory) { - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, - vk::MemoryPropertyFlagBits::eDeviceLocal}); - } else if (device->uma) { - // On UMA, prefer host-visible memory so direct tensor borrowing works. - // If unavailable, fall back to device-local memory. - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, - vk::MemoryPropertyFlagBits::eDeviceLocal, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); - } else if (device->disable_host_visible_vidmem) { - if (device->allow_sysmem_fallback) { - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + auto const &ggml_vk_create_pipeline2 = [&](vk_device& device, vk_pipeline& pipeline, const std::string &name, size_t spv_size, const void* spv_data, const char *entrypoint, + uint32_t parameter_count, uint32_t push_constant_size, std::array wg_denoms, const std::vector& specialization_constants, + uint32_t align, bool disable_robustness = false, bool require_full_subgroups = false, uint32_t required_subgroup_size = 0) { + return ggml_vk_create_pipeline(device, pipeline, name.c_str(), spv_size, spv_data, entrypoint, + parameter_count, push_constant_size, wg_denoms, specialization_constants, + align, disable_robustness, require_full_subgroups, required_subgroup_size); + }; + + // FA scalar has two SPIR-V modules (MMQ vs non-MMQ); FA cm1 has one. K/V + // quant type is selected at runtime via the FaTypeK / FaTypeV spec constants. + + for (auto &fa : device->pipeline_flash_attn_f32_f16) { + if (fa.first.path != FA_SCALAR) continue; + const uint32_t Br = fa.first.Br; + const uint32_t Bc = fa.first.Bc; + const bool aligned = fa.first.aligned; + const bool f32acc = fa.first.f32acc; + const uint32_t fa_sgs = fa.first.subgroup_size; + const bool fa_ds = fa.first.subgroup_size == 0; + + const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; + const bool use_mmq = ggml_vk_fa_scalar_uses_mmq(device, fa.first.k_type, fa.first.v_type); + const void * spv_data = nullptr; + size_t spv_size = 0; + const char *name = nullptr; + if (bf16_kv) { + spv_data = flash_attn_f32_f16_fp32_data; + spv_size = flash_attn_f32_f16_fp32_len; + name = aligned ? "flash_attn_f32_bf16_aligned" : "flash_attn_f32_bf16"; + } else if (use_mmq) { +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->fp16) { + if (f32acc) { spv_data = flash_attn_f32_f16_int8_data; spv_size = flash_attn_f32_f16_int8_len; } + else { spv_data = flash_attn_f32_f16_f16acc_int8_data; spv_size = flash_attn_f32_f16_f16acc_int8_len; } } else { - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal}); + spv_data = flash_attn_f32_f16_fp32_int8_data; + spv_size = flash_attn_f32_f16_fp32_int8_len; } +#endif + name = aligned ? "flash_attn_f32_f16_aligned" : "flash_attn_f32_f16"; } else { - // use rebar if available, otherwise fallback to device only visible memory - if (device->allow_sysmem_fallback) { - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, - vk::MemoryPropertyFlagBits::eDeviceLocal, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); + if (device->fp16) { + if (device->dot2_f16) { + if (f32acc) { spv_data = flash_attn_f32_f16_dot2_data; spv_size = flash_attn_f32_f16_dot2_len; } + else { spv_data = flash_attn_f32_f16_dot2_f16acc_data; spv_size = flash_attn_f32_f16_dot2_f16acc_len; } + } else { + if (f32acc) { spv_data = flash_attn_f32_f16_data; spv_size = flash_attn_f32_f16_len; } + else { spv_data = flash_attn_f32_f16_f16acc_data; spv_size = flash_attn_f32_f16_f16acc_len; } + } } else { - buf = ggml_vk_create_buffer(device, size, {vk::MemoryPropertyFlagBits::eDeviceLocal | vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent, - vk::MemoryPropertyFlagBits::eDeviceLocal}); + spv_data = flash_attn_f32_f16_fp32_data; + spv_size = flash_attn_f32_f16_fp32_len; } + name = aligned ? "flash_attn_f32_f16_aligned" : "flash_attn_f32_f16"; } - } catch (const vk::SystemError& e) { - std::cerr << "ggml_vulkan: Device memory allocation of size " << size << " failed." << std::endl; - std::cerr << "ggml_vulkan: " << e.what() << std::endl; - throw e; + ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 8, + sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, + get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, + !fa_ds, !fa_ds ? fa_sgs : 0); } - return buf; -} +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + if (device->coopmat1_fa_support) { + for (auto &fa : device->pipeline_flash_attn_f32_f16) { + if (fa.first.path != FA_COOPMAT1) continue; + const uint32_t Br = fa.first.Br; + const uint32_t Bc = fa.first.Bc; + const bool aligned = fa.first.aligned; + const bool f32acc = fa.first.f32acc; + const uint32_t fa_sgs = fa.first.subgroup_size; + const bool fa_ds = fa.first.subgroup_size == 0; -static void ggml_vk_destroy_buffer(vk_buffer& buf) { - if (buf == nullptr) { - return; - } + const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; - if (buf->device != nullptr) { - buf->device->memory_logger->log_deallocation(buf); + const void * spv_data; + size_t spv_size; + const char *name; + if (bf16_kv) { +#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (!device->coopmat_bf16_support) continue; + spv_data = flash_attn_f32_f16_bf16_cm1_data; + spv_size = flash_attn_f32_f16_bf16_cm1_len; + name = aligned ? "flash_attn_f32_bf16_aligned_cm1" : "flash_attn_f32_bf16_cm1"; +#else + continue; +#endif + } else { + if (f32acc) { spv_data = flash_attn_f32_f16_cm1_data; spv_size = flash_attn_f32_f16_cm1_len; } + else { spv_data = flash_attn_f32_f16_f16acc_cm1_data; spv_size = flash_attn_f32_f16_f16acc_cm1_len; } + name = aligned ? "flash_attn_f32_f16_aligned_cm1" : "flash_attn_f32_f16_cm1"; + } + ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 8, + sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, + get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, + !fa_ds, !fa_ds ? fa_sgs : 0); + } } +#endif - buf.reset(); -} +#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (device->coopmat2) { + for (auto &fa : device->pipeline_flash_attn_f32_f16) { + if (fa.first.path != FA_COOPMAT2) continue; + const uint32_t Br = fa.first.Br; + const uint32_t Bc = fa.first.Bc; + const bool aligned = fa.first.aligned; + const bool f32acc = fa.first.f32acc; -static vk_subbuffer ggml_vk_subbuffer(const ggml_backend_vk_context* ctx, const vk_buffer& buf, size_t offset = 0) { - return { buf, offset, ggml_vk_get_max_buffer_range(ctx, buf, offset) }; -} + const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; + const void * spv_data; + size_t spv_size; + const char * name; + if (bf16_kv) { +#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (!device->coopmat2_bf16_support) continue; + spv_data = flash_attn_f32_f16_bf16_cm2_data; + spv_size = flash_attn_f32_f16_bf16_cm2_len; + name = aligned ? "flash_attn_f32_bf16_aligned_cm2" : "flash_attn_f32_bf16_cm2"; +#else + continue; +#endif + } else if (aligned) { + if (f32acc) { spv_data = flash_attn_f32_f16_cm2_data; spv_size = flash_attn_f32_f16_cm2_len; name = "flash_attn_f32_f16_aligned_f32acc_cm2"; } + else { spv_data = flash_attn_f32_f16_f16acc_cm2_data; spv_size = flash_attn_f32_f16_f16acc_cm2_len; name = "flash_attn_f32_f16_aligned_f16acc_cm2"; } + } else { + if (f32acc) { spv_data = flash_attn_f32_f16_cm2_data; spv_size = flash_attn_f32_f16_cm2_len; name = "flash_attn_f32_f16_f32acc_cm2"; } + else { spv_data = flash_attn_f32_f16_f16acc_cm2_data; spv_size = flash_attn_f32_f16_f16acc_cm2_len; name = "flash_attn_f32_f16_f16acc_cm2"; } + } + ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 8, + sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, + get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, false, 0); + } + } +#endif -static void ggml_vk_sync_buffers(ggml_backend_vk_context* ctx, vk_context& subctx) { - VK_LOG_DEBUG("ggml_vk_sync_buffers()"); + auto const &ggml_vk_mul_mm_spec = [&device](std::vector spec, bool aligned) { + spec.push_back(aligned ? 1u : 0u); // constantID=11: ALIGNED + if (device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support && + device->driver_id == vk::DriverId::eIntelProprietaryWindows) { + spec.push_back(0u); // constantID=12: SHMEM_STRIDE_PAD = 0 + spec.push_back(1u); // constantID=13: APPLY_SLM_A_RESHAPE = true + } + return spec; + }; - const bool transfer_queue = subctx->p->q->transfer_only; + auto const &ggml_vk_mul_mm_spec_quant = [&device](std::vector spec, bool aligned, uint32_t type) { + spec.push_back(aligned ? 1u : 0u); // constantID=11: ALIGNED + spec.push_back(type); // constantID=12: MmTypeA + if (device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support && + device->driver_id == vk::DriverId::eIntelProprietaryWindows) { + spec.push_back(0u); // constantID=13: SHMEM_STRIDE_PAD = 0 + spec.push_back(1u); // constantID=14: APPLY_SLM_A_RESHAPE = true + } + return spec; + }; - if (ctx) { - ctx->prealloc_x_need_sync = ctx->prealloc_y_need_sync = ctx->prealloc_split_k_need_sync = false; - } + static const ggml_type non_lut_quant_types[] = { + GGML_TYPE_Q1_0, GGML_TYPE_Q2_0, GGML_TYPE_Q4_0, GGML_TYPE_Q4_1, GGML_TYPE_Q5_0, GGML_TYPE_Q5_1, GGML_TYPE_Q8_0, + GGML_TYPE_Q2_K, GGML_TYPE_Q3_K, GGML_TYPE_Q4_K, GGML_TYPE_Q5_K, GGML_TYPE_Q6_K, GGML_TYPE_TQ1_0, GGML_TYPE_TQ2_0, + }; - subctx->s->buffer->buf.pipelineBarrier( - subctx->p->q->stage_flags, - subctx->p->q->stage_flags, - {}, - { { - { !transfer_queue ? (vk::AccessFlagBits::eShaderRead | vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) : (vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) }, - { !transfer_queue ? (vk::AccessFlagBits::eShaderRead | vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) : (vk::AccessFlagBits::eTransferRead | vk::AccessFlagBits::eTransferWrite) } - } }, - {}, - {} - ); -} +#define FOR_EACH_LUT_TYPE_NONFP4(X) \ + X(GGML_TYPE_IQ1_S, iq1_s) \ + X(GGML_TYPE_IQ1_M, iq1_m) \ + X(GGML_TYPE_IQ2_XXS, iq2_xxs) \ + X(GGML_TYPE_IQ2_XS, iq2_xs) \ + X(GGML_TYPE_IQ2_S, iq2_s) \ + X(GGML_TYPE_IQ3_XXS, iq3_xxs) \ + X(GGML_TYPE_IQ3_S, iq3_s) \ + X(GGML_TYPE_IQ4_XS, iq4_xs) \ + X(GGML_TYPE_IQ4_NL, iq4_nl) +#define FOR_EACH_LUT_FP4_TYPE(X) \ + X(GGML_TYPE_MXFP4, mxfp4) \ + X(GGML_TYPE_NVFP4, nvfp4) +#define FOR_EACH_LUT_TYPE(X) \ + FOR_EACH_LUT_TYPE_NONFP4(X) \ + FOR_EACH_LUT_FP4_TYPE(X) -static void ggml_vk_reset_event(vk_context& ctx, vk::Event& event) { - VK_LOG_DEBUG("ggml_vk_set_event()"); + const int mul_mat_id_param_count = 5; - ctx->s->buffer->buf.resetEvent( - event, - ctx->p->q->stage_flags - ); -} + using spec_fn_t = std::function(const std::vector&, bool)>; + auto const &create_mm_pipelines = [&]( + const vk_matmul_pipeline_key& key, + const std::vector& tile_configs, + const std::string& shader_name, size_t spv_len, const void* spv_data, + uint32_t push_constant_size, uint32_t param_count, + const spec_fn_t& spec_fn, + bool disable_robustness = false, bool require_full_subgroups = false, uint32_t required_subgroup_size = 0, + bool create_aligned = true, bool pin_subgroup_to_warp = false + ) { + auto& vec = device->pipeline_matmul[key]; + const bool first_call = vec.empty(); + for (size_t i = 0; i < tile_configs.size(); i++) { + const auto& tc = tile_configs[i]; + + // Intel coopmat1 pins the required subgroup size to each warptile's WARP element. + const uint32_t rsgs = pin_subgroup_to_warp ? tc.warptile[WARP_SIZE_IDX] : required_subgroup_size; + const bool rfs = require_full_subgroups || pin_subgroup_to_warp; + + if (first_call) { + vk_matmul_pipeline_pair pair{}; + pair.align = tc.align; + std::string suffix = "_" + std::to_string(i); + pair.unaligned = std::make_shared(); + if (create_aligned) { + pair.aligned = std::make_shared(); + } + vec.push_back(pair); + } -static void ggml_vk_set_event(vk_context& ctx, vk::Event& event) { - VK_LOG_DEBUG("ggml_vk_set_event()"); + ggml_vk_create_pipeline(device, vec[i].unaligned, + vec[i].unaligned->name.empty() ? (shader_name + "_" + std::to_string(i)).c_str() : vec[i].unaligned->name.c_str(), + spv_len, spv_data, "main", param_count, push_constant_size, + tc.wg_denoms, spec_fn(tc.warptile, false), 1, + disable_robustness, rfs, rsgs); - ctx->s->buffer->buf.setEvent( - event, - ctx->p->q->stage_flags - ); -} + if (vec[i].aligned) { + ggml_vk_create_pipeline(device, vec[i].aligned, + vec[i].aligned->name.empty() ? (shader_name + "_aligned_" + std::to_string(i)).c_str() : vec[i].aligned->name.c_str(), + spv_len, spv_data, "main", param_count, push_constant_size, + tc.wg_denoms, spec_fn(tc.warptile, true), tc.align, + disable_robustness, rfs, rsgs); + } + } + }; -static void ggml_vk_wait_events(vk_context& ctx, std::vector&& events) { - VK_LOG_DEBUG("ggml_vk_wait_events()"); - if (events.empty()) { - return; - } + auto filter_tc = [&](const std::vector& configs, ggml_type type, bool is_id, bool is_int = false) -> std::vector { + std::vector result; + bool enabled[3]; + if (is_int) { + enabled[0] = is_id ? device->mul_mat_id_s_int[type] : device->mul_mat_s_int[type]; + enabled[1] = is_id ? device->mul_mat_id_m_int[type] : device->mul_mat_m_int[type]; + enabled[2] = is_id ? device->mul_mat_id_l_int[type] : device->mul_mat_l_int[type]; + } else { + enabled[0] = is_id ? device->mul_mat_id_s[type] : device->mul_mat_s[type]; + enabled[1] = is_id ? device->mul_mat_id_m[type] : device->mul_mat_m[type]; + enabled[2] = is_id ? device->mul_mat_id_l[type] : device->mul_mat_l[type]; + } + for (size_t i = 0; i < configs.size() && i < 3; i++) { + if (enabled[i]) result.push_back(configs[i]); + } + return result; + }; - ctx->s->buffer->buf.waitEvents( - events, - ctx->p->q->stage_flags, - ctx->p->q->stage_flags, - {}, - {}, - {} - ); -} + std::vector tc_mm = {{s_warptile, s_wg_denoms, s_align}, {m_warptile, m_wg_denoms, m_align}, {l_warptile, l_wg_denoms, l_align}}; + std::vector tc_mmq = {{s_warptile_mmq, s_mmq_wg_denoms, s_align}, {m_warptile_mmq, m_mmq_wg_denoms, m_align}, {l_warptile_mmq, l_mmq_wg_denoms, l_align}}; -struct vk_fa_tuning_params { - FaCodePath path; - uint32_t workgroup_size; - uint32_t subgroup_size; - uint32_t block_rows; - uint32_t block_cols; - uint32_t d_split; - uint32_t row_split; - bool shmem_staging; - bool disable_subgroups; - uint32_t limit_occupancy_shmem; - - void print() const { - std::cerr << "path=" << path << " workgroup_size=" << workgroup_size << " subgroup_size=" << subgroup_size << - " block_rows=" << block_rows << " block_cols=" << block_cols << " d_split=" << d_split << - " row_split=" << row_split << " shmem_staging=" << shmem_staging << " disable_subgroups=" << disable_subgroups << - " limit_occupancy_shmem=" << limit_occupancy_shmem << std::endl; - } -}; +#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (device->coopmat2) { + auto const &ggml_vk_mul_mm_cm2_spec = [&](std::vector spec, bool aligned, uint32_t type = UINT32_MAX) { + spec.push_back(aligned ? 1u : 0u); // ALIGNED + spec.push_back(device->subgroup_size); // subgroup_size + if (type != UINT32_MAX) { + spec.push_back(type); // MmTypeA + spec.push_back((uint32_t)ggml_type_size((ggml_type)type)); // MmABlockBytes + } + return spec; + }; -static bool ggml_vk_flash_attn_scalar_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type); -static bool ggml_vk_flash_attn_coopmat_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type = GGML_TYPE_F16, ggml_type v_type = GGML_TYPE_F16); + std::vector tc_mmq_k = {{s_warptile_mmq_k, s_mmq_wg_denoms_k, s_align}, {m_warptile_mmq_k, m_mmq_wg_denoms_k, m_align}, {l_warptile_mmq_k, l_mmq_wg_denoms_k, l_align}}; + std::vector tc_mmqid = {{s_warptile_mmqid, s_mmqid_wg_denoms, s_align}, {m_warptile_mmqid, m_mmqid_wg_denoms, m_align}, {l_warptile_mmqid, l_mmqid_wg_denoms, l_align}}; -static vk_fa_tuning_params get_fa_tuning_params_scalar(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { + spec_fn_t cm2_spec = [&](const std::vector& wt, bool a) { return ggml_vk_mul_mm_cm2_spec(wt, a); }; - vk_fa_tuning_params result{}; - result.path = FA_SCALAR; + // F16 x F16 + create_mm_pipelines({GGML_TYPE_F16, GGML_TYPE_F16, false, true}, tc_mm, "matmul_f16_f16acc", matmul_f16_f16acc_cm2_len, matmul_f16_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_F16, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f16", matmul_f16_cm2_len, matmul_f16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (device->coopmat_bf16_support) { + create_mm_pipelines({GGML_TYPE_BF16, GGML_TYPE_BF16, false, false}, tc_mm, "matmul_bf16", matmul_bf16_cm2_len, matmul_bf16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); + } +#endif + for (const auto type : non_lut_quant_types) { + // regression in unified shader on Ampere + if (type == GGML_TYPE_Q4_K || type == GGML_TYPE_Q5_K) { + continue; + } + auto& tc = ((type >= GGML_TYPE_Q2_K && type <= GGML_TYPE_Q6_K) || type == GGML_TYPE_TQ1_0 || type == GGML_TYPE_TQ2_0) ? tc_mmq_k : tc_mmq; + spec_fn_t qs = [&, type](const std::vector& wt, bool a) { return ggml_vk_mul_mm_cm2_spec(wt, a, (uint32_t)type); }; + create_mm_pipelines({type, GGML_TYPE_F16, false, true}, tc, "matmul_quant_f16_f16acc", matmul_quant_f16_f16acc_cm2_len, matmul_quant_f16_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, qs, true); + create_mm_pipelines({type, GGML_TYPE_F16, false, false}, tc, "matmul_quant_f16", matmul_quant_f16_cm2_len, matmul_quant_f16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, qs, true); + } + create_mm_pipelines({GGML_TYPE_Q4_K, GGML_TYPE_F16, false, true}, tc_mmq_k, "matmul_q4_k_f16_f16acc", matmul_q4_k_f16_f16acc_cm2_len, matmul_q4_k_f16_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q4_K, GGML_TYPE_F16, false, false}, tc_mmq_k, "matmul_q4_k_f16", matmul_q4_k_f16_cm2_len, matmul_q4_k_f16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q5_K, GGML_TYPE_F16, false, true}, tc_mmq_k, "matmul_q5_k_f16_f16acc", matmul_q5_k_f16_f16acc_cm2_len, matmul_q5_k_f16_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q5_K, GGML_TYPE_F16, false, false}, tc_mmq_k, "matmul_q5_k_f16", matmul_q5_k_f16_cm2_len, matmul_q5_k_f16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); +#define X_CM2(TYPE, tstr) \ + { auto tc = filter_tc(tc_mmq, TYPE, false); \ + if (!tc.empty()) { \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, false, true}, tc, "matmul_" #tstr "_f16_f16acc", matmul_##tstr##_f16_f16acc_cm2_len, matmul_##tstr##_f16_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, false, false}, tc, "matmul_" #tstr "_f16", matmul_##tstr##_f16_cm2_len, matmul_##tstr##_f16_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); \ + } } + FOR_EACH_LUT_TYPE_NONFP4(X_CM2) +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + if (device->ocp_fp4) { +#define X_CM2_OCP(TYPE, tstr) \ + { auto tc = filter_tc(tc_mmq, TYPE, false); \ + if (!tc.empty()) { \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, false, true}, tc, "matmul_" #tstr "_f16_ocp_f16acc", matmul_##tstr##_f16_ocp_f16acc_cm2_len, matmul_##tstr##_f16_ocp_f16acc_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, false, false}, tc, "matmul_" #tstr "_f16_ocp", matmul_##tstr##_f16_ocp_cm2_len, matmul_##tstr##_f16_ocp_cm2_data, sizeof(vk_mat_mat_push_constants), 3, cm2_spec, true); \ + } } + FOR_EACH_LUT_FP4_TYPE(X_CM2_OCP) +#undef X_CM2_OCP + } else +#endif + { + FOR_EACH_LUT_FP4_TYPE(X_CM2) + } +#undef X_CM2 - if (device->vendor_id == VK_VENDOR_ID_INTEL) { - // Disable subgroup use due to performance issues when enforcing subgroup sizes - result.subgroup_size = 32; - result.disable_subgroups = true; - } else if (device->vendor_id == VK_VENDOR_ID_AMD && device->architecture != AMD_GCN) { - result.subgroup_size = n_rows < 4 ? 32 : device->subgroup_size; - } else { - result.subgroup_size = device->subgroup_size; - } + GGML_ASSERT(device->subgroup_ballot); - // Row split splits the workgroup so that synchronization only has to happen within subgroups, which avoids barriers - uint32_t row_split_max_hsk = 64; - if (device->vendor_id == VK_VENDOR_ID_AMD && device->architecture != AMD_GCN && !device->uma) { - row_split_max_hsk = n_rows <= 8 ? 64 : 128; - } - result.row_split = (n_rows < 4 || hsk <= row_split_max_hsk) ? 1 : 4; - - if (result.subgroup_size > 32 && (n_rows < 4 || hsk < (result.row_split == 1 ? 128 : 64))) { - result.workgroup_size = result.subgroup_size * 2; - } else { - result.workgroup_size = result.subgroup_size * 4; - } - - const uint32_t D = hsk | hsv; - - const bool reduce_block_rows = D & 8 || n_kv < 1024 || device->vendor_id == VK_VENDOR_ID_INTEL; - - if (n_rows == 1) { - result.block_rows = 1; - result.block_cols = 64; - } else { - // row_split 1 means higher register use per row, so block size has to be adjusted - if (result.row_split == 1) { - result.block_rows = n_rows == 2 ? 2 : ((n_rows <= 4 || reduce_block_rows) ? 4 : 8); - } else { - result.block_rows = n_rows <= 4 ? 4 : ((n_rows <= 8 || reduce_block_rows) ? 8 : 16); + create_mm_pipelines({GGML_TYPE_F16, GGML_TYPE_F16, true, true}, tc_mm, "matmul_id_subgroup_f16_f16acc", matmul_id_subgroup_f16_f16acc_cm2_len, matmul_id_subgroup_f16_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_mm, "matmul_id_subgroup_f16", matmul_id_subgroup_f16_cm2_len, matmul_id_subgroup_f16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (device->coopmat_bf16_support) { + create_mm_pipelines({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_mm, "matmul_id_subgroup_bf16", matmul_id_subgroup_bf16_cm2_len, matmul_id_subgroup_bf16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); } +#endif + for (const auto type : non_lut_quant_types) { + if (type == GGML_TYPE_Q4_K || type == GGML_TYPE_Q5_K) { + continue; + } + spec_fn_t qs_id = [&, type](const std::vector& wt, bool a) { return ggml_vk_mul_mm_cm2_spec(wt, a, (uint32_t)type); }; + create_mm_pipelines({type, GGML_TYPE_F16, true, true}, tc_mmqid, "matmul_id_subgroup_quant_f16_f16acc", matmul_id_subgroup_quant_f16_f16acc_cm2_len, matmul_id_subgroup_quant_f16_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, qs_id, true); + create_mm_pipelines({type, GGML_TYPE_F16, true, false}, tc_mmqid, "matmul_id_subgroup_quant_f16", matmul_id_subgroup_quant_f16_cm2_len, matmul_id_subgroup_quant_f16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, qs_id, true); + } + create_mm_pipelines({GGML_TYPE_Q4_K, GGML_TYPE_F16, true, true}, tc_mmqid, "matmul_id_subgroup_q4_k_f16_f16acc", matmul_id_subgroup_q4_k_f16_f16acc_cm2_len, matmul_id_subgroup_q4_k_f16_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q4_K, GGML_TYPE_F16, true, false}, tc_mmqid, "matmul_id_subgroup_q4_k_f16", matmul_id_subgroup_q4_k_f16_cm2_len, matmul_id_subgroup_q4_k_f16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q5_K, GGML_TYPE_F16, true, true}, tc_mmqid, "matmul_id_subgroup_q5_k_f16_f16acc", matmul_id_subgroup_q5_k_f16_f16acc_cm2_len, matmul_id_subgroup_q5_k_f16_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); + create_mm_pipelines({GGML_TYPE_Q5_K, GGML_TYPE_F16, true, false}, tc_mmqid, "matmul_id_subgroup_q5_k_f16", matmul_id_subgroup_q5_k_f16_cm2_len, matmul_id_subgroup_q5_k_f16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); +#define X_CM2_ID(TYPE, tstr) \ + { auto tc = filter_tc(tc_mmqid, TYPE, true); \ + if (!tc.empty()) { \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, true, true}, tc, "matmul_id_subgroup_" #tstr "_f16_f16acc", matmul_id_subgroup_##tstr##_f16_f16acc_cm2_len, matmul_id_subgroup_##tstr##_f16_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, true, false}, tc, "matmul_id_subgroup_" #tstr "_f16", matmul_id_subgroup_##tstr##_f16_cm2_len, matmul_id_subgroup_##tstr##_f16_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); \ + } } + FOR_EACH_LUT_TYPE_NONFP4(X_CM2_ID) +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + if (device->ocp_fp4) { +#define X_CM2_ID_OCP(TYPE, tstr) \ + { auto tc = filter_tc(tc_mmqid, TYPE, true); \ + if (!tc.empty()) { \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, true, true}, tc, "matmul_id_subgroup_" #tstr "_f16_ocp_f16acc", matmul_id_subgroup_##tstr##_f16_ocp_f16acc_cm2_len, matmul_id_subgroup_##tstr##_f16_ocp_f16acc_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); \ + create_mm_pipelines({TYPE, GGML_TYPE_F16, true, false}, tc, "matmul_id_subgroup_" #tstr "_f16_ocp", matmul_id_subgroup_##tstr##_f16_ocp_cm2_len, matmul_id_subgroup_##tstr##_f16_ocp_cm2_data, sizeof(vk_mat_mat_id_push_constants), 5, cm2_spec, true); \ + } } + FOR_EACH_LUT_FP4_TYPE(X_CM2_ID_OCP) +#undef X_CM2_ID_OCP + } else +#endif + { + FOR_EACH_LUT_FP4_TYPE(X_CM2_ID) + } +#undef X_CM2_ID + } else +#endif // defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + if (device->coopmat_support) { + spec_fn_t cm1_spec = [&](const std::vector& wt, bool a) { return ggml_vk_mul_mm_spec(wt, a); }; - result.block_cols = (D & 8) ? 64 : 32; - } - - const uint32_t D_lsb = D ^ (D & (D-1)); // extract lowest set bit - - result.d_split = std::min(std::min(result.subgroup_size, 8u), D_lsb / 4); + // Intel coopmat1 pins each pipeline's required subgroup size to its warptile WARP element. + const bool cm1_pin = device->vendor_id == VK_VENDOR_ID_INTEL; - result.shmem_staging = (device->vendor_id == VK_VENDOR_ID_NVIDIA && hsk < 256 && hsv < 256) ? 1 : 0; + // Intel coopmat1 uses a dedicated large-tile config for quant matmul_id. + std::vector tc_mmq_id = tc_mmq; + if (cm1_pin) { + tc_mmq_id[2] = { { 512, 128, 128, 32, 32, 32, 2, device->coopmat_m, device->coopmat_n, device->coopmat_k, 32 }, { 128, 128, 1 }, 32 }; + } - if (!reduce_block_rows && !ggml_vk_flash_attn_scalar_shmem_support(device, result, hsk, hsv, f32acc, k_type, v_type)) { - result.block_rows /= 2; - } + auto cm1_create = [&](vk_matmul_pipeline_key key, const std::vector& tc_base, + const std::string& name, size_t len, const void* data, uint32_t pc_size, uint32_t pc) { + auto tc = filter_tc(tc_base, key.type_a, key.mul_mat_id); + if (!tc.empty()) create_mm_pipelines(key, tc, name, len, data, pc_size, pc, cm1_spec, false, true, 0, true, cm1_pin); + }; + auto cm1_create_quant = [&](vk_matmul_pipeline_key key, const std::vector& tc_base, + const std::string& name, size_t len, const void* data, uint32_t pc_size, uint32_t pc) { + spec_fn_t qs = [&, type_a=key.type_a](const std::vector& wt, bool a) { return ggml_vk_mul_mm_spec_quant(wt, a, (uint32_t)type_a); }; + auto tc = filter_tc(tc_base, key.type_a, key.mul_mat_id); + if (!tc.empty()) create_mm_pipelines(key, tc, name, len, data, pc_size, pc, qs, false, true, 0, true, cm1_pin); + }; - // On AMD RDNA, for small head sizes and big batch size the shader uses few registers, so too many subgroups get scheduled - // at once and end up thrashing the cache. Fix this by setting a large (unused) shmem buffer that reduces occupancy. - // This targets an occupancy of 4 subgroups per SIMD. - if (device->vendor_id == VK_VENDOR_ID_AMD && device->properties.limits.maxComputeSharedMemorySize == 65536) { - if (device->architecture != AMD_GCN && n_rows >= 64 && hsk <= 128) { - // 30kb target for hsk > 64, 26kb for <= 64 due to smaller workgroup size - // Values are guessed, tested on RDNA2 - result.limit_occupancy_shmem = (hsk <= 64 ? 26 : 30) * 1024 / 4 / 4; - } else if (device->architecture == AMD_GCN && n_rows <= 8 && hsk >= 256) { - // Same thing for GCN, with an occupancy target of 2 subgroups per SIMD. - // Here low-batch FA with large head size is affected. - // n_rows < 4 switch because workgroup size switches from 128 to 256 there. - result.limit_occupancy_shmem = (n_rows < 4 ? 14 : 26) * 1024 / 4 / 4; + cm1_create({GGML_TYPE_F32, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f32_f32", matmul_f32_f32_cm1_len, matmul_f32_f32_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + cm1_create({GGML_TYPE_F32, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f32_f16", matmul_f32_f16_cm1_len, matmul_f32_f16_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + if (device->coopmat_acc_f16_support) { + cm1_create({GGML_TYPE_F16, GGML_TYPE_F16, false, true}, tc_mm, "matmul_f16_f16acc", matmul_f16_f16acc_cm1_len, matmul_f16_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + cm1_create({GGML_TYPE_F16, GGML_TYPE_F32, false, true}, tc_mm, "matmul_f16_f32_f16acc", matmul_f16_f32_f16acc_cm1_len, matmul_f16_f32_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); } - } - - return result; -} + if (device->coopmat_acc_f32_support) { + cm1_create({GGML_TYPE_F16, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f16", matmul_f16_cm1_len, matmul_f16_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + cm1_create({GGML_TYPE_F16, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f16_f32", matmul_f16_f32_cm1_len, matmul_f16_f32_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + } +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (device->coopmat_bf16_support) { + cm1_create({GGML_TYPE_BF16, GGML_TYPE_BF16, false, false}, tc_mm, "matmul_bf16", matmul_bf16_cm1_len, matmul_bf16_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + } +#endif + for (const auto type : non_lut_quant_types) { + if (device->coopmat_acc_f16_support) { + cm1_create_quant({type, GGML_TYPE_F32, false, true}, tc_mmq, "matmul_quant_f32_f16acc", matmul_quant_f32_f16acc_cm1_len, matmul_quant_f32_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + cm1_create_quant({type, GGML_TYPE_F16, false, true}, tc_mmq, "matmul_quant_f16_f16acc", matmul_quant_f16_f16acc_cm1_len, matmul_quant_f16_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + } + if (device->coopmat_acc_f32_support) { + cm1_create_quant({type, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_quant_f32", matmul_quant_f32_cm1_len, matmul_quant_f32_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + cm1_create_quant({type, GGML_TYPE_F16, false, false}, tc_mmq, "matmul_quant_f16", matmul_quant_f16_cm1_len, matmul_quant_f16_cm1_data, sizeof(vk_mat_mat_push_constants), 3); + } + } + // The _f16 variants provide the f16 B-type pipeline used when y_non_contig converts f32->f16. +#define X_CM1(TYPE, tstr) \ + if (device->coopmat_acc_f16_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, false, true}, tc_mmq, "matmul_" #tstr "_f32_f16acc", matmul_##tstr##_f32_f16acc_cm1_len, matmul_##tstr##_f32_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + cm1_create({TYPE, GGML_TYPE_F16, false, true}, tc_mmq, "matmul_" #tstr "_f16_f16acc", matmul_##tstr##_f16_f16acc_cm1_len, matmul_##tstr##_f16_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + } \ + if (device->coopmat_acc_f32_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_" #tstr "_f32", matmul_##tstr##_f32_cm1_len, matmul_##tstr##_f32_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + cm1_create({TYPE, GGML_TYPE_F16, false, false}, tc_mmq, "matmul_" #tstr "_f16", matmul_##tstr##_f16_cm1_len, matmul_##tstr##_f16_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + } + FOR_EACH_LUT_TYPE_NONFP4(X_CM1) +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + if (device->ocp_fp4) { +#define X_CM1_OCP(TYPE, tstr) \ + if (device->coopmat_acc_f16_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, false, true}, tc_mmq, "matmul_" #tstr "_f32_ocp_f16acc", matmul_##tstr##_f32_ocp_f16acc_cm1_len, matmul_##tstr##_f32_ocp_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + cm1_create({TYPE, GGML_TYPE_F16, false, true}, tc_mmq, "matmul_" #tstr "_f16_ocp_f16acc", matmul_##tstr##_f16_ocp_f16acc_cm1_len, matmul_##tstr##_f16_ocp_f16acc_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + } \ + if (device->coopmat_acc_f32_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_" #tstr "_f32_ocp", matmul_##tstr##_f32_ocp_cm1_len, matmul_##tstr##_f32_ocp_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + cm1_create({TYPE, GGML_TYPE_F16, false, false}, tc_mmq, "matmul_" #tstr "_f16_ocp", matmul_##tstr##_f16_ocp_cm1_len, matmul_##tstr##_f16_ocp_cm1_data, sizeof(vk_mat_mat_push_constants), 3); \ + } + FOR_EACH_LUT_FP4_TYPE(X_CM1_OCP) +#undef X_CM1_OCP + } else +#endif + { + FOR_EACH_LUT_FP4_TYPE(X_CM1) + } +#undef X_CM1 -static vk_fa_tuning_params get_fa_tuning_params_coopmat1(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { - GGML_UNUSED(n_rows); - GGML_UNUSED(n_kv); - GGML_UNUSED(k_type); - GGML_UNUSED(v_type); - GGML_UNUSED(f32acc); + GGML_ASSERT(device->subgroup_ballot); - vk_fa_tuning_params result{}; - result.path = FA_COOPMAT1; + cm1_create({GGML_TYPE_F32, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_subgroup_f32_f32", matmul_id_subgroup_f32_f32_cm1_len, matmul_id_subgroup_f32_f32_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + if (device->coopmat_acc_f16_support) { + cm1_create({GGML_TYPE_F16, GGML_TYPE_F16, true, true}, tc_mm, "matmul_id_subgroup_f16_f16acc", matmul_id_subgroup_f16_f16acc_cm1_len, matmul_id_subgroup_f16_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + cm1_create({GGML_TYPE_F16, GGML_TYPE_F32, true, true}, tc_mm, "matmul_id_subgroup_f16_f32_f16acc", matmul_id_subgroup_f16_f32_f16acc_cm1_len, matmul_id_subgroup_f16_f32_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } + if (device->coopmat_acc_f32_support) { + cm1_create({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_mm, "matmul_id_subgroup_f16", matmul_id_subgroup_f16_cm1_len, matmul_id_subgroup_f16_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + cm1_create({GGML_TYPE_F16, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_subgroup_f16_f32", matmul_id_subgroup_f16_f32_cm1_len, matmul_id_subgroup_f16_f32_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (device->coopmat_bf16_support) { + cm1_create({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_mm, "matmul_id_subgroup_bf16", matmul_id_subgroup_bf16_cm1_len, matmul_id_subgroup_bf16_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } +#endif + for (const auto type : non_lut_quant_types) { + if (device->coopmat_acc_f16_support) { + cm1_create_quant({type, GGML_TYPE_F32, true, true}, tc_mmq_id, "matmul_id_subgroup_quant_f32_f16acc", matmul_id_subgroup_quant_f32_f16acc_cm1_len, matmul_id_subgroup_quant_f32_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + cm1_create_quant({type, GGML_TYPE_F16, true, true}, tc_mmq_id, "matmul_id_subgroup_quant_f16_f16acc", matmul_id_subgroup_quant_f16_f16acc_cm1_len, matmul_id_subgroup_quant_f16_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } + if (device->coopmat_acc_f32_support) { + cm1_create_quant({type, GGML_TYPE_F32, true, false}, tc_mmq_id, "matmul_id_subgroup_quant_f32", matmul_id_subgroup_quant_f32_cm1_len, matmul_id_subgroup_quant_f32_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + cm1_create_quant({type, GGML_TYPE_F16, true, false}, tc_mmq_id, "matmul_id_subgroup_quant_f16", matmul_id_subgroup_quant_f16_cm1_len, matmul_id_subgroup_quant_f16_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } + } + // The _f16 variants provide the f16 B-type pipeline used when y_non_contig converts f32->f16. +#define X_CM1_ID(TYPE, tstr) \ + if (device->coopmat_acc_f16_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, true, true}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f32_f16acc", matmul_id_subgroup_##tstr##_f32_f16acc_cm1_len, matmul_id_subgroup_##tstr##_f32_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + cm1_create({TYPE, GGML_TYPE_F16, true, true}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f16_f16acc", matmul_id_subgroup_##tstr##_f16_f16acc_cm1_len, matmul_id_subgroup_##tstr##_f16_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + } \ + if (device->coopmat_acc_f32_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f32", matmul_id_subgroup_##tstr##_f32_cm1_len, matmul_id_subgroup_##tstr##_f32_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + cm1_create({TYPE, GGML_TYPE_F16, true, false}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f16", matmul_id_subgroup_##tstr##_f16_cm1_len, matmul_id_subgroup_##tstr##_f16_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + } + FOR_EACH_LUT_TYPE_NONFP4(X_CM1_ID) +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + if (device->ocp_fp4) { +#define X_CM1_ID_OCP(TYPE, tstr) \ + if (device->coopmat_acc_f16_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, true, true}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f32_ocp_f16acc", matmul_id_subgroup_##tstr##_f32_ocp_f16acc_cm1_len, matmul_id_subgroup_##tstr##_f32_ocp_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + cm1_create({TYPE, GGML_TYPE_F16, true, true}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f16_ocp_f16acc", matmul_id_subgroup_##tstr##_f16_ocp_f16acc_cm1_len, matmul_id_subgroup_##tstr##_f16_ocp_f16acc_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + } \ + if (device->coopmat_acc_f32_support) { \ + cm1_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f32_ocp", matmul_id_subgroup_##tstr##_f32_ocp_cm1_len, matmul_id_subgroup_##tstr##_f32_ocp_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + cm1_create({TYPE, GGML_TYPE_F16, true, false}, tc_mmq_id, "matmul_id_subgroup_" #tstr "_f16_ocp", matmul_id_subgroup_##tstr##_f16_ocp_cm1_len, matmul_id_subgroup_##tstr##_f16_ocp_cm1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + } + FOR_EACH_LUT_FP4_TYPE(X_CM1_ID_OCP) +#undef X_CM1_ID_OCP + } else +#endif + { + FOR_EACH_LUT_FP4_TYPE(X_CM1_ID) + } +#undef X_CM1_ID + } else +#endif // defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + { + // Helper for subgroup path with dot2 selection and filtering + auto sg_create = [&](vk_matmul_pipeline_key key, const std::vector& tc_base, + const std::string& name, size_t len, const void* data, uint32_t pc_size, uint32_t pc, + uint32_t rsgs = 0) { + auto tc = filter_tc(tc_base, key.type_a, key.mul_mat_id); + if (!tc.empty()) create_mm_pipelines(key, tc, name, len, data, pc_size, pc, + [&](const std::vector& wt, bool a) { return ggml_vk_mul_mm_spec(wt, a); }, + false, rsgs > 0, rsgs); + }; + auto sg_create_quant = [&](vk_matmul_pipeline_key key, const std::vector& tc_base, + const std::string& name, size_t len, const void* data, uint32_t pc_size, uint32_t pc, + uint32_t rsgs = 0) { + auto tc = filter_tc(tc_base, key.type_a, key.mul_mat_id); + if (!tc.empty()) { + spec_fn_t qs = [&, type_a=key.type_a](const std::vector& wt, bool a) { return ggml_vk_mul_mm_spec_quant(wt, a, (uint32_t)type_a); }; + create_mm_pipelines(key, tc, name, len, data, pc_size, pc, qs, false, rsgs > 0, rsgs); + } + }; +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + auto sg_create_mmq = [&](vk_matmul_pipeline_key key, const std::vector& tc_base, + const std::string& name, size_t len, const void* data, uint32_t pc_size, uint32_t pc, + uint32_t rsgs = 0) { + auto tc = filter_tc(tc_base, key.type_a, key.mul_mat_id, true); + if (!tc.empty()) { + spec_fn_t identity = [](const std::vector& wt, bool) { return wt; }; + create_mm_pipelines(key, tc, name, len, data, pc_size, pc, identity, false, rsgs > 0, rsgs, false); + } + }; +#endif - const uint32_t D = hsk | hsv; + std::vector tc_id = {{s_warptile_id, s_wg_denoms, s_align}, {m_warptile_id, m_wg_denoms, m_align}, {l_warptile_id, l_wg_denoms, l_align}}; + std::vector tc_mmqid = {{s_warptile_mmqid, s_mmq_wg_denoms, s_align}, {m_warptile_mmqid, m_mmq_wg_denoms, m_align}, {l_warptile_mmqid, l_mmq_wg_denoms, l_align}}; - const uint32_t coopmat_block_rows = 16; - const uint32_t coopmat_block_cols = 16; + if (device->fp16) { + // FP16 subgroup path - with dot2 runtime selection + #define SPV_DOT2(NAME) (device->dot2_f16 ? NAME ## _dot2_len : NAME ## _len), (device->dot2_f16 ? NAME ## _dot2_data : NAME ## _data) + #define SPV_DOT2_F16ACC(NAME) (device->dot2_f16 ? NAME ## _dot2_f16acc_len : NAME ## _f16acc_len), (device->dot2_f16 ? NAME ## _dot2_f16acc_data : NAME ## _f16acc_data) + + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f32_f32", SPV_DOT2(matmul_f32_f32), sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F32, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f32_f16", SPV_DOT2(matmul_f32_f16), sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, false, true}, tc_mm, "matmul_f16_f16acc", SPV_DOT2_F16ACC(matmul_f16), sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f16", SPV_DOT2(matmul_f16), sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, false, true}, tc_mm, "matmul_f16_f32_f16acc", SPV_DOT2_F16ACC(matmul_f16_f32), sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f16_f32", SPV_DOT2(matmul_f16_f32), sizeof(vk_mat_mat_push_constants), 3); + // BF16 - no dot2 + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, false, false}, tc_mm, "matmul_bf16", matmul_bf16_len, matmul_bf16_data, sizeof(vk_mat_mat_push_constants), 3); + + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, false, true}, tc_mmq, "matmul_quant_f32_f16acc", SPV_DOT2_F16ACC(matmul_quant_f32), sizeof(vk_mat_mat_push_constants), 3); + sg_create_quant({type, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_quant_f32", SPV_DOT2(matmul_quant_f32), sizeof(vk_mat_mat_push_constants), 3); + } + #define X_SG(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, false, true}, tc_mmq, "matmul_" #tstr "_f32_f16acc", SPV_DOT2_F16ACC(matmul_##tstr##_f32), sizeof(vk_mat_mat_push_constants), 3); \ + sg_create({TYPE, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_" #tstr "_f32", SPV_DOT2(matmul_##tstr##_f32), sizeof(vk_mat_mat_push_constants), 3); + FOR_EACH_LUT_TYPE(X_SG) +#undef X_SG - const uint32_t num_subgroups = 4; +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + std::vector tc_mmq_int = {{s_warptile_mmq_int, s_mmq_wg_denoms, s_align}, {m_warptile_mmq_int, m_mmq_wg_denoms, m_align}, {l_warptile_mmq_int, l_mmq_wg_denoms, l_align}}; + std::vector tc_mmq_int_k = {{s_warptile_mmq_int_k, s_mmq_wg_denoms, s_align}, {m_warptile_mmq_int_k, m_mmq_wg_denoms, m_align}, {l_warptile_mmq_int_k, l_mmq_wg_denoms, l_align}}; + sg_create_mmq({GGML_TYPE_Q2_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q2_0_q8_1", matmul_q2_0_q8_1_len, matmul_q2_0_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q4_0_q8_1", matmul_q4_0_q8_1_len, matmul_q4_0_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_1, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q4_1_q8_1", matmul_q4_1_q8_1_len, matmul_q4_1_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q5_0_q8_1", matmul_q5_0_q8_1_len, matmul_q5_0_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_1, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q5_1_q8_1", matmul_q5_1_q8_1_len, matmul_q5_1_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q8_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q8_0_q8_1", matmul_q8_0_q8_1_len, matmul_q8_0_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_MXFP4, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_mxfp4_q8_1", matmul_mxfp4_q8_1_len, matmul_mxfp4_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_IQ4_XS, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_iq4_xs_q8_1", matmul_iq4_xs_q8_1_len, matmul_iq4_xs_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q2_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q2_k_q8_1", matmul_q2_k_q8_1_len, matmul_q2_k_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q3_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q3_k_q8_1", matmul_q3_k_q8_1_len, matmul_q3_k_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q4_k_q8_1", matmul_q4_k_q8_1_len, matmul_q4_k_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q5_k_q8_1", matmul_q5_k_q8_1_len, matmul_q5_k_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q6_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q6_k_q8_1", matmul_q6_k_q8_1_len, matmul_q6_k_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_IQ3_S, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_iq3_s_q8_1", matmul_iq3_s_q8_1_len, matmul_iq3_s_q8_1_data, sizeof(vk_mat_mat_push_constants), 3); + } +#endif - result.block_rows = coopmat_block_rows; - result.block_cols = coopmat_block_cols * num_subgroups; - result.row_split = num_subgroups; - result.subgroup_size = device->subgroup_size; - result.workgroup_size = num_subgroups * result.subgroup_size; + if (device->subgroup_ballot && device->subgroup_require_full_support && subgroup_min_size_16) { + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, true, false}, tc_id, "matmul_id_subgroup_f32_f32", SPV_DOT2(matmul_id_subgroup_f32_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, true}, tc_id, "matmul_id_subgroup_f16_f16acc", SPV_DOT2_F16ACC(matmul_id_subgroup_f16), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_id, "matmul_id_subgroup_f16", SPV_DOT2(matmul_id_subgroup_f16), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, true}, tc_id, "matmul_id_subgroup_f16_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_subgroup_f16_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, false}, tc_id, "matmul_id_subgroup_f16_f32", SPV_DOT2(matmul_id_subgroup_f16_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + // BF16 id - no dot2 + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_id, "matmul_id_subgroup_bf16", matmul_id_subgroup_bf16_len, matmul_id_subgroup_bf16_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, true, true}, tc_mmqid, "matmul_id_subgroup_quant_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_subgroup_quant_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_quant({type, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_subgroup_quant_f32", SPV_DOT2(matmul_id_subgroup_quant_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + } + #define X_SG_ID_SUB(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, true, true}, tc_mmqid, "matmul_id_subgroup_" #tstr "_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_subgroup_##tstr##_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); \ + sg_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_subgroup_" #tstr "_f32", SPV_DOT2(matmul_id_subgroup_##tstr##_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + FOR_EACH_LUT_TYPE(X_SG_ID_SUB) +#undef X_SG_ID_SUB +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + std::vector tc_mmqid_int = {{s_warptile_mmqid_int, s_mmq_wg_denoms, s_align}, {m_warptile_mmqid_int, m_mmq_wg_denoms, m_align}, {l_warptile_mmqid_int, l_mmq_wg_denoms, l_align}}; + std::vector tc_mmqid_int_k = {{s_warptile_mmqid_int_k, s_mmq_wg_denoms, s_align}, {m_warptile_mmqid_int_k, m_mmq_wg_denoms, m_align}, {l_warptile_mmqid_int_k, l_mmq_wg_denoms, l_align}}; + sg_create_mmq({GGML_TYPE_Q2_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q2_0_q8_1", matmul_id_subgroup_q2_0_q8_1_len, matmul_id_subgroup_q2_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q4_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q4_0_q8_1", matmul_id_subgroup_q4_0_q8_1_len, matmul_id_subgroup_q4_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q4_1, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q4_1_q8_1", matmul_id_subgroup_q4_1_q8_1_len, matmul_id_subgroup_q4_1_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q5_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q5_0_q8_1", matmul_id_subgroup_q5_0_q8_1_len, matmul_id_subgroup_q5_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q5_1, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q5_1_q8_1", matmul_id_subgroup_q5_1_q8_1_len, matmul_id_subgroup_q5_1_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q8_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_q8_0_q8_1", matmul_id_subgroup_q8_0_q8_1_len, matmul_id_subgroup_q8_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_MXFP4, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_mxfp4_q8_1", matmul_id_subgroup_mxfp4_q8_1_len, matmul_id_subgroup_mxfp4_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_IQ4_XS, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_subgroup_iq4_xs_q8_1", matmul_id_subgroup_iq4_xs_q8_1_len, matmul_id_subgroup_iq4_xs_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + sg_create_mmq({GGML_TYPE_Q2_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_q2_k_q8_1", matmul_id_subgroup_q2_k_q8_1_len, matmul_id_subgroup_q2_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create_mmq({GGML_TYPE_Q3_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_q3_k_q8_1", matmul_id_subgroup_q3_k_q8_1_len, matmul_id_subgroup_q3_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create_mmq({GGML_TYPE_Q4_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_q4_k_q8_1", matmul_id_subgroup_q4_k_q8_1_len, matmul_id_subgroup_q4_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create_mmq({GGML_TYPE_Q5_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_q5_k_q8_1", matmul_id_subgroup_q5_k_q8_1_len, matmul_id_subgroup_q5_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create_mmq({GGML_TYPE_Q6_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_q6_k_q8_1", matmul_id_subgroup_q6_k_q8_1_len, matmul_id_subgroup_q6_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create_mmq({GGML_TYPE_IQ3_S, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_subgroup_iq3_s_q8_1", matmul_id_subgroup_iq3_s_q8_1_len, matmul_id_subgroup_iq3_s_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + } +#endif + } else { + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_f32_f32", SPV_DOT2(matmul_id_f32_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, true}, tc_mm, "matmul_id_f16_f16acc", SPV_DOT2_F16ACC(matmul_id_f16), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_mm, "matmul_id_f16", SPV_DOT2(matmul_id_f16), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, true}, tc_mm, "matmul_id_f16_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_f16_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_f16_f32", SPV_DOT2(matmul_id_f16_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + // BF16 id - no dot2 + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_mm, "matmul_id_bf16", matmul_id_bf16_len, matmul_id_bf16_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, true, true}, tc_mmqid, "matmul_id_quant_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_quant_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_quant({type, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_quant_f32", SPV_DOT2(matmul_id_quant_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } + #define X_SG_ID(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, true, true}, tc_mmqid, "matmul_id_" #tstr "_f32_f16acc", SPV_DOT2_F16ACC(matmul_id_##tstr##_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); \ + sg_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_" #tstr "_f32", SPV_DOT2(matmul_id_##tstr##_f32), sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + FOR_EACH_LUT_TYPE(X_SG_ID) +#undef X_SG_ID +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + std::vector tc_mmqid_int = {{s_warptile_mmqid_int, s_mmq_wg_denoms, s_align}, {m_warptile_mmqid_int, m_mmq_wg_denoms, m_align}, {l_warptile_mmqid_int, l_mmq_wg_denoms, l_align}}; + std::vector tc_mmqid_int_k = {{s_warptile_mmqid_int_k, s_mmq_wg_denoms, s_align}, {m_warptile_mmqid_int_k, m_mmq_wg_denoms, m_align}, {l_warptile_mmqid_int_k, l_mmq_wg_denoms, l_align}}; + sg_create_mmq({GGML_TYPE_Q2_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q2_0_q8_1", matmul_id_q2_0_q8_1_len, matmul_id_q2_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q4_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q4_0_q8_1", matmul_id_q4_0_q8_1_len, matmul_id_q4_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q4_1, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q4_1_q8_1", matmul_id_q4_1_q8_1_len, matmul_id_q4_1_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q5_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q5_0_q8_1", matmul_id_q5_0_q8_1_len, matmul_id_q5_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q5_1, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q5_1_q8_1", matmul_id_q5_1_q8_1_len, matmul_id_q5_1_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q8_0, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_q8_0_q8_1", matmul_id_q8_0_q8_1_len, matmul_id_q8_0_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_MXFP4, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_mxfp4_q8_1", matmul_id_mxfp4_q8_1_len, matmul_id_mxfp4_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_IQ4_XS, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int, "matmul_id_iq4_xs_q8_1", matmul_id_iq4_xs_q8_1_len, matmul_id_iq4_xs_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q2_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_q2_k_q8_1", matmul_id_q2_k_q8_1_len, matmul_id_q2_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q3_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_q3_k_q8_1", matmul_id_q3_k_q8_1_len, matmul_id_q3_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q4_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_q4_k_q8_1", matmul_id_q4_k_q8_1_len, matmul_id_q4_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q5_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_q5_k_q8_1", matmul_id_q5_k_q8_1_len, matmul_id_q5_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_Q6_K, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_q6_k_q8_1", matmul_id_q6_k_q8_1_len, matmul_id_q6_k_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create_mmq({GGML_TYPE_IQ3_S, GGML_TYPE_Q8_1, true, false}, tc_mmqid_int_k, "matmul_id_iq3_s_q8_1", matmul_id_iq3_s_q8_1_len, matmul_id_iq3_s_q8_1_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } +#endif + } + #undef SPV_DOT2 + #undef SPV_DOT2_F16ACC + } else { + // FP32-only fallback path + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f32_f32", matmul_f32_f32_fp32_len, matmul_f32_f32_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F32, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f32_f16", matmul_f32_f16_fp32_len, matmul_f32_f16_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, false, false}, tc_mm, "matmul_f16", matmul_f16_fp32_len, matmul_f16_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, false, false}, tc_mm, "matmul_f16_f32", matmul_f16_f32_fp32_len, matmul_f16_f32_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, false, false}, tc_mm, "matmul_bf16", matmul_bf16_fp32_len, matmul_bf16_fp32_data, sizeof(vk_mat_mat_push_constants), 3); - const uint32_t D_lsb = D ^ (D & (D-1)); // extract lowest set bit - result.d_split = std::min(std::min(result.subgroup_size, 8u), D_lsb / 4); + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_quant_f32", matmul_quant_f32_fp32_len, matmul_quant_f32_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + } + #define X_SG_FP32(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, false, false}, tc_mmq, "matmul_" #tstr "_f32", matmul_##tstr##_f32_fp32_len, matmul_##tstr##_f32_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + FOR_EACH_LUT_TYPE(X_SG_FP32) +#undef X_SG_FP32 - result.shmem_staging = (device->vendor_id == VK_VENDOR_ID_NVIDIA && hsk < 256 && hsv < 256) ? 1 : 0; +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + std::vector tc_mmq_int = {{s_warptile_mmq_int, s_mmq_wg_denoms, s_align}, {m_warptile_mmq_int, m_mmq_wg_denoms, m_align}, {l_warptile_mmq_int, l_mmq_wg_denoms, l_align}}; + std::vector tc_mmq_int_k = {{s_warptile_mmq_int_k, s_mmq_wg_denoms, s_align}, {m_warptile_mmq_int_k, m_mmq_wg_denoms, m_align}, {l_warptile_mmq_int_k, l_mmq_wg_denoms, l_align}}; + sg_create_mmq({GGML_TYPE_Q2_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q2_0_q8_1", matmul_q2_0_q8_1_fp32_len, matmul_q2_0_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q4_0_q8_1", matmul_q4_0_q8_1_fp32_len, matmul_q4_0_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_1, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q4_1_q8_1", matmul_q4_1_q8_1_fp32_len, matmul_q4_1_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q5_0_q8_1", matmul_q5_0_q8_1_fp32_len, matmul_q5_0_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_1, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q5_1_q8_1", matmul_q5_1_q8_1_fp32_len, matmul_q5_1_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q8_0, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_q8_0_q8_1", matmul_q8_0_q8_1_fp32_len, matmul_q8_0_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_IQ4_XS, GGML_TYPE_Q8_1, false, false}, tc_mmq_int, "matmul_iq4_xs_q8_1", matmul_iq4_xs_q8_1_fp32_len, matmul_iq4_xs_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q2_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q2_k_q8_1", matmul_q2_k_q8_1_fp32_len, matmul_q2_k_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q3_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q3_k_q8_1", matmul_q3_k_q8_1_fp32_len, matmul_q3_k_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q4_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q4_k_q8_1", matmul_q4_k_q8_1_fp32_len, matmul_q4_k_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q5_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q5_k_q8_1", matmul_q5_k_q8_1_fp32_len, matmul_q5_k_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_Q6_K, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_q6_k_q8_1", matmul_q6_k_q8_1_fp32_len, matmul_q6_k_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + sg_create_mmq({GGML_TYPE_IQ3_S, GGML_TYPE_Q8_1, false, false}, tc_mmq_int_k, "matmul_iq3_s_q8_1", matmul_iq3_s_q8_1_fp32_len, matmul_iq3_s_q8_1_fp32_data, sizeof(vk_mat_mat_push_constants), 3); + } +#endif - return result; -} + if (device->subgroup_ballot && device->subgroup_require_full_support && subgroup_min_size_16) { + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, true, false}, tc_id, "matmul_id_subgroup_f32_f32", matmul_id_subgroup_f32_f32_fp32_len, matmul_id_subgroup_f32_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_id, "matmul_id_subgroup_f16", matmul_id_subgroup_f16_fp32_len, matmul_id_subgroup_f16_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, false}, tc_id, "matmul_id_subgroup_f16_f32", matmul_id_subgroup_f16_f32_fp32_len, matmul_id_subgroup_f16_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_id, "matmul_id_subgroup_bf16", matmul_id_subgroup_bf16_fp32_len, matmul_id_subgroup_bf16_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size_16); + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_subgroup_quant_f32", matmul_id_subgroup_quant_f32_fp32_len, matmul_id_subgroup_quant_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + } + #define X_SG_ID_SUB_FP32(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_subgroup_" #tstr "_f32", matmul_id_subgroup_##tstr##_f32_fp32_len, matmul_id_subgroup_##tstr##_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, mul_mat_subgroup_size); + FOR_EACH_LUT_TYPE(X_SG_ID_SUB_FP32) +#undef X_SG_ID_SUB_FP32 + } else { + sg_create({GGML_TYPE_F32, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_f32_f32", matmul_id_f32_f32_fp32_len, matmul_id_f32_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F16, true, false}, tc_mm, "matmul_id_f16", matmul_id_f16_fp32_len, matmul_id_f16_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_F16, GGML_TYPE_F32, true, false}, tc_mm, "matmul_id_f16_f32", matmul_id_f16_f32_fp32_len, matmul_id_f16_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + sg_create({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_mm, "matmul_id_bf16", matmul_id_bf16_fp32_len, matmul_id_bf16_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + for (const auto type : non_lut_quant_types) { + sg_create_quant({type, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_quant_f32", matmul_id_quant_f32_fp32_len, matmul_id_quant_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + } + #define X_SG_ID_FP32(TYPE, tstr) \ + sg_create({TYPE, GGML_TYPE_F32, true, false}, tc_mmqid, "matmul_id_" #tstr "_f32", matmul_id_##tstr##_f32_fp32_len, matmul_id_##tstr##_f32_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count); + FOR_EACH_LUT_TYPE(X_SG_ID_FP32) +#undef X_SG_ID_FP32 + } + } + } +#undef FOR_EACH_LUT_TYPE +#undef FOR_EACH_LUT_TYPE_NONFP4 +#undef FOR_EACH_LUT_FP4_TYPE + // BF16 fallback for coopmat devices without bf16 coopmat support + if ((device->coopmat2 || device->coopmat_support) +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + && !device->coopmat_bf16_support +#endif + ) { + const uint32_t s_warptile_wm_bf16 = device->subgroup_size == 8 ? 8 : 32; + std::vector tc_bf16_fb = { + {{ subgroup_size_32, 32, 32, 16, s_warptile_wm_bf16, 32, 2, 2, 2, 1, subgroup_size_8 }, {32, 32, 1}, s_align}, + {{ 128, 64, 64, 16, mm_warp_8, 32, 2, 4, 2, 1, mm_warp_8 }, {64, 64, 1}, m_align}, + {{ 128, 128, 128, 16, mm_warp_8 * 2, 64, 2, 4, 4, 1, mm_warp_8 }, {128, 128, 1}, l_align}, + }; + auto tc_bf16_filtered = filter_tc(tc_bf16_fb, GGML_TYPE_BF16, false); + auto tc_bf16_id_filtered = filter_tc(tc_bf16_fb, GGML_TYPE_BF16, true); + spec_fn_t bf16_spec = [&](const std::vector& wt, bool a) { return ggml_vk_mul_mm_spec(wt, a); }; + if (!tc_bf16_filtered.empty()) { + create_mm_pipelines({GGML_TYPE_BF16, GGML_TYPE_BF16, false, false}, tc_bf16_filtered, "matmul_bf16", matmul_bf16_fp32_len, matmul_bf16_fp32_data, sizeof(vk_mat_mat_push_constants), 3, bf16_spec); + } + if (!tc_bf16_id_filtered.empty()) { + create_mm_pipelines({GGML_TYPE_BF16, GGML_TYPE_BF16, true, false}, tc_bf16_id_filtered, "matmul_id_bf16", matmul_id_bf16_fp32_len, matmul_id_bf16_fp32_data, sizeof(vk_mat_mat_id_push_constants), mul_mat_id_param_count, bf16_spec); + } + } -static vk_fa_tuning_params get_fa_tuning_params_coopmat2(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { - GGML_UNUSED(n_kv); - GGML_UNUSED(f32acc); + // Set up tile selector functions + if (device->coopmat2) { + device->matmul_tile_selector = [](uint32_t m, uint32_t n, uint32_t /*k*/, uint32_t shader_core_count, + const std::vector& configs) -> uint32_t { + if (configs.size() <= 1) return 0; + uint32_t last = (uint32_t)configs.size() - 1; + if (configs.size() == 2) { + uint32_t crossover = configs[0].unaligned->wg_denoms[1]; + return (n > crossover) ? 1 : 0; + } + // 3+ configs: s=0, m=1, l=2 + const uint32_t tiles_l = CEIL_DIV(m, configs[last].unaligned->wg_denoms[0]) * CEIL_DIV(n, configs[last].unaligned->wg_denoms[1]); + const uint32_t tiles_m = CEIL_DIV(m, configs[1].unaligned->wg_denoms[0]) * CEIL_DIV(n, configs[1].unaligned->wg_denoms[1]); + uint32_t crossover_large = configs[1].unaligned->wg_denoms[1]; + bool prefer_large = tiles_m > shader_core_count || tiles_l > shader_core_count || + (tiles_l <= shader_core_count / 3 && tiles_m > shader_core_count / 2); + if (n > crossover_large && prefer_large) return last; + uint32_t crossover_medium_m = configs[0].unaligned->wg_denoms[0]; + uint32_t crossover_medium_n = configs[0].unaligned->wg_denoms[1]; + if (m > crossover_medium_m && n > crossover_medium_n) return 1; + return 0; + }; + device->matmul_id_tile_selector = [](uint32_t /*m*/, uint32_t n, uint32_t /*k*/, uint32_t /*shader_core_count*/, + const std::vector& configs) -> uint32_t { + if (configs.size() <= 1) return 0; + uint32_t last = (uint32_t)configs.size() - 1; + if (configs.size() == 2) { + uint32_t crossover = configs[0].unaligned->wg_denoms[1]; + return (n > crossover) ? 1 : 0; + } + uint32_t crossover_large = configs[1].unaligned->wg_denoms[1]; + if (n > crossover_large) return last; + uint32_t crossover_medium = configs[0].unaligned->wg_denoms[1]; + if (n > crossover_medium) return 1; + return 0; + }; + } else { + device->matmul_tile_selector = [](uint32_t m, uint32_t n, uint32_t /*k*/, uint32_t /*shader_core_count*/, + const std::vector& configs) -> uint32_t { + if (configs.size() <= 1) return 0; + if (m <= 32 || n <= 32) return 0; + if (configs.size() == 2) return 1; + if (m <= 64 || n <= 64) return 1; + return (uint32_t)configs.size() - 1; + }; + device->matmul_id_tile_selector = device->matmul_tile_selector; + } - vk_fa_tuning_params result{}; - result.path = FA_COOPMAT2; + // mul mat vec - const uint32_t D = hsk | hsv; + // the number of rows computed per shader depends on GPU model and quant + uint32_t rm_stdq = 1; + uint32_t rm_kq = 2; + uint32_t rm_stdq_int = 1; + uint32_t rm_kq_int = 1; + auto const &rm_iq_int = [](uint32_t i) { return i == 0 ? 8u : 4u; }; + if (device->vendor_id == VK_VENDOR_ID_AMD) { + if (device->architecture == AMD_GCN) { + rm_stdq = 2; + rm_kq = 4; + rm_stdq_int = 4; + } + } else if (device->vendor_id == VK_VENDOR_ID_INTEL) { + rm_stdq = 2; + rm_stdq_int = 2; + } + // RDNA3: above four columns, static 4 rows for all types bench faster than the default + const bool is_rdna3 = device->vendor_id == VK_VENDOR_ID_AMD && device->architecture == AMD_RDNA3; + auto const &rm_int_n = [&](uint32_t rows, uint32_t i) { return (is_rdna3 && i >= 4) ? 4u : rows; }; + // RDNA3: Static 4 rows for all types bench faster than the default + auto const &rm_id = [&](uint32_t rows) { return is_rdna3 ? 4u : rows; }; + uint32_t rm_iq = 2 * rm_kq; - const bool small_rows = n_rows < 32; + const bool use_subgroups = device->subgroup_arithmetic; + // The Imagination proprietary compiler rejects the subgroup-only dequant mul_mat_vec + // shaders that require a subgroup size >= 16; fall back to shared-memory reduction. + const bool is_imagination_proprietary = + device->driver_id == vk::DriverId::eImaginationProprietary; + // Ensure a subgroup size >= 16 is available + const bool use_subgroups16 = use_subgroups && subgroup_min_size_16 && !is_imagination_proprietary; - if (small_rows) { - result.block_rows = 32; - result.block_cols = 32; - } else if (ggml_is_quantized(k_type) || ggml_is_quantized(v_type) || hsk >= 256 || hsv >= 256) { - result.block_rows = (hsk >= 512 || hsv >= 512) ? 32 : 64; - result.block_cols = 32; - } else { - result.block_rows = 64; - result.block_cols = 64; - } + const uint32_t subgroup_size = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control && device->subgroup_min_size <= 16 && device->subgroup_max_size >= 16) ? 16 : device->subgroup_size; + const uint32_t subgroup_size16 = std::max(subgroup_size, 16u); - result.subgroup_size = device->subgroup_size; - result.workgroup_size = (small_rows && (D % 32) == 0) ? 256 : 128; + const uint32_t force_subgroup_size = use_subgroups ? subgroup_size : 0; + const uint32_t force_subgroup_size16 = use_subgroups16 ? subgroup_size16 : 0; + static constexpr uint32_t mul_mat_vec_num_bindings = 5; + static constexpr uint32_t mul_mat_vec_id_num_bindings = 6; - return result; -} +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) +#define OCP_DMMV_LEN(NAME, REDUC) (device->ocp_fp4 ? NAME ## _ocp_len[REDUC] : NAME ## _len[REDUC]) +#define OCP_DMMV_DATA(NAME, REDUC) (device->ocp_fp4 ? NAME ## _ocp_data[REDUC] : NAME ## _data[REDUC]) +#else +#define OCP_DMMV_LEN(NAME, REDUC) NAME ## _len[REDUC] +#define OCP_DMMV_DATA(NAME, REDUC) NAME ## _data[REDUC] +#endif -static vk_fa_tuning_params get_fa_tuning_params(const vk_device& device, uint32_t hsk, uint32_t hsv, uint32_t n_rows, uint32_t n_kv, ggml_type k_type, ggml_type v_type, bool f32acc) { - FaCodePath path = device->coopmat2 ? FA_COOPMAT2 : - device->coopmat1_fa_support ? FA_COOPMAT1 : FA_SCALAR; + for (uint32_t w = 0; w < DMMV_WG_SIZE_COUNT; ++w) { + const uint32_t wg_size_subgroup = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size : (subgroup_size * 4); + const uint32_t wg_size_subgroup16 = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size16 : (subgroup_size16 * 4); - if (path == FA_COOPMAT2 && k_type == GGML_TYPE_BF16 && !device->coopmat2_bf16_support) { - path = FA_COOPMAT1; - } - if (path == FA_COOPMAT1 && k_type == GGML_TYPE_BF16 && !device->coopmat_bf16_support) { - path = FA_SCALAR; - } + const shader_reduction_mode reduc = (use_subgroups && w == DMMV_WG_SIZE_SUBGROUP) ? SHADER_REDUCTION_MODE_SUBGROUP : + (use_subgroups && w == DMMV_WG_SIZE_LARGE) ? SHADER_REDUCTION_MODE_HYBRID : + SHADER_REDUCTION_MODE_SHMEM; - if (path == FA_COOPMAT1 && device->architecture == vk_device_architecture::NVIDIA_TURING) { - // Nvidia compiler bug, see https://github.com/ggml-org/llama.cpp/pull/19075#issuecomment-3820716090 - path = FA_SCALAR; - } + const shader_reduction_mode reduc16 = (use_subgroups16 && w == DMMV_WG_SIZE_SUBGROUP) ? SHADER_REDUCTION_MODE_SUBGROUP : + (use_subgroups16 && w == DMMV_WG_SIZE_LARGE) ? SHADER_REDUCTION_MODE_HYBRID : + SHADER_REDUCTION_MODE_SHMEM; - if (path == FA_COOPMAT1) { - bool shape_ok = (f32acc && device->coopmat_support_16x16x16_f32acc) || - (!f32acc && device->coopmat_support_16x16x16_f16acc); - const vk_fa_tuning_params params = get_fa_tuning_params_coopmat1(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); - bool shmem_ok = ggml_vk_flash_attn_coopmat_shmem_support(device, params, hsk, hsv, f32acc, k_type, v_type); - - if (!shape_ok || !shmem_ok) { - path = FA_SCALAR; - } - } - - // scalar is faster than coopmat when N==1 - if (n_rows == 1 && (path == FA_COOPMAT1 || path == FA_COOPMAT2)) { - path = FA_SCALAR; - } - - switch (path) { - case FA_SCALAR: - return get_fa_tuning_params_scalar(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); - case FA_COOPMAT1: - return get_fa_tuning_params_coopmat1(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); - case FA_COOPMAT2: - return get_fa_tuning_params_coopmat2(device, hsk, hsv, n_rows, n_kv, k_type, v_type, f32acc); - default: - throw std::runtime_error("unsupported FaCodePath"); - } -} - -static vk_fa_pipeline_state get_fa_pipeline_state(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool aligned, bool f32acc, - bool use_mask, bool use_mask_opt, bool use_logit_softcap, ggml_type k_type, ggml_type v_type) { - const bool old_amd_windows = device->vendor_id == VK_VENDOR_ID_AMD && device->driver_id == vk::DriverId::eAmdProprietary && - (device->architecture == AMD_GCN || device->architecture == AMD_RDNA1 || device->architecture == AMD_RDNA2); - - uint32_t flags = (use_mask_opt ? 1 : 0) | - (use_mask ? 2 : 0) | - (use_logit_softcap ? 4 : 0) | - (old_amd_windows ? 8 : 0); + for (uint32_t i = 0; i < mul_mat_vec_max_cols; ++i) { + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_F32 ][i], "mul_mat_vec_f32_f32_f32", arr_dmmv_f32_f32_f32_len[reduc], arr_dmmv_f32_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1, 1, 1}, {wg_size_subgroup, 1, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_F16 ][i], "mul_mat_vec_f16_f32_f32", arr_dmmv_f16_f32_f32_len[reduc], arr_dmmv_f16_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_BF16][i], "mul_mat_vec_bf16_f32_f32", arr_dmmv_bf16_f32_f32_len[reduc], arr_dmmv_bf16_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q1_0][i], "mul_mat_vec_q1_0_f32_f32", arr_dmmv_q1_0_f32_f32_len[reduc], arr_dmmv_q1_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_f32_f32", arr_dmmv_q2_0_f32_f32_len[reduc], arr_dmmv_q2_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_f32_f32", arr_dmmv_q4_0_f32_f32_len[reduc], arr_dmmv_q4_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_f32_f32", arr_dmmv_q4_1_f32_f32_len[reduc], arr_dmmv_q4_1_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_f32_f32", arr_dmmv_q5_0_f32_f32_len[reduc], arr_dmmv_q5_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_f32_f32", arr_dmmv_q5_1_f32_f32_len[reduc], arr_dmmv_q5_1_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_f32_f32", arr_dmmv_q8_0_f32_f32_len[reduc], arr_dmmv_q8_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_f32_f32", arr_dmmv_q2_k_f32_f32_len[reduc16], arr_dmmv_q2_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_f32_f32", arr_dmmv_q3_k_f32_f32_len[reduc16], arr_dmmv_q3_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_f32_f32", arr_dmmv_q4_k_f32_f32_len[reduc16], arr_dmmv_q4_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_f32_f32", arr_dmmv_q5_k_f32_f32_len[reduc16], arr_dmmv_q5_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_f32_f32", arr_dmmv_q6_k_f32_f32_len[reduc16], arr_dmmv_q6_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_TQ1_0][i], "mul_mat_vec_tq1_0_f32_f32", arr_dmmv_tq1_0_f32_f32_len[reduc16], arr_dmmv_tq1_0_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_TQ2_0][i], "mul_mat_vec_tq2_0_f32_f32", arr_dmmv_tq2_0_f32_f32_len[reduc16], arr_dmmv_tq2_0_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_f32_f32", arr_dmmv_iq1_s_f32_f32_len[reduc16], arr_dmmv_iq1_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_f32_f32", arr_dmmv_iq1_m_f32_f32_len[reduc16], arr_dmmv_iq1_m_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_XXS][i], "mul_mat_vec_iq2_xxs_f32_f32", arr_dmmv_iq2_xxs_f32_f32_len[reduc16], arr_dmmv_iq2_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_XS][i], "mul_mat_vec_iq2_xs_f32_f32", arr_dmmv_iq2_xs_f32_f32_len[reduc16], arr_dmmv_iq2_xs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_S][i], "mul_mat_vec_iq2_s_f32_f32", arr_dmmv_iq2_s_f32_f32_len[reduc16], arr_dmmv_iq2_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ3_XXS][i], "mul_mat_vec_iq3_xxs_f32_f32", arr_dmmv_iq3_xxs_f32_f32_len[reduc16], arr_dmmv_iq3_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ3_S][i], "mul_mat_vec_iq3_s_f32_f32", arr_dmmv_iq3_s_f32_f32_len[reduc16], arr_dmmv_iq3_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ4_XS][i], "mul_mat_vec_iq4_xs_f32_f32", arr_dmmv_iq4_xs_f32_f32_len[reduc16], arr_dmmv_iq4_xs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ4_NL][i], "mul_mat_vec_iq4_nl_f32_f32", arr_dmmv_iq4_nl_f32_f32_len[reduc16], arr_dmmv_iq4_nl_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_f32_f32", OCP_DMMV_LEN(arr_dmmv_mxfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_mxfp4_f32_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_NVFP4][i], "mul_mat_vec_nvfp4_f32_f32", OCP_DMMV_LEN(arr_dmmv_nvfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_nvfp4_f32_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - const uint32_t subgroup_size = params.disable_subgroups ? 0 : params.subgroup_size; + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_F32 ][i], "mul_mat_vec_f32_f16_f32", arr_dmmv_f32_f16_f32_len[reduc], arr_dmmv_f32_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1, 1, 1}, {wg_size_subgroup, 1, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_F16 ][i], "mul_mat_vec_f16_f16_f32", arr_dmmv_f16_f16_f32_len[reduc], arr_dmmv_f16_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_BF16][i], "mul_mat_vec_bf16_f16_f32", arr_dmmv_bf16_f16_f32_len[reduc], arr_dmmv_bf16_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q1_0][i], "mul_mat_vec_q1_0_f16_f32", arr_dmmv_q1_0_f16_f32_len[reduc], arr_dmmv_q1_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_f16_f32", arr_dmmv_q2_0_f16_f32_len[reduc], arr_dmmv_q2_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_f16_f32", arr_dmmv_q4_0_f16_f32_len[reduc], arr_dmmv_q4_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_f16_f32", arr_dmmv_q4_1_f16_f32_len[reduc], arr_dmmv_q4_1_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_f16_f32", arr_dmmv_q5_0_f16_f32_len[reduc], arr_dmmv_q5_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_f16_f32", arr_dmmv_q5_1_f16_f32_len[reduc], arr_dmmv_q5_1_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_f16_f32", arr_dmmv_q8_0_f16_f32_len[reduc], arr_dmmv_q8_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_f16_f32", arr_dmmv_q2_k_f16_f32_len[reduc16], arr_dmmv_q2_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_f16_f32", arr_dmmv_q3_k_f16_f32_len[reduc16], arr_dmmv_q3_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_f16_f32", arr_dmmv_q4_k_f16_f32_len[reduc16], arr_dmmv_q4_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_f16_f32", arr_dmmv_q5_k_f16_f32_len[reduc16], arr_dmmv_q5_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_f16_f32", arr_dmmv_q6_k_f16_f32_len[reduc16], arr_dmmv_q6_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_TQ1_0][i], "mul_mat_vec_tq1_0_f16_f32", arr_dmmv_tq1_0_f16_f32_len[reduc16], arr_dmmv_tq1_0_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_TQ2_0][i], "mul_mat_vec_tq2_0_f16_f32", arr_dmmv_tq2_0_f16_f32_len[reduc16], arr_dmmv_tq2_0_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_f16_f32", arr_dmmv_iq1_s_f16_f32_len[reduc16], arr_dmmv_iq1_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_f16_f32", arr_dmmv_iq1_m_f16_f32_len[reduc16], arr_dmmv_iq1_m_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_XXS][i], "mul_mat_vec_iq2_xxs_f16_f32", arr_dmmv_iq2_xxs_f16_f32_len[reduc16], arr_dmmv_iq2_xxs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_XS][i], "mul_mat_vec_iq2_xs_f16_f32", arr_dmmv_iq2_xs_f16_f32_len[reduc16], arr_dmmv_iq2_xs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_S][i], "mul_mat_vec_iq2_s_f16_f32", arr_dmmv_iq2_s_f16_f32_len[reduc16], arr_dmmv_iq2_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ3_XXS][i], "mul_mat_vec_iq3_xxs_f16_f32", arr_dmmv_iq3_xxs_f16_f32_len[reduc16], arr_dmmv_iq3_xxs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ3_S][i], "mul_mat_vec_iq3_s_f16_f32", arr_dmmv_iq3_s_f16_f32_len[reduc16], arr_dmmv_iq3_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ4_XS][i], "mul_mat_vec_iq4_xs_f16_f32", arr_dmmv_iq4_xs_f16_f32_len[reduc16], arr_dmmv_iq4_xs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ4_NL][i], "mul_mat_vec_iq4_nl_f16_f32", arr_dmmv_iq4_nl_f16_f32_len[reduc16], arr_dmmv_iq4_nl_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_f16_f32", OCP_DMMV_LEN(arr_dmmv_mxfp4_f16_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_mxfp4_f16_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_NVFP4][i], "mul_mat_vec_nvfp4_f16_f32", OCP_DMMV_LEN(arr_dmmv_nvfp4_f16_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_nvfp4_f16_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - return vk_fa_pipeline_state{hsk, hsv, params.block_rows, params.block_cols, params.d_split, params.row_split, params.shmem_staging, params.path, params.workgroup_size, subgroup_size, aligned, f32acc, flags, params.limit_occupancy_shmem, k_type, v_type}; -} +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + const uint32_t subgroup_size_int = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; + const uint32_t wg_size_subgroup_int = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size_int : (subgroup_size_int * 4); -static std::vector get_fa_spec_constants(const vk_fa_pipeline_state& state) { - const auto fa_block_bytes = [](ggml_type t) -> uint32_t { - if (t == GGML_TYPE_F32) return 16u; - return (uint32_t) ggml_type_size(t); - }; - return { - /* 0 WorkGroupSize */ state.workgroup_size, - /* 1 Br */ state.Br, - /* 2 Bc */ state.Bc, - /* 3 HSK */ state.HSK, - /* 4 HSV */ state.HSV, - /* 5 Clamp */ static_cast(!state.aligned), - /* 6 D_split */ state.D_split, - /* 7 row_split */ state.row_split, - /* 8 SubGroupSize */ state.subgroup_size, - /* 9 SHMEM_STAGING */ state.shmem_staging ? 1u : 0u, - /*10 Flags */ state.flags, - /*11 LIMIT_OCCUPANCY_SHMEM */ state.limit_occupancy_shmem, - /*12 FaTypeK */ static_cast(state.k_type), - /*13 FaTypeV */ static_cast(state.v_type), - /*14 FaBlockBytesK */ fa_block_bytes(state.k_type), - /*15 FaBlockBytesV */ fa_block_bytes(state.v_type), - }; -} + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_q8_1_f32", arr_dmmv_q2_0_q8_1_f32_len[reduc], arr_dmmv_q2_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(2*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(2*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_q8_1_f32", arr_dmmv_q4_0_q8_1_f32_len[reduc], arr_dmmv_q4_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_q8_1_f32", arr_dmmv_q4_1_q8_1_f32_len[reduc], arr_dmmv_q4_1_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_q8_1_f32", arr_dmmv_q5_0_q8_1_f32_len[reduc], arr_dmmv_q5_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_q8_1_f32", arr_dmmv_q5_1_q8_1_f32_len[reduc], arr_dmmv_q5_1_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_q8_1_f32", arr_dmmv_q8_0_q8_1_f32_len[reduc], arr_dmmv_q8_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); -static bool ggml_vk_matmul_shmem_support(const vk_device& device, const std::vector& warptile, bool mul_mat_id, ggml_type src0_type) { + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_q8_1_f32", arr_dmmv_mxfp4_q8_1_f32_len[reduc], arr_dmmv_mxfp4_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(2*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(2*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); - uint32_t lut_size = 0; - switch (src0_type) { - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - // Regular matmul uses the compact uint16_t IQ1 grid; the expanded - // uint32_t grid is only enabled for the q8_1/int-dot vector path. - lut_size = 2*2048; - break; - case GGML_TYPE_IQ2_XXS: - lut_size = 8*256; - break; - case GGML_TYPE_IQ2_XS: - lut_size = 8*512; - break; - case GGML_TYPE_IQ2_S: - lut_size = 8*1024; - break; - case GGML_TYPE_IQ3_XXS: - lut_size = 4*256; - break; - case GGML_TYPE_IQ3_S: - lut_size = 4*512; - break; - case GGML_TYPE_IQ4_NL: - case GGML_TYPE_IQ4_XS: - case GGML_TYPE_MXFP4: - lut_size = 4*16; - break; - case GGML_TYPE_NVFP4: - // Same kvalues budget as MXFP4 plus ue4m3_fp32_lut[128] (types.glsl, DATA_A_NVFP4). - lut_size = 4*16 + 128u * (uint32_t)sizeof(float); - break; - default: - break; - } + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_q8_1_f32", arr_dmmv_q2_k_q8_1_f32_len[reduc], arr_dmmv_q2_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(2*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(2*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_q8_1_f32", arr_dmmv_q3_k_q8_1_f32_len[reduc], arr_dmmv_q3_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_q8_1_f32", arr_dmmv_q4_k_q8_1_f32_len[reduc], arr_dmmv_q4_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_q8_1_f32", arr_dmmv_q5_k_q8_1_f32_len[reduc], arr_dmmv_q5_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_q8_1_f32", arr_dmmv_q6_k_q8_1_f32_len[reduc], arr_dmmv_q6_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_kq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_kq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); - // Needs to be kept up to date on shader changes - // Needs to stay aligned with ggml_vk_mul_mm_spec. - const bool intel_shmem_stride_pad_zero = device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support && - device->driver_id == vk::DriverId::eIntelProprietaryWindows; - const uint32_t bank_conflict_offset = intel_shmem_stride_pad_zero ? 0 : (device->coopmat_support ? 8 : 1); - const uint32_t type_size = device->fp16 ? sizeof(ggml_fp16_t) : sizeof(float); - const uint32_t warps = warptile[0] / warptile[10]; + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_q8_1_f32", arr_dmmv_iq1_s_q8_1_f32_len[reduc], arr_dmmv_iq1_s_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_iq_int(i), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_q8_1_f32", arr_dmmv_iq1_m_q8_1_f32_len[reduc], arr_dmmv_iq1_m_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_iq_int(i), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(i), i+1}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_IQ4_XS][i], "mul_mat_vec_iq4_xs_q8_1_f32", arr_dmmv_iq4_xs_q8_1_f32_len[reduc], arr_dmmv_iq4_xs_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_int_n(1*rm_stdq_int, i), 1, 1}, {wg_size_subgroup_int, rm_int_n(1*rm_stdq_int, i), i+1}, 1, true, use_subgroups, subgroup_size_int); - const uint32_t load_bufs = (warptile[1] + warptile[2]) * (warptile[3] + bank_conflict_offset) * type_size; - const uint32_t mmid_row_ids = mul_mat_id ? (warptile[2] * 2 * sizeof(uint16_t)) : 0; - const uint32_t coopmat_stage = device->coopmat_support ? warptile[7] * warptile[8] / warps * sizeof(float) : 0; - const uint32_t ballots_sh = mul_mat_id ? (warps * 4 * sizeof(uint32_t)) : 0; + } +#endif // GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT + } - const uint32_t total_size = load_bufs + mmid_row_ids + coopmat_stage + lut_size + ballots_sh; - const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_F32 ], "mul_mat_vec_id_f32_f32", arr_dmmv_id_f32_f32_f32_len[reduc], arr_dmmv_id_f32_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1, 1, 1}, {wg_size_subgroup, 1}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_F16 ], "mul_mat_vec_id_f16_f32", arr_dmmv_id_f16_f32_f32_len[reduc], arr_dmmv_id_f16_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2, 1, 1}, {wg_size_subgroup, 2}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_BF16], "mul_mat_vec_id_bf16_f32", arr_dmmv_id_bf16_f32_f32_len[reduc], arr_dmmv_id_bf16_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2, 1, 1}, {wg_size_subgroup, 2}, 1, false, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q1_0], "mul_mat_vec_id_q1_0_f32", arr_dmmv_id_q1_0_f32_f32_len[reduc], arr_dmmv_id_q1_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q2_0], "mul_mat_vec_id_q2_0_f32", arr_dmmv_id_q2_0_f32_f32_len[reduc], arr_dmmv_id_q2_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_0], "mul_mat_vec_id_q4_0_f32", arr_dmmv_id_q4_0_f32_f32_len[reduc], arr_dmmv_id_q4_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_1], "mul_mat_vec_id_q4_1_f32", arr_dmmv_id_q4_1_f32_f32_len[reduc], arr_dmmv_id_q4_1_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_0], "mul_mat_vec_id_q5_0_f32", arr_dmmv_id_q5_0_f32_f32_len[reduc], arr_dmmv_id_q5_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_1], "mul_mat_vec_id_q5_1_f32", arr_dmmv_id_q5_1_f32_f32_len[reduc], arr_dmmv_id_q5_1_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q8_0], "mul_mat_vec_id_q8_0_f32", arr_dmmv_id_q8_0_f32_f32_len[reduc], arr_dmmv_id_q8_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q2_K], "mul_mat_vec_id_q2_k_f32", arr_dmmv_id_q2_k_f32_f32_len[reduc16], arr_dmmv_id_q2_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q3_K], "mul_mat_vec_id_q3_k_f32", arr_dmmv_id_q3_k_f32_f32_len[reduc16], arr_dmmv_id_q3_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_K], "mul_mat_vec_id_q4_k_f32", arr_dmmv_id_q4_k_f32_f32_len[reduc16], arr_dmmv_id_q4_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_K], "mul_mat_vec_id_q5_k_f32", arr_dmmv_id_q5_k_f32_f32_len[reduc16], arr_dmmv_id_q5_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q6_K], "mul_mat_vec_id_q6_k_f32", arr_dmmv_id_q6_k_f32_f32_len[reduc16], arr_dmmv_id_q6_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_TQ1_0], "mul_mat_vec_id_tq1_0_f32", arr_dmmv_id_tq1_0_f32_f32_len[reduc16], arr_dmmv_id_tq1_0_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_TQ2_0], "mul_mat_vec_id_tq2_0_f32", arr_dmmv_id_tq2_0_f32_f32_len[reduc16], arr_dmmv_id_tq2_0_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ1_S], "mul_mat_vec_id_iq1_s_f32", arr_dmmv_id_iq1_s_f32_f32_len[reduc16], arr_dmmv_id_iq1_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ1_M], "mul_mat_vec_id_iq1_m_f32", arr_dmmv_id_iq1_m_f32_f32_len[reduc16], arr_dmmv_id_iq1_m_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_XXS], "mul_mat_vec_id_iq2_xxs_f32", arr_dmmv_id_iq2_xxs_f32_f32_len[reduc16], arr_dmmv_id_iq2_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_XS], "mul_mat_vec_id_iq2_xs_f32", arr_dmmv_id_iq2_xs_f32_f32_len[reduc16], arr_dmmv_id_iq2_xs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_S], "mul_mat_vec_id_iq2_s_f32", arr_dmmv_id_iq2_s_f32_f32_len[reduc16], arr_dmmv_id_iq2_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ3_XXS], "mul_mat_vec_id_iq3_xxs_f32", arr_dmmv_id_iq3_xxs_f32_f32_len[reduc16], arr_dmmv_id_iq3_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ3_S], "mul_mat_vec_id_iq3_s_f32", arr_dmmv_id_iq3_s_f32_f32_len[reduc16], arr_dmmv_id_iq3_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ4_XS], "mul_mat_vec_id_iq4_xs_f32", arr_dmmv_id_iq4_xs_f32_f32_len[reduc16], arr_dmmv_id_iq4_xs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ4_NL], "mul_mat_vec_id_iq4_nl_f32", arr_dmmv_id_iq4_nl_f32_f32_len[reduc16], arr_dmmv_id_iq4_nl_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_MXFP4], "mul_mat_vec_id_mxfp4_f32", OCP_DMMV_LEN(arr_dmmv_id_mxfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_id_mxfp4_f32_f32, reduc16), "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_NVFP4], "mul_mat_vec_id_nvfp4_f32", OCP_DMMV_LEN(arr_dmmv_id_nvfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_id_nvfp4_f32_f32, reduc16), "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - VK_LOG_DEBUG("ggml_vk_matmul_shmem_support(warptile=(" << warptile[0] << "," << warptile[1] << "," << warptile[2] << "), " - "mul_mat_id=" << mul_mat_id << ", src0_type=" << ggml_type_name(src0_type) << ", supported=" << supported); +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (device->integer_dot_product) { + const uint32_t subgroup_size_int = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; + const uint32_t wg_size_subgroup_int = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size_int : (subgroup_size_int * 4); - return supported; -} + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q2_0], "mul_mat_vec_id_q2_0_q8_1_f32", arr_dmmv_id_q2_0_q8_1_f32_len[reduc], arr_dmmv_id_q2_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(2*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(2*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_0], "mul_mat_vec_id_q4_0_q8_1_f32", arr_dmmv_id_q4_0_q8_1_f32_len[reduc], arr_dmmv_id_q4_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_1], "mul_mat_vec_id_q4_1_q8_1_f32", arr_dmmv_id_q4_1_q8_1_f32_len[reduc], arr_dmmv_id_q4_1_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_0], "mul_mat_vec_id_q5_0_q8_1_f32", arr_dmmv_id_q5_0_q8_1_f32_len[reduc], arr_dmmv_id_q5_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_1], "mul_mat_vec_id_q5_1_q8_1_f32", arr_dmmv_id_q5_1_q8_1_f32_len[reduc], arr_dmmv_id_q5_1_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q8_0], "mul_mat_vec_id_q8_0_q8_1_f32", arr_dmmv_id_q8_0_q8_1_f32_len[reduc], arr_dmmv_id_q8_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); -// Shmem usage for the q8_1 mmq shader (mul_mmq.comp), which uses -// block_a_cache / block_b_cache layouts (see mul_mmq_shmem_types.glsl) rather -// than the float load buffers checked by ggml_vk_matmul_shmem_support. -// Sizes follow std430 rules. Returns false for types without a q8_1 pipeline. -static bool ggml_vk_matmul_int_shmem_support(const vk_device& device, const std::vector& warptile, bool mul_mat_id, ggml_type src0_type) { + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_MXFP4], "mul_mat_vec_id_mxfp4_q8_1_f32", arr_dmmv_id_mxfp4_q8_1_f32_len[reduc], arr_dmmv_id_mxfp4_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(2*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(2*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); - // FLOAT_TYPE in the shader is float16_t with fp16 support, otherwise float. - const uint32_t fp_size = device->fp16 ? 2u : 4u; - const uint32_t fp_align = fp_size; - const uint32_t fp2_size = 2u * fp_size; - const uint32_t fp2_align = device->fp16 ? 4u : 8u; + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q2_K], "mul_mat_vec_id_q2_k_q8_1_f32", arr_dmmv_id_q2_k_q8_1_f32_len[reduc], arr_dmmv_id_q2_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(2*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(2*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q3_K], "mul_mat_vec_id_q3_k_q8_1_f32", arr_dmmv_id_q3_k_q8_1_f32_len[reduc], arr_dmmv_id_q3_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_K], "mul_mat_vec_id_q4_k_q8_1_f32", arr_dmmv_id_q4_k_q8_1_f32_len[reduc], arr_dmmv_id_q4_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_K], "mul_mat_vec_id_q5_k_q8_1_f32", arr_dmmv_id_q5_k_q8_1_f32_len[reduc], arr_dmmv_id_q5_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q6_K], "mul_mat_vec_id_q6_k_q8_1_f32", arr_dmmv_id_q6_k_q8_1_f32_len[reduc], arr_dmmv_id_q6_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_kq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_kq_int)}, 1, true, use_subgroups, subgroup_size_int); - struct member { uint32_t size, align; }; - auto std430_size = [](std::initializer_list members) { - uint32_t off = 0, struct_align = 1; - for (const auto &m : members) { - off = (off + m.align - 1) & ~(m.align - 1); - off += m.size; - struct_align = std::max(struct_align, m.align); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_IQ1_S], "mul_mat_vec_id_iq1_s_q8_1_f32", arr_dmmv_id_iq1_s_q8_1_f32_len[reduc], arr_dmmv_id_iq1_s_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_iq_int(0), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(0)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_IQ1_M], "mul_mat_vec_id_iq1_m_q8_1_f32", arr_dmmv_id_iq1_m_q8_1_f32_len[reduc], arr_dmmv_id_iq1_m_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_iq_int(0), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(0)}, 1, true, use_subgroups, subgroup_size_int); + ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_IQ4_XS], "mul_mat_vec_id_iq4_xs_q8_1_f32", arr_dmmv_id_iq4_xs_q8_1_f32_len[reduc], arr_dmmv_id_iq4_xs_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_id(1*rm_stdq_int), 1, 1}, {wg_size_subgroup_int, rm_id(1*rm_stdq_int)}, 1, true, use_subgroups, subgroup_size_int); } - return (off + struct_align - 1) & ~(struct_align - 1); - }; - - uint32_t block_a_size = 0; - switch (src0_type) { - case GGML_TYPE_Q2_0: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + dm - case GGML_TYPE_Q4_0: block_a_size = std430_size({{16, 4}, {fp_size, fp_align}}); break; // qs[16/4] + dm - case GGML_TYPE_Q4_1: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[16/4] + dm(vec2) - case GGML_TYPE_Q5_0: block_a_size = std430_size({{16, 4}, {4, 4}, {fp_size, fp_align}}); break; // qs[16/4] + qh + dm - case GGML_TYPE_Q5_1: block_a_size = std430_size({{16, 4}, {4, 4}, {fp2_size, fp2_align}}); break; // qs[16/4] + qh + dm(vec2) - case GGML_TYPE_Q8_0: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + dm - case GGML_TYPE_MXFP4: block_a_size = std430_size({{32, 4}, {fp_size, fp_align}}); break; // qs[8] + d - case GGML_TYPE_Q2_K: block_a_size = std430_size({{ 8, 4}, {2, 2}, {fp2_size, fp2_align}}); break; // qs[2] + scales(u8vec2) + dm(vec2) - case GGML_TYPE_Q3_K: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[4] + d_scales(vec2) - case GGML_TYPE_Q4_K: block_a_size = std430_size({{16, 4}, {fp2_size, fp2_align}}); break; // qs[4] + dm(vec2) - case GGML_TYPE_Q5_K: block_a_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); break; // qs[8] + dm(vec2) - case GGML_TYPE_Q6_K: block_a_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); break; // qs[8] + d_scales(vec2) - default: - return false; +#endif // GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT } - // block_b_cache: { int32_t qs[8]; FLOAT_TYPEV2 ds; } - const uint32_t block_b_size = std430_size({{32, 4}, {fp2_size, fp2_align}}); +#undef OCP_DMMV_DATA +#undef OCP_DMMV_LEN - const uint32_t BM = warptile[1]; - const uint32_t BN = warptile[2]; - // mul_mmq.comp: BK_STEP=1 for MUL_MAT_ID, 4 otherwise. - const uint32_t BK_STEP = mul_mat_id ? 1u : 4u; +#if !defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + GGML_UNUSED(rm_stdq_int); + GGML_UNUSED(rm_kq_int); + GGML_UNUSED(is_rdna3); + GGML_UNUSED(rm_int_n); + GGML_UNUSED(rm_id); + GGML_UNUSED(rm_iq_int); +#endif - const uint32_t buf_a_size = BM * BK_STEP * block_a_size; - const uint32_t buf_b_size = BN * BK_STEP * block_b_size; - const uint32_t mmid_row_ids = mul_mat_id ? (BN * 2u * (uint32_t)sizeof(uint16_t)) : 0u; - - const uint32_t warps = warptile[0] / warptile[10]; - const uint32_t ballots_sh = mul_mat_id ? (warps * 4u * (uint32_t)sizeof(uint32_t)) : 0u; - - const uint32_t total_size = buf_a_size + buf_b_size + mmid_row_ids + ballots_sh; - const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - - VK_LOG_DEBUG("ggml_vk_matmul_int_shmem_support(warptile=(" << warptile[0] << "," << warptile[1] << "," << warptile[2] << "), " - "mul_mat_id=" << mul_mat_id << ", src0_type=" << ggml_type_name(src0_type) << ", total=" << total_size << ", supported=" << supported); - - return supported; -} + // dequant shaders + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_F32 ], "f32_to_f16", dequant_f32_len, dequant_f32_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q1_0], "dequant_q1_0", dequant_q1_0_len, dequant_q1_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 8, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q2_0], "dequant_q2_0", dequant_q2_0_len, dequant_q2_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 4, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_0], "dequant_q4_0", dequant_q4_0_len, dequant_q4_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_1], "dequant_q4_1", dequant_q4_1_len, dequant_q4_1_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_0], "dequant_q5_0", dequant_q5_0_len, dequant_q5_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_1], "dequant_q5_1", dequant_q5_1_len, dequant_q5_1_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q8_0], "dequant_q8_0", dequant_q8_0_len, dequant_q8_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant_transpose[GGML_TYPE_Q8_0], "dequant_q8_0_transpose", dequant_q8_0_transpose_len, dequant_q8_0_transpose_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q2_K], "dequant_q2_k", dequant_q2_k_len, dequant_q2_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q3_K], "dequant_q3_k", dequant_q3_k_len, dequant_q3_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_K], "dequant_q4_k", dequant_q4_k_len, dequant_q4_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_K], "dequant_q5_k", dequant_q5_k_len, dequant_q5_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q6_K], "dequant_q6_k", dequant_q6_k_len, dequant_q6_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_TQ1_0], "dequant_tq1_0", dequant_tq1_0_len, dequant_tq1_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 4, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_TQ2_0], "dequant_tq2_0", dequant_tq2_0_len, dequant_tq2_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ1_S], "dequant_iq1_s", dequant_iq1_s_len, dequant_iq1_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ1_M], "dequant_iq1_m", dequant_iq1_m_len, dequant_iq1_m_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_XXS], "dequant_iq2_xxs", dequant_iq2_xxs_len, dequant_iq2_xxs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_XS], "dequant_iq2_xs", dequant_iq2_xs_len, dequant_iq2_xs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_S], "dequant_iq2_s", dequant_iq2_s_len, dequant_iq2_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ3_XXS], "dequant_iq3_xxs", dequant_iq3_xxs_len, dequant_iq3_xxs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ3_S], "dequant_iq3_s", dequant_iq3_s_len, dequant_iq3_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ4_XS], "dequant_iq4_xs", dequant_iq4_xs_len, dequant_iq4_xs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ4_NL], "dequant_iq4_nl", dequant_iq4_nl_len, dequant_iq4_nl_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_MXFP4], "dequant_mxfp4", dequant_mxfp4_len, dequant_mxfp4_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_NVFP4], "dequant_nvfp4", dequant_nvfp4_len, dequant_nvfp4_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); -struct GpuPipelineConfig { - // GPU architecture identifier. - // Example: vk_device_architecture::AMD_GCN - vk_device_architecture arch; + // get_rows + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_F32 ], "get_rows_f32", get_rows_f32_len, get_rows_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_F16 ], "get_rows_f16", get_rows_f16_len, get_rows_f16_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_BF16], "get_rows_bf16", get_rows_bf16_len, get_rows_bf16_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q1_0], "get_rows_q1_0", get_rows_q1_0_len, get_rows_q1_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q2_0], "get_rows_q2_0", get_rows_q2_0_len, get_rows_q2_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_0], "get_rows_q4_0", get_rows_q4_0_len, get_rows_q4_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_1], "get_rows_q4_1", get_rows_q4_1_len, get_rows_q4_1_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_0], "get_rows_q5_0", get_rows_q5_0_len, get_rows_q5_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_1], "get_rows_q5_1", get_rows_q5_1_len, get_rows_q5_1_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q8_0], "get_rows_q8_0", get_rows_q8_0_len, get_rows_q8_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q2_K], "get_rows_q2_k", get_rows_q2_k_len, get_rows_q2_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q3_K], "get_rows_q3_k", get_rows_q3_k_len, get_rows_q3_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_K], "get_rows_q4_k", get_rows_q4_k_len, get_rows_q4_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_K], "get_rows_q5_k", get_rows_q5_k_len, get_rows_q5_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q6_K], "get_rows_q6_k", get_rows_q6_k_len, get_rows_q6_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_TQ1_0], "get_rows_tq1_0", get_rows_tq1_0_len, get_rows_tq1_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_TQ2_0], "get_rows_tq2_0", get_rows_tq2_0_len, get_rows_tq2_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ1_S], "get_rows_iq1_s", get_rows_iq1_s_len, get_rows_iq1_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ1_M], "get_rows_iq1_m", get_rows_iq1_m_len, get_rows_iq1_m_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_XXS], "get_rows_iq2_xxs", get_rows_iq2_xxs_len, get_rows_iq2_xxs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_XS], "get_rows_iq2_xs", get_rows_iq2_xs_len, get_rows_iq2_xs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_S], "get_rows_iq2_s", get_rows_iq2_s_len, get_rows_iq2_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ3_XXS], "get_rows_iq3_xxs", get_rows_iq3_xxs_len, get_rows_iq3_xxs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ3_S], "get_rows_iq3_s", get_rows_iq3_s_len, get_rows_iq3_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ4_XS], "get_rows_iq4_xs", get_rows_iq4_xs_len, get_rows_iq4_xs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ4_NL], "get_rows_iq4_nl", get_rows_iq4_nl_len, get_rows_iq4_nl_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_MXFP4], "get_rows_mxfp4", get_rows_mxfp4_len, get_rows_mxfp4_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_NVFP4], "get_rows_nvfp4", get_rows_nvfp4_len, get_rows_nvfp4_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_I32], "get_rows_i32", get_rows_i32_len, get_rows_i32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - // Mapping of pipeline names to their specific subgroup sizes. - // Example: {"soft_max_f32", 64} - std::unordered_map pipelines; + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_F32 ], "get_rows_f32_f32", get_rows_f32_f32_len, get_rows_f32_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_F16 ], "get_rows_f16_f32", get_rows_f16_f32_len, get_rows_f16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_BF16], "get_rows_bf16_f32", get_rows_bf16_f32_len, get_rows_bf16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q1_0], "get_rows_q1_0_f32", get_rows_q1_0_f32_len, get_rows_q1_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q2_0], "get_rows_q2_0_f32", get_rows_q2_0_f32_len, get_rows_q2_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_0], "get_rows_q4_0_f32", get_rows_q4_0_f32_len, get_rows_q4_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_1], "get_rows_q4_1_f32", get_rows_q4_1_f32_len, get_rows_q4_1_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_0], "get_rows_q5_0_f32", get_rows_q5_0_f32_len, get_rows_q5_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_1], "get_rows_q5_1_f32", get_rows_q5_1_f32_len, get_rows_q5_1_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q8_0], "get_rows_q8_0_f32", get_rows_q8_0_f32_len, get_rows_q8_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q2_K], "get_rows_q2_k_f32", get_rows_q2_k_f32_len, get_rows_q2_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q3_K], "get_rows_q3_k_f32", get_rows_q3_k_f32_len, get_rows_q3_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_K], "get_rows_q4_k_f32", get_rows_q4_k_f32_len, get_rows_q4_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_K], "get_rows_q5_k_f32", get_rows_q5_k_f32_len, get_rows_q5_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q6_K], "get_rows_q6_k_f32", get_rows_q6_k_f32_len, get_rows_q6_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_TQ1_0], "get_rows_tq1_0_f32", get_rows_tq1_0_f32_len, get_rows_tq1_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_TQ2_0], "get_rows_tq2_0_f32", get_rows_tq2_0_f32_len, get_rows_tq2_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ1_S], "get_rows_iq1_s_f32", get_rows_iq1_s_f32_len, get_rows_iq1_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ1_M], "get_rows_iq1_m_f32", get_rows_iq1_m_f32_len, get_rows_iq1_m_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_XXS], "get_rows_iq2_xxs_f32", get_rows_iq2_xxs_f32_len, get_rows_iq2_xxs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_XS], "get_rows_iq2_xs_f32", get_rows_iq2_xs_f32_len, get_rows_iq2_xs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_S], "get_rows_iq2_s_f32", get_rows_iq2_s_f32_len, get_rows_iq2_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ3_XXS], "get_rows_iq3_xxs_f32", get_rows_iq3_xxs_f32_len, get_rows_iq3_xxs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ3_S], "get_rows_iq3_s_f32", get_rows_iq3_s_f32_len, get_rows_iq3_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ4_XS], "get_rows_iq4_xs_f32", get_rows_iq4_xs_f32_len, get_rows_iq4_xs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ4_NL], "get_rows_iq4_nl_f32", get_rows_iq4_nl_f32_len, get_rows_iq4_nl_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_MXFP4], "get_rows_mxfp4_f32", get_rows_mxfp4_f32_len, get_rows_mxfp4_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_NVFP4], "get_rows_nvfp4_f32", get_rows_nvfp4_f32_len, get_rows_nvfp4_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_get_rows_back_f32, "get_rows_back_f32", get_rows_back_f32_len, get_rows_back_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {256, 1, 1}, {}, 1, true); - // Default subgroup size for this GPU. - // Defaults to 0 if not explicitly provided. - uint32_t default_subgroup_size = 0; -}; + ggml_vk_create_pipeline(device, device->pipeline_matmul_split_k_reduce, "split_k_reduce", split_k_reduce_len, split_k_reduce_data, "main", 2, 2 * sizeof(uint32_t), {256 * 4, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_flash_attn_split_k_reduce, "fa_split_k_reduce", fa_split_k_reduce_len, fa_split_k_reduce_data, "main", 3, sizeof(vk_op_flash_attn_split_k_reduce_push_constants), {1, device->subgroup_size, 1}, {device->subgroup_size}, 1, true); -// Pipeline configuration for RDNA1 GPUs. -static const std::unordered_map rdna1_pipelines = { - {"soft_max", 64}, {"im2col", 64}, - {"argmax", 64}, {"mul_mat_vec", 64}, - {"mul_mat_vec_f16", 32}, {"mul_mat_vec_f32_f16", 32} -}; + if (device->vendor_id == VK_VENDOR_ID_INTEL && (device->architecture == INTEL_XE2 || (device->architecture == INTEL_XE1 && device->coopmat_support && device->uma))) { + auto upper_power_of_2 = [&](uint32_t in) { + GGML_ASSERT(in != 0); + if (in <= 1) return 1u; + uint32_t ret = in - 1; + ret |= ret >> 1; + ret |= ret >> 2; + ret |= ret >> 4; + ret |= ret >> 8; + ret |= ret >> 16; + return ret + 1; + }; -// Pipeline configuration for RDNA2 GPUs. -static const std::unordered_map rdna2_pipelines = { - {"soft_max", 64}, {"im2col", 64}, -}; + uint32_t xe_native_sub_group_size = 16; + if (device->architecture == INTEL_XE1) { + xe_native_sub_group_size = 8; + } + + for (auto& it : device->pipeline_xe_fa_decode_dual_phases) { + const uint32_t split_p_chunk = 32; + auto HdQk = it.first; + auto& pipelines = it.second; + uint32_t head_dim_qk = std::get<0>(HdQk); + uint32_t head_dim_pv = std::get<1>(HdQk); + uint32_t gqa_ratio = std::get<2>(HdQk); + uint32_t q_len = std::get<3>(HdQk); + const uint32_t out_dim_per_wg = gqa_ratio > 16 ? 8 : 16; + uint32_t aligned_q_len = upper_power_of_2(q_len); + uint32_t group_sz_ph1 = std::min(std::max(aligned_q_len * xe_native_sub_group_size, 64u), 256u); + uint32_t out_per_wg_ph1 = std::min(q_len, 256u / xe_native_sub_group_size); + uint32_t aligned_gqa_ratio = upper_power_of_2(gqa_ratio); + uint32_t split_p_per_iter_ph2 = 256; + uint32_t split_p_per_warp = 16; + uint32_t group_sz_ph2 = (split_p_per_iter_ph2 / split_p_per_warp) * xe_native_sub_group_size; + uint32_t out_per_wg_ph2 = std::min(std::max(16u / aligned_gqa_ratio, 1u), q_len); + ggml_vk_create_pipeline(device, pipelines.first, "xe_fa_decode_ph1", fa_decode_ph1_cm1_len, fa_decode_ph1_cm1_data, "main", 5, sizeof(vk_fa_xe_opt_push_constants), { 1, 32, 1 }, { group_sz_ph1, gqa_ratio, head_dim_qk, xe_native_sub_group_size, split_p_chunk, out_per_wg_ph1 }, 1, false, true, xe_native_sub_group_size); + ggml_vk_create_pipeline(device, pipelines.second, "xe_fa_decode_ph2", fa_decode_ph2_cm1_len, fa_decode_ph2_cm1_data, "main", 5, sizeof(vk_fa_xe_opt_push_constants), { 1, 1, 1 }, { group_sz_ph2, gqa_ratio, head_dim_pv, out_per_wg_ph2, xe_native_sub_group_size, split_p_per_iter_ph2, split_p_chunk, out_dim_per_wg }, 1, false, true, xe_native_sub_group_size); + } + } -static constexpr uint32_t RDNA_DEFAULT_SUBGROUP_SIZE = 32; + for (auto &it : device->pipeline_fa_mask_opt) { + auto BrBc = it.first; + ggml_vk_create_pipeline(device, it.second, "fa_mask_opt", fa_mask_opt_len, fa_mask_opt_data, "main", 2, sizeof(vk_op_flash_attn_mask_opt_push_constants), {1, 1, 1}, {128, 128 / device->subgroup_size, BrBc.first, BrBc.second}, 1, true, true, device->subgroup_size); + } -// Define configurations for different GPUs. -static std::vector gpu_pipeline_configs = { - { - vk_device_architecture::AMD_RDNA1, - { - rdna1_pipelines, - }, - RDNA_DEFAULT_SUBGROUP_SIZE - }, { - vk_device_architecture::AMD_RDNA2, - { - rdna2_pipelines, - }, - RDNA_DEFAULT_SUBGROUP_SIZE - }, -}; - -static uint32_t get_subgroup_size(const std::string &pipeline_name, const vk_device_architecture &arch) { - for (const auto &config : gpu_pipeline_configs) { - if (config.arch == arch) { - auto pipIt = config.pipelines.find(pipeline_name); - if (pipIt != config.pipelines.end()) { - return pipIt->second; - } - std::vector> sorted_pipelines(config.pipelines.begin(), config.pipelines.end()); - std::sort(sorted_pipelines.begin(), sorted_pipelines.end(), - [](const auto &a, const auto &b) { return a.first.size() > b.first.size(); }); - for (const auto &entry : sorted_pipelines) { - if (pipeline_name.find(entry.first) != std::string::npos) { - return entry.second; - } - } - return config.default_subgroup_size; + // Large workgroup so the per-row KV scan parallelizes; capped to device limits. + const uint32_t compact_max = std::min({1024u, device->properties.limits.maxComputeWorkGroupInvocations, device->properties.limits.maxComputeWorkGroupSize[0]}); + + // Fast ballot prefix-sum path when the device supports full subgroups; otherwise + // a shared-memory prefix-sum fallback. Both emit a deterministic ascending list. + device->fa_sparse_compact_use_subgroups = device->subgroup_ballot && device->subgroup_require_full_support; + if (device->fa_sparse_compact_use_subgroups) { + const uint32_t compact_wg = std::max(device->subgroup_size, (compact_max / device->subgroup_size) * device->subgroup_size); + const uint32_t compact_num_sg = compact_wg / device->subgroup_size; + ggml_vk_create_pipeline(device, device->pipeline_fa_sparse_compact_subgroup, "fa_sparse_compact_subgroup", fa_sparse_compact_subgroup_len, fa_sparse_compact_subgroup_data, "main", 2, sizeof(vk_op_flash_attn_sparse_compact_push_constants), {1, 1, 1}, {compact_wg, compact_num_sg}, 1, true, true, device->subgroup_size); + } else { + ggml_vk_create_pipeline(device, device->pipeline_fa_sparse_compact, "fa_sparse_compact", fa_sparse_compact_len, fa_sparse_compact_data, "main", 2, sizeof(vk_op_flash_attn_sparse_compact_push_constants), {1, 1, 1}, {compact_max}, 1, true); } } - return 0; // If no matching configuration is found -} -// Whether scalar flash attention will use the MMQ path for the given K/V types. -static bool ggml_vk_fa_type_needs_shmem(ggml_type type) { - switch (type) { - case GGML_TYPE_IQ4_NL: - return true; - default: - return false; + if (device->subgroup_clustered && device->subgroup_require_full_support) { + ggml_vk_create_pipeline(device, device->pipeline_quantize_q8_1_x4, "quantize_q8_1_x4", quantize_q8_1_x4_subgroup_len, quantize_q8_1_x4_subgroup_data, "main", 2, sizeof(vk_quantize_q8_1_push_constants), {32 * device->subgroup_size / 8, 1, 1}, { device->subgroup_size }, 1, true, true); + } else { + ggml_vk_create_pipeline(device, device->pipeline_quantize_q8_1_x4, "quantize_q8_1_x4", quantize_q8_1_x4_len, quantize_q8_1_x4_data, "main", 2, sizeof(vk_quantize_q8_1_push_constants), {32 * device->subgroup_size / 8, 1, 1}, { device->subgroup_size }, 1); } -} - -static bool ggml_vk_fa_scalar_uses_mmq(const vk_device& device, ggml_type k_type, ggml_type v_type) { -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - return device->integer_dot_product && device->subgroup_clustered && - !ggml_vk_fa_type_needs_shmem(v_type) && - (k_type == GGML_TYPE_Q4_0 || k_type == GGML_TYPE_Q4_1 || - k_type == GGML_TYPE_Q5_0 || k_type == GGML_TYPE_Q5_1 || - k_type == GGML_TYPE_Q8_0); -#else - GGML_UNUSED(device); - GGML_UNUSED(k_type); - GGML_UNUSED(v_type); - return false; -#endif -} -// load_shaders walks the pipeline list under compile_mutex and either claims -// the requested pipeline for compilation or, if another thread is already -// compiling it, drops the lock and waits on compile_cv. Compiles themselves -// run unlocked. -struct CompileTask { - vk_pipeline pipeline; - size_t spv_size; - const void * spv_data; - std::string entrypoint; - uint32_t parameter_count; - std::array wg_denoms; - std::vector specialization_constants; - bool disable_robustness; - bool require_full_subgroups; - uint32_t required_subgroup_size; -}; + for (uint32_t i = 0; i < p021_max_gqa_ratio; ++i) { + if (device->subgroup_arithmetic && device->subgroup_require_full_support) { + ggml_vk_create_pipeline2(device, device->pipeline_mul_mat_vec_p021_f16_f32[i], "mul_mat_vec_p021_f16_f32"+std::to_string(i+1), mul_mat_vec_p021_f16_f32_subgroup_add_len, mul_mat_vec_p021_f16_f32_subgroup_add_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_p021_push_constants), {1, 1, 1}, {device->subgroup_size, i + 1}, 1, true, true); + } else { + ggml_vk_create_pipeline2(device, device->pipeline_mul_mat_vec_p021_f16_f32[i], "mul_mat_vec_p021_f16_f32"+std::to_string(i+1), mul_mat_vec_p021_f16_f32_len, mul_mat_vec_p021_f16_f32_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_p021_push_constants), {1, 1, 1}, {device->subgroup_size, i + 1}, 1, true); + } + } + ggml_vk_create_pipeline(device, device->pipeline_mul_mat_vec_nc_f16_f32, "mul_mat_vec_nc_f16_f32", mul_mat_vec_nc_f16_f32_len, mul_mat_vec_nc_f16_f32_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_nc_push_constants), {1, 1, 1}, {}, 1); -static void ggml_vk_load_shaders(vk_device& device, vk_pipeline requested) { - VK_LOG_DEBUG("ggml_vk_load_shaders(" << device->name << ")"); + ggml_vk_create_pipeline(device, device->pipeline_norm_f32, "norm_f32", norm_f32_len, norm_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_group_norm_f32, "group_norm_f32", group_norm_f32_len, group_norm_f32_data, "main", 2, sizeof(vk_op_push_constants), {1, 1, 1}, {}, 1); - // some shaders have a minimum subgroup size - const uint32_t subgroup_size_8 = std::max(device->subgroup_size, 8u); - const uint32_t subgroup_size_16 = std::max(device->subgroup_size, 16u); - const uint32_t subgroup_size_32 = std::max(device->subgroup_size, 32u); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_f32, "rms_norm_f32", rms_norm_f32_len, rms_norm_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_f32, "rms_norm_mul_f32", rms_norm_f32_len, rms_norm_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_add_f32, "rms_norm_mul_add_f32", rms_norm_mul_add_f32_len, rms_norm_mul_add_f32_data, "main", 5, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_add_mul_f32, "rms_norm_mul_add_mul_f32", rms_norm_mul_add_f32_len, rms_norm_mul_add_f32_data, "main", 5, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1, 1}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_add_partials_f32, "rms_norm_mul_add_partials_f32", rms_norm_mul_add_partials_f32_len, rms_norm_mul_add_partials_f32_data, "main", 6, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_add_mul_partials_f32, "rms_norm_mul_add_mul_partials_f32", rms_norm_mul_add_partials_f32_len, rms_norm_mul_add_partials_f32_data, "main", 6, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1, 1}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_set_rows_f32_f32, "rms_norm_set_rows_f32_f32", rms_norm_set_rows_f32_f32_len, rms_norm_set_rows_f32_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_set_rows_f32_f16, "rms_norm_set_rows_f32_f16", rms_norm_set_rows_f32_f16_len, rms_norm_set_rows_f32_f16_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_partials_f32, "rms_norm_partials_f32", rms_norm_partials_f32_len, rms_norm_partials_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_partials_f32, "rms_norm_mul_partials_f32", rms_norm_partials_f32_len, rms_norm_partials_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1}, 1, true); - const uint32_t mul_mat_subgroup_size = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; - const uint32_t mul_mat_subgroup_size_8 = std::max(mul_mat_subgroup_size, 8u); - const uint32_t mul_mat_subgroup_size_16 = std::max(mul_mat_subgroup_size, 16u); - const uint32_t mul_mat_subgroup_size_32 = std::max(mul_mat_subgroup_size, 32u); + if (sizeof(vk_op_rms_norm_mul_rope_push_constants) <= device->properties.limits.maxPushConstantsSize) { + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_rope_f32_f32, "rms_norm_mul_rope_f32_f32", rms_norm_mul_rope_f32_f32_len, rms_norm_mul_rope_f32_f32_data, "main", 7, sizeof(vk_op_rms_norm_mul_rope_push_constants), {1, 1, 1}, {0, 1}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_rope_f32_f16, "rms_norm_mul_rope_f32_f16", rms_norm_mul_rope_f32_f16_len, rms_norm_mul_rope_f32_f16_data, "main", 7, sizeof(vk_op_rms_norm_mul_rope_push_constants), {1, 1, 1}, {0, 1}, 1, true); + } - const bool subgroup_min_size_16 = (!device->subgroup_size_control && device->subgroup_size >= 16) || - (device->subgroup_size_control && device->subgroup_max_size >= 16); + ggml_vk_create_pipeline(device, device->pipeline_rms_norm_back_f32, "rms_norm_back_f32", rms_norm_back_f32_len, rms_norm_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_l2_norm_f32, "l2_norm_f32", l2_norm_f32_len, l2_norm_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); - // mulmat - std::vector l_warptile, m_warptile, s_warptile, - l_warptile_id, m_warptile_id, s_warptile_id, - l_warptile_mmq, m_warptile_mmq, s_warptile_mmq, - l_warptile_mmq_int, m_warptile_mmq_int, s_warptile_mmq_int, - l_warptile_mmq_int_k, m_warptile_mmq_int_k, s_warptile_mmq_int_k, - l_warptile_mmq_k, m_warptile_mmq_k, s_warptile_mmq_k, - l_warptile_mmqid, m_warptile_mmqid, s_warptile_mmqid, - l_warptile_mmqid_int, m_warptile_mmqid_int, s_warptile_mmqid_int, - l_warptile_mmqid_int_k, m_warptile_mmqid_int_k, s_warptile_mmqid_int_k; - std::array l_wg_denoms, m_wg_denoms, s_wg_denoms, - l_mmq_wg_denoms, m_mmq_wg_denoms, s_mmq_wg_denoms, - l_mmq_wg_denoms_k, m_mmq_wg_denoms_k, s_mmq_wg_denoms_k, - l_mmqid_wg_denoms, m_mmqid_wg_denoms, s_mmqid_wg_denoms; + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_f32, "cpy_f32_f32", cpy_f32_f32_len, cpy_f32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_f16, "cpy_f32_f16", cpy_f32_f16_len, cpy_f32_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f16_f16, "cpy_f16_f16", cpy_f16_f16_len, cpy_f16_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f16_f32, "cpy_f16_f32", cpy_f16_f32_len, cpy_f16_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_bf16,"cpy_f32_bf16",cpy_f32_bf16_len,cpy_f32_bf16_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_bf16_f32,"cpy_bf16_f32",cpy_bf16_f32_len,cpy_bf16_f32_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_i32_f32, "cpy_i32_f32", cpy_i32_f32_len, cpy_i32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_i32, "cpy_f32_i32", cpy_f32_i32_len, cpy_f32_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - uint32_t l_align, m_align, s_align; + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_f32, "contig_cpy_f32_f32", contig_cpy_f32_f32_len, contig_cpy_f32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_f16, "contig_cpy_f32_f16", contig_cpy_f32_f16_len, contig_cpy_f32_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f16_f16, "contig_cpy_f16_f16", contig_cpy_f16_f16_len, contig_cpy_f16_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f16_f32, "contig_cpy_f16_f32", contig_cpy_f16_f32_len, contig_cpy_f16_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_bf16,"contig_cpy_f32_bf16",contig_cpy_f32_bf16_len,contig_cpy_f32_bf16_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_bf16_f32,"contig_cpy_bf16_f32",contig_cpy_bf16_f32_len,contig_cpy_bf16_f32_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_i32_f32, "contig_cpy_i32_f32", contig_cpy_i32_f32_len, contig_cpy_i32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_i32, "contig_cpy_f32_i32", contig_cpy_f32_i32_len, contig_cpy_f32_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - vk_pipeline wait_pipeline; - CompileTask claimed_task {}; - bool has_claimed_task = false; + ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_32, "cpy_transpose_32", cpy_transpose_32_len, cpy_transpose_32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_16, "cpy_transpose_16", cpy_transpose_16_len, cpy_transpose_16_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_02_32, "cpy_transpose_02_32", cpy_transpose_02_32_len, cpy_transpose_02_32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_02_16, "cpy_transpose_02_16", cpy_transpose_02_16_len, cpy_transpose_02_16_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); - // The rest of the walk reads and writes shared device state, so hold the - // lock until we're done deciding what to compile. - std::unique_lock compile_lock(device->compile_mutex); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q1_0], "cpy_f32_q1_0", cpy_f32_q1_0_len, cpy_f32_q1_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q2_0], "cpy_f32_q2_0", cpy_f32_q2_0_len, cpy_f32_q2_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q4_0], "cpy_f32_q4_0", cpy_f32_q4_0_len, cpy_f32_q4_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q4_1], "cpy_f32_q4_1", cpy_f32_q4_1_len, cpy_f32_q4_1_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q5_0], "cpy_f32_q5_0", cpy_f32_q5_0_len, cpy_f32_q5_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q5_1], "cpy_f32_q5_1", cpy_f32_q5_1_len, cpy_f32_q5_1_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q8_0], "cpy_f32_q8_0", cpy_f32_q8_0_len, cpy_f32_q8_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_IQ4_NL], "cpy_f32_iq4_nl", cpy_f32_iq4_nl_len, cpy_f32_iq4_nl_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - if (device->coopmat2) { - // spec constants and tile sizes for non-quant matmul/matmul_id - l_warptile = { 256, 128, 256, 64, 1 }; - m_warptile = { 256, 128, 128, 64, 0 }; - s_warptile = { 128, 64, 64, 64, 0 }; - l_wg_denoms = {128, 256, 1 }; - m_wg_denoms = {128, 128, 1 }; - s_wg_denoms = { 64, 64, 1 }; +#define SET_ROWS(src_idx, src, itype) \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_F32], "set_rows_" #src "_f32" #itype, set_rows_ ## src ## _f32 ## itype ## _len, set_rows_ ## src ## _f32 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_F16], "set_rows_" #src "_f16" #itype, set_rows_ ## src ## _f16 ## itype ## _len, set_rows_ ## src ## _f16 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_BF16], "set_rows_" #src "_bf16" #itype, set_rows_ ## src ## _bf16 ## itype ## _len, set_rows_ ## src ## _bf16 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q1_0], "set_rows_" #src "_q1_0" #itype, set_rows_ ## src ## _q1_0 ## itype ## _len, set_rows_ ## src ## _q1_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q2_0], "set_rows_" #src "_q2_0" #itype, set_rows_ ## src ## _q2_0 ## itype ## _len, set_rows_ ## src ## _q2_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q4_0], "set_rows_" #src "_q4_0" #itype, set_rows_ ## src ## _q4_0 ## itype ## _len, set_rows_ ## src ## _q4_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q4_1], "set_rows_" #src "_q4_1" #itype, set_rows_ ## src ## _q4_1 ## itype ## _len, set_rows_ ## src ## _q4_1 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q5_0], "set_rows_" #src "_q5_0" #itype, set_rows_ ## src ## _q5_0 ## itype ## _len, set_rows_ ## src ## _q5_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q5_1], "set_rows_" #src "_q5_1" #itype, set_rows_ ## src ## _q5_1 ## itype ## _len, set_rows_ ## src ## _q5_1 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q8_0], "set_rows_" #src "_q8_0" #itype, set_rows_ ## src ## _q8_0 ## itype ## _len, set_rows_ ## src ## _q8_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_IQ4_NL], "set_rows_" #src "_iq4_nl" #itype, set_rows_ ## src ## _iq4_nl ## itype ## _len, set_rows_ ## src ## _iq4_nl ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); - // spec constants and tile sizes for quant matmul (non-Qi_K) - l_warptile_mmq = { 256, 128, 256, 64, 1 }; - m_warptile_mmq = { 256, 128, 128, 64, 1 }; - s_warptile_mmq = { 256, 32, 64, 128, 0 }; - l_mmq_wg_denoms = { 128, 256, 1 }; - m_mmq_wg_denoms = { 128, 128, 1 }; - s_mmq_wg_denoms = { 32, 64, 1 }; + SET_ROWS(0, f32, _i32) + SET_ROWS(0, f32, _i64) + SET_ROWS(1, f16, _i32) + SET_ROWS(1, f16, _i64) +#undef SET_ROWS - // spec constants and tile sizes for quant matmul (Qi_K) - l_warptile_mmq_k = { 256, 128, 256, 64, 1 }; - m_warptile_mmq_k = { 256, 128, 128, 64, 1 }; - s_warptile_mmq_k = { 256, 32, 64, 128, 0 }; - l_mmq_wg_denoms_k = { 128, 256, 1 }; - m_mmq_wg_denoms_k = { 128, 128, 1 }; - s_mmq_wg_denoms_k = { 32, 64, 1 }; - // spec constants and tile sizes for quant matmul_id - const uint32_t mmqid_bk = device->coopmat2_decode_vector ? 64u : 32u; - l_warptile_mmqid = { 256, 128, 128, mmqid_bk, 1, device->subgroup_size }; - m_warptile_mmqid = { 256, 128, 64, mmqid_bk, 0, device->subgroup_size }; - s_warptile_mmqid = { 256, 128, 64, mmqid_bk, 0, device->subgroup_size }; - l_mmqid_wg_denoms = { 128, 128, 1 }; - m_mmqid_wg_denoms = { 128, 64, 1 }; - s_mmqid_wg_denoms = { 128, 64, 1 }; + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q1_0], "cpy_q1_0_f32", cpy_q1_0_f32_len, cpy_q1_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q1_0), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q2_0], "cpy_q2_0_f32", cpy_q2_0_f32_len, cpy_q2_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q2_0), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q4_0], "cpy_q4_0_f32", cpy_q4_0_f32_len, cpy_q4_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q4_0), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q4_1], "cpy_q4_1_f32", cpy_q4_1_f32_len, cpy_q4_1_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q4_1), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q5_0], "cpy_q5_0_f32", cpy_q5_0_f32_len, cpy_q5_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q5_0), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q5_1], "cpy_q5_1_f32", cpy_q5_1_f32_len, cpy_q5_1_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q5_1), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q8_0], "cpy_q8_0_f32", cpy_q8_0_f32_len, cpy_q8_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q8_0), 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_IQ4_NL], "cpy_iq4_nl_f32", cpy_iq4_nl_f32_len, cpy_iq4_nl_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_IQ4_NL), 1, 1}, {}, 1); - l_align = 128; - m_align = 64; - s_align = 32; - } else { - // Matrix cores require different warp group sizes - const uint32_t tm_l = device->coopmat_support ? device->coopmat_m : 4; - const uint32_t tm_m = device->coopmat_support ? device->coopmat_m : 4; - const uint32_t tm_s = device->coopmat_support ? device->coopmat_m : 2; - const uint32_t tn_l = device->coopmat_support ? device->coopmat_n : 4; - const uint32_t tn_m = device->coopmat_support ? device->coopmat_n : 2; - const uint32_t tn_s = device->coopmat_support ? device->coopmat_n : 2; - const uint32_t tk_l = device->coopmat_support ? device->coopmat_k : 1; - const uint32_t tk_m = device->coopmat_support ? device->coopmat_k : 1; - const uint32_t tk_s = device->coopmat_support ? device->coopmat_k : 1; + auto get_suffix = [](bool src0_f16, bool src1_f16, bool dst_f16) { + std::string s; + s += std::string(src0_f16 ? "_f16" : "_f32"); + s += std::string(src1_f16 ? "_f16" : "_f32"); + s += std::string(dst_f16 ? "_f16" : "_f32"); + return s; + }; - const uint32_t s_warptile_wm = device->subgroup_size == 8 ? 8 : 32; +#define CREATE_BINARY(name, namemod, spec, bindings) \ + for (int s0 : {0,1}) for (int s1 : {0,1}) for (int d : {0,1}) \ + ggml_vk_create_pipeline2(device, device->pipeline_ ## name ## namemod[s0][s1][d], \ + #name + get_suffix(s0, s1, d) + #namemod, name ## _len[s0][s1][d], name ## _data[s0][s1][d], \ + "main", (bindings), sizeof(vk_op_binary_push_constants), {512, 1, 1}, spec, 1); - l_warptile = { 128, 128, 128, 16, subgroup_size_8 * 2, 64, 2, tm_l, tn_l, tk_l, subgroup_size_8 }; - m_warptile = { 128, 64, 64, 16, subgroup_size_8, 32, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - s_warptile = { subgroup_size_32, 32, 32, 16, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, subgroup_size_8 }; + CREATE_BINARY(add, , {0}, 4) + CREATE_BINARY(add, _norepeat, {1}, 4) + CREATE_BINARY(sub, , {0}, 3) + CREATE_BINARY(sub, _norepeat, {1}, 3) + CREATE_BINARY(mul, , {0}, 3) + CREATE_BINARY(mul, _norepeat, {1}, 3) + CREATE_BINARY(div, , {0}, 3) + CREATE_BINARY(div, _norepeat, {1}, 3) + CREATE_BINARY(add_rms, , {0}, 4) + CREATE_BINARY(add_rms, _norepeat, {1}, 4) +#undef CREATE_BINARY - l_warptile_mmq = { 128, 128, 128, 32, subgroup_size_8 * 2, 64, 2, tm_l, tn_l, tk_l, subgroup_size_8 }; - m_warptile_mmq = { 128, 64, 64, 32, subgroup_size_8, 32, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - s_warptile_mmq = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, subgroup_size_8 }; + if (device->multi_add) { + for (uint32_t i = 0; i < MAX_FUSED_ADDS; ++i) { + ggml_vk_create_pipeline2(device, device->pipeline_multi_add[i], "multi_add_f32_" + std::to_string(i+1), multi_add_f32_len, multi_add_f32_data, "main", MAX_PARAMETER_COUNT, sizeof(vk_op_multi_add_push_constants), {512, 1, 1}, {i+2}, 1); + ggml_vk_create_pipeline2(device, device->pipeline_multi_add_rms[i], "multi_add_rms_f32_" + std::to_string(i+1), multi_add_rms_f32_len, multi_add_rms_f32_data, "main", MAX_PARAMETER_COUNT, sizeof(vk_op_multi_add_push_constants), {512, 1, 1}, {i+2}, 1); + } + } - // Integer MMQ has a smaller shared memory profile, but heavier register use - l_warptile_mmq_int = { 128, 128, 128, 32, subgroup_size_8 * 2, 64, 2, 4, 4, 1, subgroup_size_8 }; - m_warptile_mmq_int = { 128, 64, 64, 32, subgroup_size_8, 32, 2, 2, 2, 1, subgroup_size_8 }; - s_warptile_mmq_int = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, 2, 1, 1, subgroup_size_8 }; + ggml_vk_create_pipeline(device, device->pipeline_add_id_f32, "add_id_f32", add_id_f32_len, add_id_f32_data, "main", 4, sizeof(vk_op_add_id_push_constants), {1, 1, 1}, {}, 1); - // K-quants use even more registers, mitigate by setting WMITER to 1 - l_warptile_mmq_int_k = { 128, 128, 128, 32, subgroup_size_8 * 2, 64, 1, 4, 4, 1, subgroup_size_8 }; - m_warptile_mmq_int_k = { 128, 64, 64, 32, subgroup_size_8, 32, 1, 2, 2, 1, subgroup_size_8 }; - s_warptile_mmq_int_k = { subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 1, 2, 1, 1, subgroup_size_8 }; + ggml_vk_create_pipeline(device, device->pipeline_acc_f32, "acc_f32", acc_f32_len, acc_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {0, 1}, 1); + ggml_vk_create_pipeline(device, device->pipeline_set_f32, "set_f32", acc_f32_len, acc_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {0, 0}, 1); - l_warptile_id = { 128, 128, 128, 16, mul_mat_subgroup_size_16 * 2, 64, 2, tm_l, tn_l, tk_l, mul_mat_subgroup_size_16 }; - m_warptile_id = { 128, 64, 64, 16, mul_mat_subgroup_size_16, 32, 2, tm_m, tn_m, tk_m, mul_mat_subgroup_size_16 }; - s_warptile_id = { mul_mat_subgroup_size_16, 32, 32, 16, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, mul_mat_subgroup_size_16 }; + ggml_vk_create_pipeline(device, device->pipeline_concat_i8, "concat_i8", concat_i8_len, concat_i8_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_concat_i16, "concat_i16", concat_i16_len, concat_i16_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_concat_i32, "concat_i32", concat_i32_len, concat_i32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_concat_i64, "concat_i64", concat_i64_len, concat_i64_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - l_warptile_mmqid = { 128, 128, 128, 32, mul_mat_subgroup_size_8 * 2, 64, 2, tm_l, tn_l, tk_l, mul_mat_subgroup_size_8 }; - m_warptile_mmqid = { 128, 64, 64, 32, mul_mat_subgroup_size_8, 32, 2, tm_m, tn_m, tk_m, mul_mat_subgroup_size_8 }; - s_warptile_mmqid = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, tm_s, tn_s, tk_s, mul_mat_subgroup_size_8 }; + ggml_vk_create_pipeline(device, device->pipeline_upscale_nearest_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_NEAREST}, 1); + ggml_vk_create_pipeline(device, device->pipeline_upscale_bilinear_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BILINEAR}, 1); + ggml_vk_create_pipeline(device, device->pipeline_upscale_bicubic_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BICUBIC}, 1); + ggml_vk_create_pipeline(device, device->pipeline_upscale_bilinear_antialias_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BILINEAR | GGML_SCALE_FLAG_ANTIALIAS}, 1); - l_warptile_mmqid_int = { 128, 128, 128, 32, mul_mat_subgroup_size_8 * 2, 64, 2, 4, 4, 1, mul_mat_subgroup_size_8 }; - m_warptile_mmqid_int = { 128, 64, 64, 32, mul_mat_subgroup_size_8, 32, 2, 2, 2, 1, mul_mat_subgroup_size_8 }; - s_warptile_mmqid_int = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 2, 2, 1, 1, mul_mat_subgroup_size_8 }; + ggml_vk_create_pipeline(device, device->pipeline_scale_f32, "scale_f32", scale_f32_len, scale_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - l_warptile_mmqid_int_k = { 128, 128, 128, 32, mul_mat_subgroup_size_16 * 2, 64, 1, 4, 4, 1, mul_mat_subgroup_size_16 }; - m_warptile_mmqid_int_k = { 128, 64, 64, 32, mul_mat_subgroup_size_16, 32, 1, 2, 2, 1, mul_mat_subgroup_size_16 }; - s_warptile_mmqid_int_k = { mul_mat_subgroup_size_32, 32, 32, 32, s_warptile_wm, 32, 1, 2, 1, 1, mul_mat_subgroup_size_16 }; + ggml_vk_create_pipeline(device, device->pipeline_log[0], "log_f32", log_f32_len, log_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_log[1], "log_f16", log_f16_len, log_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - // chip specific tuning - if ((device->architecture == AMD_GCN) && (device->driver_id != vk::DriverId::eAmdProprietary)) { - m_warptile_mmq = m_warptile_mmq_int = { 256, 64, 64, 32, 16, 16, 2, 2, 2, 1, 16 }; - m_warptile_mmqid = m_warptile_mmqid_int = { 256, 64, 64, 32, 16, 16, 2, 2, 2, 1, 16 }; - } else if (device->vendor_id == VK_VENDOR_ID_AMD && device->coopmat_support && device->driver_id != vk::DriverId::eAmdProprietary) { - // This is intentionally using tx_m values, slight performance increase - l_warptile = { 256, 128, 128, 16, subgroup_size_8, 64, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - l_warptile_mmq = l_warptile_mmq_int = { 256, 128, 128, 32, subgroup_size_8, 64, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - l_warptile_mmq_int_k = { 256, 128, 128, 32, subgroup_size_16, 64, 1, 4, 2, 1, subgroup_size_16 }; - } else if (device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support) { - // Xe2/Xe3 with coopmat enabled - warptile performance tuning - l_warptile = { 512, 128, 128, 16, subgroup_size_8, 32, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - l_warptile_mmq = { 512, 128, 128, 32, subgroup_size_8, 32, 2, tm_m, tn_m, tk_m, subgroup_size_8 }; - } + ggml_vk_create_pipeline(device, device->pipeline_tri[0], "tri_f32", tri_f32_len, tri_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_tri[1], "tri_f16", tri_f16_len, tri_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - l_mmq_wg_denoms = l_wg_denoms = {128, 128, 1 }; - m_mmq_wg_denoms = m_wg_denoms = { 64, 64, 1 }; - s_mmq_wg_denoms = s_wg_denoms = { 32, 32, 1 }; - l_align = 128; - m_align = 64; - s_align = 32; + ggml_vk_create_pipeline(device, device->pipeline_diag[0], "diag_f32", diag_f32_len, diag_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_diag[1], "diag_f16", diag_f16_len, diag_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - for (uint32_t i = 0; i < GGML_TYPE_COUNT; ++i) { - ggml_type t = (ggml_type)i; - // Disable medium and large matrix multiplication if not enough shared memory is available - // Check mmq warptiles as the largest configuration - // Throw an error if not enough for any matrix multiplication is available - if (!ggml_vk_matmul_shmem_support(device, s_warptile_mmq, false, t)) { - std::cerr << "ggml_vulkan: Error: Shared memory size too small for matrix multiplication." << std::endl; - throw std::runtime_error("Shared memory size too small for matrix multiplication."); - } else if (!ggml_vk_matmul_shmem_support(device, m_warptile_mmq, false, t)) { - device->mul_mat_m[i] = false; - device->mul_mat_l[i] = false; - } else if (!ggml_vk_matmul_shmem_support(device, l_warptile_mmq, false, t)) { - device->mul_mat_l[i] = false; - } + ggml_vk_create_pipeline(device, device->pipeline_pad_f32, "pad_f32", pad_f32_len, pad_f32_data, "main", 2, sizeof(vk_op_pad_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_pad_reflect_1d_f32, "pad_reflect_1d_f32", pad_reflect_1d_f32_len, pad_reflect_1d_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - // Disable mul_mat_id if not enough shared memory is available - if (!ggml_vk_matmul_shmem_support(device, s_warptile_mmqid, true, t)) { - device->mul_mat_id_s[i] = false; - device->mul_mat_id_m[i] = false; - device->mul_mat_id_l[i] = false; - } else if (!ggml_vk_matmul_shmem_support(device, m_warptile_mmqid, true, t)) { - device->mul_mat_id_m[i] = false; - device->mul_mat_id_l[i] = false; - } else if (!ggml_vk_matmul_shmem_support(device, l_warptile_mmqid, true, t)) { - device->mul_mat_id_l[i] = false; - } + ggml_vk_create_pipeline(device, device->pipeline_roll_f32, "roll_f32", roll_f32_len, roll_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - // The q8_1 mmq path has its own (larger) shmem layout, check it separately. - // K-quants use the _int_k warptiles, others use _int. - const bool is_k_quant = (t == GGML_TYPE_Q2_K || t == GGML_TYPE_Q3_K || - t == GGML_TYPE_Q4_K || t == GGML_TYPE_Q5_K || - t == GGML_TYPE_Q6_K); - const auto & s_int = is_k_quant ? s_warptile_mmq_int_k : s_warptile_mmq_int; - const auto & m_int = is_k_quant ? m_warptile_mmq_int_k : m_warptile_mmq_int; - const auto & l_int = is_k_quant ? l_warptile_mmq_int_k : l_warptile_mmq_int; - const auto & s_intid = is_k_quant ? s_warptile_mmqid_int_k : s_warptile_mmqid_int; - const auto & m_intid = is_k_quant ? m_warptile_mmqid_int_k : m_warptile_mmqid_int; - const auto & l_intid = is_k_quant ? l_warptile_mmqid_int_k : l_warptile_mmqid_int; + ggml_vk_create_pipeline(device, device->pipeline_repeat_i32, "repeat_i32", repeat_i32_len, repeat_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_repeat_back_f32, "repeat_back_f32", repeat_back_f32_len, repeat_back_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - if (!ggml_vk_matmul_int_shmem_support(device, s_int, false, t)) { - device->mul_mat_s_int[i] = false; - device->mul_mat_m_int[i] = false; - device->mul_mat_l_int[i] = false; - } else if (!ggml_vk_matmul_int_shmem_support(device, m_int, false, t)) { - device->mul_mat_m_int[i] = false; - device->mul_mat_l_int[i] = false; - } else if (!ggml_vk_matmul_int_shmem_support(device, l_int, false, t)) { - device->mul_mat_l_int[i] = false; - } + ggml_vk_create_pipeline(device, device->pipeline_repeat_i16, "repeat_i16", repeat_i16_len, repeat_i16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - if (!ggml_vk_matmul_int_shmem_support(device, s_intid, true, t)) { - device->mul_mat_id_s_int[i] = false; - device->mul_mat_id_m_int[i] = false; - device->mul_mat_id_l_int[i] = false; - } else if (!ggml_vk_matmul_int_shmem_support(device, m_intid, true, t)) { - device->mul_mat_id_m_int[i] = false; - device->mul_mat_id_l_int[i] = false; - } else if (!ggml_vk_matmul_int_shmem_support(device, l_intid, true, t)) { - device->mul_mat_id_l_int[i] = false; - } - } - } +#define CREATE_UNARY(name) \ + ggml_vk_create_pipeline(device, device->pipeline_ ## name [0], #name "_f32", name ## _f32_len, name ## _f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); \ + ggml_vk_create_pipeline(device, device->pipeline_ ## name [1], #name "_f16", name ## _f16_len, name ## _f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - if (!device->pipeline_matmul_f32) { - device->pipeline_matmul_f32 = std::make_shared(); - } - if (!device->pipeline_matmul_f32_f16) { - device->pipeline_matmul_f32_f16 = std::make_shared(); - } - if (!device->pipeline_matmul_id_f32) { - device->pipeline_matmul_id_f32 = std::make_shared(); - } - if (!device->pipeline_matmul_bf16) { - device->pipeline_matmul_bf16 = std::make_shared(); - } - if (!device->pipeline_matmul_id_bf16) { - device->pipeline_matmul_id_bf16 = std::make_shared(); - } + CREATE_UNARY(elu) + CREATE_UNARY(gelu) + CREATE_UNARY(gelu_erf) + CREATE_UNARY(gelu_quick) + CREATE_UNARY(silu) + CREATE_UNARY(relu) + CREATE_UNARY(sqr) + CREATE_UNARY(sqrt) + CREATE_UNARY(sin) + CREATE_UNARY(cos) + CREATE_UNARY(clamp) + CREATE_UNARY(leaky_relu) + CREATE_UNARY(xielu) + CREATE_UNARY(neg) + CREATE_UNARY(tanh) + CREATE_UNARY(sigmoid) + CREATE_UNARY(hardsigmoid) + CREATE_UNARY(hardswish) + CREATE_UNARY(abs) + CREATE_UNARY(softplus) + CREATE_UNARY(step) + CREATE_UNARY(round) + CREATE_UNARY(ceil) + CREATE_UNARY(floor) + CREATE_UNARY(trunc) + CREATE_UNARY(sgn) + CREATE_UNARY(exp) + CREATE_UNARY(expm1) +#undef CREATE_UNARY - auto const &ggml_vk_create_pipeline = [&](vk_device& device, vk_pipeline& base_pipeline, const char *name, size_t spv_size, const void* spv_data, const char *entrypoint, - uint32_t parameter_count, uint32_t push_constant_size, std::array wg_denoms, const std::vector& specialization_constants, - uint32_t align, bool disable_robustness = false, bool require_full_subgroups = false, uint32_t required_subgroup_size = 0) { +// spec constants: {norepeat, op_on_b} +#define CREATE_UNARY_MUL(name, idx) \ + for (int dt = 0; dt < 2; ++dt) { \ + const size_t len_ = dt ? name ## _mul_f16_len : name ## _mul_f32_len; \ + const unsigned char * data_ = dt ? name ## _mul_f16_data : name ## _mul_f32_data; \ + const std::string dts_ = dt ? "f16" : "f32"; \ + for (int ob = 0; ob < 2; ++ob) \ + for (int nr = 0; nr < 2; ++nr) \ + ggml_vk_create_pipeline(device, device->pipeline_unary_mul[(idx)][dt][nr][ob], \ + (#name "_mul" + std::string(ob ? "_b" : "") + "_" + dts_ + (nr ? "_norepeat" : "")).c_str(), \ + len_, data_, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, \ + { (uint32_t) nr, (uint32_t) ob }, 1); \ + } + + CREATE_UNARY_MUL(gelu, 0) + CREATE_UNARY_MUL(sigmoid, 1) + CREATE_UNARY_MUL(silu, 2) + CREATE_UNARY_MUL(softplus, 3) +#undef CREATE_UNARY_MUL - if (!require_full_subgroups && required_subgroup_size == 0) { - required_subgroup_size = get_subgroup_size(name, device->architecture); - } + ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f16, "add1_f16_f16", add1_f16_f16_len, add1_f16_f16_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f32, "add1_f16_f32", add1_f16_f32_len, add1_f16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_add1_f32_f32, "add1_f32_f32", add1_f32_f32_len, add1_f32_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - vk_pipeline *ptr = &base_pipeline; + ggml_vk_create_pipeline(device, device->pipeline_arange_f32, "arange_f32", arange_f32_len, arange_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); - int num_pipelines = 1; -#if defined(VK_EXT_shader_64bit_indexing) - if (device->shader_64b_indexing) { - num_pipelines = 2; - } -#endif - for (int i = 0; i < num_pipelines; ++i, ptr = &(*ptr)->next) { - vk_pipeline &pipeline = *ptr; - if (!pipeline) { - pipeline = std::make_shared(); - } - if (!pipeline->initialized) { - pipeline->name = name; - pipeline->parameter_count = parameter_count; - pipeline->push_constant_size = push_constant_size; - pipeline->wg_denoms = wg_denoms; - pipeline->align = align; - pipeline->initialized = true; -#if defined(VK_EXT_shader_64bit_indexing) - pipeline->is_64b_indexing = (i == 1); -#endif - } - - // We only care about the pipeline this call asked for; the rest - // (including the 64-bit indexing variant) are handled by their - // own request_descriptor_sets / load_shaders calls. - if (pipeline.get() != requested.get()) { - continue; - } + ggml_vk_create_pipeline(device, device->pipeline_fill_f32, "fill_f32", fill_f32_len, fill_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_fill_f16, "fill_f16", fill_f16_len, fill_f16_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); - if (pipeline->compiled) { - continue; - } +#define CREATE_GLU(name) \ + ggml_vk_create_pipeline(device, device->pipeline_ ## name [0], #name "_f32", name ## _f32_len, name ## _f32_data, "main", 3, sizeof(vk_op_glu_push_constants), {512, 1, 1}, {}, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_ ## name [1], #name "_f16", name ## _f16_len, name ## _f16_data, "main", 3, sizeof(vk_op_glu_push_constants), {512, 1, 1}, {}, 1, true); - wait_pipeline = pipeline; + CREATE_GLU(geglu) + CREATE_GLU(reglu) + CREATE_GLU(swiglu) + CREATE_GLU(swiglu_oai) + CREATE_GLU(swiglu_clamp) + CREATE_GLU(geglu_erf) + CREATE_GLU(geglu_quick) +#undef CREATE_GLU - if (!pipeline->compile_pending) { - pipeline->compile_pending = true; - claimed_task.pipeline = pipeline; - claimed_task.spv_size = spv_size; - claimed_task.spv_data = spv_data; - claimed_task.entrypoint = entrypoint; - claimed_task.parameter_count = parameter_count; - claimed_task.wg_denoms = wg_denoms; - claimed_task.specialization_constants = specialization_constants; - claimed_task.disable_robustness = disable_robustness; - claimed_task.require_full_subgroups = require_full_subgroups; - claimed_task.required_subgroup_size = required_subgroup_size; - has_claimed_task = true; - } - } - }; + ggml_vk_create_pipeline(device, device->pipeline_silu_back_f32, "silu_back_f32", silu_back_f32_len, silu_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); - auto const &ggml_vk_create_pipeline2 = [&](vk_device& device, vk_pipeline& pipeline, const std::string &name, size_t spv_size, const void* spv_data, const char *entrypoint, - uint32_t parameter_count, uint32_t push_constant_size, std::array wg_denoms, const std::vector& specialization_constants, - uint32_t align, bool disable_robustness = false, bool require_full_subgroups = false, uint32_t required_subgroup_size = 0) { - return ggml_vk_create_pipeline(device, pipeline, name.c_str(), spv_size, spv_data, entrypoint, - parameter_count, push_constant_size, wg_denoms, specialization_constants, - align, disable_robustness, require_full_subgroups, required_subgroup_size); - }; + ggml_vk_create_pipeline(device, device->pipeline_diag_mask_inf_f32, "diag_mask_inf_f32", diag_mask_inf_f32_len, diag_mask_inf_f32_data, "main", 2, sizeof(vk_op_diag_mask_push_constants), {1, 512, 1}, {}, 1, true); - // FA scalar has two SPIR-V modules (MMQ vs non-MMQ); FA cm1 has one. K/V - // quant type is selected at runtime via the FaTypeK / FaTypeV spec constants. + ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32, "soft_max_f32", soft_max_f32_len, soft_max_f32_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_wg512, "soft_max_f32_wg512", soft_max_f32_len, soft_max_f32_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 512 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_f16, "soft_max_f32_f16", soft_max_f32_f16_len, soft_max_f32_f16_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_f16_wg512, "soft_max_f32_f16_wg512", soft_max_f32_f16_len, soft_max_f32_f16_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 512 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_back_f32, "soft_max_back_f32", soft_max_back_f32_len, soft_max_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1, true); - for (auto &fa : device->pipeline_flash_attn_f32_f16) { - if (fa.first.path != FA_SCALAR) continue; - const uint32_t Br = fa.first.Br; - const uint32_t Bc = fa.first.Bc; - const bool aligned = fa.first.aligned; - const bool f32acc = fa.first.f32acc; - const uint32_t fa_sgs = fa.first.subgroup_size; - const bool fa_ds = fa.first.subgroup_size == 0; + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large1_f32, "soft_max_large1_f32", soft_max_large1_f32_len, soft_max_large1_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large2_f32, "soft_max_large2_f32", soft_max_large2_f32_len, soft_max_large2_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large3_f32, "soft_max_large3_f32", soft_max_large3_f32_len, soft_max_large3_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large1_f32_f16, "soft_max_large1_f32_f16", soft_max_large1_f32_f16_len, soft_max_large1_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large2_f32_f16, "soft_max_large2_f32_f16", soft_max_large2_f32_f16_len, soft_max_large2_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_soft_max_large3_f32_f16, "soft_max_large3_f32_f16", soft_max_large3_f32_f16_len, soft_max_large3_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; - const bool use_mmq = ggml_vk_fa_scalar_uses_mmq(device, fa.first.k_type, fa.first.v_type); - const void * spv_data = nullptr; - size_t spv_size = 0; - const char *name = nullptr; - if (bf16_kv) { - spv_data = flash_attn_f32_f16_fp32_data; - spv_size = flash_attn_f32_f16_fp32_len; - name = aligned ? "flash_attn_f32_bf16_aligned" : "flash_attn_f32_bf16"; - } else if (use_mmq) { -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->fp16) { - if (f32acc) { spv_data = flash_attn_f32_f16_int8_data; spv_size = flash_attn_f32_f16_int8_len; } - else { spv_data = flash_attn_f32_f16_f16acc_int8_data; spv_size = flash_attn_f32_f16_f16acc_int8_len; } - } else { - spv_data = flash_attn_f32_f16_fp32_int8_data; - spv_size = flash_attn_f32_f16_fp32_int8_len; - } -#endif - name = aligned ? "flash_attn_f32_f16_aligned" : "flash_attn_f32_f16"; - } else { - if (device->fp16) { - if (device->dot2_f16) { - if (f32acc) { spv_data = flash_attn_f32_f16_dot2_data; spv_size = flash_attn_f32_f16_dot2_len; } - else { spv_data = flash_attn_f32_f16_dot2_f16acc_data; spv_size = flash_attn_f32_f16_dot2_f16acc_len; } - } else { - if (f32acc) { spv_data = flash_attn_f32_f16_data; spv_size = flash_attn_f32_f16_len; } - else { spv_data = flash_attn_f32_f16_f16acc_data; spv_size = flash_attn_f32_f16_f16acc_len; } - } - } else { - spv_data = flash_attn_f32_f16_fp32_data; - spv_size = flash_attn_f32_f16_fp32_len; - } - name = aligned ? "flash_attn_f32_f16_aligned" : "flash_attn_f32_f16"; - } - ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 7, - sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, - get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, - !fa_ds, !fa_ds ? fa_sgs : 0); - } + ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f32, "rope_norm_f32", rope_norm_f32_len, rope_norm_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f32, "rope_neox_f32", rope_neox_f32_len, rope_neox_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f32, "rope_multi_f32", rope_multi_f32_len, rope_multi_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_vision_f32, "rope_vision_f32", rope_vision_f32_len, rope_vision_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (device->coopmat1_fa_support) { - for (auto &fa : device->pipeline_flash_attn_f32_f16) { - if (fa.first.path != FA_COOPMAT1) continue; - const uint32_t Br = fa.first.Br; - const uint32_t Bc = fa.first.Bc; - const bool aligned = fa.first.aligned; - const bool f32acc = fa.first.f32acc; - const uint32_t fa_sgs = fa.first.subgroup_size; - const bool fa_ds = fa.first.subgroup_size == 0; + ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f16, "rope_norm_f16", rope_norm_f16_len, rope_norm_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f16, "rope_neox_f16", rope_neox_f16_len, rope_neox_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f16, "rope_multi_f16", rope_multi_f16_len, rope_multi_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_vision_f16, "rope_vision_f16", rope_vision_f16_len, rope_vision_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; + ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f32_f16, "rope_norm_f32_f16", rope_norm_f32_f16_len, rope_norm_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f32_f16, "rope_neox_f32_f16", rope_neox_f32_f16_len, rope_neox_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f32_f16, "rope_multi_f32_f16", rope_multi_f32_f16_len, rope_multi_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - const void * spv_data; - size_t spv_size; - const char *name; - if (bf16_kv) { -#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (!device->coopmat_bf16_support) continue; - spv_data = flash_attn_f32_f16_bf16_cm1_data; - spv_size = flash_attn_f32_f16_bf16_cm1_len; - name = aligned ? "flash_attn_f32_bf16_aligned_cm1" : "flash_attn_f32_bf16_cm1"; -#else - continue; -#endif - } else { - if (f32acc) { spv_data = flash_attn_f32_f16_cm1_data; spv_size = flash_attn_f32_f16_cm1_len; } - else { spv_data = flash_attn_f32_f16_f16acc_cm1_data; spv_size = flash_attn_f32_f16_f16acc_cm1_len; } - name = aligned ? "flash_attn_f32_f16_aligned_cm1" : "flash_attn_f32_f16_cm1"; - } - ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 7, - sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, - get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, - !fa_ds, !fa_ds ? fa_sgs : 0); + for (uint32_t i = 0; i < num_argsort_pipelines; ++i) { + uint32_t BLOCK_SIZE = 1u << std::min(i, device->max_workgroup_size_log2); + if (i <= device->max_workgroup_size_log2 && + 2 * sizeof(int) * BLOCK_SIZE <= device->properties.limits.maxComputeSharedMemorySize) { + const uint32_t NCOLS_PADDED_LOG2 = i; + ggml_vk_create_pipeline2(device, device->pipeline_argsort_f32[i], "argsort_f32_"+std::to_string(i), argsort_f32_len, argsort_f32_data, "main", 3, sizeof(vk_op_argsort_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, NCOLS_PADDED_LOG2}, 1, true); } + const uint32_t WG_UNROLL_FACTOR = BLOCK_SIZE > 1 ? 2 : 1; + BLOCK_SIZE /= WG_UNROLL_FACTOR; + ggml_vk_create_pipeline2(device, device->pipeline_argsort_large_f32[i], "argsort_large_f32_"+std::to_string(i), argsort_large_f32_len, argsort_large_f32_data, "main", 3, sizeof(vk_op_argsort_push_constants), {BLOCK_SIZE * WG_UNROLL_FACTOR, 1, 1}, {BLOCK_SIZE, WG_UNROLL_FACTOR}, 1, true); } -#endif - -#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (device->coopmat2) { - for (auto &fa : device->pipeline_flash_attn_f32_f16) { - if (fa.first.path != FA_COOPMAT2) continue; - const uint32_t Br = fa.first.Br; - const uint32_t Bc = fa.first.Bc; - const bool aligned = fa.first.aligned; - const bool f32acc = fa.first.f32acc; - const bool bf16_kv = fa.first.k_type == GGML_TYPE_BF16; - const void * spv_data; - size_t spv_size; - const char * name; - if (bf16_kv) { -#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (!device->coopmat2_bf16_support) continue; - spv_data = flash_attn_f32_f16_bf16_cm2_data; - spv_size = flash_attn_f32_f16_bf16_cm2_len; - name = aligned ? "flash_attn_f32_bf16_aligned_cm2" : "flash_attn_f32_bf16_cm2"; -#else - continue; -#endif - } else if (aligned) { - if (f32acc) { spv_data = flash_attn_f32_f16_cm2_data; spv_size = flash_attn_f32_f16_cm2_len; name = "flash_attn_f32_f16_aligned_f32acc_cm2"; } - else { spv_data = flash_attn_f32_f16_f16acc_cm2_data; spv_size = flash_attn_f32_f16_f16acc_cm2_len; name = "flash_attn_f32_f16_aligned_f16acc_cm2"; } - } else { - if (f32acc) { spv_data = flash_attn_f32_f16_cm2_data; spv_size = flash_attn_f32_f16_cm2_len; name = "flash_attn_f32_f16_f32acc_cm2"; } - else { spv_data = flash_attn_f32_f16_f16acc_cm2_data; spv_size = flash_attn_f32_f16_f16acc_cm2_len; name = "flash_attn_f32_f16_f16acc_cm2"; } + for (uint32_t i = 0; i < num_topk_pipelines; ++i) { + const uint32_t BLOCK_SIZE = 1u << i; + const uint32_t NCOLS_PADDED_LOG2 = i; + if (i <= device->max_workgroup_size_log2) { + uint32_t nary_shmem = 2 * sizeof(int) * BLOCK_SIZE + + sizeof(int) * device->subgroup_size + + 2 * sizeof(int) + + 2 * (BLOCK_SIZE / device->subgroup_size) * sizeof(int); + if (device->subgroup_arithmetic && device->subgroup_require_full_support && device->subgroup_shuffle && device->subgroup_ballot && + nary_shmem <= device->properties.limits.maxComputeSharedMemorySize) { + ggml_vk_create_pipeline2(device, device->pipeline_topk_f32[i], "topk_f32_"+std::to_string(i), topk_nary_search_f32_len, topk_nary_search_f32_data, "main", 2, sizeof(vk_op_topk_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, device->subgroup_size, device->subgroup_size_log2}, 1, true, true, device->subgroup_size); + } else if (2 * sizeof(int) * BLOCK_SIZE <= device->properties.limits.maxComputeSharedMemorySize) { + ggml_vk_create_pipeline2(device, device->pipeline_topk_f32[i], "topk_f32_"+std::to_string(i), topk_argsort_f32_len, topk_argsort_f32_data, "main", 2, sizeof(vk_op_topk_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, NCOLS_PADDED_LOG2}, 1, true); } - ggml_vk_create_pipeline(device, fa.second, name, spv_size, spv_data, "main", 7, - sizeof(vk_flash_attn_push_constants), {Br, 1, 1}, - get_fa_spec_constants(fa.first), aligned ? Bc : 1, true, false, 0); } } -#endif - auto const &ggml_vk_mul_mm_spec = [&device](std::vector spec, bool aligned) { - spec.push_back(aligned ? 1u : 0u); // constantID=11: ALIGNED - if (device->vendor_id == VK_VENDOR_ID_INTEL && device->coopmat_support && - device->driver_id == vk::DriverId::eIntelProprietaryWindows) { - spec.push_back(0u); // constantID=12: SHMEM_STRIDE_PAD = 0 - spec.push_back(1u); // constantID=13: APPLY_SLM_A_RESHAPE = true - } - return spec; - }; + // large-k fallback: one workgroup per row, radix-select instead of a full sort. The QSA + // variant (spec constant 1) additionally gathers the qwen4 indexer input on the fly. + { + const uint32_t BLOCK_SIZE = 1u << std::min(10u, device->max_workgroup_size_log2); + ggml_vk_create_pipeline2(device, device->pipeline_topk_radix_f32, "topk_radix_f32", topk_radix_select_f32_len, topk_radix_select_f32_data, "main", 5, sizeof(vk_op_topk_radix_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, 0}, 1, true); + ggml_vk_create_pipeline2(device, device->pipeline_topk_radix_qsa, "topk_radix_qsa", topk_radix_select_f32_len, topk_radix_select_f32_data, "main", 5, sizeof(vk_op_topk_radix_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, 1}, 1, true); + } - const int mul_mat_id_param_count = 5; + ggml_vk_create_pipeline(device, device->pipeline_argmax_f32, "argmax_f32", argmax_f32_len, argmax_f32_data, "main", 2, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); -#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (device->coopmat2) { - auto const &ggml_vk_mul_mm_cm2_spec = [](std::vector spec, bool aligned, bool mul_mat_id) { - if (mul_mat_id && spec.size() > 5) { - spec.insert(spec.begin() + 5, aligned ? 1u : 0u); - } else { - spec.push_back(aligned ? 1u : 0u); - } - if (mul_mat_id && spec.size() == 6) { - spec.push_back(32); + ggml_vk_create_pipeline(device, device->pipeline_sum_rows_f32, "sum_rows_f32", sum_rows_f32_len, sum_rows_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + ggml_vk_create_pipeline(device, device->pipeline_cross_entropy_loss_f32, "cross_entropy_loss_f32", cross_entropy_loss_f32_len, cross_entropy_loss_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + ggml_vk_create_pipeline(device, device->pipeline_cross_entropy_loss_f32_wg512, "cross_entropy_loss_f32_wg512", cross_entropy_loss_f32_len, cross_entropy_loss_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, { 512 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_cross_entropy_loss_back_f32, "cross_entropy_loss_back_f32", cross_entropy_loss_back_f32_len, cross_entropy_loss_back_f32_data, "main", 4, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + ggml_vk_create_pipeline(device, device->pipeline_cross_entropy_loss_back_f32_wg512, "cross_entropy_loss_back_f32_wg512", cross_entropy_loss_back_f32_len, cross_entropy_loss_back_f32_data, "main", 4, sizeof(vk_op_push_constants), {1, 1, 1}, { 512 }, 1); + // Intel Windows driver in range [32.0.101.8509, 32.0.101.8860) will crash when using fwht kernels so we gate that here + const bool can_use_fwht = device->driver_id != vk::DriverId::eIntelProprietaryWindows || + !ggml_vk_intel_windows_driver_in_range(device->properties.driverVersion, 101, 8509, 101, 8860); + if (can_use_fwht && device->subgroup_basic && device->subgroup_shuffle) { + int idx = 0; + for (uint32_t n : {64, 128, 256, 512}) { + if (device->subgroup_size <= n) { + ggml_vk_create_pipeline(device, device->pipeline_fwht_f32[idx], "fwht_f32", fwht_f32_len, fwht_f32_data, "main", 2, sizeof(vk_op_fwht_push_constants), {1, 1, 1}, { device->subgroup_size, n }, 1, true, true, device->subgroup_size); } - return spec; - }; - - // Create 6 variants, {s,m,l}x{unaligned,aligned} -#define CREATE_MM(PIPELINE_NAME, NAMELC, F16ACC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC #F16ACC "_l", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(l_ ## WARPTILE, false, PARAMCOUNT == mul_mat_id_param_count), 1, true); \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC #F16ACC "_m", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(m_ ## WARPTILE, false, PARAMCOUNT == mul_mat_id_param_count), 1, true); \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC #F16ACC "_s", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(s_ ## WARPTILE, false, PARAMCOUNT == mul_mat_id_param_count), 1, true); \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_l, #NAMELC #F16ACC "_aligned_l", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(l_ ## WARPTILE, true, PARAMCOUNT == mul_mat_id_param_count), l_align, true); \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_m, #NAMELC #F16ACC "_aligned_m", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(m_ ## WARPTILE, true, PARAMCOUNT == mul_mat_id_param_count), m_align, true); \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_s, #NAMELC #F16ACC "_aligned_s", NAMELC ## F16ACC ## _cm2_len, NAMELC ## F16ACC ## _cm2_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_cm2_spec(s_ ## WARPTILE, true, PARAMCOUNT == mul_mat_id_param_count), s_align, true); \ - - // Create 2 variants, {f16,f32} accumulator -#define CREATE_MM2(PIPELINE_NAME, NAMELC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT) \ - CREATE_MM(PIPELINE_NAME . f16acc, NAMELC, _f16acc, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT) \ - CREATE_MM(PIPELINE_NAME . f32acc, NAMELC, , WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT) \ - - CREATE_MM2(pipeline_matmul_f16, matmul_f16, wg_denoms, warptile, vk_mat_mat_push_constants, 3) -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (device->coopmat_bf16_support) { - CREATE_MM(pipeline_matmul_bf16, matmul_bf16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3) + ++idx; } -#endif - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q1_0], matmul_q1_0_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q2_0], matmul_q2_0_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q4_0], matmul_q4_0_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q4_1], matmul_q4_1_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q5_0], matmul_q5_0_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q5_1], matmul_q5_1_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q8_0], matmul_q8_0_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q2_K], matmul_q2_k_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_TQ2_0], matmul_tq2_0_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q3_K], matmul_q3_k_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q4_K], matmul_q4_k_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q5_K], matmul_q5_k_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_Q6_K], matmul_q6_k_f16, mmq_wg_denoms_k, warptile_mmq_k, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ1_S], matmul_iq1_s_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ1_M], matmul_iq1_m_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ2_XXS], matmul_iq2_xxs_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ2_XS], matmul_iq2_xs_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ2_S], matmul_iq2_s_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ3_XXS], matmul_iq3_xxs_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ3_S], matmul_iq3_s_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ4_XS], matmul_iq4_xs_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_IQ4_NL], matmul_iq4_nl_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - if (device->ocp_fp4) { - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_MXFP4], matmul_mxfp4_f16_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_NVFP4], matmul_nvfp4_f16_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - } else -#endif - { - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_MXFP4], matmul_mxfp4_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) - CREATE_MM2(pipeline_dequant_mul_mat_mat_f16[GGML_TYPE_NVFP4], matmul_nvfp4_f16, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3) + } else if (can_use_fwht) { + int idx = 0; + for (uint32_t n : {64, 128, 256, 512}) { + const uint32_t block_size = std::min(device->subgroup_size, n); + ggml_vk_create_pipeline(device, device->pipeline_fwht_f32[idx], "fwht_shmem_f32", fwht_shmem_f32_len, fwht_shmem_f32_data, "main", 2, sizeof(vk_op_fwht_push_constants), {1, 1, 1}, { block_size, n }, 1); + ++idx; } + } - GGML_ASSERT(device->subgroup_ballot); + const uint32_t cumsum_elem_per_thread = (device->vendor_id == VK_VENDOR_ID_AMD || device->vendor_id == VK_VENDOR_ID_INTEL) ? 2 : 4; + ggml_vk_create_pipeline(device, device->pipeline_cumsum_f32, "cumsum_f32", cumsum_f32_len, cumsum_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { 256, device->subgroup_size, cumsum_elem_per_thread }, 1, true, true, device->subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_cumsum_small_f32, "cumsum_f32", cumsum_f32_len, cumsum_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { 128, device->subgroup_size, 1 }, 1, true, true, device->subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_cumsum_multipass1_f32, "cumsum_multipass1_f32", cumsum_multipass1_f32_len, cumsum_multipass1_f32_data, "main", 3, sizeof(vk_op_sum_rows_push_constants), {256, 1, 1}, { 256, device->subgroup_size }, 1, true, true, device->subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_cumsum_multipass2_f32, "cumsum_multipass2_f32", cumsum_multipass2_f32_len, cumsum_multipass2_f32_data, "main", 3, sizeof(vk_op_sum_rows_push_constants), {256, 1, 1}, { 256, device->subgroup_size }, 1, true, true, device->subgroup_size); - CREATE_MM2(pipeline_matmul_id_f16, matmul_id_subgroup_f16, wg_denoms, warptile, vk_mat_mat_id_push_constants, 5) -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (device->coopmat_bf16_support) { - CREATE_MM(pipeline_matmul_id_bf16, matmul_id_subgroup_bf16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, 5) - } -#endif - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0], matmul_id_subgroup_q1_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0], matmul_id_subgroup_q2_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0], matmul_id_subgroup_q4_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1], matmul_id_subgroup_q4_1_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0], matmul_id_subgroup_q5_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1], matmul_id_subgroup_q5_1_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0], matmul_id_subgroup_q8_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K], matmul_id_subgroup_q2_k_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0], matmul_id_subgroup_tq2_0_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K], matmul_id_subgroup_q3_k_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K], matmul_id_subgroup_q4_k_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K], matmul_id_subgroup_q5_k_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K], matmul_id_subgroup_q6_k_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S], matmul_id_subgroup_iq1_s_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M], matmul_id_subgroup_iq1_m_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS], matmul_id_subgroup_iq2_xxs_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS], matmul_id_subgroup_iq2_xs_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S], matmul_id_subgroup_iq2_s_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS], matmul_id_subgroup_iq3_xxs_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S], matmul_id_subgroup_iq3_s_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS], matmul_id_subgroup_iq4_xs_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL], matmul_id_subgroup_iq4_nl_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - if (device->ocp_fp4) { - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_f16_ocp, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_subgroup_nvfp4_f16_ocp, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - } else -#endif - { - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - CREATE_MM2(pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_subgroup_nvfp4_f16, mmqid_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, 5) - } -#undef CREATE_MM -#undef CREATE_MM2 - } else -#endif // defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (device->coopmat_support) { - // Create 6 variants, {s,m,l}x{unaligned,aligned} -#define CREATE_MM(TYPE, PIPELINE_NAME, NAMELC, F16ACC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID) \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC #F16ACC "_l", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, false), 1, false, true); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC #F16ACC "_m", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, false), 1, false, true); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC #F16ACC "_s", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, false), 1, false, true); \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_l, #NAMELC #F16ACC "_aligned_l", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, true), l_align, false, true); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_m, #NAMELC #F16ACC "_aligned_m", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, true), m_align, false, true); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_s, #NAMELC #F16ACC "_aligned_s", NAMELC ## F16ACC ## _cm1_len, NAMELC ## F16ACC ## _cm1_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, true), s_align, false, true); \ - - // Create 2 variants, {f16,f32} accumulator -#define CREATE_MM2(TYPE, PIPELINE_NAME, NAMELC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID) \ - if (device->coopmat_acc_f16_support) { \ - CREATE_MM(TYPE, PIPELINE_NAME . f16acc, NAMELC, _f16acc, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID) \ - } \ - if (device->coopmat_acc_f32_support) { \ - CREATE_MM(TYPE, PIPELINE_NAME . f32acc, NAMELC, , WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID) \ - } \ + ggml_vk_create_pipeline(device, device->pipeline_count_equal_i32, "count_equal_i32", count_equal_i32_len, count_equal_i32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, { device->subgroup_size }, 1); - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32, matmul_f32_f32, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, ); - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32_f16, matmul_f32_f16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_f16, matmul_f16, wg_denoms, warptile, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_f16_f32, matmul_f16_f32, wg_denoms, warptile, vk_mat_mat_push_constants, 3, ); -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (device->coopmat_bf16_support) { - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_bf16, matmul_bf16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, ) - } -#endif + if (device->subgroup_arithmetic && device->subgroup_require_full_support) { + ggml_vk_create_pipeline(device, device->pipeline_count_experts, "count_experts", count_experts_subgroup_len, count_experts_subgroup_data, "main", 2, sizeof(vk_op_count_experts_push_constants), {1, 1, 1}, {}, 1, true, true); + } else { + ggml_vk_create_pipeline(device, device->pipeline_count_experts, "count_experts", count_experts_len, count_experts_data, "main", 2, sizeof(vk_op_count_experts_push_constants), {1, 1, 1}, {}, 1, true); + } - CREATE_MM2(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q1_0], matmul_q1_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_0], matmul_q2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_0], matmul_q4_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_1], matmul_q4_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_0], matmul_q5_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_1], matmul_q5_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q8_0], matmul_q8_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - - CREATE_MM2(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_K], matmul_q2_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_TQ2_0], matmul_tq2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q3_K], matmul_q3_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_K], matmul_q4_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_K], matmul_q5_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q6_K], matmul_q6_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_S], matmul_iq1_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_M], matmul_iq1_m_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XXS], matmul_iq2_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XS], matmul_iq2_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_S], matmul_iq2_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_XXS], matmul_iq3_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_S], matmul_iq3_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_XS], matmul_iq4_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_NL], matmul_iq4_nl_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); + // comb holds a token's 4x4 matrix in one 16-lane slice of a subgroup, so it + // needs at least 16 lanes, pinned to a known size. + if (device->subgroup_basic && device->subgroup_shuffle && device->subgroup_require_full_support && device->subgroup_size >= 16) { + const uint32_t tokens_per_workgroup = 4 * (device->subgroup_size / 16); + ggml_vk_create_pipeline(device, device->pipeline_dsv4_hc_comb_f32, "dsv4_hc_comb_f32", dsv4_hc_comb_f32_len, dsv4_hc_comb_f32_data, "main", 4, sizeof(vk_op_dsv4_hc_comb_push_constants), {tokens_per_workgroup, 1, 1}, { device->subgroup_size }, 1, true, true, device->subgroup_size); + } -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - if (device->ocp_fp4) { - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_MXFP4], matmul_mxfp4_f32_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_NVFP4], matmul_nvfp4_f32_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - } else -#endif - { - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_MXFP4], matmul_mxfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_NVFP4], matmul_nvfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, ); - } + ggml_vk_create_pipeline(device, device->pipeline_dsv4_hc_pre_f32, "dsv4_hc_pre_f32", dsv4_hc_pre_f32_len, dsv4_hc_pre_f32_data, "main", 3, sizeof(vk_op_dsv4_hc_pre_push_constants), {256, 1, 1}, { 256, 0 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_dsv4_hc_pre_gated_f32, "dsv4_hc_pre_gated_f32", dsv4_hc_pre_f32_len, dsv4_hc_pre_f32_data, "main", 3, sizeof(vk_op_dsv4_hc_pre_push_constants), {256, 1, 1}, { 256, 1 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_dsv4_hc_post_f32, "dsv4_hc_post_f32", dsv4_hc_post_f32_len, dsv4_hc_post_f32_data, "main", 5, sizeof(vk_op_dsv4_hc_post_push_constants), {256, 1, 1}, { 256, 1 }, 1); + ggml_vk_create_pipeline(device, device->pipeline_dsv4_hc_post_nocomb_f32,"dsv4_hc_post_nocomb_f32",dsv4_hc_post_f32_len, dsv4_hc_post_f32_data, "main", 5, sizeof(vk_op_dsv4_hc_post_push_constants), {256, 1, 1}, { 256, 0 }, 1); - GGML_ASSERT(device->subgroup_ballot); + for (auto &s : device->pipeline_solve_tri_f32) { + const vk_solve_tri_pipeline_state &state = s.first; - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_id_f32, matmul_id_subgroup_f32_f32, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16, matmul_id_subgroup_f16, wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16_f32, matmul_id_subgroup_f16_f32, wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (device->coopmat_bf16_support) { - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_subgroup_bf16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - } -#endif + // Max number of rows to load at a time, limited by shared memory + const uint32_t batch_N = device->properties.limits.maxComputeSharedMemorySize / ((state.N + state.K) * sizeof(float)); + // Need at least K invocations, and prefer a minimum of 128 to spread out loading shared memory + const uint32_t block_size = std::max(128u, 1u << (uint32_t)ceilf(log2f(float(state.K)))); - CREATE_MM2(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0], matmul_id_subgroup_q1_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0], matmul_id_subgroup_q2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0], matmul_id_subgroup_q4_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1], matmul_id_subgroup_q4_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0], matmul_id_subgroup_q5_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1], matmul_id_subgroup_q5_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0], matmul_id_subgroup_q8_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K], matmul_id_subgroup_q2_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0], matmul_id_subgroup_tq2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K], matmul_id_subgroup_q3_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K], matmul_id_subgroup_q4_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K], matmul_id_subgroup_q5_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K], matmul_id_subgroup_q6_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S], matmul_id_subgroup_iq1_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M], matmul_id_subgroup_iq1_m_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS], matmul_id_subgroup_iq2_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS], matmul_id_subgroup_iq2_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S], matmul_id_subgroup_iq2_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS], matmul_id_subgroup_iq3_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S], matmul_id_subgroup_iq3_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS], matmul_id_subgroup_iq4_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL], matmul_id_subgroup_iq4_nl_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - if (device->ocp_fp4) { - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_f32_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_subgroup_nvfp4_f32_ocp, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - } else -#endif - { - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_subgroup_nvfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id); - } -#undef CREATE_MM2 -#undef CREATE_MM - } else -#endif // defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (device->fp16) { - // Create 6 variants, {s,m,l}x{unaligned,aligned} - // Selects dot2 SPIR-V variant at runtime when device->dot2_f16 is true -#define CREATE_MM(TYPE, PIPELINE_NAME, NAMELC, F16ACC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC #F16ACC "_l", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC #F16ACC "_m", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC #F16ACC "_s", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_l, #NAMELC #F16ACC "_aligned_l", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, true), l_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_m, #NAMELC #F16ACC "_aligned_m", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, true), m_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_s, #NAMELC #F16ACC "_aligned_s", (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _len : NAMELC ## F16ACC ## _len), (device->dot2_f16 ? NAMELC ## _dot2 ## F16ACC ## _data : NAMELC ## F16ACC ## _data), "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, true), s_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - - // bf16 scalar path promotes to f32, no dot2 variant -#define CREATE_MM_NODOT2(TYPE, PIPELINE_NAME, NAMELC, F16ACC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC #F16ACC "_l", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC #F16ACC "_m", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC #F16ACC "_s", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_l, #NAMELC #F16ACC "_aligned_l", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, true), l_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_m, #NAMELC #F16ACC "_aligned_m", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, true), m_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_s, #NAMELC #F16ACC "_aligned_s", NAMELC ## F16ACC ## _len, NAMELC ## F16ACC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, true), s_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - -#define CREATE_MMQ(TYPE, PIPELINE_NAME, NAMELC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - if (device->mul_mat ## ID ## _l_int[TYPE]) { \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME .f32acc->l, #NAMELC "_l", NAMELC ## _len, NAMELC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, l_ ## WARPTILE, 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - } \ - if (device->mul_mat ## ID ## _m_int[TYPE]) { \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME .f32acc->m, #NAMELC "_m", NAMELC ## _len, NAMELC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, m_ ## WARPTILE, 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - } \ - if (device->mul_mat ## ID ## _s_int[TYPE]) { \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME .f32acc->s, #NAMELC "_s", NAMELC ## _len, NAMELC ## _data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, s_ ## WARPTILE, 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - } \ + ggml_vk_create_pipeline( + device, s.second, "solve_tri_f32", + solve_tri_f32_len, solve_tri_f32_data, "main", 3, + sizeof(vk_op_binary_push_constants), {1, 1, 1}, { 0, state.N, state.K, batch_N, block_size }, 1, true); + } + +#define IM2COL(bda) \ + ggml_vk_create_pipeline(device, device->pipeline_im2col_f32, "im2col_f32", im2col_f32 ## bda ## _len, im2col_f32 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_push_constants), {512, 1, 1}, { device->subgroup_size }, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_im2col_3d_f32, "im2col_3d_f32", im2col_3d_f32 ## bda ## _len, im2col_3d_f32 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_3d_push_constants), {512, 1, 1}, { 512 }, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_im2col_f32_f16, "im2col_f32_f16", im2col_f32_f16 ## bda ## _len, im2col_f32_f16 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_push_constants), {512, 1, 1}, { device->subgroup_size }, 1, true); \ + ggml_vk_create_pipeline(device, device->pipeline_im2col_3d_f32_f16, "im2col_3d_f32_f16", im2col_3d_f32_f16 ## bda ## _len, im2col_3d_f32_f16 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_3d_push_constants), {512, 1, 1}, { 512 }, 1, true); + if (device->shader_int64 && device->buffer_device_address) { + IM2COL(_bda) + } else { + IM2COL() + } - // Create 2 variants, {f16,f32} accumulator -#define CREATE_MM2(TYPE, PIPELINE_NAME, NAMELC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - CREATE_MM(TYPE, PIPELINE_NAME . f16acc, NAMELC, _f16acc, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - CREATE_MM(TYPE, PIPELINE_NAME . f32acc, NAMELC, , WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32, matmul_f32_f32, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32_f16, matmul_f32_f16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_f16, matmul_f16, wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_f16_f32, matmul_f16_f32, wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - - CREATE_MM_NODOT2(GGML_TYPE_BF16, pipeline_matmul_bf16, matmul_bf16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - - CREATE_MM2(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q1_0], matmul_q1_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_0], matmul_q2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_0], matmul_q4_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_1], matmul_q4_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_0], matmul_q5_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_1], matmul_q5_1_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q8_0], matmul_q8_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_K], matmul_q2_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_TQ2_0], matmul_tq2_0_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q3_K], matmul_q3_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_K], matmul_q4_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_K], matmul_q5_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q6_K], matmul_q6_k_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_S], matmul_iq1_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_M], matmul_iq1_m_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XXS], matmul_iq2_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XS], matmul_iq2_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_S], matmul_iq2_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_XXS], matmul_iq3_xxs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_S], matmul_iq3_s_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_XS], matmul_iq4_xs_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_NL], matmul_iq4_nl_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_MXFP4], matmul_mxfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_NVFP4], matmul_nvfp4_f32, mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); + ggml_vk_create_pipeline(device, device->pipeline_timestep_embedding_f32, "timestep_embedding_f32", timestep_embedding_f32_len, timestep_embedding_f32_data, "main", 2, sizeof(vk_op_timestep_embedding_push_constants), {256, 1, 1}, {}, 1); -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - CREATE_MMQ(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q2_0], matmul_q2_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_0], matmul_q4_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_1], matmul_q4_1_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_0], matmul_q5_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_1], matmul_q5_1_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q8_0], matmul_q8_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); + ggml_vk_create_pipeline(device, device->pipeline_conv_transpose_1d_f32, "conv_transpose_1d_f32", conv_transpose_1d_f32_len, conv_transpose_1d_f32_data, "main", 3, sizeof(vk_op_conv_transpose_1d_push_constants), {1, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_f32, "col2im_1d_f32", col2im_1d_f32_len, col2im_1d_f32_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_f16, "col2im_1d_f16", col2im_1d_f16_len, col2im_1d_f16_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_bf16, "col2im_1d_bf16", col2im_1d_bf16_len, col2im_1d_bf16_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); - CREATE_MMQ(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_MXFP4], matmul_mxfp4_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, , 0); + ggml_vk_create_pipeline(device, device->pipeline_out_prod_f32, "out_prod_f32", out_prod_f32_len, out_prod_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {256, 1, 1}, {}, 1); - CREATE_MMQ(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q2_K], matmul_q2_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q3_K], matmul_q3_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_K], matmul_q4_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_K], matmul_q5_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, , 0); - CREATE_MMQ(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q6_K], matmul_q6_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, , 0); - } -#endif + ggml_vk_create_pipeline(device, device->pipeline_snake_f32, "snake_f32", snake_f32_len, snake_f32_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_snake_f16, "snake_f16", snake_f16_len, snake_f16_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_snake_bf16, "snake_bf16", snake_bf16_len, snake_bf16_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); - if (device->subgroup_ballot && device->subgroup_require_full_support && subgroup_min_size_16) { - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_id_f32, matmul_id_subgroup_f32_f32, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16, matmul_id_subgroup_f16, wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16_f32, matmul_id_subgroup_f16_f32, wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM_NODOT2(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_subgroup_bf16, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM2(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0], matmul_id_subgroup_q1_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0], matmul_id_subgroup_q2_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0], matmul_id_subgroup_q4_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1], matmul_id_subgroup_q4_1_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0], matmul_id_subgroup_q5_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1], matmul_id_subgroup_q5_1_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0], matmul_id_subgroup_q8_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K], matmul_id_subgroup_q2_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0], matmul_id_subgroup_tq2_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K], matmul_id_subgroup_q3_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K], matmul_id_subgroup_q4_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K], matmul_id_subgroup_q5_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K], matmul_id_subgroup_q6_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S], matmul_id_subgroup_iq1_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M], matmul_id_subgroup_iq1_m_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS], matmul_id_subgroup_iq2_xxs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS], matmul_id_subgroup_iq2_xs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S], matmul_id_subgroup_iq2_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS], matmul_id_subgroup_iq3_xxs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S], matmul_id_subgroup_iq3_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS], matmul_id_subgroup_iq4_xs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL], matmul_id_subgroup_iq4_nl_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_subgroup_nvfp4_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); + ggml_vk_create_pipeline(device, device->pipeline_pool1d_f32, "pool1d_f32", pool1d_f32_len, pool1d_f32_data, "main", 2, sizeof(vk_op_pool1d_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_pool2d_f32, "pool2d_f32", pool2d_f32_len, pool2d_f32_data, "main", 2, sizeof(vk_op_pool2d_push_constants), {512, 1, 1}, {}, 1); -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - CREATE_MMQ(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q2_0], matmul_id_subgroup_q2_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MMQ(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_0], matmul_id_subgroup_q4_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MMQ(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_1], matmul_id_subgroup_q4_1_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MMQ(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_0], matmul_id_subgroup_q5_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MMQ(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_1], matmul_id_subgroup_q5_1_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MMQ(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q8_0], matmul_id_subgroup_q8_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - - CREATE_MMQ(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_MXFP4], matmul_id_subgroup_mxfp4_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - - CREATE_MMQ(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q2_K], matmul_id_subgroup_q2_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MMQ(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q3_K], matmul_id_subgroup_q3_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MMQ(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_K], matmul_id_subgroup_q4_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MMQ(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_K], matmul_id_subgroup_q5_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MMQ(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q6_K], matmul_id_subgroup_q6_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - } -#endif - } else { - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_id_f32, matmul_id_f32_f32, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16, matmul_id_f16, wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_F16, pipeline_matmul_id_f16_f32, matmul_id_f16_f32, wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM_NODOT2(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_bf16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0], matmul_id_q1_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0], matmul_id_q2_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0], matmul_id_q4_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1], matmul_id_q4_1_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0], matmul_id_q5_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1], matmul_id_q5_1_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0], matmul_id_q8_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K], matmul_id_q2_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0], matmul_id_tq2_0_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K], matmul_id_q3_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K], matmul_id_q4_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K], matmul_id_q5_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K], matmul_id_q6_k_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S], matmul_id_iq1_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M], matmul_id_iq1_m_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS], matmul_id_iq2_xxs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS], matmul_id_iq2_xs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S], matmul_id_iq2_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS], matmul_id_iq3_xxs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S], matmul_id_iq3_s_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS], matmul_id_iq4_xs_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL], matmul_id_iq4_nl_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4], matmul_id_mxfp4_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM2(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4], matmul_id_nvfp4_f32, mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); + ggml_vk_create_pipeline(device, device->pipeline_rwkv_wkv6_f32, "rwkv_wkv6_f32", rwkv_wkv6_f32_len, rwkv_wkv6_f32_data, "main", 7, sizeof(vk_op_rwkv_wkv6_push_constants), {1, 1, 1}, {device->subgroup_size}, 1); -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - CREATE_MMQ(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q2_0], matmul_id_q2_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_0], matmul_id_q4_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_1], matmul_id_q4_1_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_0], matmul_id_q5_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_1], matmul_id_q5_1_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q8_0], matmul_id_q8_0_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - - CREATE_MMQ(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_MXFP4], matmul_id_mxfp4_q8_1, mmq_wg_denoms, warptile_mmqid_int, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - - CREATE_MMQ(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q2_K], matmul_id_q2_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q3_K], matmul_id_q3_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q4_K], matmul_id_q4_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q5_K], matmul_id_q5_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MMQ(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id_q8_1[GGML_TYPE_Q6_K], matmul_id_q6_k_q8_1, mmq_wg_denoms, warptile_mmqid_int_k, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - } -#endif - } -#undef CREATE_MM2 -#undef CREATE_MMQ -#undef CREATE_MM -#undef CREATE_MM_NODOT2 - } else { - // Create 6 variants, {s,m,l}x{unaligned,aligned} -#define CREATE_MM(TYPE, PIPELINE_NAME, NAMELC, F16ACC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID, REQSUBGROUPSIZE) \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC #F16ACC "_l", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC #F16ACC "_m", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC #F16ACC "_s", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, false), 1, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _l[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_l, #NAMELC #F16ACC "_aligned_l", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, ggml_vk_mul_mm_spec(l_ ## WARPTILE, true), l_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _m[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_m, #NAMELC #F16ACC "_aligned_m", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, ggml_vk_mul_mm_spec(m_ ## WARPTILE, true), m_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - if (device->mul_mat ## ID ## _s[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->a_s, #NAMELC #F16ACC "_aligned_s", NAMELC ## F16ACC ## _fp32_len, NAMELC ## F16ACC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, ggml_vk_mul_mm_spec(s_ ## WARPTILE, true), s_align, false, REQSUBGROUPSIZE > 0, REQSUBGROUPSIZE); \ - -#define CREATE_MMQ(TYPE, PIPELINE_NAME, NAMELC, WG_DENOMS, WARPTILE, PUSHCONST, PARAMCOUNT, ID) \ - if (device->mul_mat ## ID ## _l_int[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->l, #NAMELC "_l", NAMELC ## _fp32_len, NAMELC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), l_ ## WG_DENOMS, l_ ## WARPTILE, 1); \ - if (device->mul_mat ## ID ## _m_int[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->m, #NAMELC "_m", NAMELC ## _fp32_len, NAMELC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), m_ ## WG_DENOMS, m_ ## WARPTILE, 1); \ - if (device->mul_mat ## ID ## _s_int[TYPE]) \ - ggml_vk_create_pipeline(device, device-> PIPELINE_NAME ->s, #NAMELC "_s", NAMELC ## _fp32_len, NAMELC ## _fp32_data, "main", PARAMCOUNT, sizeof(PUSHCONST), s_ ## WG_DENOMS, s_ ## WARPTILE, 1); \ - - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32, matmul_f32_f32, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_f32_f16, matmul_f32_f16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_f16.f32acc, matmul_f16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_f16_f32.f32acc, matmul_f16_f32, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_bf16, matmul_bf16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - - CREATE_MM(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q1_0].f32acc, matmul_q1_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_0].f32acc, matmul_q2_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_0].f32acc, matmul_q4_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_1].f32acc, matmul_q4_1_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_0].f32acc, matmul_q5_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_1].f32acc, matmul_q5_1_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q8_0].f32acc, matmul_q8_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - - CREATE_MM(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q2_K].f32acc, matmul_q2_k_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat[GGML_TYPE_TQ2_0].f32acc, matmul_tq2_0_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q3_K].f32acc, matmul_q3_k_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q4_K].f32acc, matmul_q4_k_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q5_K].f32acc, matmul_q5_k_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat[GGML_TYPE_Q6_K].f32acc, matmul_q6_k_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_S].f32acc, matmul_iq1_s_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ1_M].f32acc, matmul_iq1_m_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XXS].f32acc, matmul_iq2_xxs_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_XS].f32acc, matmul_iq2_xs_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ2_S].f32acc, matmul_iq2_s_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_XXS].f32acc, matmul_iq3_xxs_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ3_S].f32acc, matmul_iq3_s_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_XS].f32acc, matmul_iq4_xs_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat[GGML_TYPE_IQ4_NL].f32acc, matmul_iq4_nl_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_MXFP4].f32acc, matmul_mxfp4_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat[GGML_TYPE_NVFP4].f32acc, matmul_nvfp4_f32, , mmq_wg_denoms, warptile_mmq, vk_mat_mat_push_constants, 3, , 0); + ggml_vk_create_pipeline(device, device->pipeline_rwkv_wkv7_f32, "rwkv_wkv7_f32", rwkv_wkv7_f32_len, rwkv_wkv7_f32_data, "main", 8, sizeof(vk_op_rwkv_wkv7_push_constants), {1, 1, 1}, {device->subgroup_size}, 1); -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - CREATE_MMQ(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q2_0].f32acc, matmul_q2_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_0].f32acc, matmul_q4_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_1].f32acc, matmul_q4_1_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_0].f32acc, matmul_q5_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_1].f32acc, matmul_q5_1_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q8_0].f32acc, matmul_q8_0_q8_1, mmq_wg_denoms, warptile_mmq_int, vk_mat_mat_push_constants, 3, ); - - CREATE_MMQ(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q2_K].f32acc, matmul_q2_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q3_K].f32acc, matmul_q3_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q4_K].f32acc, matmul_q4_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q5_K].f32acc, matmul_q5_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, ); - CREATE_MMQ(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_q8_1[GGML_TYPE_Q6_K].f32acc, matmul_q6_k_q8_1, mmq_wg_denoms, warptile_mmq_int_k, vk_mat_mat_push_constants, 3, ); + ggml_vk_create_pipeline(device, device->pipeline_gated_linear_attn_f32, "gated_linear_attn_f32", gated_linear_attn_f32_len, gated_linear_attn_f32_data, "main", 6, sizeof(vk_op_gated_linear_attn_push_constants), {1, 1, 1}, {}, 1); + + { + const bool li_subgroup = device->subgroup_arithmetic && device->subgroup_require_full_support; + const size_t li_len = li_subgroup ? lightning_indexer_subgroup_f32_len : lightning_indexer_f32_len; + const void * li_data = li_subgroup ? (const void *)lightning_indexer_subgroup_f32_data : (const void *)lightning_indexer_f32_data; + + for (ggml_type k_type : lightning_indexer_k_types) { + const std::string name = "lightning_indexer_" + std::string(ggml_type_name(k_type)) + "_k_f32"; + ggml_vk_create_pipeline(device, device->pipeline_lightning_indexer_f32[k_type], name.c_str(), li_len, li_data, "main", 5, sizeof(vk_op_lightning_indexer_push_constants), {1, 1, 1}, {(uint32_t)k_type, fa_block_bytes(k_type), device->subgroup_size}, 1, true, li_subgroup); } -#endif + } - if (device->subgroup_ballot && device->subgroup_require_full_support && subgroup_min_size_16) { - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_id_f32, matmul_id_subgroup_f32_f32, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_id_f16.f32acc, matmul_id_subgroup_f16, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_id_f16_f32.f32acc, matmul_id_subgroup_f16_f32, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_subgroup_bf16, , wg_denoms, warptile_id, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size_16); - - CREATE_MM(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0].f32acc, matmul_id_subgroup_q1_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0].f32acc, matmul_id_subgroup_q2_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0].f32acc, matmul_id_subgroup_q4_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1].f32acc, matmul_id_subgroup_q4_1_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0].f32acc, matmul_id_subgroup_q5_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1].f32acc, matmul_id_subgroup_q5_1_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0].f32acc, matmul_id_subgroup_q8_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K].f32acc, matmul_id_subgroup_q2_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0].f32acc, matmul_id_subgroup_tq2_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K].f32acc, matmul_id_subgroup_q3_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K].f32acc, matmul_id_subgroup_q4_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K].f32acc, matmul_id_subgroup_q5_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K].f32acc, matmul_id_subgroup_q6_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S].f32acc, matmul_id_subgroup_iq1_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M].f32acc, matmul_id_subgroup_iq1_m_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS].f32acc, matmul_id_subgroup_iq2_xxs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS].f32acc, matmul_id_subgroup_iq2_xs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S].f32acc, matmul_id_subgroup_iq2_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS].f32acc, matmul_id_subgroup_iq3_xxs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S].f32acc, matmul_id_subgroup_iq3_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS].f32acc, matmul_id_subgroup_iq4_xs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL].f32acc, matmul_id_subgroup_iq4_nl_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4].f32acc, matmul_id_subgroup_mxfp4_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - CREATE_MM(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4].f32acc, matmul_id_subgroup_nvfp4_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, mul_mat_subgroup_size); - } else { - CREATE_MM(GGML_TYPE_F32, pipeline_matmul_id_f32, matmul_id_f32_f32, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_id_f16.f32acc, matmul_id_f16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_F16, pipeline_matmul_id_f16_f32.f32acc, matmul_id_f16_f32, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_bf16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - - CREATE_MM(GGML_TYPE_Q1_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q1_0].f32acc, matmul_id_q1_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_0].f32acc, matmul_id_q2_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q4_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_0].f32acc, matmul_id_q4_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q4_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_1].f32acc, matmul_id_q4_1_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q5_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_0].f32acc, matmul_id_q5_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q5_1, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_1].f32acc, matmul_id_q5_1_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q8_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q8_0].f32acc, matmul_id_q8_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q2_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q2_K].f32acc, matmul_id_q2_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_TQ2_0, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_TQ2_0].f32acc, matmul_id_tq2_0_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q3_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q3_K].f32acc, matmul_id_q3_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q4_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q4_K].f32acc, matmul_id_q4_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q5_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q5_K].f32acc, matmul_id_q5_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_Q6_K, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_Q6_K].f32acc, matmul_id_q6_k_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ1_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_S].f32acc, matmul_id_iq1_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ1_M, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ1_M].f32acc, matmul_id_iq1_m_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ2_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XXS].f32acc, matmul_id_iq2_xxs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ2_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_XS].f32acc, matmul_id_iq2_xs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ2_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ2_S].f32acc, matmul_id_iq2_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ3_XXS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_XXS].f32acc, matmul_id_iq3_xxs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ3_S, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ3_S].f32acc, matmul_id_iq3_s_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ4_XS, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_XS].f32acc, matmul_id_iq4_xs_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_IQ4_NL, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_IQ4_NL].f32acc, matmul_id_iq4_nl_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_MXFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_MXFP4].f32acc, matmul_id_mxfp4_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - CREATE_MM(GGML_TYPE_NVFP4, pipeline_dequant_mul_mat_mat_id[GGML_TYPE_NVFP4].f32acc, matmul_id_nvfp4_f32, , mmq_wg_denoms, warptile_mmqid, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - } - } - // reusing CREATE_MM from the fp32 path - if ((device->coopmat2 || device->coopmat_support) -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - && !device->coopmat_bf16_support -#endif - ) { - const uint32_t s_warptile_wm = device->subgroup_size == 8 ? 8 : 32; + { + const uint32_t gdn_sizes[] = {16, 32, 64, 128}; + const char * gdn_names[][2] = { + {"gated_delta_net_f32_d16", "gated_delta_net_f32_d16_kda"}, + {"gated_delta_net_f32_d32", "gated_delta_net_f32_d32_kda"}, + {"gated_delta_net_f32_d64", "gated_delta_net_f32_d64_kda"}, + {"gated_delta_net_f32_d128", "gated_delta_net_f32_d128_kda"}, + }; + for (uint32_t si = 0; si < 4; si++) { + const uint32_t S_V = gdn_sizes[si]; + GGML_ASSERT(is_pow2(S_V)); + + uint32_t lanes_per_column; + if (S_V >= 128u && device->subgroup_clustered) { + lanes_per_column = 8u; + } else { + // Use largest power-of-two that divides both S_V and subgroup_size so that + // (1) S_V % lanes_per_column == 0 and (2) S_V % (subgroup_size / lanes_per_column) == 0. + // This means we don't need extra bounds checking logic in the shader. + lanes_per_column = std::min(S_V, device->subgroup_size); + } - // use scalar tile sizes - l_warptile = { 128, 128, 128, 16, subgroup_size_8 * 2, 64, 2, 4, 4, 1, subgroup_size_8 }; - m_warptile = { 128, 64, 64, 16, subgroup_size_8, 32, 2, 4, 2, 1, subgroup_size_8 }; - s_warptile = { subgroup_size_32, 32, 32, 16, s_warptile_wm, 32, 2, 2, 2, 1, subgroup_size_8 }; + // gated_delta_net.comp relies on S_V % COLS_PER_WG == 0 and + // S_V % LANES_PER_COLUMN == 0 to avoid bounds checks. + while (lanes_per_column > 1u) { + const bool valid_lanes = (device->subgroup_size % lanes_per_column) == 0 && + (S_V % lanes_per_column) == 0; + const uint32_t cols_per_wg = valid_lanes ? device->subgroup_size / lanes_per_column : 0; + if (valid_lanes && cols_per_wg > 0 && (S_V % cols_per_wg) == 0) { + break; + } + lanes_per_column >>= 1u; + } - l_wg_denoms = {128, 128, 1 }; - m_wg_denoms = { 64, 64, 1 }; - s_wg_denoms = { 32, 32, 1 }; + GGML_ASSERT((device->subgroup_size % lanes_per_column) == 0); + GGML_ASSERT((S_V % lanes_per_column) == 0); + GGML_ASSERT((S_V % (device->subgroup_size / lanes_per_column)) == 0); - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_bf16, matmul_bf16, , wg_denoms, warptile, vk_mat_mat_push_constants, 3, , 0); - CREATE_MM(GGML_TYPE_BF16, pipeline_matmul_id_bf16, matmul_id_bf16, , wg_denoms, warptile, vk_mat_mat_id_push_constants, mul_mat_id_param_count, _id, 0); - } -#undef CREATE_MM + const bool need_partial_subgroup_reduce = lanes_per_column != 1u && lanes_per_column < device->subgroup_size; + const bool use_clustered_reduce = device->subgroup_arithmetic && device->subgroup_clustered && need_partial_subgroup_reduce; + const bool use_subgroup_reduce = device->subgroup_arithmetic && !need_partial_subgroup_reduce; + const bool use_subgroup_ops = use_clustered_reduce || use_subgroup_reduce; + size_t gdn_len; + const void * gdn_data; + if (use_clustered_reduce) { + gdn_len = gated_delta_net_f32_len; + gdn_data = (const void *)gated_delta_net_f32_data; + } else if (use_subgroup_reduce) { + gdn_len = gated_delta_net_f32_nocluster_len; + gdn_data = (const void *)gated_delta_net_f32_nocluster_data; + } else { + gdn_len = gated_delta_net_f32_shmem_len; + gdn_data = (const void *)gated_delta_net_f32_shmem_data; + } - // mul mat vec + const uint32_t cols_per_wg = device->subgroup_size / lanes_per_column; + const std::array wg_denoms = {1u, 1u, cols_per_wg}; - // the number of rows computed per shader depends on GPU model and quant - uint32_t rm_stdq = 1; - uint32_t rm_kq = 2; - uint32_t rm_stdq_int = 1; - uint32_t rm_kq_int = 1; - auto const &rm_iq_int = [](uint32_t i) { return i == 0 ? 8u : 4u; }; - if (device->vendor_id == VK_VENDOR_ID_AMD) { - if (device->architecture == AMD_GCN) { - rm_stdq = 2; - rm_kq = 4; - rm_stdq_int = 4; + for (uint32_t kda = 0; kda < 2; kda++) { + ggml_vk_create_pipeline(device, device->pipeline_gated_delta_net[si][kda], + gdn_names[si][kda], gdn_len, gdn_data, "main", 7, sizeof(vk_op_gated_delta_net_push_constants), + wg_denoms, {S_V, kda, device->subgroup_size, lanes_per_column}, 1, true, use_subgroup_ops, device->subgroup_size); + } } - } else if (device->vendor_id == VK_VENDOR_ID_INTEL) { - rm_stdq = 2; - rm_stdq_int = 2; } - uint32_t rm_iq = 2 * rm_kq; - const bool use_subgroups = device->subgroup_arithmetic; - // Ensure a subgroup size >= 16 is available - const bool use_subgroups16 = use_subgroups && subgroup_min_size_16; + if (device->subgroup_arithmetic && device->subgroup_require_full_support) { + ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d128, "ssm_scan_128_f32", ssm_scan_subgroup_f32_len, ssm_scan_subgroup_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {128, device->subgroup_size}, 1, true, true); + ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d256, "ssm_scan_256_f32", ssm_scan_subgroup_f32_len, ssm_scan_subgroup_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {256, device->subgroup_size}, 1, true, true); + } else { + ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d128, "ssm_scan_128_f32", ssm_scan_f32_len, ssm_scan_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {128, device->subgroup_size, 16}, 1, true, true); + ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d256, "ssm_scan_256_f32", ssm_scan_f32_len, ssm_scan_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {256, device->subgroup_size, 16}, 1, true, true); + } - const uint32_t subgroup_size = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control && device->subgroup_min_size <= 16 && device->subgroup_max_size >= 16) ? 16 : device->subgroup_size; - const uint32_t subgroup_size16 = std::max(subgroup_size, 16u); + ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_f32, "ssm_conv_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 0, 0}, 1); + ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_silu_f32, "ssm_conv_silu_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 0, 1}, 1); + ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_bias_silu_f32, "ssm_conv_bias_silu_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 1, 1}, 1); - const uint32_t force_subgroup_size = use_subgroups ? subgroup_size : 0; - const uint32_t force_subgroup_size16 = use_subgroups16 ? subgroup_size16 : 0; - static constexpr uint32_t mul_mat_vec_num_bindings = 5; - static constexpr uint32_t mul_mat_vec_id_num_bindings = 6; + ggml_vk_create_pipeline(device, device->pipeline_opt_step_adamw_f32, "opt_step_adamw_f32", opt_step_adamw_f32_len, opt_step_adamw_f32_data, "main", 5, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) -#define OCP_DMMV_LEN(NAME, REDUC) (device->ocp_fp4 ? NAME ## _ocp_len[REDUC] : NAME ## _len[REDUC]) -#define OCP_DMMV_DATA(NAME, REDUC) (device->ocp_fp4 ? NAME ## _ocp_data[REDUC] : NAME ## _data[REDUC]) -#else -#define OCP_DMMV_LEN(NAME, REDUC) NAME ## _len[REDUC] -#define OCP_DMMV_DATA(NAME, REDUC) NAME ## _data[REDUC] -#endif + ggml_vk_create_pipeline(device, device->pipeline_opt_step_sgd_f32, "opt_step_sgd_f32", opt_step_sgd_f32_len, opt_step_sgd_f32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); - for (uint32_t w = 0; w < DMMV_WG_SIZE_COUNT; ++w) { - const uint32_t wg_size_subgroup = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size : (subgroup_size * 4); - const uint32_t wg_size_subgroup16 = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size16 : (subgroup_size16 * 4); + // conv2d, conv_transpose_2d, conv3d + for (uint32_t s = 0; s < CONV_SHAPE_COUNT; ++s) { + // smaller WG for the small-tile fallback gives more concurrent WGs per SM + uint32_t conv2d_WG_SIZE = (s == CONV_SHAPE_64x32) ? 128 : 256; + uint32_t use_collectives = 0; // Enables subgroup ops for preventing the re-calculation of indices. + uint32_t conv2d_TS_K = (s == CONV_SHAPE_64x32) ? 4 : 8; + uint32_t conv2d_SHMEM_PAD = 4; + vk_conv_block_size conv2d_BS = vk_conv_block_sizes[s]; + bool conv2d_UNROLL = true; - const shader_reduction_mode reduc = (use_subgroups && w == DMMV_WG_SIZE_SUBGROUP) ? SHADER_REDUCTION_MODE_SUBGROUP : - (use_subgroups && w == DMMV_WG_SIZE_LARGE) ? SHADER_REDUCTION_MODE_HYBRID : - SHADER_REDUCTION_MODE_SHMEM; +#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (device->coopmat2) { + conv2d_SHMEM_PAD = 8; // 8 float16_t + } +#endif - const shader_reduction_mode reduc16 = (use_subgroups16 && w == DMMV_WG_SIZE_SUBGROUP) ? SHADER_REDUCTION_MODE_SUBGROUP : - (use_subgroups16 && w == DMMV_WG_SIZE_LARGE) ? SHADER_REDUCTION_MODE_HYBRID : - SHADER_REDUCTION_MODE_SHMEM; + if (device->vendor_id == VK_VENDOR_ID_INTEL) { + conv2d_SHMEM_PAD = 0; + conv2d_UNROLL = false; + } else if (device->vendor_id == VK_VENDOR_ID_AMD) { + conv2d_SHMEM_PAD = device->architecture == vk_device_architecture::AMD_GCN ? 1 : 4; + if (s == CONV_SHAPE_128x128 && device->architecture != vk_device_architecture::AMD_GCN) { + conv2d_UNROLL = false; + } + } - for (uint32_t i = 0; i < mul_mat_vec_max_cols; ++i) { - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_F32 ][i], "mul_mat_vec_f32_f32_f32", arr_dmmv_f32_f32_f32_len[reduc], arr_dmmv_f32_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1, 1, 1}, {wg_size_subgroup, 1, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_F16 ][i], "mul_mat_vec_f16_f32_f32", arr_dmmv_f16_f32_f32_len[reduc], arr_dmmv_f16_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_BF16][i], "mul_mat_vec_bf16_f32_f32", arr_dmmv_bf16_f32_f32_len[reduc], arr_dmmv_bf16_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q1_0][i], "mul_mat_vec_q1_0_f32_f32", arr_dmmv_q1_0_f32_f32_len[reduc], arr_dmmv_q1_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_f32_f32", arr_dmmv_q2_0_f32_f32_len[reduc], arr_dmmv_q2_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_f32_f32", arr_dmmv_q4_0_f32_f32_len[reduc], arr_dmmv_q4_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_f32_f32", arr_dmmv_q4_1_f32_f32_len[reduc], arr_dmmv_q4_1_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_f32_f32", arr_dmmv_q5_0_f32_f32_len[reduc], arr_dmmv_q5_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_f32_f32", arr_dmmv_q5_1_f32_f32_len[reduc], arr_dmmv_q5_1_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_f32_f32", arr_dmmv_q8_0_f32_f32_len[reduc], arr_dmmv_q8_0_f32_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_f32_f32", arr_dmmv_q2_k_f32_f32_len[reduc16], arr_dmmv_q2_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_TQ2_0][i], "mul_mat_vec_tq2_0_f32_f32", arr_dmmv_tq2_0_f32_f32_len[reduc16], arr_dmmv_tq2_0_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_f32_f32", arr_dmmv_q3_k_f32_f32_len[reduc16], arr_dmmv_q3_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_f32_f32", arr_dmmv_q4_k_f32_f32_len[reduc16], arr_dmmv_q4_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_f32_f32", arr_dmmv_q5_k_f32_f32_len[reduc16], arr_dmmv_q5_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_f32_f32", arr_dmmv_q6_k_f32_f32_len[reduc16], arr_dmmv_q6_k_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_f32_f32", arr_dmmv_iq1_s_f32_f32_len[reduc16], arr_dmmv_iq1_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_f32_f32", arr_dmmv_iq1_m_f32_f32_len[reduc16], arr_dmmv_iq1_m_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_XXS][i], "mul_mat_vec_iq2_xxs_f32_f32", arr_dmmv_iq2_xxs_f32_f32_len[reduc16], arr_dmmv_iq2_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_XS][i], "mul_mat_vec_iq2_xs_f32_f32", arr_dmmv_iq2_xs_f32_f32_len[reduc16], arr_dmmv_iq2_xs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ2_S][i], "mul_mat_vec_iq2_s_f32_f32", arr_dmmv_iq2_s_f32_f32_len[reduc16], arr_dmmv_iq2_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ3_XXS][i], "mul_mat_vec_iq3_xxs_f32_f32", arr_dmmv_iq3_xxs_f32_f32_len[reduc16], arr_dmmv_iq3_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ3_S][i], "mul_mat_vec_iq3_s_f32_f32", arr_dmmv_iq3_s_f32_f32_len[reduc16], arr_dmmv_iq3_s_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ4_XS][i], "mul_mat_vec_iq4_xs_f32_f32", arr_dmmv_iq4_xs_f32_f32_len[reduc16], arr_dmmv_iq4_xs_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_IQ4_NL][i], "mul_mat_vec_iq4_nl_f32_f32", arr_dmmv_iq4_nl_f32_f32_len[reduc16], arr_dmmv_iq4_nl_f32_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_f32_f32", OCP_DMMV_LEN(arr_dmmv_mxfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_mxfp4_f32_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f32_f32[w][GGML_TYPE_NVFP4][i], "mul_mat_vec_nvfp4_f32_f32", OCP_DMMV_LEN(arr_dmmv_nvfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_nvfp4_f32_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_F32 ][i], "mul_mat_vec_f32_f16_f32", arr_dmmv_f32_f16_f32_len[reduc], arr_dmmv_f32_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1, 1, 1}, {wg_size_subgroup, 1, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_F16 ][i], "mul_mat_vec_f16_f16_f32", arr_dmmv_f16_f16_f32_len[reduc], arr_dmmv_f16_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_BF16][i], "mul_mat_vec_bf16_f16_f32", arr_dmmv_bf16_f16_f32_len[reduc], arr_dmmv_bf16_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2, 1, 1}, {wg_size_subgroup, 2, i+1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q1_0][i], "mul_mat_vec_q1_0_f16_f32", arr_dmmv_q1_0_f16_f32_len[reduc], arr_dmmv_q1_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_f16_f32", arr_dmmv_q2_0_f16_f32_len[reduc], arr_dmmv_q2_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_f16_f32", arr_dmmv_q4_0_f16_f32_len[reduc], arr_dmmv_q4_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_f16_f32", arr_dmmv_q4_1_f16_f32_len[reduc], arr_dmmv_q4_1_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_f16_f32", arr_dmmv_q5_0_f16_f32_len[reduc], arr_dmmv_q5_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_f16_f32", arr_dmmv_q5_1_f16_f32_len[reduc], arr_dmmv_q5_1_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_f16_f32", arr_dmmv_q8_0_f16_f32_len[reduc], arr_dmmv_q8_0_f16_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq, i+1}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_f16_f32", arr_dmmv_q2_k_f16_f32_len[reduc16], arr_dmmv_q2_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_TQ2_0][i], "mul_mat_vec_tq2_0_f16_f32", arr_dmmv_tq2_0_f16_f32_len[reduc16], arr_dmmv_tq2_0_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_f16_f32", arr_dmmv_q3_k_f16_f32_len[reduc16], arr_dmmv_q3_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_f16_f32", arr_dmmv_q4_k_f16_f32_len[reduc16], arr_dmmv_q4_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_f16_f32", arr_dmmv_q5_k_f16_f32_len[reduc16], arr_dmmv_q5_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_f16_f32", arr_dmmv_q6_k_f16_f32_len[reduc16], arr_dmmv_q6_k_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_f16_f32", arr_dmmv_iq1_s_f16_f32_len[reduc16], arr_dmmv_iq1_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_f16_f32", arr_dmmv_iq1_m_f16_f32_len[reduc16], arr_dmmv_iq1_m_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_XXS][i], "mul_mat_vec_iq2_xxs_f16_f32", arr_dmmv_iq2_xxs_f16_f32_len[reduc16], arr_dmmv_iq2_xxs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_XS][i], "mul_mat_vec_iq2_xs_f16_f32", arr_dmmv_iq2_xs_f16_f32_len[reduc16], arr_dmmv_iq2_xs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ2_S][i], "mul_mat_vec_iq2_s_f16_f32", arr_dmmv_iq2_s_f16_f32_len[reduc16], arr_dmmv_iq2_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ3_XXS][i], "mul_mat_vec_iq3_xxs_f16_f32", arr_dmmv_iq3_xxs_f16_f32_len[reduc16], arr_dmmv_iq3_xxs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ3_S][i], "mul_mat_vec_iq3_s_f16_f32", arr_dmmv_iq3_s_f16_f32_len[reduc16], arr_dmmv_iq3_s_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ4_XS][i], "mul_mat_vec_iq4_xs_f16_f32", arr_dmmv_iq4_xs_f16_f32_len[reduc16], arr_dmmv_iq4_xs_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_IQ4_NL][i], "mul_mat_vec_iq4_nl_f16_f32", arr_dmmv_iq4_nl_f16_f32_len[reduc16], arr_dmmv_iq4_nl_f16_f32_data[reduc16], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_f16_f32", OCP_DMMV_LEN(arr_dmmv_mxfp4_f16_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_mxfp4_f16_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_f16_f32[w][GGML_TYPE_NVFP4][i], "mul_mat_vec_nvfp4_f16_f32", OCP_DMMV_LEN(arr_dmmv_nvfp4_f16_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_nvfp4_f16_f32, reduc16), "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq, i+1}, 1, true, use_subgroups16, force_subgroup_size16); + // Use collectives on pre-Turing NVIDIA GPUs and GCN AMD cards, which had slower integer math. + bool allow_collectives_nv = device->vendor_id != VK_VENDOR_ID_NVIDIA || + device->architecture == vk_device_architecture::NVIDIA_PRE_TURING; + bool allow_collectives_amd = device->vendor_id != VK_VENDOR_ID_AMD || + device->architecture == vk_device_architecture::AMD_GCN; -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - const uint32_t subgroup_size_int = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; - const uint32_t wg_size_subgroup_int = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size_int : (subgroup_size_int * 4); + if (device->subgroup_shuffle && + device->vendor_id != VK_VENDOR_ID_INTEL && // Do not enable collectives on Intel, see PR 14316. + allow_collectives_nv && + allow_collectives_amd) { + use_collectives = 1; + conv2d_BS.CRS = std::min( + device->subgroup_size, + conv2d_BS.CRS); // CRS block size should be capped at subgroup size for correctness when shuffle is used. + } - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q2_0][i], "mul_mat_vec_q2_0_q8_1_f32", arr_dmmv_q2_0_q8_1_f32_len[reduc], arr_dmmv_q2_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_0][i], "mul_mat_vec_q4_0_q8_1_f32", arr_dmmv_q4_0_q8_1_f32_len[reduc], arr_dmmv_q4_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_1][i], "mul_mat_vec_q4_1_q8_1_f32", arr_dmmv_q4_1_q8_1_f32_len[reduc], arr_dmmv_q4_1_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_0][i], "mul_mat_vec_q5_0_q8_1_f32", arr_dmmv_q5_0_q8_1_f32_len[reduc], arr_dmmv_q5_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_1][i], "mul_mat_vec_q5_1_q8_1_f32", arr_dmmv_q5_1_q8_1_f32_len[reduc], arr_dmmv_q5_1_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q8_0][i], "mul_mat_vec_q8_0_q8_1_f32", arr_dmmv_q8_0_q8_1_f32_len[reduc], arr_dmmv_q8_0_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); + // cm1 is used only when cm2 is unavailable; capped at 64x128 (due to shared memory size). + // Requires 16x16x16 f16-acc since that's the fragment shape hard-coded in the shader. + // Subgroup size must be 32 or 64 (to keep WG_SIZE sane) and we need + // subgroup_size_control to force the driver to actually use it. + bool conv2d_use_cm1 = false; +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + conv2d_use_cm1 = !device->coopmat2 && + device->coopmat_support && device->coopmat_support_16x16x16_f16acc && + device->subgroup_size_control && + (device->subgroup_size == 32 || device->subgroup_size == 64) && + s != CONV_SHAPE_128x128; +#endif - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_MXFP4][i], "mul_mat_vec_mxfp4_q8_1_f32", arr_dmmv_mxfp4_q8_1_f32_len[reduc], arr_dmmv_mxfp4_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_stdq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); + const uint32_t conv2d_cm1_shmem_pad = 8; - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q2_K][i], "mul_mat_vec_q2_k_q8_1_f32", arr_dmmv_q2_k_q8_1_f32_len[reduc], arr_dmmv_q2_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {2*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q3_K][i], "mul_mat_vec_q3_k_q8_1_f32", arr_dmmv_q3_k_q8_1_f32_len[reduc], arr_dmmv_q3_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q4_K][i], "mul_mat_vec_q4_k_q8_1_f32", arr_dmmv_q4_k_q8_1_f32_len[reduc], arr_dmmv_q4_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q5_K][i], "mul_mat_vec_q5_k_q8_1_f32", arr_dmmv_q5_k_q8_1_f32_len[reduc], arr_dmmv_q5_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_Q6_K][i], "mul_mat_vec_q6_k_q8_1_f32", arr_dmmv_q6_k_q8_1_f32_len[reduc], arr_dmmv_q6_k_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int, i+1}, 1, true, use_subgroups, subgroup_size_int); + auto shmem_req = [&](uint32_t pad, bool csh_store, bool fp16_shmem) { + const uint32_t elem_size = fp16_shmem ? (uint32_t)sizeof(uint16_t) : (uint32_t)sizeof(float); + const uint32_t csh_elems = csh_store ? conv2d_BS.K * conv2d_BS.NPQ : 0u; + return (conv2d_BS.K * (conv2d_BS.CRS + pad) + conv2d_BS.CRS * (conv2d_BS.NPQ + pad) + csh_elems) * elem_size; + }; - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_IQ1_S][i], "mul_mat_vec_iq1_s_q8_1_f32", arr_dmmv_iq1_s_q8_1_f32_len[reduc], arr_dmmv_iq1_s_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_iq_int(i), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(i), i+1}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_q8_1_f32[w][GGML_TYPE_IQ1_M][i], "mul_mat_vec_iq1_m_q8_1_f32", arr_dmmv_iq1_m_q8_1_f32_len[reduc], arr_dmmv_iq1_m_q8_1_f32_data[reduc], "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_push_constants), {1*rm_iq_int(i), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(i), i+1}, 1, true, use_subgroups, subgroup_size_int); + // 2D, transpose-2D, and 3D conv use the same KxCRS @ CRSxNPQ shmem + // layout. cm1 needs Csh for output, so check before applying cm1 params. + if (conv2d_use_cm1 && device->properties.limits.maxComputeSharedMemorySize < shmem_req(conv2d_cm1_shmem_pad, true, true)) { + conv2d_use_cm1 = false; + } + uint32_t conv2d_WM = 16, conv2d_WN = 16; // cm1 subgroup tile, ignored otherwise + if (conv2d_use_cm1) { + conv2d_SHMEM_PAD = conv2d_cm1_shmem_pad; + // 16x16x16 fragments; pick WM/WN to keep WG_SIZE at 256 + // (i.e. 8 subgroups for sg=32, 4 subgroups for sg=64). + const bool sg64 = (device->subgroup_size == 64); + switch (s) { + case CONV_SHAPE_64x32: conv2d_WM = sg64 ? 32 : 16; conv2d_WN = 16; break; + case CONV_SHAPE_64x128: conv2d_WM = 32; conv2d_WN = sg64 ? 64 : 32; break; + case CONV_SHAPE_32x256: conv2d_WM = sg64 ? 16 : 32; conv2d_WN = sg64 ? 128 : 32; break; + default: break; } -#endif // GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT + const uint32_t warps_M = conv2d_BS.K / conv2d_WM; + const uint32_t warps_N = conv2d_BS.NPQ / conv2d_WN; + conv2d_WG_SIZE = warps_M * warps_N * device->subgroup_size; } - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_F32 ], "mul_mat_vec_id_f32_f32", arr_dmmv_id_f32_f32_f32_len[reduc], arr_dmmv_id_f32_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1, 1, 1}, {wg_size_subgroup, 1}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_F16 ], "mul_mat_vec_id_f16_f32", arr_dmmv_id_f16_f32_f32_len[reduc], arr_dmmv_id_f16_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2, 1, 1}, {wg_size_subgroup, 2}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_BF16], "mul_mat_vec_id_bf16_f32", arr_dmmv_id_bf16_f32_f32_len[reduc], arr_dmmv_id_bf16_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2, 1, 1}, {wg_size_subgroup, 2}, 1, false, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q1_0], "mul_mat_vec_id_q1_0_f32", arr_dmmv_id_q1_0_f32_f32_len[reduc], arr_dmmv_id_q1_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q2_0], "mul_mat_vec_id_q2_0_f32", arr_dmmv_id_q2_0_f32_f32_len[reduc], arr_dmmv_id_q2_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_0], "mul_mat_vec_id_q4_0_f32", arr_dmmv_id_q4_0_f32_f32_len[reduc], arr_dmmv_id_q4_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_1], "mul_mat_vec_id_q4_1_f32", arr_dmmv_id_q4_1_f32_f32_len[reduc], arr_dmmv_id_q4_1_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_0], "mul_mat_vec_id_q5_0_f32", arr_dmmv_id_q5_0_f32_f32_len[reduc], arr_dmmv_id_q5_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_1], "mul_mat_vec_id_q5_1_f32", arr_dmmv_id_q5_1_f32_f32_len[reduc], arr_dmmv_id_q5_1_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq, 1, 1}, {wg_size_subgroup, 2*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q8_0], "mul_mat_vec_id_q8_0_f32", arr_dmmv_id_q8_0_f32_f32_len[reduc], arr_dmmv_id_q8_0_f32_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq, 1, 1}, {wg_size_subgroup, 1*rm_stdq}, 1, true, use_subgroups, force_subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q2_K], "mul_mat_vec_id_q2_k_f32", arr_dmmv_id_q2_k_f32_f32_len[reduc16], arr_dmmv_id_q2_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_TQ2_0], "mul_mat_vec_id_tq2_0_f32", arr_dmmv_id_tq2_0_f32_f32_len[reduc16], arr_dmmv_id_tq2_0_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q3_K], "mul_mat_vec_id_q3_k_f32", arr_dmmv_id_q3_k_f32_f32_len[reduc16], arr_dmmv_id_q3_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q4_K], "mul_mat_vec_id_q4_k_f32", arr_dmmv_id_q4_k_f32_f32_len[reduc16], arr_dmmv_id_q4_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q5_K], "mul_mat_vec_id_q5_k_f32", arr_dmmv_id_q5_k_f32_f32_len[reduc16], arr_dmmv_id_q5_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_Q6_K], "mul_mat_vec_id_q6_k_f32", arr_dmmv_id_q6_k_f32_f32_len[reduc16], arr_dmmv_id_q6_k_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_kq, 1, 1}, {wg_size_subgroup16, rm_kq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ1_S], "mul_mat_vec_id_iq1_s_f32", arr_dmmv_id_iq1_s_f32_f32_len[reduc16], arr_dmmv_id_iq1_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ1_M], "mul_mat_vec_id_iq1_m_f32", arr_dmmv_id_iq1_m_f32_f32_len[reduc16], arr_dmmv_id_iq1_m_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_XXS], "mul_mat_vec_id_iq2_xxs_f32", arr_dmmv_id_iq2_xxs_f32_f32_len[reduc16], arr_dmmv_id_iq2_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_XS], "mul_mat_vec_id_iq2_xs_f32", arr_dmmv_id_iq2_xs_f32_f32_len[reduc16], arr_dmmv_id_iq2_xs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ2_S], "mul_mat_vec_id_iq2_s_f32", arr_dmmv_id_iq2_s_f32_f32_len[reduc16], arr_dmmv_id_iq2_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ3_XXS], "mul_mat_vec_id_iq3_xxs_f32", arr_dmmv_id_iq3_xxs_f32_f32_len[reduc16], arr_dmmv_id_iq3_xxs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ3_S], "mul_mat_vec_id_iq3_s_f32", arr_dmmv_id_iq3_s_f32_f32_len[reduc16], arr_dmmv_id_iq3_s_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ4_XS], "mul_mat_vec_id_iq4_xs_f32", arr_dmmv_id_iq4_xs_f32_f32_len[reduc16], arr_dmmv_id_iq4_xs_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_IQ4_NL], "mul_mat_vec_id_iq4_nl_f32", arr_dmmv_id_iq4_nl_f32_f32_len[reduc16], arr_dmmv_id_iq4_nl_f32_f32_data[reduc16], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_MXFP4], "mul_mat_vec_id_mxfp4_f32", OCP_DMMV_LEN(arr_dmmv_id_mxfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_id_mxfp4_f32_f32, reduc16), "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_f32[w][GGML_TYPE_NVFP4], "mul_mat_vec_id_nvfp4_f32", OCP_DMMV_LEN(arr_dmmv_id_nvfp4_f32_f32, reduc16), OCP_DMMV_DATA(arr_dmmv_id_nvfp4_f32_f32, reduc16), "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {rm_iq, 1, 1}, {wg_size_subgroup16, rm_iq}, 1, true, use_subgroups16, force_subgroup_size16); + // stage cm2 accumulator through shmem for coalesced global stores; + // skipped on 128x128 where the extra Csh footprint hurts occupancy. + // cm1 always uses the staged path. + uint32_t conv2d_csh_store = (device->coopmat2 && s != CONV_SHAPE_128x128) ? 1u : 0u; + if (conv2d_use_cm1) { + conv2d_csh_store = 1; + } -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (device->integer_dot_product) { - const uint32_t subgroup_size_int = (device->vendor_id == VK_VENDOR_ID_INTEL && device->subgroup_size_control) ? device->subgroup_min_size : device->subgroup_size; - const uint32_t wg_size_subgroup_int = (w == DMMV_WG_SIZE_SUBGROUP) ? subgroup_size_int : (subgroup_size_int * 4); + // shmem is fp16 on cm2/cm1 (matches Csh), fp32 on scalar + const bool conv2d_use_fp16_shmem = device->coopmat2 || conv2d_use_cm1; - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q2_0], "mul_mat_vec_id_q2_0_q8_1_f32", arr_dmmv_id_q2_0_q8_1_f32_len[reduc], arr_dmmv_id_q2_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_0], "mul_mat_vec_id_q4_0_q8_1_f32", arr_dmmv_id_q4_0_q8_1_f32_len[reduc], arr_dmmv_id_q4_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_1], "mul_mat_vec_id_q4_1_q8_1_f32", arr_dmmv_id_q4_1_q8_1_f32_len[reduc], arr_dmmv_id_q4_1_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_0], "mul_mat_vec_id_q5_0_q8_1_f32", arr_dmmv_id_q5_0_q8_1_f32_len[reduc], arr_dmmv_id_q5_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_1], "mul_mat_vec_id_q5_1_q8_1_f32", arr_dmmv_id_q5_1_q8_1_f32_len[reduc], arr_dmmv_id_q5_1_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q8_0], "mul_mat_vec_id_q8_0_q8_1_f32", arr_dmmv_id_q8_0_q8_1_f32_len[reduc], arr_dmmv_id_q8_0_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); + // shrink CRS if the non-cm1 config still doesn't fit + if (device->properties.limits.maxComputeSharedMemorySize < shmem_req(conv2d_SHMEM_PAD, conv2d_csh_store, conv2d_use_fp16_shmem)) { + GGML_ASSERT(!conv2d_use_cm1); + conv2d_BS.CRS = 8; + if (use_collectives) { + conv2d_BS.CRS = std::min(device->subgroup_size, conv2d_BS.CRS); + } + conv2d_csh_store = 0; + } - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_MXFP4], "mul_mat_vec_id_mxfp4_q8_1_f32", arr_dmmv_id_mxfp4_q8_1_f32_len[reduc], arr_dmmv_id_mxfp4_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_stdq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_stdq_int}, 1, true, use_subgroups, subgroup_size_int); + std::array wg_denoms = { conv2d_BS.K, 1, 1 }; + std::vector spec_constants = { conv2d_WG_SIZE, conv2d_BS.K, conv2d_BS.CRS, conv2d_BS.NPQ, conv2d_TS_K, use_collectives, conv2d_SHMEM_PAD }; - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q2_K], "mul_mat_vec_id_q2_k_q8_1_f32", arr_dmmv_id_q2_k_q8_1_f32_len[reduc], arr_dmmv_id_q2_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {2*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 2*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q3_K], "mul_mat_vec_id_q3_k_q8_1_f32", arr_dmmv_id_q3_k_q8_1_f32_len[reduc], arr_dmmv_id_q3_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q4_K], "mul_mat_vec_id_q4_k_q8_1_f32", arr_dmmv_id_q4_k_q8_1_f32_len[reduc], arr_dmmv_id_q4_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q5_K], "mul_mat_vec_id_q5_k_q8_1_f32", arr_dmmv_id_q5_k_q8_1_f32_len[reduc], arr_dmmv_id_q5_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_Q6_K], "mul_mat_vec_id_q6_k_q8_1_f32", arr_dmmv_id_q6_k_q8_1_f32_len[reduc], arr_dmmv_id_q6_k_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_kq_int, 1, 1}, {wg_size_subgroup_int, 1*rm_kq_int}, 1, true, use_subgroups, subgroup_size_int); + // cm1 needs a fixed subgroup width to match the WG_SIZE we computed + const uint32_t conv2d_required_subgroup_size = conv2d_use_cm1 ? device->subgroup_size : 0; - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_IQ1_S], "mul_mat_vec_id_iq1_s_q8_1_f32", arr_dmmv_id_iq1_s_q8_1_f32_len[reduc], arr_dmmv_id_iq1_s_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_iq_int(0), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(0)}, 1, true, use_subgroups, subgroup_size_int); - ggml_vk_create_pipeline(device, device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[w][GGML_TYPE_IQ1_M], "mul_mat_vec_id_iq1_m_q8_1_f32", arr_dmmv_id_iq1_m_q8_1_f32_len[reduc], arr_dmmv_id_iq1_m_q8_1_f32_data[reduc], "main", mul_mat_vec_id_num_bindings, sizeof(vk_mat_vec_id_push_constants), {1*rm_iq_int(0), 1, 1}, {wg_size_subgroup_int, 1*rm_iq_int(0)}, 1, true, use_subgroups, subgroup_size_int); +#define CREATE_CONV(name, type_suffix, spv_suffix) \ + for (auto &c : device->pipeline_##name##type_suffix[s]) { \ + const vk_conv2d_pipeline_state &state = c.first; \ + std::vector spec_constants_cpy = spec_constants; \ + spec_constants_cpy.push_back(state.s0); \ + spec_constants_cpy.push_back(state.s1); \ + spec_constants_cpy.push_back(state.p0); \ + spec_constants_cpy.push_back(state.p1); \ + spec_constants_cpy.push_back(state.d0); \ + spec_constants_cpy.push_back(state.d1); \ + spec_constants_cpy.push_back(state.KW); \ + spec_constants_cpy.push_back(state.KH); \ + spec_constants_cpy.push_back(state.aligned); \ + spec_constants_cpy.push_back(conv2d_csh_store); \ + spec_constants_cpy.push_back(conv2d_WM); \ + spec_constants_cpy.push_back(conv2d_WN); \ + ggml_vk_create_pipeline( \ + device, c.second, #name #type_suffix, \ + name##type_suffix##spv_suffix##_len, name##type_suffix##spv_suffix##_data, "main", 3, \ + sizeof(vk_op_conv2d_push_constants), wg_denoms, spec_constants_cpy, 1, true, use_collectives || conv2d_required_subgroup_size, conv2d_required_subgroup_size); \ } -#endif // GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT - } - -#undef OCP_DMMV_DATA -#undef OCP_DMMV_LEN - -#if !defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - GGML_UNUSED(rm_stdq_int); - GGML_UNUSED(rm_kq_int); - GGML_UNUSED(rm_iq_int); +#define CREATE_CONVS(spv_suffix) \ + CREATE_CONV(conv2d, _f32, spv_suffix) \ + CREATE_CONV(conv2d, _f16_f32, spv_suffix) \ + CREATE_CONV(conv_transpose_2d, _f32, spv_suffix) \ + CREATE_CONV(conv_transpose_2d, _f16_f32, spv_suffix) +#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (device->coopmat2) { + CREATE_CONVS(_cm2) + } else #endif +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + if (conv2d_use_cm1) { + CREATE_CONVS(_cm1) + } else +#endif + if (conv2d_UNROLL) { + CREATE_CONVS(_unroll) + } else { + CREATE_CONVS( ) + } +#undef CREATE_CONV +#undef CREATE_CONVS - // dequant shaders - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_F32 ], "f32_to_f16", dequant_f32_len, dequant_f32_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q1_0], "dequant_q1_0", dequant_q1_0_len, dequant_q1_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 8, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q2_0], "dequant_q2_0", dequant_q2_0_len, dequant_q2_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 4, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_0], "dequant_q4_0", dequant_q4_0_len, dequant_q4_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_1], "dequant_q4_1", dequant_q4_1_len, dequant_q4_1_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_0], "dequant_q5_0", dequant_q5_0_len, dequant_q5_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_1], "dequant_q5_1", dequant_q5_1_len, dequant_q5_1_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q8_0], "dequant_q8_0", dequant_q8_0_len, dequant_q8_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q2_K], "dequant_q2_k", dequant_q2_k_len, dequant_q2_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_TQ2_0], "dequant_tq2_0", dequant_tq2_0_len, dequant_tq2_0_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q3_K], "dequant_q3_k", dequant_q3_k_len, dequant_q3_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q4_K], "dequant_q4_k", dequant_q4_k_len, dequant_q4_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q5_K], "dequant_q5_k", dequant_q5_k_len, dequant_q5_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_Q6_K], "dequant_q6_k", dequant_q6_k_len, dequant_q6_k_data, "main", 2, 5 * sizeof(uint32_t), {256 * 64, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ1_S], "dequant_iq1_s", dequant_iq1_s_len, dequant_iq1_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ1_M], "dequant_iq1_m", dequant_iq1_m_len, dequant_iq1_m_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_XXS], "dequant_iq2_xxs", dequant_iq2_xxs_len, dequant_iq2_xxs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_XS], "dequant_iq2_xs", dequant_iq2_xs_len, dequant_iq2_xs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ2_S], "dequant_iq2_s", dequant_iq2_s_len, dequant_iq2_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ3_XXS], "dequant_iq3_xxs", dequant_iq3_xxs_len, dequant_iq3_xxs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ3_S], "dequant_iq3_s", dequant_iq3_s_len, dequant_iq3_s_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ4_XS], "dequant_iq4_xs", dequant_iq4_xs_len, dequant_iq4_xs_data, "main", 2, 5 * sizeof(uint32_t), {256 * 32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_IQ4_NL], "dequant_iq4_nl", dequant_iq4_nl_len, dequant_iq4_nl_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_MXFP4], "dequant_mxfp4", dequant_mxfp4_len, dequant_mxfp4_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_dequant[GGML_TYPE_NVFP4], "dequant_nvfp4", dequant_nvfp4_len, dequant_nvfp4_data, "main", 2, 5 * sizeof(uint32_t), {256 * 16, 1, 1}, {}, 1); - - // get_rows - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_F32 ], "get_rows_f32", get_rows_f32_len, get_rows_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_F16 ], "get_rows_f16", get_rows_f16_len, get_rows_f16_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_BF16], "get_rows_bf16", get_rows_bf16_len, get_rows_bf16_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q1_0], "get_rows_q1_0", get_rows_q1_0_len, get_rows_q1_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q2_0], "get_rows_q2_0", get_rows_q2_0_len, get_rows_q2_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_0], "get_rows_q4_0", get_rows_q4_0_len, get_rows_q4_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_1], "get_rows_q4_1", get_rows_q4_1_len, get_rows_q4_1_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_0], "get_rows_q5_0", get_rows_q5_0_len, get_rows_q5_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_1], "get_rows_q5_1", get_rows_q5_1_len, get_rows_q5_1_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q8_0], "get_rows_q8_0", get_rows_q8_0_len, get_rows_q8_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q2_K], "get_rows_q2_k", get_rows_q2_k_len, get_rows_q2_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_TQ2_0], "get_rows_tq2_0", get_rows_tq2_0_len, get_rows_tq2_0_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q3_K], "get_rows_q3_k", get_rows_q3_k_len, get_rows_q3_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q4_K], "get_rows_q4_k", get_rows_q4_k_len, get_rows_q4_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q5_K], "get_rows_q5_k", get_rows_q5_k_len, get_rows_q5_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_Q6_K], "get_rows_q6_k", get_rows_q6_k_len, get_rows_q6_k_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ1_S], "get_rows_iq1_s", get_rows_iq1_s_len, get_rows_iq1_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ1_M], "get_rows_iq1_m", get_rows_iq1_m_len, get_rows_iq1_m_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_XXS], "get_rows_iq2_xxs", get_rows_iq2_xxs_len, get_rows_iq2_xxs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_XS], "get_rows_iq2_xs", get_rows_iq2_xs_len, get_rows_iq2_xs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ2_S], "get_rows_iq2_s", get_rows_iq2_s_len, get_rows_iq2_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ3_XXS], "get_rows_iq3_xxs", get_rows_iq3_xxs_len, get_rows_iq3_xxs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ3_S], "get_rows_iq3_s", get_rows_iq3_s_len, get_rows_iq3_s_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ4_XS], "get_rows_iq4_xs", get_rows_iq4_xs_len, get_rows_iq4_xs_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_IQ4_NL], "get_rows_iq4_nl", get_rows_iq4_nl_len, get_rows_iq4_nl_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_MXFP4], "get_rows_mxfp4", get_rows_mxfp4_len, get_rows_mxfp4_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_NVFP4], "get_rows_nvfp4", get_rows_nvfp4_len, get_rows_nvfp4_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows[GGML_TYPE_I32], "get_rows_i32", get_rows_i32_len, get_rows_i32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_F32 ], "get_rows_f32_f32", get_rows_f32_f32_len, get_rows_f32_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_F16 ], "get_rows_f16_f32", get_rows_f16_f32_len, get_rows_f16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_BF16], "get_rows_bf16_f32", get_rows_bf16_f32_len, get_rows_bf16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), { 512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q1_0], "get_rows_q1_0_f32", get_rows_q1_0_f32_len, get_rows_q1_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q2_0], "get_rows_q2_0_f32", get_rows_q2_0_f32_len, get_rows_q2_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_0], "get_rows_q4_0_f32", get_rows_q4_0_f32_len, get_rows_q4_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_1], "get_rows_q4_1_f32", get_rows_q4_1_f32_len, get_rows_q4_1_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_0], "get_rows_q5_0_f32", get_rows_q5_0_f32_len, get_rows_q5_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_1], "get_rows_q5_1_f32", get_rows_q5_1_f32_len, get_rows_q5_1_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q8_0], "get_rows_q8_0_f32", get_rows_q8_0_f32_len, get_rows_q8_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q2_K], "get_rows_q2_k_f32", get_rows_q2_k_f32_len, get_rows_q2_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_TQ2_0], "get_rows_tq2_0_f32", get_rows_tq2_0_f32_len, get_rows_tq2_0_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q3_K], "get_rows_q3_k_f32", get_rows_q3_k_f32_len, get_rows_q3_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q4_K], "get_rows_q4_k_f32", get_rows_q4_k_f32_len, get_rows_q4_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q5_K], "get_rows_q5_k_f32", get_rows_q5_k_f32_len, get_rows_q5_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_Q6_K], "get_rows_q6_k_f32", get_rows_q6_k_f32_len, get_rows_q6_k_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ1_S], "get_rows_iq1_s_f32", get_rows_iq1_s_f32_len, get_rows_iq1_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ1_M], "get_rows_iq1_m_f32", get_rows_iq1_m_f32_len, get_rows_iq1_m_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_XXS], "get_rows_iq2_xxs_f32", get_rows_iq2_xxs_f32_len, get_rows_iq2_xxs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_XS], "get_rows_iq2_xs_f32", get_rows_iq2_xs_f32_len, get_rows_iq2_xs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ2_S], "get_rows_iq2_s_f32", get_rows_iq2_s_f32_len, get_rows_iq2_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ3_XXS], "get_rows_iq3_xxs_f32", get_rows_iq3_xxs_f32_len, get_rows_iq3_xxs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ3_S], "get_rows_iq3_s_f32", get_rows_iq3_s_f32_len, get_rows_iq3_s_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ4_XS], "get_rows_iq4_xs_f32", get_rows_iq4_xs_f32_len, get_rows_iq4_xs_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_IQ4_NL], "get_rows_iq4_nl_f32", get_rows_iq4_nl_f32_len, get_rows_iq4_nl_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_MXFP4], "get_rows_mxfp4_f32", get_rows_mxfp4_f32_len, get_rows_mxfp4_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_f32[GGML_TYPE_NVFP4], "get_rows_nvfp4_f32", get_rows_nvfp4_f32_len, get_rows_nvfp4_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {1024, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_get_rows_back_f32, "get_rows_back_f32", get_rows_back_f32_len, get_rows_back_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {256, 1, 1}, {}, 1, true); - - ggml_vk_create_pipeline(device, device->pipeline_matmul_split_k_reduce, "split_k_reduce", split_k_reduce_len, split_k_reduce_data, "main", 2, 2 * sizeof(uint32_t), {256 * 4, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_flash_attn_split_k_reduce, "fa_split_k_reduce", fa_split_k_reduce_len, fa_split_k_reduce_data, "main", 3, sizeof(vk_op_flash_attn_split_k_reduce_push_constants), {1, device->subgroup_size, 1}, {device->subgroup_size}, 1, true); - - for (auto &it : device->pipeline_fa_mask_opt) { - auto BrBc = it.first; - ggml_vk_create_pipeline(device, it.second, "fa_mask_opt", fa_mask_opt_len, fa_mask_opt_data, "main", 2, sizeof(vk_op_flash_attn_mask_opt_push_constants), {1, 1, 1}, {128, 128 / device->subgroup_size, BrBc.first, BrBc.second}, 1, true, true, device->subgroup_size); - } - - if (device->subgroup_clustered && device->subgroup_require_full_support) { - ggml_vk_create_pipeline(device, device->pipeline_quantize_q8_1_x4, "quantize_q8_1_x4", quantize_q8_1_x4_subgroup_len, quantize_q8_1_x4_subgroup_data, "main", 2, sizeof(vk_quantize_q8_1_push_constants), {32 * device->subgroup_size / 8, 1, 1}, { device->subgroup_size }, 1, true, true); - } else { - ggml_vk_create_pipeline(device, device->pipeline_quantize_q8_1_x4, "quantize_q8_1_x4", quantize_q8_1_x4_len, quantize_q8_1_x4_data, "main", 2, sizeof(vk_quantize_q8_1_push_constants), {32 * device->subgroup_size / 8, 1, 1}, { device->subgroup_size }, 1); - } - - for (uint32_t i = 0; i < p021_max_gqa_ratio; ++i) { - if (device->subgroup_arithmetic && device->subgroup_require_full_support) { - ggml_vk_create_pipeline2(device, device->pipeline_mul_mat_vec_p021_f16_f32[i], "mul_mat_vec_p021_f16_f32"+std::to_string(i+1), mul_mat_vec_p021_f16_f32_subgroup_add_len, mul_mat_vec_p021_f16_f32_subgroup_add_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_p021_push_constants), {1, 1, 1}, {device->subgroup_size, i + 1}, 1, true, true); + std::vector conv3d_spec_constants = { conv2d_WG_SIZE, conv2d_BS.K, conv2d_BS.CRS, conv2d_BS.NPQ, conv2d_TS_K, conv2d_SHMEM_PAD }; +#define CREATE_CONV3D(type_suffix, spv_suffix) \ + for (auto &c : device->pipeline_conv3d##type_suffix[s]) { \ + const vk_conv3d_pipeline_state &state = c.first; \ + std::vector spec_constants_cpy = conv3d_spec_constants; \ + spec_constants_cpy.push_back(state.s0); \ + spec_constants_cpy.push_back(state.s1); \ + spec_constants_cpy.push_back(state.s2); \ + spec_constants_cpy.push_back(state.p0); \ + spec_constants_cpy.push_back(state.p1); \ + spec_constants_cpy.push_back(state.p2); \ + spec_constants_cpy.push_back(state.d0); \ + spec_constants_cpy.push_back(state.d1); \ + spec_constants_cpy.push_back(state.d2); \ + spec_constants_cpy.push_back(state.KW); \ + spec_constants_cpy.push_back(state.KH); \ + spec_constants_cpy.push_back(state.KD); \ + spec_constants_cpy.push_back(state.aligned); \ + spec_constants_cpy.push_back(conv2d_csh_store); \ + spec_constants_cpy.push_back(conv2d_WM); \ + spec_constants_cpy.push_back(conv2d_WN); \ + ggml_vk_create_pipeline( \ + device, c.second, "conv3d" #type_suffix, \ + conv3d##type_suffix##spv_suffix##_len, conv3d##type_suffix##spv_suffix##_data, "main", 3, \ + sizeof(vk_op_conv3d_push_constants), wg_denoms, spec_constants_cpy, 1, true, conv2d_required_subgroup_size != 0, conv2d_required_subgroup_size); \ + } +#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (device->coopmat2) { + CREATE_CONV3D(_f32, _cm2) + CREATE_CONV3D(_f16_f32, _cm2) + } else +#endif +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + if (conv2d_use_cm1) { + CREATE_CONV3D(_f32, _cm1) + CREATE_CONV3D(_f16_f32, _cm1) + } else +#endif + if (conv2d_UNROLL) { + CREATE_CONV3D(_f32, _unroll) + CREATE_CONV3D(_f16_f32, _unroll) } else { - ggml_vk_create_pipeline2(device, device->pipeline_mul_mat_vec_p021_f16_f32[i], "mul_mat_vec_p021_f16_f32"+std::to_string(i+1), mul_mat_vec_p021_f16_f32_len, mul_mat_vec_p021_f16_f32_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_p021_push_constants), {1, 1, 1}, {device->subgroup_size, i + 1}, 1, true); + CREATE_CONV3D(_f32, ) + CREATE_CONV3D(_f16_f32, ) } +#undef CREATE_CONV3D } - ggml_vk_create_pipeline(device, device->pipeline_mul_mat_vec_nc_f16_f32, "mul_mat_vec_nc_f16_f32", mul_mat_vec_nc_f16_f32_len, mul_mat_vec_nc_f16_f32_data, "main", mul_mat_vec_num_bindings, sizeof(vk_mat_vec_nc_push_constants), {1, 1, 1}, {}, 1); - - ggml_vk_create_pipeline(device, device->pipeline_norm_f32, "norm_f32", norm_f32_len, norm_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_group_norm_f32, "group_norm_f32", group_norm_f32_len, group_norm_f32_data, "main", 2, sizeof(vk_op_push_constants), {1, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_f32, "rms_norm_f32", rms_norm_f32_len, rms_norm_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_f32, "rms_norm_mul_f32", rms_norm_f32_len, rms_norm_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_partials_f32, "rms_norm_partials_f32", rms_norm_partials_f32_len, rms_norm_partials_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 0}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_partials_f32, "rms_norm_mul_partials_f32", rms_norm_partials_f32_len, rms_norm_partials_f32_data, "main", 4, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {0, 1}, 1, true); + ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_whcn_f32, "conv2d_dw_whcn_f32", conv2d_dw_whcn_f32_len, conv2d_dw_whcn_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_cwhn_f32, "conv2d_dw_cwhn_f32", conv2d_dw_cwhn_f32_len, conv2d_dw_cwhn_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_whcn_f16_f32, "conv2d_dw_whcn_f16_f32", conv2d_dw_whcn_f16_f32_len, conv2d_dw_whcn_f16_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); + ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_cwhn_f16_f32, "conv2d_dw_cwhn_f16_f32", conv2d_dw_cwhn_f16_f32_len, conv2d_dw_cwhn_f16_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); - if (sizeof(vk_op_rms_norm_mul_rope_push_constants) <= device->properties.limits.maxPushConstantsSize) { - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_rope_f32_f32, "rms_norm_mul_rope_f32_f32", rms_norm_mul_rope_f32_f32_len, rms_norm_mul_rope_f32_f32_data, "main", 7, sizeof(vk_op_rms_norm_mul_rope_push_constants), {1, 1, 1}, {0, 1}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_mul_rope_f32_f16, "rms_norm_mul_rope_f32_f16", rms_norm_mul_rope_f32_f16_len, rms_norm_mul_rope_f32_f16_data, "main", 7, sizeof(vk_op_rms_norm_mul_rope_push_constants), {1, 1, 1}, {0, 1}, 1, true); + for (uint32_t use_push = 0; use_push < 2; ++use_push) { + for (uint32_t i = 0; i < num_topk_moe_pipelines; ++i) { + ggml_vk_create_pipeline2(device, device->pipeline_topk_moe[i][use_push], "topk_moe_f32_"+std::to_string(i), topk_moe_f32_len, topk_moe_f32_data, "main", 4, sizeof(vk_op_topk_moe_push_constants), {1, 1, 1}, {device->subgroup_size, 1u<subgroup_size); + } } - ggml_vk_create_pipeline(device, device->pipeline_rms_norm_back_f32, "rms_norm_back_f32", rms_norm_back_f32_len, rms_norm_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_l2_norm_f32, "l2_norm_f32", l2_norm_f32_len, l2_norm_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); - - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_f32, "cpy_f32_f32", cpy_f32_f32_len, cpy_f32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_f16, "cpy_f32_f16", cpy_f32_f16_len, cpy_f32_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f16_f16, "cpy_f16_f16", cpy_f16_f16_len, cpy_f16_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f16_f32, "cpy_f16_f32", cpy_f16_f32_len, cpy_f16_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_bf16,"cpy_f32_bf16",cpy_f32_bf16_len,cpy_f32_bf16_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_bf16_f32,"cpy_bf16_f32",cpy_bf16_f32_len,cpy_bf16_f32_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_i32_f32, "cpy_i32_f32", cpy_i32_f32_len, cpy_i32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_i32, "cpy_f32_i32", cpy_f32_i32_len, cpy_f32_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + // Drop compile_mutex so other threads can walk while we compile. + compile_lock.unlock(); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_f32, "contig_cpy_f32_f32", contig_cpy_f32_f32_len, contig_cpy_f32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_f16, "contig_cpy_f32_f16", contig_cpy_f32_f16_len, contig_cpy_f32_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f16_f16, "contig_cpy_f16_f16", contig_cpy_f16_f16_len, contig_cpy_f16_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f16_f32, "contig_cpy_f16_f32", contig_cpy_f16_f32_len, contig_cpy_f16_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_bf16,"contig_cpy_f32_bf16",contig_cpy_f32_bf16_len,contig_cpy_f32_bf16_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_bf16_f32,"contig_cpy_bf16_f32",contig_cpy_bf16_f32_len,contig_cpy_bf16_f32_data,"main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_i32_f32, "contig_cpy_i32_f32", contig_cpy_i32_f32_len, contig_cpy_i32_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_contig_cpy_f32_i32, "contig_cpy_f32_i32", contig_cpy_f32_i32_len, contig_cpy_f32_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + // Compile what we claimed; create_pipeline_func reacquires compile_mutex + // at the end to flip compile_pending/compiled and notify waiters. + if (has_claimed_task) { + auto & task = claimed_task; + ggml_vk_create_pipeline_func(device, task.pipeline, task.spv_size, task.spv_data, + task.entrypoint, task.parameter_count, task.wg_denoms, + task.specialization_constants, task.disable_robustness, + task.require_full_subgroups, task.required_subgroup_size); + } - ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_32, "cpy_transpose_32", cpy_transpose_32_len, cpy_transpose_32_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_transpose_16, "cpy_transpose_16", cpy_transpose_16_len, cpy_transpose_16_data, "main", 2, sizeof(vk_op_unary_push_constants), {1, 1, 1}, {}, 1); + // Another thread may be compiling the pipeline we need; block on it here. + if (wait_pipeline) { + std::unique_lock wait_lock(device->compile_mutex); + device->compile_cv.wait(wait_lock, [&] { + return wait_pipeline->compiled.load(); + }); + } +} - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q1_0], "cpy_f32_q1_0", cpy_f32_q1_0_len, cpy_f32_q1_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q2_0], "cpy_f32_q2_0", cpy_f32_q2_0_len, cpy_f32_q2_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q4_0], "cpy_f32_q4_0", cpy_f32_q4_0_len, cpy_f32_q4_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q4_1], "cpy_f32_q4_1", cpy_f32_q4_1_len, cpy_f32_q4_1_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q5_0], "cpy_f32_q5_0", cpy_f32_q5_0_len, cpy_f32_q5_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q5_1], "cpy_f32_q5_1", cpy_f32_q5_1_len, cpy_f32_q5_1_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_Q8_0], "cpy_f32_q8_0", cpy_f32_q8_0_len, cpy_f32_q8_0_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_f32_quant[GGML_TYPE_IQ4_NL], "cpy_f32_iq4_nl", cpy_f32_iq4_nl_len, cpy_f32_iq4_nl_data, "main", 2, sizeof(vk_op_unary_push_constants), {32, 1, 1}, {}, 1); +vk_device ggml_vk_get_device(size_t idx) { + VK_LOG_DEBUG("ggml_vk_get_device(" << idx << ")"); -#define SET_ROWS(src_idx, src, itype) \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_F32], "set_rows_" #src "_f32" #itype, set_rows_ ## src ## _f32 ## itype ## _len, set_rows_ ## src ## _f32 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_F16], "set_rows_" #src "_f16" #itype, set_rows_ ## src ## _f16 ## itype ## _len, set_rows_ ## src ## _f16 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_BF16], "set_rows_" #src "_bf16" #itype, set_rows_ ## src ## _bf16 ## itype ## _len, set_rows_ ## src ## _bf16 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q1_0], "set_rows_" #src "_q1_0" #itype, set_rows_ ## src ## _q1_0 ## itype ## _len, set_rows_ ## src ## _q1_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q2_0], "set_rows_" #src "_q2_0" #itype, set_rows_ ## src ## _q2_0 ## itype ## _len, set_rows_ ## src ## _q2_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q4_0], "set_rows_" #src "_q4_0" #itype, set_rows_ ## src ## _q4_0 ## itype ## _len, set_rows_ ## src ## _q4_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q4_1], "set_rows_" #src "_q4_1" #itype, set_rows_ ## src ## _q4_1 ## itype ## _len, set_rows_ ## src ## _q4_1 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q5_0], "set_rows_" #src "_q5_0" #itype, set_rows_ ## src ## _q5_0 ## itype ## _len, set_rows_ ## src ## _q5_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q5_1], "set_rows_" #src "_q5_1" #itype, set_rows_ ## src ## _q5_1 ## itype ## _len, set_rows_ ## src ## _q5_1 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_Q8_0], "set_rows_" #src "_q8_0" #itype, set_rows_ ## src ## _q8_0 ## itype ## _len, set_rows_ ## src ## _q8_0 ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_set_rows ## itype [src_idx][GGML_TYPE_IQ4_NL], "set_rows_" #src "_iq4_nl" #itype, set_rows_ ## src ## _iq4_nl ## itype ## _len, set_rows_ ## src ## _iq4_nl ## itype ## _data, "main", 3, sizeof(vk_op_binary_push_constants), {1, 1, 1}, {1}, 1, true); + if (vk_instance.devices[idx] == nullptr) { + VK_LOG_DEBUG("Initializing new vk_device"); + vk_device device = std::make_shared(); + vk_instance.devices[idx] = device; - SET_ROWS(0, f32, _i32) - SET_ROWS(0, f32, _i64) - SET_ROWS(1, f16, _i32) - SET_ROWS(1, f16, _i64) -#undef SET_ROWS + device->memory_logger = std::unique_ptr(new vk_memory_logger()); + size_t dev_num = vk_instance.device_indices[idx]; - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q1_0], "cpy_q1_0_f32", cpy_q1_0_f32_len, cpy_q1_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q1_0), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q2_0], "cpy_q2_0_f32", cpy_q2_0_f32_len, cpy_q2_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q2_0), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q4_0], "cpy_q4_0_f32", cpy_q4_0_f32_len, cpy_q4_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q4_0), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q4_1], "cpy_q4_1_f32", cpy_q4_1_f32_len, cpy_q4_1_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q4_1), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q5_0], "cpy_q5_0_f32", cpy_q5_0_f32_len, cpy_q5_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q5_0), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q5_1], "cpy_q5_1_f32", cpy_q5_1_f32_len, cpy_q5_1_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q5_1), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_Q8_0], "cpy_q8_0_f32", cpy_q8_0_f32_len, cpy_q8_0_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_Q8_0), 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_cpy_quant_f32[GGML_TYPE_IQ4_NL], "cpy_iq4_nl_f32", cpy_iq4_nl_f32_len, cpy_iq4_nl_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {(uint32_t)ggml_blck_size(GGML_TYPE_IQ4_NL), 1, 1}, {}, 1); + std::vector physical_devices = vk_instance.instance.enumeratePhysicalDevices(); - auto get_suffix = [](bool src0_f16, bool src1_f16, bool dst_f16) { - std::string s; - s += std::string(src0_f16 ? "_f16" : "_f32"); - s += std::string(src1_f16 ? "_f16" : "_f32"); - s += std::string(dst_f16 ? "_f16" : "_f32"); - return s; - }; + if (dev_num >= physical_devices.size()) { + std::cerr << "ggml_vulkan: Device with index " << dev_num << " does not exist." << std::endl; + throw std::runtime_error("Device not found"); + } -#define CREATE_BINARY(name, namemod, spec, bindings) \ - for (int s0 : {0,1}) for (int s1 : {0,1}) for (int d : {0,1}) \ - ggml_vk_create_pipeline2(device, device->pipeline_ ## name ## namemod[s0][s1][d], \ - #name + get_suffix(s0, s1, d) + #namemod, name ## _len[s0][s1][d], name ## _data[s0][s1][d], \ - "main", (bindings), sizeof(vk_op_binary_push_constants), {512, 1, 1}, spec, 1); + device->physical_device = physical_devices[dev_num]; + const std::vector ext_props = device->physical_device.enumerateDeviceExtensionProperties(); - CREATE_BINARY(add, , {0}, 4) - CREATE_BINARY(add, _norepeat, {1}, 4) - CREATE_BINARY(sub, , {0}, 3) - CREATE_BINARY(sub, _norepeat, {1}, 3) - CREATE_BINARY(mul, , {0}, 3) - CREATE_BINARY(mul, _norepeat, {1}, 3) - CREATE_BINARY(div, , {0}, 3) - CREATE_BINARY(div, _norepeat, {1}, 3) - CREATE_BINARY(add_rms, , {0}, 4) - CREATE_BINARY(add_rms, _norepeat, {1}, 4) -#undef CREATE_BINARY + device->architecture = get_device_architecture(device->physical_device); - if (device->multi_add) { - for (uint32_t i = 0; i < MAX_FUSED_ADDS; ++i) { - ggml_vk_create_pipeline2(device, device->pipeline_multi_add[i], "multi_add_f32_" + std::to_string(i+1), multi_add_f32_len, multi_add_f32_data, "main", MAX_PARAMETER_COUNT, sizeof(vk_op_multi_add_push_constants), {512, 1, 1}, {i+2}, 1); - ggml_vk_create_pipeline2(device, device->pipeline_multi_add_rms[i], "multi_add_rms_f32_" + std::to_string(i+1), multi_add_rms_f32_len, multi_add_rms_f32_data, "main", MAX_PARAMETER_COUNT, sizeof(vk_op_multi_add_push_constants), {512, 1, 1}, {i+2}, 1); - } - } + const char* GGML_VK_PREFER_HOST_MEMORY = getenv("GGML_VK_PREFER_HOST_MEMORY"); + device->prefer_host_memory = GGML_VK_PREFER_HOST_MEMORY != nullptr; - ggml_vk_create_pipeline(device, device->pipeline_add_id_f32, "add_id_f32", add_id_f32_len, add_id_f32_data, "main", 4, sizeof(vk_op_add_id_push_constants), {1, 1, 1}, {}, 1); + const char* GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM = getenv("GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM"); + device->disable_host_visible_vidmem = GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM != nullptr; - ggml_vk_create_pipeline(device, device->pipeline_acc_f32, "acc_f32", acc_f32_len, acc_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {0, 1}, 1); - ggml_vk_create_pipeline(device, device->pipeline_set_f32, "set_f32", acc_f32_len, acc_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {0, 0}, 1); + const char* GGML_VK_ALLOW_SYSMEM_FALLBACK = getenv("GGML_VK_ALLOW_SYSMEM_FALLBACK"); + device->allow_sysmem_fallback = GGML_VK_ALLOW_SYSMEM_FALLBACK != nullptr; - ggml_vk_create_pipeline(device, device->pipeline_concat_i8, "concat_i8", concat_i8_len, concat_i8_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_concat_i16, "concat_i16", concat_i16_len, concat_i16_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_concat_i32, "concat_i32", concat_i32_len, concat_i32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_concat_i64, "concat_i64", concat_i64_len, concat_i64_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + const char* GGML_VK_DISABLE_GRAPH_OPTIMIZE = getenv("GGML_VK_DISABLE_GRAPH_OPTIMIZE"); + device->disable_graph_optimize = GGML_VK_DISABLE_GRAPH_OPTIMIZE != nullptr; - ggml_vk_create_pipeline(device, device->pipeline_upscale_nearest_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_NEAREST}, 1); - ggml_vk_create_pipeline(device, device->pipeline_upscale_bilinear_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BILINEAR}, 1); - ggml_vk_create_pipeline(device, device->pipeline_upscale_bicubic_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BICUBIC}, 1); - ggml_vk_create_pipeline(device, device->pipeline_upscale_bilinear_antialias_f32, "upscale_f32", upscale_f32_len, upscale_f32_data, "main", 2, sizeof(vk_op_upscale_push_constants), {512, 1, 1}, {GGML_SCALE_MODE_BILINEAR | GGML_SCALE_FLAG_ANTIALIAS}, 1); - - ggml_vk_create_pipeline(device, device->pipeline_scale_f32, "scale_f32", scale_f32_len, scale_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + bool fp16_storage = false; + bool fp16_compute = false; + bool maintenance4_support = false; + bool sm_builtins = false; + bool amd_shader_core_properties2 = false; + bool pipeline_robustness = false; + bool coopmat2_support = false; + bool coopmat2_decode_vector_support = false; + bool pipeline_executable_properties_support = false; + bool internally_sync_support = false; + device->coopmat_support = false; + device->integer_dot_product = false; + device->shader_64b_indexing = false; + bool bfloat16_support = false; + bool dot2_f16_support = false; + bool ocp_microscaling_extension = false; + bool shader_float8_extension = false; - ggml_vk_create_pipeline(device, device->pipeline_log[0], "log_f32", log_f32_len, log_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_log[1], "log_f16", log_f16_len, log_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + for (const auto& properties : ext_props) { + if (strcmp("VK_KHR_maintenance4", properties.extensionName) == 0) { + maintenance4_support = true; + } else if (strcmp("VK_KHR_16bit_storage", properties.extensionName) == 0) { + fp16_storage = true; + } else if (strcmp("VK_KHR_shader_float16_int8", properties.extensionName) == 0) { + fp16_compute = true; + } else if (strcmp("VK_NV_shader_sm_builtins", properties.extensionName) == 0) { + sm_builtins = true; + } else if (strcmp("VK_AMD_shader_core_properties2", properties.extensionName) == 0) { + amd_shader_core_properties2 = true; + } else if (strcmp("VK_EXT_pipeline_robustness", properties.extensionName) == 0) { + pipeline_robustness = true; + } else if (strcmp("VK_EXT_subgroup_size_control", properties.extensionName) == 0) { + device->subgroup_size_control = true; +#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_cooperative_matrix", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT")) { + device->coopmat_support = true; + device->coopmat_m = 0; + device->coopmat_n = 0; + device->coopmat_k = 0; +#endif +#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + } else if (strcmp("VK_NV_cooperative_matrix2", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT2")) { + coopmat2_support = true; +#endif + } else if (strcmp(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME, properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT2_DECODE_VECTOR")) { + coopmat2_decode_vector_support = true; +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_shader_integer_dot_product", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_INTEGER_DOT_PRODUCT")) { + device->integer_dot_product = true; +#endif +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_shader_bfloat16", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_BFLOAT16")) { + bfloat16_support = true; +#endif +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) + } else if (strcmp(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME, properties.extensionName) == 0) { + ocp_microscaling_extension = true; +#endif +#if defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + } else if (strcmp(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME, properties.extensionName) == 0) { + shader_float8_extension = true; +#endif + } else if (strcmp("VK_VALVE_shader_mixed_float_dot_product", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_DOT2")) { + dot2_f16_support = true; + } else if (strcmp("VK_KHR_pipeline_executable_properties", properties.extensionName) == 0) { + pipeline_executable_properties_support = true; + } else if (strcmp("VK_EXT_memory_priority", properties.extensionName) == 0 && + getenv("GGML_VK_ENABLE_MEMORY_PRIORITY")) { + device->memory_priority = true; + } else if (strcmp("VK_EXT_external_memory_host", properties.extensionName) == 0) { + device->external_memory_host = true; +#if defined(VK_EXT_shader_64bit_indexing) + } else if (strcmp("VK_EXT_shader_64bit_indexing", properties.extensionName) == 0) { + device->shader_64b_indexing = true; +#endif + } else if (strcmp(VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME, properties.extensionName) == 0) { + internally_sync_support = true; + } else if (strcmp("VK_EXT_device_fault", properties.extensionName) == 0) { + device->device_fault = true; + } + } - ggml_vk_create_pipeline(device, device->pipeline_tri[0], "tri_f32", tri_f32_len, tri_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_tri[1], "tri_f16", tri_f16_len, tri_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + vk::PhysicalDeviceProperties2 props2; + vk::PhysicalDeviceMaintenance3Properties props3; + vk::PhysicalDeviceMaintenance4Properties props4; + vk::PhysicalDeviceSubgroupProperties subgroup_props; + vk::PhysicalDeviceDriverProperties driver_props; + vk::PhysicalDeviceShaderSMBuiltinsPropertiesNV sm_props; + vk::PhysicalDeviceShaderCoreProperties2AMD amd_shader_core_properties2_props; + vk::PhysicalDeviceVulkan11Properties vk11_props; + vk::PhysicalDeviceVulkan12Properties vk12_props; + vk::PhysicalDeviceSubgroupSizeControlPropertiesEXT subgroup_size_control_props; + vk::PhysicalDeviceShaderIntegerDotProductPropertiesKHR shader_integer_dot_product_props; + vk::PhysicalDeviceExternalMemoryHostPropertiesEXT external_memory_host_props; - ggml_vk_create_pipeline(device, device->pipeline_diag[0], "diag_f32", diag_f32_len, diag_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_diag[1], "diag_f16", diag_f16_len, diag_f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + props2.pNext = &props3; + props3.pNext = &subgroup_props; + subgroup_props.pNext = &driver_props; + driver_props.pNext = &vk11_props; + vk11_props.pNext = &vk12_props; - ggml_vk_create_pipeline(device, device->pipeline_pad_f32, "pad_f32", pad_f32_len, pad_f32_data, "main", 2, sizeof(vk_op_pad_push_constants), {512, 1, 1}, {}, 1); + VkBaseOutStructure * last_struct = (VkBaseOutStructure *)&vk12_props; - ggml_vk_create_pipeline(device, device->pipeline_roll_f32, "roll_f32", roll_f32_len, roll_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + if (maintenance4_support) { + last_struct->pNext = (VkBaseOutStructure *)&props4; + last_struct = (VkBaseOutStructure *)&props4; + } + if (sm_builtins) { + last_struct->pNext = (VkBaseOutStructure *)&sm_props; + last_struct = (VkBaseOutStructure *)&sm_props; + } + if (amd_shader_core_properties2) { + last_struct->pNext = (VkBaseOutStructure *)&amd_shader_core_properties2_props; + last_struct = (VkBaseOutStructure *)&amd_shader_core_properties2_props; + } + if (device->subgroup_size_control) { + last_struct->pNext = (VkBaseOutStructure *)&subgroup_size_control_props; + last_struct = (VkBaseOutStructure *)&subgroup_size_control_props; + } - ggml_vk_create_pipeline(device, device->pipeline_repeat_i32, "repeat_i32", repeat_i32_len, repeat_i32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_repeat_back_f32, "repeat_back_f32", repeat_back_f32_len, repeat_back_f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); +#if defined(VK_NV_cooperative_matrix2) + vk::PhysicalDeviceCooperativeMatrix2PropertiesNV coopmat2_props; + if (coopmat2_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat2_props; + last_struct = (VkBaseOutStructure *)&coopmat2_props; + } +#endif - ggml_vk_create_pipeline(device, device->pipeline_repeat_i16, "repeat_i16", repeat_i16_len, repeat_i16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + if (device->integer_dot_product) { + last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_props; + last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_props; + } -#define CREATE_UNARY(name) \ - ggml_vk_create_pipeline(device, device->pipeline_ ## name [0], #name "_f32", name ## _f32_len, name ## _f32_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); \ - ggml_vk_create_pipeline(device, device->pipeline_ ## name [1], #name "_f16", name ## _f16_len, name ## _f16_data, "main", 2, sizeof(vk_op_unary_push_constants), {512, 1, 1}, {}, 1); + if (device->external_memory_host) { + last_struct->pNext = (VkBaseOutStructure *)&external_memory_host_props; + last_struct = (VkBaseOutStructure *)&external_memory_host_props; + } - CREATE_UNARY(elu) - CREATE_UNARY(gelu) - CREATE_UNARY(gelu_erf) - CREATE_UNARY(gelu_quick) - CREATE_UNARY(silu) - CREATE_UNARY(relu) - CREATE_UNARY(sqr) - CREATE_UNARY(sqrt) - CREATE_UNARY(sin) - CREATE_UNARY(cos) - CREATE_UNARY(clamp) - CREATE_UNARY(leaky_relu) - CREATE_UNARY(xielu) - CREATE_UNARY(neg) - CREATE_UNARY(tanh) - CREATE_UNARY(sigmoid) - CREATE_UNARY(hardsigmoid) - CREATE_UNARY(hardswish) - CREATE_UNARY(abs) - CREATE_UNARY(softplus) - CREATE_UNARY(step) - CREATE_UNARY(round) - CREATE_UNARY(ceil) - CREATE_UNARY(floor) - CREATE_UNARY(trunc) - CREATE_UNARY(sgn) - CREATE_UNARY(exp) - CREATE_UNARY(expm1) -#undef CREATE_UNARY + device->physical_device.getProperties2(&props2); + device->properties = props2.properties; + device->vendor_id = device->properties.vendorID; + device->driver_id = driver_props.driverID; - ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f16, "add1_f16_f16", add1_f16_f16_len, add1_f16_f16_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_add1_f16_f32, "add1_f16_f32", add1_f16_f32_len, add1_f16_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_add1_f32_f32, "add1_f32_f32", add1_f32_f32_len, add1_f32_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {512, 1, 1}, {}, 1); + if (device->driver_id == vk::DriverId::eMoltenvk) { + // Disable external_memory_host until https://github.com/KhronosGroup/MoltenVK/pull/2622 + // is available in the Vulkan SDK. + device->external_memory_host = false; + } - ggml_vk_create_pipeline(device, device->pipeline_arange_f32, "arange_f32", arange_f32_len, arange_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + // Implementing the async backend interfaces seems broken on older Intel HW, + // see https://github.com/ggml-org/llama.cpp/issues/17302. + device->support_async = (device->vendor_id != VK_VENDOR_ID_INTEL || + std::string(device->properties.deviceName.data()).find("(DG1)") == std::string::npos) && + getenv("GGML_VK_DISABLE_ASYNC") == nullptr; - ggml_vk_create_pipeline(device, device->pipeline_fill_f32, "fill_f32", fill_f32_len, fill_f32_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_fill_f16, "fill_f16", fill_f16_len, fill_f16_data, "main", 1, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + if (!device->support_async) { + GGML_LOG_DEBUG("ggml_vulkan: WARNING: Async execution disabled on certain Intel devices.\n"); + } -#define CREATE_GLU(name) \ - ggml_vk_create_pipeline(device, device->pipeline_ ## name [0], #name "_f32", name ## _f32_len, name ## _f32_data, "main", 3, sizeof(vk_op_glu_push_constants), {512, 1, 1}, {}, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_ ## name [1], #name "_f16", name ## _f16_len, name ## _f16_data, "main", 3, sizeof(vk_op_glu_push_constants), {512, 1, 1}, {}, 1, true); + const char* GGML_VK_FORCE_MAX_ALLOCATION_SIZE = getenv("GGML_VK_FORCE_MAX_ALLOCATION_SIZE"); - CREATE_GLU(geglu) - CREATE_GLU(reglu) - CREATE_GLU(swiglu) - CREATE_GLU(swiglu_oai) - CREATE_GLU(geglu_erf) - CREATE_GLU(geglu_quick) -#undef CREATE_GLU + if (GGML_VK_FORCE_MAX_ALLOCATION_SIZE != nullptr) { + device->max_memory_allocation_size = std::stoull(GGML_VK_FORCE_MAX_ALLOCATION_SIZE); + } else if (maintenance4_support) { + device->max_memory_allocation_size = std::min(props3.maxMemoryAllocationSize, props4.maxBufferSize); + } else { + device->max_memory_allocation_size = props3.maxMemoryAllocationSize; + } - ggml_vk_create_pipeline(device, device->pipeline_silu_back_f32, "silu_back_f32", silu_back_f32_len, silu_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + const char* GGML_VK_FORCE_MAX_BUFFER_SIZE = getenv("GGML_VK_FORCE_MAX_BUFFER_SIZE"); - ggml_vk_create_pipeline(device, device->pipeline_diag_mask_inf_f32, "diag_mask_inf_f32", diag_mask_inf_f32_len, diag_mask_inf_f32_data, "main", 2, sizeof(vk_op_diag_mask_push_constants), {1, 512, 1}, {}, 1, true); + if (GGML_VK_FORCE_MAX_BUFFER_SIZE != nullptr) { + device->max_buffer_size = std::stoull(GGML_VK_FORCE_MAX_BUFFER_SIZE); + } else if (maintenance4_support) { + device->max_buffer_size = props4.maxBufferSize; + } else { + device->max_buffer_size = device->max_memory_allocation_size; + } - ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32, "soft_max_f32", soft_max_f32_len, soft_max_f32_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_wg512, "soft_max_f32_wg512", soft_max_f32_len, soft_max_f32_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 512 }, 1); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_f16, "soft_max_f32_f16", soft_max_f32_f16_len, soft_max_f32_f16_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_f32_f16_wg512, "soft_max_f32_f16_wg512", soft_max_f32_f16_len, soft_max_f32_f16_data, "main", 4, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 512 }, 1); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_back_f32, "soft_max_back_f32", soft_max_back_f32_len, soft_max_back_f32_data, "main", 3, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1, true); + const char* GGML_VK_SUBALLOCATION_BLOCK_SIZE = getenv("GGML_VK_SUBALLOCATION_BLOCK_SIZE"); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large1_f32, "soft_max_large1_f32", soft_max_large1_f32_len, soft_max_large1_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large2_f32, "soft_max_large2_f32", soft_max_large2_f32_len, soft_max_large2_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large3_f32, "soft_max_large3_f32", soft_max_large3_f32_len, soft_max_large3_f32_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large1_f32_f16, "soft_max_large1_f32_f16", soft_max_large1_f32_f16_len, soft_max_large1_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large2_f32_f16, "soft_max_large2_f32_f16", soft_max_large2_f32_f16_len, soft_max_large2_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_soft_max_large3_f32_f16, "soft_max_large3_f32_f16", soft_max_large3_f32_f16_len, soft_max_large3_f32_f16_data, "main", 6, sizeof(vk_op_soft_max_push_constants), {1, 1, 1}, { 128, 4 }, 1, true); + if (GGML_VK_SUBALLOCATION_BLOCK_SIZE != nullptr) { + device->suballocation_block_size = std::stoull(GGML_VK_SUBALLOCATION_BLOCK_SIZE); + } else { + // Limit batching of allocations to 1GB by default to avoid fragmentation issues + device->suballocation_block_size = 1024*1024*1024; + } + device->suballocation_block_size = std::min(device->suballocation_block_size, device->max_memory_allocation_size); - ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f32, "rope_norm_f32", rope_norm_f32_len, rope_norm_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f32, "rope_neox_f32", rope_neox_f32_len, rope_neox_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f32, "rope_multi_f32", rope_multi_f32_len, rope_multi_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_vision_f32, "rope_vision_f32", rope_vision_f32_len, rope_vision_f32_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + device->subgroup_size = subgroup_props.subgroupSize; + device->subgroup_size_log2 = uint32_t(log2f(float(device->subgroup_size))); + device->uma = device->properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu; + if (sm_builtins) { + device->shader_core_count = sm_props.shaderSMCount; + } else if (amd_shader_core_properties2) { + device->shader_core_count = amd_shader_core_properties2_props.activeComputeUnitCount; + } else if (device->vendor_id == VK_VENDOR_ID_INTEL) { + device->shader_core_count = ggml_vk_intel_shader_core_count(device->physical_device); + } else { + device->shader_core_count = 0; + } + device->float_controls_rte_fp16 = vk12_props.shaderRoundingModeRTEFloat16; + device->float_controls_denorm_preserve_fp16 = vk12_props.shaderDenormPreserveFloat16; - ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f16, "rope_norm_f16", rope_norm_f16_len, rope_norm_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f16, "rope_neox_f16", rope_neox_f16_len, rope_neox_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f16, "rope_multi_f16", rope_multi_f16_len, rope_multi_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_vision_f16, "rope_vision_f16", rope_vision_f16_len, rope_vision_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + device->subgroup_basic = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eBasic); + device->subgroup_arithmetic = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eArithmetic); +#ifdef __APPLE__ + // Workaround for subgroup arithmetic failing on MoltenVK with AMD GPUs (issue 15846) + if (device->vendor_id == VK_VENDOR_ID_AMD) { + device->subgroup_arithmetic = false; + } +#endif + device->subgroup_shuffle = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eShuffle); +#ifdef __APPLE__ + if (device->vendor_id == VK_VENDOR_ID_AMD) { + device->subgroup_shuffle = false; + } +#endif + device->subgroup_clustered = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eClustered); - ggml_vk_create_pipeline(device, device->pipeline_rope_norm_f32_f16, "rope_norm_f32_f16", rope_norm_f32_f16_len, rope_norm_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_neox_f32_f16, "rope_neox_f32_f16", rope_neox_f32_f16_len, rope_neox_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_rope_multi_f32_f16, "rope_multi_f32_f16", rope_multi_f32_f16_len, rope_multi_f32_f16_data, "main", 5, sizeof(vk_op_rope_push_constants), {1, 512, 1}, {}, 1); + device->subgroup_ballot = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eBallot); - for (uint32_t i = 0; i < num_argsort_pipelines; ++i) { - uint32_t BLOCK_SIZE = 1u << std::min(i, device->max_workgroup_size_log2); - if (i <= device->max_workgroup_size_log2 && - 2 * sizeof(int) * BLOCK_SIZE <= device->properties.limits.maxComputeSharedMemorySize) { - const uint32_t NCOLS_PADDED_LOG2 = i; - ggml_vk_create_pipeline2(device, device->pipeline_argsort_f32[i], "argsort_f32_"+std::to_string(i), argsort_f32_len, argsort_f32_data, "main", 3, sizeof(vk_op_argsort_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, NCOLS_PADDED_LOG2}, 1, true); - } - const uint32_t WG_UNROLL_FACTOR = BLOCK_SIZE > 1 ? 2 : 1; - BLOCK_SIZE /= WG_UNROLL_FACTOR; - ggml_vk_create_pipeline2(device, device->pipeline_argsort_large_f32[i], "argsort_large_f32_"+std::to_string(i), argsort_large_f32_len, argsort_large_f32_data, "main", 3, sizeof(vk_op_argsort_push_constants), {BLOCK_SIZE * WG_UNROLL_FACTOR, 1, 1}, {BLOCK_SIZE, WG_UNROLL_FACTOR}, 1, true); - } + device->subgroup_vote = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && + (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eVote); - for (uint32_t i = 0; i < num_topk_pipelines; ++i) { - const uint32_t BLOCK_SIZE = 1u << i; - const uint32_t NCOLS_PADDED_LOG2 = i; - if (i <= device->max_workgroup_size_log2) { - uint32_t nary_shmem = 2 * sizeof(int) * BLOCK_SIZE + - sizeof(int) * device->subgroup_size + - 2 * sizeof(int) + - 2 * (BLOCK_SIZE / device->subgroup_size) * sizeof(int); - if (device->subgroup_arithmetic && device->subgroup_require_full_support && device->subgroup_shuffle && device->subgroup_ballot && - nary_shmem <= device->properties.limits.maxComputeSharedMemorySize) { - ggml_vk_create_pipeline2(device, device->pipeline_topk_f32[i], "topk_f32_"+std::to_string(i), topk_nary_search_f32_len, topk_nary_search_f32_data, "main", 2, sizeof(vk_op_topk_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, device->subgroup_size, device->subgroup_size_log2}, 1, true, true, device->subgroup_size); - } else if (2 * sizeof(int) * BLOCK_SIZE <= device->properties.limits.maxComputeSharedMemorySize) { - ggml_vk_create_pipeline2(device, device->pipeline_topk_f32[i], "topk_f32_"+std::to_string(i), topk_argsort_f32_len, topk_argsort_f32_data, "main", 2, sizeof(vk_op_topk_push_constants), {BLOCK_SIZE, 1, 1}, {BLOCK_SIZE, NCOLS_PADDED_LOG2}, 1, true); - } + // Submit at least every 100 nodes, in case there are workloads without as much matmul. + device->max_nodes_per_submit = 100; + const char* GGML_VK_MAX_NODES_PER_SUBMIT = getenv("GGML_VK_MAX_NODES_PER_SUBMIT"); + if (GGML_VK_MAX_NODES_PER_SUBMIT != nullptr) { + uint32_t max_nodes_per_submit = std::stoul(GGML_VK_MAX_NODES_PER_SUBMIT); + device->max_nodes_per_submit = std::max(max_nodes_per_submit, 1u); } - } - ggml_vk_create_pipeline(device, device->pipeline_argmax_f32, "argmax_f32", argmax_f32_len, argmax_f32_data, "main", 2, sizeof(vk_op_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); + const bool force_disable_f16 = getenv("GGML_VK_DISABLE_F16") != nullptr; - ggml_vk_create_pipeline(device, device->pipeline_sum_rows_f32, "sum_rows_f32", sum_rows_f32_len, sum_rows_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { device->subgroup_size }, 1); - // Intel Windows driver in range [32.0.101.8509, 32.0.101.8860) will crash when using fwht kernels so we gate that here - const bool can_use_fwht = device->driver_id != vk::DriverId::eIntelProprietaryWindows || - !ggml_vk_intel_windows_driver_in_range(device->properties.driverVersion, 101, 8509, 101, 8860); - if (can_use_fwht && device->subgroup_basic && device->subgroup_shuffle) { - int idx = 0; - for (uint32_t n : {64, 128, 256, 512}) { - if (device->subgroup_size <= n) { - ggml_vk_create_pipeline(device, device->pipeline_fwht_f32[idx], "fwht_f32", fwht_f32_len, fwht_f32_data, "main", 2, sizeof(vk_op_fwht_push_constants), {1, 1, 1}, { device->subgroup_size, n }, 1, true, true, device->subgroup_size); - } - ++idx; - } - } else if (can_use_fwht) { - int idx = 0; - for (uint32_t n : {64, 128, 256, 512}) { - const uint32_t block_size = std::min(device->subgroup_size, n); - ggml_vk_create_pipeline(device, device->pipeline_fwht_f32[idx], "fwht_shmem_f32", fwht_shmem_f32_len, fwht_shmem_f32_data, "main", 2, sizeof(vk_op_fwht_push_constants), {1, 1, 1}, { block_size, n }, 1); - ++idx; + device->fp16 = !force_disable_f16 && fp16_storage && fp16_compute; + + if (!ggml_vk_khr_cooperative_matrix_support(device->properties, driver_props, device->architecture)) { + device->coopmat_support = false; } - } - const uint32_t cumsum_elem_per_thread = (device->vendor_id == VK_VENDOR_ID_AMD || device->vendor_id == VK_VENDOR_ID_INTEL) ? 2 : 4; - ggml_vk_create_pipeline(device, device->pipeline_cumsum_f32, "cumsum_f32", cumsum_f32_len, cumsum_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { 256, device->subgroup_size, cumsum_elem_per_thread }, 1, true, true, device->subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_cumsum_small_f32, "cumsum_f32", cumsum_f32_len, cumsum_f32_data, "main", 2, sizeof(vk_op_sum_rows_push_constants), {1, 1, 1}, { 128, device->subgroup_size, 1 }, 1, true, true, device->subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_cumsum_multipass1_f32, "cumsum_multipass1_f32", cumsum_multipass1_f32_len, cumsum_multipass1_f32_data, "main", 3, sizeof(vk_op_sum_rows_push_constants), {256, 1, 1}, { 256, device->subgroup_size }, 1, true, true, device->subgroup_size); - ggml_vk_create_pipeline(device, device->pipeline_cumsum_multipass2_f32, "cumsum_multipass2_f32", cumsum_multipass2_f32_len, cumsum_multipass2_f32_data, "main", 3, sizeof(vk_op_sum_rows_push_constants), {256, 1, 1}, { 256, device->subgroup_size }, 1, true, true, device->subgroup_size); + device->integer_dot_product = device->integer_dot_product && shader_integer_dot_product_props.integerDotProduct4x8BitPackedSignedAccelerated; - ggml_vk_create_pipeline(device, device->pipeline_count_equal_i32, "count_equal_i32", count_equal_i32_len, count_equal_i32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, { device->subgroup_size }, 1); + device->min_imported_host_pointer_alignment = external_memory_host_props.minImportedHostPointerAlignment; - ggml_vk_create_pipeline(device, device->pipeline_count_experts, "count_experts", count_experts_len, count_experts_data, "main", 2, sizeof(vk_op_count_experts_push_constants), {1, 1, 1}, {}, 1, true); + device->max_workgroup_size_log2 = uint32_t(log2f(float(device->properties.limits.maxComputeWorkGroupInvocations))); - for (auto &s : device->pipeline_solve_tri_f32) { - const vk_solve_tri_pipeline_state &state = s.first; + std::vector queue_family_props = device->physical_device.getQueueFamilyProperties(); - // Max number of rows to load at a time, limited by shared memory - const uint32_t batch_N = device->properties.limits.maxComputeSharedMemorySize / ((state.N + state.K) * sizeof(float)); - // Need at least K invocations, and prefer a minimum of 128 to spread out loading shared memory - const uint32_t block_size = std::max(128u, 1u << (uint32_t)ceilf(log2f(float(state.K)))); + // Try to find a non-graphics compute queue and transfer-focused queues + // Allow overriding avoiding the graphics queue because it can increase performance on RADV + const bool allow_graphics_queue = (getenv("GGML_VK_ALLOW_GRAPHICS_QUEUE") != nullptr); + const vk::QueueFlagBits graphics_flag = allow_graphics_queue ? (vk::QueueFlagBits)0 : vk::QueueFlagBits::eGraphics; + const uint32_t compute_queue_family_index = ggml_vk_find_queue_family_index(queue_family_props, vk::QueueFlagBits::eCompute, graphics_flag, -1, 1); + const uint32_t transfer_queue_family_index = ggml_vk_find_queue_family_index(queue_family_props, vk::QueueFlagBits::eTransfer, vk::QueueFlagBits::eCompute | graphics_flag, compute_queue_family_index, 1); - ggml_vk_create_pipeline( - device, s.second, "solve_tri_f32", - solve_tri_f32_len, solve_tri_f32_data, "main", 3, - sizeof(vk_op_binary_push_constants), {1, 1, 1}, { 0, state.N, state.K, batch_N, block_size }, 1, true); - } + const float priorities[] = { 1.0f, 1.0f }; + device->single_queue = compute_queue_family_index == transfer_queue_family_index && queue_family_props[compute_queue_family_index].queueCount == 1; -#define IM2COL(bda) \ - ggml_vk_create_pipeline(device, device->pipeline_im2col_f32, "im2col_f32", im2col_f32 ## bda ## _len, im2col_f32 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_push_constants), {512, 1, 1}, { device->subgroup_size }, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_im2col_3d_f32, "im2col_3d_f32", im2col_3d_f32 ## bda ## _len, im2col_3d_f32 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_3d_push_constants), {512, 1, 1}, { 512 }, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_im2col_f32_f16, "im2col_f32_f16", im2col_f32_f16 ## bda ## _len, im2col_f32_f16 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_push_constants), {512, 1, 1}, { device->subgroup_size }, 1, true); \ - ggml_vk_create_pipeline(device, device->pipeline_im2col_3d_f32_f16, "im2col_3d_f32_f16", im2col_3d_f32_f16 ## bda ## _len, im2col_3d_f32_f16 ## bda ## _data, "main", 2, sizeof(vk_op_im2col_3d_push_constants), {512, 1, 1}, { 512 }, 1, true); - if (device->shader_int64 && device->buffer_device_address) { - IM2COL(_bda) - } else { - IM2COL() - } + std::vector device_queue_create_infos; + vk::DeviceCreateInfo device_create_info{}; + std::vector device_extensions; + vk::PhysicalDeviceFeatures device_features = device->physical_device.getFeatures(); - ggml_vk_create_pipeline(device, device->pipeline_timestep_embedding_f32, "timestep_embedding_f32", timestep_embedding_f32_len, timestep_embedding_f32_data, "main", 2, sizeof(vk_op_timestep_embedding_push_constants), {256, 1, 1}, {}, 1); + VkPhysicalDeviceFeatures2 device_features2; + device_features2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2; + device_features2.pNext = nullptr; + device_features2.features = (VkPhysicalDeviceFeatures)device_features; - ggml_vk_create_pipeline(device, device->pipeline_conv_transpose_1d_f32, "conv_transpose_1d_f32", conv_transpose_1d_f32_len, conv_transpose_1d_f32_data, "main", 3, sizeof(vk_op_conv_transpose_1d_push_constants), {1, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_f32, "col2im_1d_f32", col2im_1d_f32_len, col2im_1d_f32_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_f16, "col2im_1d_f16", col2im_1d_f16_len, col2im_1d_f16_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); - ggml_vk_create_pipeline(device, device->pipeline_col2im_1d_bf16, "col2im_1d_bf16", col2im_1d_bf16_len, col2im_1d_bf16_data, "main", 2, sizeof(vk_op_col2im_1d_push_constants), {256, 1, 1}, {}, 1, true); + VkPhysicalDeviceVulkan11Features vk11_features; + vk11_features.pNext = nullptr; + vk11_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES; + device_features2.pNext = &vk11_features; - ggml_vk_create_pipeline(device, device->pipeline_out_prod_f32, "out_prod_f32", out_prod_f32_len, out_prod_f32_data, "main", 3, sizeof(vk_op_binary_push_constants), {256, 1, 1}, {}, 1); + VkPhysicalDeviceVulkan12Features vk12_features; + vk12_features.pNext = nullptr; + vk12_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES; + vk11_features.pNext = &vk12_features; - ggml_vk_create_pipeline(device, device->pipeline_snake_f32, "snake_f32", snake_f32_len, snake_f32_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_snake_f16, "snake_f16", snake_f16_len, snake_f16_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_snake_bf16, "snake_bf16", snake_bf16_len, snake_bf16_data, "main", 4, sizeof(vk_op_snake_push_constants), {256, 1, 1}, {}, 1); + last_struct = (VkBaseOutStructure *)&vk12_features; - ggml_vk_create_pipeline(device, device->pipeline_pool1d_f32, "pool1d_f32", pool1d_f32_len, pool1d_f32_data, "main", 2, sizeof(vk_op_pool1d_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_pool2d_f32, "pool2d_f32", pool2d_f32_len, pool2d_f32_data, "main", 2, sizeof(vk_op_pool2d_push_constants), {512, 1, 1}, {}, 1); + VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR internally_synchronized_queues_features{}; + internally_synchronized_queues_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INTERNALLY_SYNCHRONIZED_QUEUES_FEATURES_KHR; + internally_synchronized_queues_features.pNext = nullptr; + internally_synchronized_queues_features.internallySynchronizedQueues = VK_FALSE; - ggml_vk_create_pipeline(device, device->pipeline_rwkv_wkv6_f32, "rwkv_wkv6_f32", rwkv_wkv6_f32_len, rwkv_wkv6_f32_data, "main", 7, sizeof(vk_op_rwkv_wkv6_push_constants), {1, 1, 1}, {device->subgroup_size}, 1); + if (internally_sync_support) { + last_struct->pNext = (VkBaseOutStructure *)&internally_synchronized_queues_features; + last_struct = (VkBaseOutStructure *)&internally_synchronized_queues_features; + device_extensions.push_back(VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME); + } - ggml_vk_create_pipeline(device, device->pipeline_rwkv_wkv7_f32, "rwkv_wkv7_f32", rwkv_wkv7_f32_len, rwkv_wkv7_f32_data, "main", 8, sizeof(vk_op_rwkv_wkv7_push_constants), {1, 1, 1}, {device->subgroup_size}, 1); + VkPhysicalDevicePipelineRobustnessFeaturesEXT pl_robustness_features; + pl_robustness_features.pNext = nullptr; + pl_robustness_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PIPELINE_ROBUSTNESS_FEATURES_EXT; + pl_robustness_features.pipelineRobustness = VK_FALSE; - ggml_vk_create_pipeline(device, device->pipeline_gated_linear_attn_f32, "gated_linear_attn_f32", gated_linear_attn_f32_len, gated_linear_attn_f32_data, "main", 6, sizeof(vk_op_gated_linear_attn_push_constants), {1, 1, 1}, {}, 1); + if (pipeline_robustness) { + last_struct->pNext = (VkBaseOutStructure *)&pl_robustness_features; + last_struct = (VkBaseOutStructure *)&pl_robustness_features; + device_extensions.push_back("VK_EXT_pipeline_robustness"); + } - { - const uint32_t gdn_sizes[] = {16, 32, 64, 128}; - const char * gdn_names[][2] = { - {"gated_delta_net_f32_d16", "gated_delta_net_f32_d16_kda"}, - {"gated_delta_net_f32_d32", "gated_delta_net_f32_d32_kda"}, - {"gated_delta_net_f32_d64", "gated_delta_net_f32_d64_kda"}, - {"gated_delta_net_f32_d128", "gated_delta_net_f32_d128_kda"}, - }; - for (uint32_t si = 0; si < 4; si++) { - const uint32_t S_V = gdn_sizes[si]; - GGML_ASSERT(is_pow2(S_V)); + VkPhysicalDeviceMemoryPriorityFeaturesEXT memory_priority_features; + memory_priority_features.pNext = nullptr; + memory_priority_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MEMORY_PRIORITY_FEATURES_EXT; + memory_priority_features.memoryPriority = VK_FALSE; + if (device->memory_priority) { + last_struct->pNext = (VkBaseOutStructure *)&memory_priority_features; + last_struct = (VkBaseOutStructure *)&memory_priority_features; + device_extensions.push_back("VK_EXT_memory_priority"); + } - uint32_t lanes_per_column; - if (S_V >= 128u && device->subgroup_clustered) { - lanes_per_column = 8u; - } else { - // Use largest power-of-two that divides both S_V and subgroup_size so that - // (1) S_V % lanes_per_column == 0 and (2) S_V % (subgroup_size / lanes_per_column) == 0. - // This means we don't need extra bounds checking logic in the shader. - lanes_per_column = std::min(S_V, device->subgroup_size); - } + VkPhysicalDeviceSubgroupSizeControlFeaturesEXT subgroup_size_control_features; + subgroup_size_control_features.pNext = nullptr; + subgroup_size_control_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SUBGROUP_SIZE_CONTROL_FEATURES_EXT; + subgroup_size_control_features.computeFullSubgroups = false; + subgroup_size_control_features.subgroupSizeControl = false; - // gated_delta_net.comp relies on S_V % COLS_PER_WG == 0 and - // S_V % LANES_PER_COLUMN == 0 to avoid bounds checks. - while (lanes_per_column > 1u) { - const bool valid_lanes = (device->subgroup_size % lanes_per_column) == 0 && - (S_V % lanes_per_column) == 0; - const uint32_t cols_per_wg = valid_lanes ? device->subgroup_size / lanes_per_column : 0; - if (valid_lanes && cols_per_wg > 0 && (S_V % cols_per_wg) == 0) { - break; - } - lanes_per_column >>= 1u; - } + if (device->subgroup_size_control) { + last_struct->pNext = (VkBaseOutStructure *)&subgroup_size_control_features; + last_struct = (VkBaseOutStructure *)&subgroup_size_control_features; + } - GGML_ASSERT((device->subgroup_size % lanes_per_column) == 0); - GGML_ASSERT((S_V % lanes_per_column) == 0); - GGML_ASSERT((S_V % (device->subgroup_size / lanes_per_column)) == 0); +#if defined(VK_KHR_cooperative_matrix) + VkPhysicalDeviceCooperativeMatrixFeaturesKHR coopmat_features; + coopmat_features.pNext = nullptr; + coopmat_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_FEATURES_KHR; + coopmat_features.cooperativeMatrix = VK_FALSE; - const bool need_partial_subgroup_reduce = lanes_per_column != 1u && lanes_per_column < device->subgroup_size; - const bool use_clustered_reduce = device->subgroup_arithmetic && device->subgroup_clustered && need_partial_subgroup_reduce; - const bool use_subgroup_reduce = device->subgroup_arithmetic && !need_partial_subgroup_reduce; - const bool use_subgroup_ops = use_clustered_reduce || use_subgroup_reduce; - size_t gdn_len; - const void * gdn_data; - if (use_clustered_reduce) { - gdn_len = gated_delta_net_f32_len; - gdn_data = (const void *)gated_delta_net_f32_data; - } else if (use_subgroup_reduce) { - gdn_len = gated_delta_net_f32_nocluster_len; - gdn_data = (const void *)gated_delta_net_f32_nocluster_data; - } else { - gdn_len = gated_delta_net_f32_shmem_len; - gdn_data = (const void *)gated_delta_net_f32_shmem_data; - } - - const uint32_t cols_per_wg = device->subgroup_size / lanes_per_column; - const std::array wg_denoms = {1u, 1u, cols_per_wg}; + if (device->coopmat_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat_features; + last_struct = (VkBaseOutStructure *)&coopmat_features; + } +#endif - for (uint32_t kda = 0; kda < 2; kda++) { - ggml_vk_create_pipeline(device, device->pipeline_gated_delta_net[si][kda], - gdn_names[si][kda], gdn_len, gdn_data, "main", 7, sizeof(vk_op_gated_delta_net_push_constants), - wg_denoms, {S_V, kda, device->subgroup_size, lanes_per_column}, 1, true, use_subgroup_ops, device->subgroup_size); - } +#if defined(VK_NV_cooperative_matrix2) + VkPhysicalDeviceCooperativeMatrix2FeaturesNV coopmat2_features {}; + coopmat2_features.pNext = nullptr; + coopmat2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_2_FEATURES_NV; + if (coopmat2_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat2_features; + last_struct = (VkBaseOutStructure *)&coopmat2_features; + device_extensions.push_back("VK_NV_cooperative_matrix2"); } - } +#endif - if (device->subgroup_arithmetic && device->subgroup_require_full_support) { - ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d128, "ssm_scan_128_f32", ssm_scan_subgroup_f32_len, ssm_scan_subgroup_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {128, device->subgroup_size}, 1, true, true); - ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d256, "ssm_scan_256_f32", ssm_scan_subgroup_f32_len, ssm_scan_subgroup_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {256, device->subgroup_size}, 1, true, true); - } else { - ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d128, "ssm_scan_128_f32", ssm_scan_f32_len, ssm_scan_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {128, device->subgroup_size, 16}, 1, true, true); - ggml_vk_create_pipeline(device, device->pipeline_ssm_scan_f32_d256, "ssm_scan_256_f32", ssm_scan_f32_len, ssm_scan_f32_data, "main", 8, sizeof(vk_op_ssm_scan_push_constants), {1, 1, 1}, {256, device->subgroup_size, 16}, 1, true, true); - } + VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV coopmat2_decode_vector_features {}; + coopmat2_decode_vector_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV; + if (coopmat2_decode_vector_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat2_decode_vector_features; + last_struct = (VkBaseOutStructure *)&coopmat2_decode_vector_features; + device_extensions.push_back(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME); + } - ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_f32, "ssm_conv_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 0, 0}, 1); - ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_silu_f32, "ssm_conv_silu_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 0, 1}, 1); - ggml_vk_create_pipeline(device, device->pipeline_ssm_conv_bias_silu_f32, "ssm_conv_bias_silu_f32", ssm_conv_f32_len, ssm_conv_f32_data, "main", 4, sizeof(vk_op_ssm_conv_push_constants), {32, 16, 1}, {32, 16, 1, 1}, 1); +#if defined(VK_KHR_shader_bfloat16) + VkPhysicalDeviceShaderBfloat16FeaturesKHR bfloat16_features {}; + bfloat16_features.pNext = nullptr; + bfloat16_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR; + if (bfloat16_support) { + last_struct->pNext = (VkBaseOutStructure *)&bfloat16_features; + last_struct = (VkBaseOutStructure *)&bfloat16_features; + device_extensions.push_back("VK_KHR_shader_bfloat16"); + } +#endif - ggml_vk_create_pipeline(device, device->pipeline_opt_step_adamw_f32, "opt_step_adamw_f32", opt_step_adamw_f32_len, opt_step_adamw_f32_data, "main", 5, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT ocp_microscaling_features {}; + ocp_microscaling_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT; + if (ocp_microscaling_extension) { + last_struct->pNext = (VkBaseOutStructure *)&ocp_microscaling_features; + last_struct = (VkBaseOutStructure *)&ocp_microscaling_features; + device_extensions.push_back(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME); + } - ggml_vk_create_pipeline(device, device->pipeline_opt_step_sgd_f32, "opt_step_sgd_f32", opt_step_sgd_f32_len, opt_step_sgd_f32_data, "main", 3, sizeof(vk_op_push_constants), {512, 1, 1}, {}, 1); + VkPhysicalDeviceShaderFloat8FeaturesEXT shader_float8_features {}; + shader_float8_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT; + if (shader_float8_extension) { + last_struct->pNext = (VkBaseOutStructure *)&shader_float8_features; + last_struct = (VkBaseOutStructure *)&shader_float8_features; + device_extensions.push_back(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME); + } - // conv2d, conv_transpose_2d, conv3d - for (uint32_t s = 0; s < CONV_SHAPE_COUNT; ++s) { - // smaller WG for the small-tile fallback gives more concurrent WGs per SM - uint32_t conv2d_WG_SIZE = (s == CONV_SHAPE_64x32) ? 128 : 256; - uint32_t use_collectives = 0; // Enables subgroup ops for preventing the re-calculation of indices. - uint32_t conv2d_TS_K = (s == CONV_SHAPE_64x32) ? 4 : 8; - uint32_t conv2d_SHMEM_PAD = 4; - vk_conv_block_size conv2d_BS = vk_conv_block_sizes[s]; - bool conv2d_UNROLL = true; + VkPhysicalDeviceMaintenance4Features maint4_features {}; + maint4_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MAINTENANCE_4_FEATURES; + if (maintenance4_support) { + last_struct->pNext = (VkBaseOutStructure *)&maint4_features; + last_struct = (VkBaseOutStructure *)&maint4_features; + device_extensions.push_back("VK_KHR_maintenance4"); + } -#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (device->coopmat2) { - conv2d_SHMEM_PAD = 8; // 8 float16_t + VkPhysicalDeviceShaderIntegerDotProductFeaturesKHR shader_integer_dot_product_features {}; + shader_integer_dot_product_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_INTEGER_DOT_PRODUCT_FEATURES_KHR; + if (device->integer_dot_product) { + last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_features; + last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_features; + device_extensions.push_back("VK_KHR_shader_integer_dot_product"); } -#endif - if (device->vendor_id == VK_VENDOR_ID_INTEL) { - conv2d_SHMEM_PAD = 0; - conv2d_UNROLL = false; - } else if (device->vendor_id == VK_VENDOR_ID_AMD) { - conv2d_SHMEM_PAD = device->architecture == vk_device_architecture::AMD_GCN ? 1 : 4; - if (s == CONV_SHAPE_128x128 && device->architecture != vk_device_architecture::AMD_GCN) { - conv2d_UNROLL = false; - } + VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE dot2_features {}; + dot2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE; + if (dot2_f16_support) { + last_struct->pNext = (VkBaseOutStructure *)&dot2_features; + last_struct = (VkBaseOutStructure *)&dot2_features; + device_extensions.push_back("VK_VALVE_shader_mixed_float_dot_product"); } - // Use collectives on pre-Turing NVIDIA GPUs and GCN AMD cards, which had slower integer math. - bool allow_collectives_nv = device->vendor_id != VK_VENDOR_ID_NVIDIA || - device->architecture == vk_device_architecture::NVIDIA_PRE_TURING; - bool allow_collectives_amd = device->vendor_id != VK_VENDOR_ID_AMD || - device->architecture == vk_device_architecture::AMD_GCN; + VkPhysicalDevicePipelineExecutablePropertiesFeaturesKHR pep_features {}; + pep_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PIPELINE_EXECUTABLE_PROPERTIES_FEATURES_KHR; + if (pipeline_executable_properties_support) { + last_struct->pNext = (VkBaseOutStructure *)&pep_features; + last_struct = (VkBaseOutStructure *)&pep_features; + device_extensions.push_back("VK_KHR_pipeline_executable_properties"); + } - if (device->subgroup_shuffle && - device->vendor_id != VK_VENDOR_ID_INTEL && // Do not enable collectives on Intel, see PR 14316. - allow_collectives_nv && - allow_collectives_amd) { - use_collectives = 1; - conv2d_BS.CRS = std::min( - device->subgroup_size, - conv2d_BS.CRS); // CRS block size should be capped at subgroup size for correctness when shuffle is used. + if (device->external_memory_host) { + device_extensions.push_back("VK_EXT_external_memory_host"); } - // cm1 is used only when cm2 is unavailable; capped at 64x128 (due to shared memory size). - // Requires 16x16x16 f16-acc since that's the fragment shape hard-coded in the shader. - // Subgroup size must be 32 or 64 (to keep WG_SIZE sane) and we need - // subgroup_size_control to force the driver to actually use it. - bool conv2d_use_cm1 = false; -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - conv2d_use_cm1 = !device->coopmat2 && - device->coopmat_support && device->coopmat_support_16x16x16_f16acc && - device->subgroup_size_control && - (device->subgroup_size == 32 || device->subgroup_size == 64) && - s != CONV_SHAPE_128x128; +#if defined(VK_EXT_shader_64bit_indexing) + VkPhysicalDeviceShader64BitIndexingFeaturesEXT shader_64bit_indexing_features {}; + shader_64bit_indexing_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_64_BIT_INDEXING_FEATURES_EXT; + if (device->shader_64b_indexing) { + last_struct->pNext = (VkBaseOutStructure *)&shader_64bit_indexing_features; + last_struct = (VkBaseOutStructure *)&shader_64bit_indexing_features; + device_extensions.push_back("VK_EXT_shader_64bit_indexing"); + } #endif - const uint32_t conv2d_cm1_shmem_pad = 8; + VkPhysicalDeviceFaultFeaturesEXT fault_features {}; + fault_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FAULT_FEATURES_EXT; + if (device->device_fault) { + last_struct->pNext = (VkBaseOutStructure *)&fault_features; + last_struct = (VkBaseOutStructure *)&fault_features; + device_extensions.push_back("VK_EXT_device_fault"); + } - auto shmem_req = [&](uint32_t pad, bool csh_store, bool fp16_shmem) { - const uint32_t elem_size = fp16_shmem ? (uint32_t)sizeof(uint16_t) : (uint32_t)sizeof(float); - const uint32_t csh_elems = csh_store ? conv2d_BS.K * conv2d_BS.NPQ : 0u; - return (conv2d_BS.K * (conv2d_BS.CRS + pad) + conv2d_BS.CRS * (conv2d_BS.NPQ + pad) + csh_elems) * elem_size; - }; + vkGetPhysicalDeviceFeatures2(device->physical_device, &device_features2); - // 2D, transpose-2D, and 3D conv use the same KxCRS @ CRSxNPQ shmem - // layout. cm1 needs Csh for output, so check before applying cm1 params. - if (conv2d_use_cm1 && device->properties.limits.maxComputeSharedMemorySize < shmem_req(conv2d_cm1_shmem_pad, true, true)) { - conv2d_use_cm1 = false; - } + device->device_fault = device->device_fault && fault_features.deviceFault; - uint32_t conv2d_WM = 16, conv2d_WN = 16; // cm1 subgroup tile, ignored otherwise - if (conv2d_use_cm1) { - conv2d_SHMEM_PAD = conv2d_cm1_shmem_pad; - // 16x16x16 fragments; pick WM/WN to keep WG_SIZE at 256 - // (i.e. 8 subgroups for sg=32, 4 subgroups for sg=64). - const bool sg64 = (device->subgroup_size == 64); - switch (s) { - case CONV_SHAPE_64x32: conv2d_WM = sg64 ? 32 : 16; conv2d_WN = 16; break; - case CONV_SHAPE_64x128: conv2d_WM = 32; conv2d_WN = sg64 ? 64 : 32; break; - case CONV_SHAPE_32x256: conv2d_WM = sg64 ? 16 : 32; conv2d_WN = sg64 ? 128 : 32; break; - default: break; - } - const uint32_t warps_M = conv2d_BS.K / conv2d_WM; - const uint32_t warps_N = conv2d_BS.NPQ / conv2d_WN; - conv2d_WG_SIZE = warps_M * warps_N * device->subgroup_size; - } + device->has_internally_synchronized_queues = internally_synchronized_queues_features.internallySynchronizedQueues; - // stage cm2 accumulator through shmem for coalesced global stores; - // skipped on 128x128 where the extra Csh footprint hurts occupancy. - // cm1 always uses the staged path. - uint32_t conv2d_csh_store = (device->coopmat2 && s != CONV_SHAPE_128x128) ? 1u : 0u; - if (conv2d_use_cm1) { - conv2d_csh_store = 1; + // Build queue create infos only after querying whether internally synchronized queues are enabled. + // getQueue2() later uses the same flag, so creation/retrieval must stay consistent. + vk::DeviceQueueCreateFlags queue_flags = device->has_internally_synchronized_queues ? + eInternallySynchronizedKHR : + vk::DeviceQueueCreateFlags(); + + if (compute_queue_family_index != transfer_queue_family_index) { + device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 1, priorities}); + device_queue_create_infos.push_back({queue_flags, transfer_queue_family_index, 1, priorities + 1}); + } else if(!device->single_queue) { + device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 2, priorities}); + } else { + device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 1, priorities}); } - // shmem is fp16 on cm2/cm1 (matches Csh), fp32 on scalar - const bool conv2d_use_fp16_shmem = device->coopmat2 || conv2d_use_cm1; + device->pipeline_executable_properties_support = pipeline_executable_properties_support; - // shrink CRS if the non-cm1 config still doesn't fit - if (device->properties.limits.maxComputeSharedMemorySize < shmem_req(conv2d_SHMEM_PAD, conv2d_csh_store, conv2d_use_fp16_shmem)) { - GGML_ASSERT(!conv2d_use_cm1); - conv2d_BS.CRS = 8; - if (use_collectives) { - conv2d_BS.CRS = std::min(device->subgroup_size, conv2d_BS.CRS); - } - conv2d_csh_store = 0; + device->fp16 = device->fp16 && vk12_features.shaderFloat16; + +#if defined(VK_KHR_shader_bfloat16) + device->bf16 = bfloat16_support && bfloat16_features.shaderBFloat16Type; +#else + device->bf16 = false; +#endif + + device->dot2_f16 = dot2_f16_support && dot2_features.shaderMixedFloatDotProductFloat16AccFloat32; + device->ocp_fp4 = ocp_microscaling_extension && ocp_microscaling_features.shaderFloat4 && + shader_float8_extension && shader_float8_features.shaderFloat8 && + !getenv("GGML_VK_DISABLE_OCP_FP4"); + + device->pipeline_robustness = pl_robustness_features.pipelineRobustness; + + device->multi_add = vk12_props.shaderRoundingModeRTEFloat16 && + device->properties.limits.maxPushConstantsSize >= sizeof(vk_op_multi_add_push_constants) && + getenv("GGML_VK_DISABLE_MULTI_ADD") == nullptr; + + device->shader_int64 = device_features2.features.shaderInt64; + device->buffer_device_address = vk12_features.bufferDeviceAddress; + device->vulkan_memory_model = vk12_features.vulkanMemoryModel; + + if (device->subgroup_size_control) { + device->subgroup_min_size = subgroup_size_control_props.minSubgroupSize; + device->subgroup_max_size = subgroup_size_control_props.maxSubgroupSize; + device_extensions.push_back("VK_EXT_subgroup_size_control"); } - std::array wg_denoms = { conv2d_BS.K, 1, 1 }; - std::vector spec_constants = { conv2d_WG_SIZE, conv2d_BS.K, conv2d_BS.CRS, conv2d_BS.NPQ, conv2d_TS_K, use_collectives, conv2d_SHMEM_PAD }; + device->subgroup_size_control = device->subgroup_size_control && + (subgroup_size_control_props.requiredSubgroupSizeStages & vk::ShaderStageFlagBits::eCompute) && + subgroup_size_control_features.subgroupSizeControl; - // cm1 needs a fixed subgroup width to match the WG_SIZE we computed - const uint32_t conv2d_required_subgroup_size = conv2d_use_cm1 ? device->subgroup_size : 0; + device->subgroup_require_full_support = subgroup_size_control_features.computeFullSubgroups; -#define CREATE_CONV(name, type_suffix, spv_suffix) \ - for (auto &c : device->pipeline_##name##type_suffix[s]) { \ - const vk_conv2d_pipeline_state &state = c.first; \ - std::vector spec_constants_cpy = spec_constants; \ - spec_constants_cpy.push_back(state.s0); \ - spec_constants_cpy.push_back(state.s1); \ - spec_constants_cpy.push_back(state.p0); \ - spec_constants_cpy.push_back(state.p1); \ - spec_constants_cpy.push_back(state.d0); \ - spec_constants_cpy.push_back(state.d1); \ - spec_constants_cpy.push_back(state.KW); \ - spec_constants_cpy.push_back(state.KH); \ - spec_constants_cpy.push_back(state.aligned); \ - spec_constants_cpy.push_back(conv2d_csh_store); \ - spec_constants_cpy.push_back(conv2d_WM); \ - spec_constants_cpy.push_back(conv2d_WN); \ - ggml_vk_create_pipeline( \ - device, c.second, #name #type_suffix, \ - name##type_suffix##spv_suffix##_len, name##type_suffix##spv_suffix##_data, "main", 3, \ - sizeof(vk_op_conv2d_push_constants), wg_denoms, spec_constants_cpy, 1, true, use_collectives || conv2d_required_subgroup_size, conv2d_required_subgroup_size); \ - } -#define CREATE_CONVS(spv_suffix) \ - CREATE_CONV(conv2d, _f32, spv_suffix) \ - CREATE_CONV(conv2d, _f16_f32, spv_suffix) \ - CREATE_CONV(conv_transpose_2d, _f32, spv_suffix) \ - CREATE_CONV(conv_transpose_2d, _f16_f32, spv_suffix) -#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (device->coopmat2) { - CREATE_CONVS(_cm2) - } else -#endif -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (conv2d_use_cm1) { - CREATE_CONVS(_cm1) - } else -#endif - if (conv2d_UNROLL) { - CREATE_CONVS(_unroll) - } else { - CREATE_CONVS( ) - } -#undef CREATE_CONV -#undef CREATE_CONVS - - std::vector conv3d_spec_constants = { conv2d_WG_SIZE, conv2d_BS.K, conv2d_BS.CRS, conv2d_BS.NPQ, conv2d_TS_K, conv2d_SHMEM_PAD }; -#define CREATE_CONV3D(type_suffix, spv_suffix) \ - for (auto &c : device->pipeline_conv3d##type_suffix[s]) { \ - const vk_conv3d_pipeline_state &state = c.first; \ - std::vector spec_constants_cpy = conv3d_spec_constants; \ - spec_constants_cpy.push_back(state.s0); \ - spec_constants_cpy.push_back(state.s1); \ - spec_constants_cpy.push_back(state.s2); \ - spec_constants_cpy.push_back(state.p0); \ - spec_constants_cpy.push_back(state.p1); \ - spec_constants_cpy.push_back(state.p2); \ - spec_constants_cpy.push_back(state.d0); \ - spec_constants_cpy.push_back(state.d1); \ - spec_constants_cpy.push_back(state.d2); \ - spec_constants_cpy.push_back(state.KW); \ - spec_constants_cpy.push_back(state.KH); \ - spec_constants_cpy.push_back(state.KD); \ - spec_constants_cpy.push_back(state.aligned); \ - spec_constants_cpy.push_back(conv2d_csh_store); \ - spec_constants_cpy.push_back(conv2d_WM); \ - spec_constants_cpy.push_back(conv2d_WN); \ - ggml_vk_create_pipeline( \ - device, c.second, "conv3d" #type_suffix, \ - conv3d##type_suffix##spv_suffix##_len, conv3d##type_suffix##spv_suffix##_data, "main", 3, \ - sizeof(vk_op_conv3d_push_constants), wg_denoms, spec_constants_cpy, 1, true, conv2d_required_subgroup_size != 0, conv2d_required_subgroup_size); \ - } -#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (device->coopmat2) { - CREATE_CONV3D(_f32, _cm2) - CREATE_CONV3D(_f16_f32, _cm2) - } else -#endif -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (conv2d_use_cm1) { - CREATE_CONV3D(_f32, _cm1) - CREATE_CONV3D(_f16_f32, _cm1) - } else +#if defined(VK_KHR_cooperative_matrix) + device->coopmat_support = device->coopmat_support && coopmat_features.cooperativeMatrix; + device->coopmat1_fa_support = device->coopmat_support && device->subgroup_require_full_support; #endif - if (conv2d_UNROLL) { - CREATE_CONV3D(_f32, _unroll) - CREATE_CONV3D(_f16_f32, _unroll) - } else { - CREATE_CONV3D(_f32, ) - CREATE_CONV3D(_f16_f32, ) - } -#undef CREATE_CONV3D - } - ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_whcn_f32, "conv2d_dw_whcn_f32", conv2d_dw_whcn_f32_len, conv2d_dw_whcn_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_cwhn_f32, "conv2d_dw_cwhn_f32", conv2d_dw_cwhn_f32_len, conv2d_dw_cwhn_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_whcn_f16_f32, "conv2d_dw_whcn_f16_f32", conv2d_dw_whcn_f16_f32_len, conv2d_dw_whcn_f16_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); - ggml_vk_create_pipeline(device, device->pipeline_conv2d_dw_cwhn_f16_f32, "conv2d_dw_cwhn_f16_f32", conv2d_dw_cwhn_f16_f32_len, conv2d_dw_cwhn_f16_f32_data, "main", 3, sizeof(vk_op_conv2d_dw_push_constants), {512, 1, 1}, {}, 1); + if (coopmat2_support) { +#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + if (coopmat2_features.cooperativeMatrixWorkgroupScope && + coopmat2_features.cooperativeMatrixFlexibleDimensions && + coopmat2_features.cooperativeMatrixReductions && + coopmat2_features.cooperativeMatrixConversions && + coopmat2_features.cooperativeMatrixPerElementOperations && + coopmat2_features.cooperativeMatrixTensorAddressing && + coopmat2_features.cooperativeMatrixBlockLoads && + vk12_features.bufferDeviceAddress) { - for (uint32_t use_push = 0; use_push < 2; ++use_push) { - for (uint32_t i = 0; i < num_topk_moe_pipelines; ++i) { - ggml_vk_create_pipeline2(device, device->pipeline_topk_moe[i][use_push], "topk_moe_f32_"+std::to_string(i), topk_moe_f32_len, topk_moe_f32_data, "main", 4, sizeof(vk_op_topk_moe_push_constants), {1, 1, 1}, {device->subgroup_size, 1u<subgroup_size); - } - } + std::vector flexible_dimensions; + uint32_t count = 0; - // Drop compile_mutex so other threads can walk while we compile. - compile_lock.unlock(); + PFN_vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV + _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV = + (PFN_vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV) + vk_instance.instance.getProcAddr("vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV"); - // Compile what we claimed; create_pipeline_func reacquires compile_mutex - // at the end to flip compile_pending/compiled and notify waiters. - if (has_claimed_task) { - auto & task = claimed_task; - ggml_vk_create_pipeline_func(device, task.pipeline, task.spv_size, task.spv_data, - task.entrypoint, task.parameter_count, task.wg_denoms, - task.specialization_constants, task.disable_robustness, - task.require_full_subgroups, task.required_subgroup_size); - } + _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV(device->physical_device, &count, nullptr); - // Another thread may be compiling the pipeline we need; block on it here. - if (wait_pipeline) { - std::unique_lock wait_lock(device->compile_mutex); - device->compile_cv.wait(wait_lock, [&] { - return wait_pipeline->compiled.load(); - }); - } -} + VkCooperativeMatrixFlexibleDimensionsPropertiesNV empty_prop {}; + empty_prop.sType = VK_STRUCTURE_TYPE_COOPERATIVE_MATRIX_FLEXIBLE_DIMENSIONS_PROPERTIES_NV; + flexible_dimensions.resize(count, empty_prop); -static bool ggml_vk_khr_cooperative_matrix_support(const vk::PhysicalDeviceProperties& props, const vk::PhysicalDeviceDriverProperties& driver_props, vk_device_architecture arch); -static uint32_t ggml_vk_intel_shader_core_count(const vk::PhysicalDevice& vkdev); + _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV(device->physical_device, &count, flexible_dimensions.data()); -static vk_device ggml_vk_get_device(size_t idx) { - VK_LOG_DEBUG("ggml_vk_get_device(" << idx << ")"); + bool found_fp16_128 = false, + found_fp16_256 = false, + found_fp32_128 = false, + found_fp32_256 = false; + bool found_bf16_128 = false, + found_bf16_256 = false; + // need to support fp16*fp16 with fp16/fp32 accumulator, for workgroupsize 128 + // with 32x16x16 and 256 with 32x32x16. + for (auto &prop : flexible_dimensions) { + if (prop.saturatingAccumulation == VK_FALSE && + prop.scope == VK_SCOPE_WORKGROUP_KHR) { - if (vk_instance.devices[idx] == nullptr) { - VK_LOG_DEBUG("Initializing new vk_device"); - vk_device device = std::make_shared(); - vk_instance.devices[idx] = device; + if (prop.AType == VK_COMPONENT_TYPE_FLOAT16_KHR && + prop.BType == VK_COMPONENT_TYPE_FLOAT16_KHR) { - device->memory_logger = std::unique_ptr(new vk_memory_logger()); + if (prop.workgroupInvocations == 128 && + prop.MGranularity <= 32 && + prop.NGranularity <= 16 && + prop.KGranularity <= 16) { + if (prop.CType == VK_COMPONENT_TYPE_FLOAT16_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT16_KHR) { + found_fp16_128 = true; + } + if (prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { + found_fp32_128 = true; + } + } + if (prop.workgroupInvocations == 256 && + prop.MGranularity <= 32 && + prop.NGranularity <= 32 && + prop.KGranularity <= 16) { + if (prop.CType == VK_COMPONENT_TYPE_FLOAT16_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT16_KHR) { + found_fp16_256 = true; + } + if (prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { + found_fp32_256 = true; + } + } + } - size_t dev_num = vk_instance.device_indices[idx]; +#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (bfloat16_support && + prop.AType == VK_COMPONENT_TYPE_BFLOAT16_KHR && + prop.BType == VK_COMPONENT_TYPE_BFLOAT16_KHR && + prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { - std::vector physical_devices = vk_instance.instance.enumeratePhysicalDevices(); + if (prop.workgroupInvocations == 128 && + prop.MGranularity <= 32 && + prop.NGranularity <= 16 && + prop.KGranularity <= 16) { + found_bf16_128 = true; + } + if (prop.workgroupInvocations == 256 && + prop.MGranularity <= 32 && + prop.NGranularity <= 32 && + prop.KGranularity <= 16) { + found_bf16_256 = true; + } + } +#endif + } + } + if (found_fp16_128 && found_fp16_256 && + found_fp32_128 && found_fp32_256 && + coopmat2_props.cooperativeMatrixFlexibleDimensionsMaxDimension >= 512) { + device->coopmat2 = true; + device->coopmat2_bf16_support = found_bf16_128 && found_bf16_256; + device->coopmat2_decode_vector = coopmat2_decode_vector_support && coopmat2_decode_vector_features.cooperativeMatrixDecodeVector; + } + } +#endif + } - if (dev_num >= physical_devices.size()) { - std::cerr << "ggml_vulkan: Device with index " << dev_num << " does not exist." << std::endl; - throw std::runtime_error("Device not found"); + if (!vk11_features.storageBuffer16BitAccess) { + std::cerr << "ggml_vulkan: device " << GGML_VK_NAME << idx << " does not support 16-bit storage." << std::endl; + throw std::runtime_error("Unsupported device"); } - device->physical_device = physical_devices[dev_num]; - const std::vector ext_props = device->physical_device.enumerateDeviceExtensionProperties(); + device_extensions.push_back("VK_KHR_16bit_storage"); - device->architecture = get_device_architecture(device->physical_device); +#ifdef GGML_VULKAN_VALIDATE + device_extensions.push_back("VK_KHR_shader_non_semantic_info"); +#endif - const char* GGML_VK_PREFER_HOST_MEMORY = getenv("GGML_VK_PREFER_HOST_MEMORY"); - device->prefer_host_memory = GGML_VK_PREFER_HOST_MEMORY != nullptr; + if (device->fp16) { + device_extensions.push_back("VK_KHR_shader_float16_int8"); + } - const char* GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM = getenv("GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM"); - device->disable_host_visible_vidmem = GGML_VK_DISABLE_HOST_VISIBLE_VIDMEM != nullptr; +#if defined(VK_KHR_cooperative_matrix) + if (device->coopmat_support) { + // Query supported shapes + std::vector cm_props; - const char* GGML_VK_ALLOW_SYSMEM_FALLBACK = getenv("GGML_VK_ALLOW_SYSMEM_FALLBACK"); - device->allow_sysmem_fallback = GGML_VK_ALLOW_SYSMEM_FALLBACK != nullptr; + PFN_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR = + (PFN_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR)vkGetInstanceProcAddr(vk_instance.instance, "vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR"); - const char* GGML_VK_DISABLE_GRAPH_OPTIMIZE = getenv("GGML_VK_DISABLE_GRAPH_OPTIMIZE"); - device->disable_graph_optimize = GGML_VK_DISABLE_GRAPH_OPTIMIZE != nullptr; + uint32_t cm_props_num; - bool fp16_storage = false; - bool fp16_compute = false; - bool maintenance4_support = false; - bool sm_builtins = false; - bool amd_shader_core_properties2 = false; - bool pipeline_robustness = false; - bool coopmat2_support = false; - bool coopmat2_decode_vector_support = false; - bool pipeline_executable_properties_support = false; - bool internally_sync_support = false; - device->coopmat_support = false; - device->integer_dot_product = false; - device->shader_64b_indexing = false; - bool bfloat16_support = false; - bool dot2_f16_support = false; - bool ocp_microscaling_extension = false; - bool shader_float8_extension = false; + pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(device->physical_device, &cm_props_num, nullptr); - for (const auto& properties : ext_props) { - if (strcmp("VK_KHR_maintenance4", properties.extensionName) == 0) { - maintenance4_support = true; - } else if (strcmp("VK_KHR_16bit_storage", properties.extensionName) == 0) { - fp16_storage = true; - } else if (strcmp("VK_KHR_shader_float16_int8", properties.extensionName) == 0) { - fp16_compute = true; - } else if (strcmp("VK_NV_shader_sm_builtins", properties.extensionName) == 0) { - sm_builtins = true; - } else if (strcmp("VK_AMD_shader_core_properties2", properties.extensionName) == 0) { - amd_shader_core_properties2 = true; - } else if (strcmp("VK_EXT_pipeline_robustness", properties.extensionName) == 0) { - pipeline_robustness = true; - } else if (strcmp("VK_EXT_subgroup_size_control", properties.extensionName) == 0) { - device->subgroup_size_control = true; -#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_cooperative_matrix", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT")) { - device->coopmat_support = true; - device->coopmat_m = 0; - device->coopmat_n = 0; - device->coopmat_k = 0; -#endif -#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - } else if (strcmp("VK_NV_cooperative_matrix2", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT2")) { - coopmat2_support = true; -#endif - } else if (strcmp(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME, properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT2_DECODE_VECTOR")) { - coopmat2_decode_vector_support = true; -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_shader_integer_dot_product", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_INTEGER_DOT_PRODUCT")) { - device->integer_dot_product = true; -#endif -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_shader_bfloat16", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_BFLOAT16")) { - bfloat16_support = true; -#endif -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) - } else if (strcmp(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME, properties.extensionName) == 0) { - ocp_microscaling_extension = true; -#endif -#if defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - } else if (strcmp(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME, properties.extensionName) == 0) { - shader_float8_extension = true; -#endif - } else if (strcmp("VK_VALVE_shader_mixed_float_dot_product", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_DOT2")) { - dot2_f16_support = true; - } else if (strcmp("VK_KHR_pipeline_executable_properties", properties.extensionName) == 0) { - pipeline_executable_properties_support = true; - } else if (strcmp("VK_EXT_memory_priority", properties.extensionName) == 0 && - getenv("GGML_VK_ENABLE_MEMORY_PRIORITY")) { - device->memory_priority = true; - } else if (strcmp("VK_EXT_external_memory_host", properties.extensionName) == 0) { - device->external_memory_host = true; -#if defined(VK_EXT_shader_64bit_indexing) - } else if (strcmp("VK_EXT_shader_64bit_indexing", properties.extensionName) == 0) { - device->shader_64b_indexing = true; -#endif - } else if (strcmp(VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME, properties.extensionName) == 0) { - internally_sync_support = true; - } else if (strcmp("VK_EXT_device_fault", properties.extensionName) == 0) { - device->device_fault = true; - } - } + cm_props.resize(cm_props_num); - vk::PhysicalDeviceProperties2 props2; - vk::PhysicalDeviceMaintenance3Properties props3; - vk::PhysicalDeviceMaintenance4Properties props4; - vk::PhysicalDeviceSubgroupProperties subgroup_props; - vk::PhysicalDeviceDriverProperties driver_props; - vk::PhysicalDeviceShaderSMBuiltinsPropertiesNV sm_props; - vk::PhysicalDeviceShaderCoreProperties2AMD amd_shader_core_properties2_props; - vk::PhysicalDeviceVulkan11Properties vk11_props; - vk::PhysicalDeviceVulkan12Properties vk12_props; - vk::PhysicalDeviceSubgroupSizeControlPropertiesEXT subgroup_size_control_props; - vk::PhysicalDeviceShaderIntegerDotProductPropertiesKHR shader_integer_dot_product_props; - vk::PhysicalDeviceExternalMemoryHostPropertiesEXT external_memory_host_props; + for (auto& prop : cm_props) { + prop.sType = VK_STRUCTURE_TYPE_COOPERATIVE_MATRIX_PROPERTIES_KHR; + } - props2.pNext = &props3; - props3.pNext = &subgroup_props; - subgroup_props.pNext = &driver_props; - driver_props.pNext = &vk11_props; - vk11_props.pNext = &vk12_props; + pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(device->physical_device, &cm_props_num, cm_props.data()); - VkBaseOutStructure * last_struct = (VkBaseOutStructure *)&vk12_props; + VK_LOG_DEBUG("ggml_vulkan: Cooperative Matrix Shapes: " << cm_props.size()); - if (maintenance4_support) { - last_struct->pNext = (VkBaseOutStructure *)&props4; - last_struct = (VkBaseOutStructure *)&props4; - } - if (sm_builtins) { - last_struct->pNext = (VkBaseOutStructure *)&sm_props; - last_struct = (VkBaseOutStructure *)&sm_props; - } - if (amd_shader_core_properties2) { - last_struct->pNext = (VkBaseOutStructure *)&amd_shader_core_properties2_props; - last_struct = (VkBaseOutStructure *)&amd_shader_core_properties2_props; - } - if (device->subgroup_size_control) { - last_struct->pNext = (VkBaseOutStructure *)&subgroup_size_control_props; - last_struct = (VkBaseOutStructure *)&subgroup_size_control_props; - } + for (auto& prop : cm_props) { + VK_LOG_DEBUG("ggml_vulkan: M: " << prop.MSize << " N: " << prop.NSize << " K: " << prop.KSize << " A: " << vk::to_string((vk::ComponentTypeKHR)prop.AType) << " B: " << vk::to_string((vk::ComponentTypeKHR)prop.BType) << " C: " << vk::to_string((vk::ComponentTypeKHR)prop.CType) << " Result: " << vk::to_string((vk::ComponentTypeKHR)prop.ResultType) << " saturatingAccumulation: " << prop.saturatingAccumulation << " scope: " << vk::to_string((vk::ScopeKHR)prop.scope)); -#if defined(VK_NV_cooperative_matrix2) - vk::PhysicalDeviceCooperativeMatrix2PropertiesNV coopmat2_props; - if (coopmat2_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat2_props; - last_struct = (VkBaseOutStructure *)&coopmat2_props; - } + if ((vk::ComponentTypeKHR)prop.AType == vk::ComponentTypeKHR::eFloat16 && + (vk::ComponentTypeKHR)prop.BType == vk::ComponentTypeKHR::eFloat16 && + (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup + ) { + if ((vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eFloat32 && + (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eFloat32) { + // coopmat sizes not set yet + if (device->coopmat_m == 0) { + device->coopmat_acc_f32_support = true; + device->coopmat_m = prop.MSize; + device->coopmat_n = prop.NSize; + device->coopmat_k = prop.KSize; + } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { + // Only enable if shape is identical + device->coopmat_acc_f32_support = true; + } + if (prop.MSize == 16 && prop.NSize == 16 && prop.KSize == 16) { + device->coopmat_support_16x16x16_f32acc = true; + } + } else if ((vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eFloat16 && + (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eFloat16) { + // coopmat sizes not set yet + if (device->coopmat_m == 0) { + device->coopmat_acc_f16_support = true; + device->coopmat_m = prop.MSize; + device->coopmat_n = prop.NSize; + device->coopmat_k = prop.KSize; + } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { + // Only enable if shape is identical + device->coopmat_acc_f16_support = true; + } + if (prop.MSize == 16 && prop.NSize == 16 && prop.KSize == 16) { + device->coopmat_support_16x16x16_f16acc = true; + } + } + } else if ((vk::ComponentTypeKHR)prop.AType == vk::ComponentTypeKHR::eSint8 && + (vk::ComponentTypeKHR)prop.BType == vk::ComponentTypeKHR::eSint8 && + (vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eSint32 && + (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eSint32 && + (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup && + device->coopmat_int_m == 0 + ) { + device->coopmat_int_support = true; + device->coopmat_int_m = prop.MSize; + device->coopmat_int_n = prop.NSize; + device->coopmat_int_k = prop.KSize; + } +#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + if (bfloat16_support && + prop.AType == VK_COMPONENT_TYPE_BFLOAT16_KHR && + prop.BType == VK_COMPONENT_TYPE_BFLOAT16_KHR && + prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && + prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR && + (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup + ) { + // coopmat sizes not set yet + if (device->coopmat_m == 0) { + device->coopmat_bf16_support = true; + device->coopmat_m = prop.MSize; + device->coopmat_n = prop.NSize; + device->coopmat_k = prop.KSize; + } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { + // Only enable if shape is identical + device->coopmat_bf16_support = true; + } + } #endif + } - if (device->integer_dot_product) { - last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_props; - last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_props; - } - - if (device->external_memory_host) { - last_struct->pNext = (VkBaseOutStructure *)&external_memory_host_props; - last_struct = (VkBaseOutStructure *)&external_memory_host_props; - } - - device->physical_device.getProperties2(&props2); - device->properties = props2.properties; - device->vendor_id = device->properties.vendorID; - device->driver_id = driver_props.driverID; - - if (device->driver_id == vk::DriverId::eMoltenvk) { - // Disable external_memory_host until https://github.com/KhronosGroup/MoltenVK/pull/2622 - // is available in the Vulkan SDK. - device->external_memory_host = false; + if (device->coopmat_m == 0 || !device->coopmat_acc_f32_support) { + // No suitable matmul mode found + GGML_LOG_DEBUG("ggml_vulkan: WARNING: No suitable matrix core mode found. Disabling matrix cores.\n"); + device->coopmat_support = false; + } } - // Implementing the async backend interfaces seems broken on older Intel HW, - // see https://github.com/ggml-org/llama.cpp/issues/17302. - device->support_async = (device->vendor_id != VK_VENDOR_ID_INTEL || - std::string(device->properties.deviceName.data()).find("(DG1)") == std::string::npos) && - getenv("GGML_VK_DISABLE_ASYNC") == nullptr; - - if (!device->support_async) { - GGML_LOG_DEBUG("ggml_vulkan: WARNING: Async execution disabled on certain Intel devices.\n"); + if (device->coopmat_support) { + device_extensions.push_back("VK_KHR_cooperative_matrix"); } +#endif + device->name = GGML_VK_NAME + std::to_string(idx); - const char* GGML_VK_FORCE_MAX_ALLOCATION_SIZE = getenv("GGML_VK_FORCE_MAX_ALLOCATION_SIZE"); + device_create_info + .setFlags(vk::DeviceCreateFlags()) + .setQueueCreateInfos(device_queue_create_infos) + .setPEnabledExtensionNames(device_extensions); + device_create_info.setPNext(&device_features2); + device->device = device->physical_device.createDevice(device_create_info); - if (GGML_VK_FORCE_MAX_ALLOCATION_SIZE != nullptr) { - device->max_memory_allocation_size = std::stoull(GGML_VK_FORCE_MAX_ALLOCATION_SIZE); - } else if (maintenance4_support) { - device->max_memory_allocation_size = std::min(props3.maxMemoryAllocationSize, props4.maxBufferSize); - } else { - device->max_memory_allocation_size = props3.maxMemoryAllocationSize; + if (device->device_fault) { + device->pfn_vkGetDeviceFaultInfoEXT = (PFN_vkGetDeviceFaultInfoEXT) + vkGetDeviceProcAddr(device->device, "vkGetDeviceFaultInfoEXT"); } - const char* GGML_VK_FORCE_MAX_BUFFER_SIZE = getenv("GGML_VK_FORCE_MAX_BUFFER_SIZE"); + // Queues + device->compute_queue = ggml_vk_create_queue(device, compute_queue_family_index, 0, { vk::PipelineStageFlagBits::eComputeShader | vk::PipelineStageFlagBits::eTransfer }, false); - if (GGML_VK_FORCE_MAX_BUFFER_SIZE != nullptr) { - device->max_buffer_size = std::stoull(GGML_VK_FORCE_MAX_BUFFER_SIZE); - } else if (maintenance4_support) { - device->max_buffer_size = props4.maxBufferSize; - } else { - device->max_buffer_size = device->max_memory_allocation_size; - } + // Shaders + // Disable matmul tile sizes early if performance low or not supported + for (uint32_t i = 0; i < GGML_TYPE_COUNT; ++i) { + switch (device->vendor_id) { +#ifndef GGML_VULKAN_RUN_TESTS + case VK_VENDOR_ID_AMD: + device->mul_mat_l[i] = device->coopmat_support && device->driver_id != vk::DriverId::eAmdProprietary; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = true; + device->mul_mat_id_l[i] = false; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = true; + break; + case VK_VENDOR_ID_INTEL: { + // Current Windows driver does not expose BF16 support. + // We only want to use l_warptile if coopmat is available + const bool use_l_warptile = (i == GGML_TYPE_BF16) ? (device->coopmat_bf16_support && device->coopmat_support) : device->coopmat_support; + device->mul_mat_l[i] = use_l_warptile; + device->mul_mat_id_l[i] = use_l_warptile; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = true; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = true; + break; + } + case VK_VENDOR_ID_APPLE: + device->mul_mat_l[i] = false; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = false; + device->mul_mat_id_l[i] = false; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = false; + break; + case VK_VENDOR_ID_QUALCOMM: + device->mul_mat_l[i] = false; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = true; + device->mul_mat_id_l[i] = false; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = true; + break; +#endif + default: + device->mul_mat_l[i] = true; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = true; + device->mul_mat_id_l[i] = true; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = true; + break; + } - const char* GGML_VK_SUBALLOCATION_BLOCK_SIZE = getenv("GGML_VK_SUBALLOCATION_BLOCK_SIZE"); +#if VK_HEADER_VERSION >= 287 + // Honeykrisp driver for Asahi Linux doesn't report VK_VENDOR_ID_APPLE. + // Check for Honeykrisp driver and force same configuration as the VK_VENDOR_ID_APPLE case. + if (device->driver_id == vk::DriverId::eMesaHoneykrisp) { + device->mul_mat_l[i] = false; + device->mul_mat_m[i] = true; + device->mul_mat_s[i] = false; + device->mul_mat_id_l[i] = false; + device->mul_mat_id_m[i] = true; + device->mul_mat_id_s[i] = false; + } +#endif - if (GGML_VK_SUBALLOCATION_BLOCK_SIZE != nullptr) { - device->suballocation_block_size = std::stoull(GGML_VK_SUBALLOCATION_BLOCK_SIZE); - } else { - // Limit batching of allocations to 1GB by default to avoid fragmentation issues - device->suballocation_block_size = 1024*1024*1024; + device->mul_mat_l_int[i] = device->mul_mat_l[i]; + device->mul_mat_m_int[i] = device->mul_mat_m[i]; + device->mul_mat_s_int[i] = device->mul_mat_s[i]; + device->mul_mat_id_l_int[i] = device->mul_mat_id_l[i]; + device->mul_mat_id_m_int[i] = device->mul_mat_id_m[i]; + device->mul_mat_id_s_int[i] = device->mul_mat_id_s[i]; } - device->suballocation_block_size = std::min(device->suballocation_block_size, device->max_memory_allocation_size); - device->subgroup_size = subgroup_props.subgroupSize; - device->subgroup_size_log2 = uint32_t(log2f(float(device->subgroup_size))); - device->uma = device->properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu; - if (sm_builtins) { - device->shader_core_count = sm_props.shaderSMCount; - } else if (amd_shader_core_properties2) { - device->shader_core_count = amd_shader_core_properties2_props.activeComputeUnitCount; - } else if (device->vendor_id == VK_VENDOR_ID_INTEL) { - device->shader_core_count = ggml_vk_intel_shader_core_count(device->physical_device); - } else { - device->shader_core_count = 0; - } - device->float_controls_rte_fp16 = vk12_props.shaderRoundingModeRTEFloat16; - device->float_controls_denorm_preserve_fp16 = vk12_props.shaderDenormPreserveFloat16; - device->subgroup_basic = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eBasic); - device->subgroup_arithmetic = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eArithmetic); -#ifdef __APPLE__ - // Workaround for subgroup arithmetic failing on MoltenVK with AMD GPUs (issue 15846) - if (device->vendor_id == VK_VENDOR_ID_AMD) { - device->subgroup_arithmetic = false; - } -#endif - device->subgroup_shuffle = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eShuffle); -#ifdef __APPLE__ - if (device->vendor_id == VK_VENDOR_ID_AMD) { - device->subgroup_shuffle = false; + std::vector dsl_binding; + std::vector dsl_binding_flags; + for (uint32_t i = 0; i < MAX_PARAMETER_COUNT; i++) { + dsl_binding.push_back({i, vk::DescriptorType::eStorageBuffer, 1, vk::ShaderStageFlagBits::eCompute}); + dsl_binding_flags.push_back({}); } -#endif - device->subgroup_clustered = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eClustered); - - device->subgroup_ballot = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eBallot); - device->subgroup_vote = (vk11_props.subgroupSupportedStages & vk::ShaderStageFlagBits::eCompute) && - (vk11_props.subgroupSupportedOperations & vk::SubgroupFeatureFlagBits::eVote); + vk::DescriptorSetLayoutBindingFlagsCreateInfo dslbfci = { dsl_binding_flags }; - // Submit at least every 100 nodes, in case there are workloads without as much matmul. - device->max_nodes_per_submit = 100; - const char* GGML_VK_MAX_NODES_PER_SUBMIT = getenv("GGML_VK_MAX_NODES_PER_SUBMIT"); - if (GGML_VK_MAX_NODES_PER_SUBMIT != nullptr) { - uint32_t max_nodes_per_submit = std::stoul(GGML_VK_MAX_NODES_PER_SUBMIT); - device->max_nodes_per_submit = std::max(max_nodes_per_submit, 1u); - } + vk::DescriptorSetLayoutCreateInfo descriptor_set_layout_create_info( + {}, + dsl_binding); + descriptor_set_layout_create_info.setPNext(&dslbfci); + device->dsl = device->device.createDescriptorSetLayout(descriptor_set_layout_create_info); - const bool force_disable_f16 = getenv("GGML_VK_DISABLE_F16") != nullptr; + ggml_vk_load_shaders(device); - device->fp16 = !force_disable_f16 && fp16_storage && fp16_compute; + // Prefer a dedicated transfer queue on AMD dGPUs (non-GCN) when graphics queue use is disabled. + const bool prefers_transfer_queue = + device->vendor_id == VK_VENDOR_ID_AMD && + device->architecture != AMD_GCN && + !device->uma && + !allow_graphics_queue; - if (!ggml_vk_khr_cooperative_matrix_support(device->properties, driver_props, device->architecture)) { - device->coopmat_support = false; - } + if (!device->single_queue) { + const uint32_t transfer_queue_index = compute_queue_family_index == transfer_queue_family_index ? 1 : 0; + device->transfer_queue = ggml_vk_create_queue(device, transfer_queue_family_index, transfer_queue_index, { vk::PipelineStageFlagBits::eTransfer }, true); - device->integer_dot_product = device->integer_dot_product && shader_integer_dot_product_props.integerDotProduct4x8BitPackedSignedAccelerated; + device->async_use_transfer_queue = prefers_transfer_queue || (getenv("GGML_VK_ASYNC_USE_TRANSFER_QUEUE") != nullptr); + } else { + device->transfer_queue = ggml_vk_create_aliased_queue(device, device->compute_queue); - device->min_imported_host_pointer_alignment = external_memory_host_props.minImportedHostPointerAlignment; + device->async_use_transfer_queue = false; + } - device->max_workgroup_size_log2 = uint32_t(log2f(float(device->properties.limits.maxComputeWorkGroupInvocations))); + device->buffer_type = { + /* .iface = */ ggml_backend_vk_buffer_type_interface, + /* .device = */ ggml_backend_reg_dev_get(ggml_backend_vk_reg(), idx), + /* .context = */ new ggml_backend_vk_buffer_type_context{ device->name, device }, + }; - std::vector queue_family_props = device->physical_device.getQueueFamilyProperties(); + device->fence = device->device.createFence({}); - // Try to find a non-graphics compute queue and transfer-focused queues - // Allow overriding avoiding the graphics queue because it can increase performance on RADV - const bool allow_graphics_queue = (getenv("GGML_VK_ALLOW_GRAPHICS_QUEUE") != nullptr); - const vk::QueueFlagBits graphics_flag = allow_graphics_queue ? (vk::QueueFlagBits)0 : vk::QueueFlagBits::eGraphics; - const uint32_t compute_queue_family_index = ggml_vk_find_queue_family_index(queue_family_props, vk::QueueFlagBits::eCompute, graphics_flag, -1, 1); - const uint32_t transfer_queue_family_index = ggml_vk_find_queue_family_index(queue_family_props, vk::QueueFlagBits::eTransfer, vk::QueueFlagBits::eCompute | graphics_flag, compute_queue_family_index, 1); + device->idx = idx; - const float priorities[] = { 1.0f, 1.0f }; - device->single_queue = compute_queue_family_index == transfer_queue_family_index && queue_family_props[compute_queue_family_index].queueCount == 1; + device->serialize_submissions = getenv("GGML_VK_SERIALIZE_SUBMISSIONS") != nullptr; - std::vector device_queue_create_infos; - vk::DeviceCreateInfo device_create_info{}; - std::vector device_extensions; - vk::PhysicalDeviceFeatures device_features = device->physical_device.getFeatures(); + device->disable_fusion = getenv("GGML_VK_DISABLE_FUSION") != nullptr; - VkPhysicalDeviceFeatures2 device_features2; - device_features2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2; - device_features2.pNext = nullptr; - device_features2.features = (VkPhysicalDeviceFeatures)device_features; + device->add_rms_fusion = !device->disable_fusion && + device->subgroup_arithmetic && + device->vendor_id != VK_VENDOR_ID_INTEL; + device->partials_binding_alignment = + std::max(4u, (uint32_t)device->properties.limits.minStorageBufferOffsetAlignment); - VkPhysicalDeviceVulkan11Features vk11_features; - vk11_features.pNext = nullptr; - vk11_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES; - device_features2.pNext = &vk11_features; + device->mmvq_mode = 0; + if (getenv("GGML_VK_DISABLE_MMVQ")) { + device->mmvq_mode = -1; + } else if (getenv("GGML_VK_FORCE_MMVQ")) { + device->mmvq_mode = 1; + } - VkPhysicalDeviceVulkan12Features vk12_features; - vk12_features.pNext = nullptr; - vk12_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES; - vk11_features.pNext = &vk12_features; + return device; + } - last_struct = (VkBaseOutStructure *)&vk12_features; + return vk_instance.devices[idx]; +} - VkPhysicalDeviceInternallySynchronizedQueuesFeaturesKHR internally_synchronized_queues_features{}; - internally_synchronized_queues_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_INTERNALLY_SYNCHRONIZED_QUEUES_FEATURES_KHR; - internally_synchronized_queues_features.pNext = nullptr; - internally_synchronized_queues_features.internallySynchronizedQueues = VK_FALSE; +static void ggml_vk_print_gpu_info(size_t idx) { + GGML_ASSERT(idx < vk_instance.device_indices.size()); + size_t dev_num = vk_instance.device_indices[idx]; + VK_LOG_DEBUG("ggml_vk_print_gpu_info(" << dev_num << ")"); + GGML_ASSERT(vk_instance_initialized); - if (internally_sync_support) { - last_struct->pNext = (VkBaseOutStructure *)&internally_synchronized_queues_features; - last_struct = (VkBaseOutStructure *)&internally_synchronized_queues_features; - device_extensions.push_back(VK_KHR_INTERNALLY_SYNCHRONIZED_QUEUES_EXTENSION_NAME); - } + std::vector devices = vk_instance.instance.enumeratePhysicalDevices(); - VkPhysicalDevicePipelineRobustnessFeaturesEXT pl_robustness_features; - pl_robustness_features.pNext = nullptr; - pl_robustness_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PIPELINE_ROBUSTNESS_FEATURES_EXT; - pl_robustness_features.pipelineRobustness = VK_FALSE; + if (dev_num >= devices.size()) { + std::cerr << "ggml_vulkan: Device with index " << dev_num << " does not exist." << std::endl; + throw std::runtime_error("Device not found"); + } - if (pipeline_robustness) { - last_struct->pNext = (VkBaseOutStructure *)&pl_robustness_features; - last_struct = (VkBaseOutStructure *)&pl_robustness_features; - device_extensions.push_back("VK_EXT_pipeline_robustness"); - } + vk::PhysicalDevice physical_device = devices[dev_num]; + std::vector ext_props = physical_device.enumerateDeviceExtensionProperties(); - VkPhysicalDeviceMemoryPriorityFeaturesEXT memory_priority_features; - memory_priority_features.pNext = nullptr; - memory_priority_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MEMORY_PRIORITY_FEATURES_EXT; - memory_priority_features.memoryPriority = VK_FALSE; - if (device->memory_priority) { - last_struct->pNext = (VkBaseOutStructure *)&memory_priority_features; - last_struct = (VkBaseOutStructure *)&memory_priority_features; - device_extensions.push_back("VK_EXT_memory_priority"); - } + bool fp16_storage = false; + bool fp16_compute = false; + bool coopmat_support = false; + bool coopmat2_support = false; + bool coopmat2_decode_vector_support = false; + bool integer_dot_product = false; + bool bfloat16_support = false; + bool dot2_f16_support = false; + bool ocp_microscaling_extension = false; + bool shader_float8_extension = false; - VkPhysicalDeviceSubgroupSizeControlFeaturesEXT subgroup_size_control_features; - subgroup_size_control_features.pNext = nullptr; - subgroup_size_control_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SUBGROUP_SIZE_CONTROL_FEATURES_EXT; - subgroup_size_control_features.computeFullSubgroups = false; - subgroup_size_control_features.subgroupSizeControl = false; - - if (device->subgroup_size_control) { - last_struct->pNext = (VkBaseOutStructure *)&subgroup_size_control_features; - last_struct = (VkBaseOutStructure *)&subgroup_size_control_features; + for (auto properties : ext_props) { + if (strcmp("VK_KHR_16bit_storage", properties.extensionName) == 0) { + fp16_storage = true; + } else if (strcmp("VK_KHR_shader_float16_int8", properties.extensionName) == 0) { + fp16_compute = true; +#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_cooperative_matrix", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT")) { + coopmat_support = true; +#endif +#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + } else if (strcmp("VK_NV_cooperative_matrix2", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT2")) { + coopmat2_support = true; +#endif + } else if (strcmp(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME, properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_COOPMAT2_DECODE_VECTOR")) { + coopmat2_decode_vector_support = true; +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_shader_integer_dot_product", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_INTEGER_DOT_PRODUCT")) { + integer_dot_product = true; +#endif +#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) + } else if (strcmp("VK_KHR_shader_bfloat16", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_BFLOAT16")) { + bfloat16_support = true; +#endif +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) + } else if (strcmp(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME, properties.extensionName) == 0) { + ocp_microscaling_extension = true; +#endif +#if defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + } else if (strcmp(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME, properties.extensionName) == 0) { + shader_float8_extension = true; +#endif + } else if (strcmp("VK_VALVE_shader_mixed_float_dot_product", properties.extensionName) == 0 && + !getenv("GGML_VK_DISABLE_DOT2")) { + dot2_f16_support = true; } + } -#if defined(VK_KHR_cooperative_matrix) - VkPhysicalDeviceCooperativeMatrixFeaturesKHR coopmat_features; - coopmat_features.pNext = nullptr; - coopmat_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_FEATURES_KHR; - coopmat_features.cooperativeMatrix = VK_FALSE; + const vk_device_architecture device_architecture = get_device_architecture(physical_device); - if (device->coopmat_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat_features; - last_struct = (VkBaseOutStructure *)&coopmat_features; - } -#endif + const char* GGML_VK_DISABLE_F16 = getenv("GGML_VK_DISABLE_F16"); + bool force_disable_f16 = GGML_VK_DISABLE_F16 != nullptr; -#if defined(VK_NV_cooperative_matrix2) - VkPhysicalDeviceCooperativeMatrix2FeaturesNV coopmat2_features {}; - coopmat2_features.pNext = nullptr; - coopmat2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_2_FEATURES_NV; - if (coopmat2_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat2_features; - last_struct = (VkBaseOutStructure *)&coopmat2_features; - device_extensions.push_back("VK_NV_cooperative_matrix2"); - } -#endif + bool fp16 = !force_disable_f16 && fp16_storage && fp16_compute; - VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV coopmat2_decode_vector_features {}; - coopmat2_decode_vector_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV; - if (coopmat2_decode_vector_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat2_decode_vector_features; - last_struct = (VkBaseOutStructure *)&coopmat2_decode_vector_features; - device_extensions.push_back(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME); - } + vk::PhysicalDeviceProperties2 props2; + vk::PhysicalDeviceMaintenance3Properties props3; + vk::PhysicalDeviceSubgroupProperties subgroup_props; + vk::PhysicalDeviceDriverProperties driver_props; + vk::PhysicalDeviceShaderIntegerDotProductPropertiesKHR shader_integer_dot_product_props; + props2.pNext = &props3; + props3.pNext = &subgroup_props; + subgroup_props.pNext = &driver_props; -#if defined(VK_KHR_shader_bfloat16) - VkPhysicalDeviceShaderBfloat16FeaturesKHR bfloat16_features {}; - bfloat16_features.pNext = nullptr; - bfloat16_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR; - if (bfloat16_support) { - last_struct->pNext = (VkBaseOutStructure *)&bfloat16_features; - last_struct = (VkBaseOutStructure *)&bfloat16_features; - device_extensions.push_back("VK_KHR_shader_bfloat16"); - } -#endif + // Pointer to the last chain element + VkBaseOutStructure * last_struct = (VkBaseOutStructure *)&driver_props; - VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT ocp_microscaling_features {}; - ocp_microscaling_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT; - if (ocp_microscaling_extension) { - last_struct->pNext = (VkBaseOutStructure *)&ocp_microscaling_features; - last_struct = (VkBaseOutStructure *)&ocp_microscaling_features; - device_extensions.push_back(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME); - } + if (integer_dot_product) { + last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_props; + last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_props; + } - VkPhysicalDeviceShaderFloat8FeaturesEXT shader_float8_features {}; - shader_float8_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT; - if (shader_float8_extension) { - last_struct->pNext = (VkBaseOutStructure *)&shader_float8_features; - last_struct = (VkBaseOutStructure *)&shader_float8_features; - device_extensions.push_back(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME); - } + physical_device.getProperties2(&props2); - VkPhysicalDeviceMaintenance4Features maint4_features {}; - maint4_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MAINTENANCE_4_FEATURES; - if (maintenance4_support) { - last_struct->pNext = (VkBaseOutStructure *)&maint4_features; - last_struct = (VkBaseOutStructure *)&maint4_features; - device_extensions.push_back("VK_KHR_maintenance4"); - } + VkPhysicalDeviceFeatures2 device_features2; + device_features2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2; + device_features2.pNext = nullptr; - VkPhysicalDeviceShaderIntegerDotProductFeaturesKHR shader_integer_dot_product_features {}; - shader_integer_dot_product_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_INTEGER_DOT_PRODUCT_FEATURES_KHR; - if (device->integer_dot_product) { - last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_features; - last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_features; - device_extensions.push_back("VK_KHR_shader_integer_dot_product"); - } + VkPhysicalDeviceVulkan11Features vk11_features; + vk11_features.pNext = nullptr; + vk11_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES; + device_features2.pNext = &vk11_features; - VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE dot2_features {}; - dot2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE; - if (dot2_f16_support) { - last_struct->pNext = (VkBaseOutStructure *)&dot2_features; - last_struct = (VkBaseOutStructure *)&dot2_features; - device_extensions.push_back("VK_VALVE_shader_mixed_float_dot_product"); - } + VkPhysicalDeviceVulkan12Features vk12_features; + vk12_features.pNext = nullptr; + vk12_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES; + vk11_features.pNext = &vk12_features; - VkPhysicalDevicePipelineExecutablePropertiesFeaturesKHR pep_features {}; - pep_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PIPELINE_EXECUTABLE_PROPERTIES_FEATURES_KHR; - if (pipeline_executable_properties_support) { - last_struct->pNext = (VkBaseOutStructure *)&pep_features; - last_struct = (VkBaseOutStructure *)&pep_features; - device_extensions.push_back("VK_KHR_pipeline_executable_properties"); - } + // Pointer to the last chain element + last_struct = (VkBaseOutStructure *)&vk12_features; - if (device->external_memory_host) { - device_extensions.push_back("VK_EXT_external_memory_host"); - } +#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + VkPhysicalDeviceCooperativeMatrixFeaturesKHR coopmat_features; + coopmat_features.pNext = nullptr; + coopmat_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_FEATURES_KHR; + coopmat_features.cooperativeMatrix = VK_FALSE; -#if defined(VK_EXT_shader_64bit_indexing) - VkPhysicalDeviceShader64BitIndexingFeaturesEXT shader_64bit_indexing_features {}; - shader_64bit_indexing_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_64_BIT_INDEXING_FEATURES_EXT; - if (device->shader_64b_indexing) { - last_struct->pNext = (VkBaseOutStructure *)&shader_64bit_indexing_features; - last_struct = (VkBaseOutStructure *)&shader_64bit_indexing_features; - device_extensions.push_back("VK_EXT_shader_64bit_indexing"); - } + if (coopmat_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat_features; + last_struct = (VkBaseOutStructure *)&coopmat_features; + } #endif - VkPhysicalDeviceFaultFeaturesEXT fault_features {}; - fault_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FAULT_FEATURES_EXT; - if (device->device_fault) { - last_struct->pNext = (VkBaseOutStructure *)&fault_features; - last_struct = (VkBaseOutStructure *)&fault_features; - device_extensions.push_back("VK_EXT_device_fault"); - } + VkPhysicalDeviceShaderIntegerDotProductFeaturesKHR shader_integer_dot_product_features {}; + shader_integer_dot_product_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_INTEGER_DOT_PRODUCT_FEATURES_KHR; + if (integer_dot_product) { + last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_features; + last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_features; + } - vkGetPhysicalDeviceFeatures2(device->physical_device, &device_features2); +#if defined(VK_KHR_shader_bfloat16) + VkPhysicalDeviceShaderBfloat16FeaturesKHR bfloat16_features {}; + bfloat16_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR; + if (bfloat16_support) { + last_struct->pNext = (VkBaseOutStructure *)&bfloat16_features; + last_struct = (VkBaseOutStructure *)&bfloat16_features; + } +#endif - device->device_fault = device->device_fault && fault_features.deviceFault; +#if defined(VK_NV_cooperative_matrix2) + VkPhysicalDeviceCooperativeMatrix2FeaturesNV coopmat2_features {}; + coopmat2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_2_FEATURES_NV; + if (coopmat2_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat2_features; + last_struct = (VkBaseOutStructure *)&coopmat2_features; + } +#endif - device->has_internally_synchronized_queues = internally_synchronized_queues_features.internallySynchronizedQueues; + VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV coopmat2_decode_vector_features {}; + coopmat2_decode_vector_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV; + if (coopmat2_decode_vector_support) { + last_struct->pNext = (VkBaseOutStructure *)&coopmat2_decode_vector_features; + last_struct = (VkBaseOutStructure *)&coopmat2_decode_vector_features; + } - // Build queue create infos only after querying whether internally synchronized queues are enabled. - // getQueue2() later uses the same flag, so creation/retrieval must stay consistent. - vk::DeviceQueueCreateFlags queue_flags = device->has_internally_synchronized_queues ? - eInternallySynchronizedKHR : - vk::DeviceQueueCreateFlags(); + VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE dot2_features {}; + dot2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE; + if (dot2_f16_support) { + last_struct->pNext = (VkBaseOutStructure *)&dot2_features; + last_struct = (VkBaseOutStructure *)&dot2_features; + } - if (compute_queue_family_index != transfer_queue_family_index) { - device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 1, priorities}); - device_queue_create_infos.push_back({queue_flags, transfer_queue_family_index, 1, priorities + 1}); - } else if(!device->single_queue) { - device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 2, priorities}); - } else { - device_queue_create_infos.push_back({queue_flags, compute_queue_family_index, 1, priorities}); - } +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT ocp_microscaling_features {}; + ocp_microscaling_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT; + VkPhysicalDeviceShaderFloat8FeaturesEXT shader_float8_features {}; + shader_float8_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT; + if (ocp_microscaling_extension) { + last_struct->pNext = (VkBaseOutStructure *)&ocp_microscaling_features; + last_struct = (VkBaseOutStructure *)&ocp_microscaling_features; + } + if (shader_float8_extension) { + last_struct->pNext = (VkBaseOutStructure *)&shader_float8_features; + last_struct = (VkBaseOutStructure *)&shader_float8_features; + } +#endif - device->pipeline_executable_properties_support = pipeline_executable_properties_support; + vkGetPhysicalDeviceFeatures2(physical_device, &device_features2); - device->fp16 = device->fp16 && vk12_features.shaderFloat16; + fp16 = fp16 && vk12_features.shaderFloat16; #if defined(VK_KHR_shader_bfloat16) - device->bf16 = bfloat16_support && bfloat16_features.shaderBFloat16Type; + bool bf16 = bfloat16_support && bfloat16_features.shaderBFloat16Type; #else - device->bf16 = false; + bool bf16 = false; #endif - device->dot2_f16 = dot2_f16_support && dot2_features.shaderMixedFloatDotProductFloat16AccFloat32; - device->ocp_fp4 = ocp_microscaling_extension && ocp_microscaling_features.shaderFloat4 && - shader_float8_extension && shader_float8_features.shaderFloat8 && - !getenv("GGML_VK_DISABLE_OCP_FP4"); - - device->pipeline_robustness = pl_robustness_features.pipelineRobustness; + uint32_t default_subgroup_size = get_subgroup_size("", device_architecture); + const size_t subgroup_size = (default_subgroup_size != 0) ? default_subgroup_size : subgroup_props.subgroupSize; + const bool uma = props2.properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu; - device->multi_add = vk12_props.shaderRoundingModeRTEFloat16 && - device->properties.limits.maxPushConstantsSize >= sizeof(vk_op_multi_add_push_constants) && - getenv("GGML_VK_DISABLE_MULTI_ADD") == nullptr; + integer_dot_product = integer_dot_product + && shader_integer_dot_product_props.integerDotProduct4x8BitPackedSignedAccelerated + && shader_integer_dot_product_features.shaderIntegerDotProduct; - device->shader_int64 = device_features2.features.shaderInt64; - device->buffer_device_address = vk12_features.bufferDeviceAddress; - device->vulkan_memory_model = vk12_features.vulkanMemoryModel; + coopmat_support = coopmat_support +#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + && coopmat_features.cooperativeMatrix +#endif + && ggml_vk_khr_cooperative_matrix_support(props2.properties, driver_props, device_architecture); - if (device->subgroup_size_control) { - device->subgroup_min_size = subgroup_size_control_props.minSubgroupSize; - device->subgroup_max_size = subgroup_size_control_props.maxSubgroupSize; - device_extensions.push_back("VK_EXT_subgroup_size_control"); - } +#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) + coopmat2_support = coopmat2_support && + coopmat2_features.cooperativeMatrixWorkgroupScope && + coopmat2_features.cooperativeMatrixFlexibleDimensions && + coopmat2_features.cooperativeMatrixReductions && + coopmat2_features.cooperativeMatrixConversions && + coopmat2_features.cooperativeMatrixPerElementOperations && + coopmat2_features.cooperativeMatrixTensorAddressing && + coopmat2_features.cooperativeMatrixBlockLoads; +#else + coopmat2_support = false; +#endif - device->subgroup_size_control = device->subgroup_size_control && - (subgroup_size_control_props.requiredSubgroupSizeStages & vk::ShaderStageFlagBits::eCompute) && - subgroup_size_control_features.subgroupSizeControl; + coopmat2_decode_vector_support = coopmat2_decode_vector_support && coopmat2_decode_vector_features.cooperativeMatrixDecodeVector; +#if !defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) + coopmat2_decode_vector_support = false; +#endif - device->subgroup_require_full_support = subgroup_size_control_features.computeFullSubgroups; + std::string matrix_cores = coopmat2_support ? (coopmat2_decode_vector_support ? "NV_coopmat2v" : "NV_coopmat2") + : coopmat_support ? "KHR_coopmat" + : "none"; -#if defined(VK_KHR_cooperative_matrix) - device->coopmat_support = device->coopmat_support && coopmat_features.cooperativeMatrix; - device->coopmat1_fa_support = device->coopmat_support && device->subgroup_require_full_support; + bool dot2_f16 = dot2_f16_support && dot2_features.shaderMixedFloatDotProductFloat16AccFloat32; + const char *fp16_str = fp16 ? (dot2_f16 ? "dot2" : "1") : "0"; +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + const bool fp4 = ocp_microscaling_extension && ocp_microscaling_features.shaderFloat4 && + shader_float8_extension && shader_float8_features.shaderFloat8 && + !getenv("GGML_VK_DISABLE_OCP_FP4"); +#else + GGML_UNUSED(ocp_microscaling_extension); + GGML_UNUSED(shader_float8_extension); + const bool fp4 = false; #endif - if (coopmat2_support) { -#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - if (coopmat2_features.cooperativeMatrixWorkgroupScope && - coopmat2_features.cooperativeMatrixFlexibleDimensions && - coopmat2_features.cooperativeMatrixReductions && - coopmat2_features.cooperativeMatrixConversions && - coopmat2_features.cooperativeMatrixPerElementOperations && - coopmat2_features.cooperativeMatrixTensorAddressing && - coopmat2_features.cooperativeMatrixBlockLoads && - vk12_features.bufferDeviceAddress) { + std::string device_name = props2.properties.deviceName.data(); + GGML_LOG_DEBUG("ggml_vulkan: %zu = %s (%s) | uma: %d | fp16: %s | bf16: %d | fp4: %d | warp size: %zu | shared memory: %d | int dot: %d | matrix cores: %s\n", + idx, device_name.c_str(), driver_props.driverName.data(), uma, fp16_str, bf16, fp4, subgroup_size, + props2.properties.limits.maxComputeSharedMemorySize, integer_dot_product, matrix_cores.c_str()); - std::vector flexible_dimensions; - uint32_t count = 0; + if (props2.properties.deviceType == vk::PhysicalDeviceType::eCpu) { + GGML_LOG_DEBUG("ggml_vulkan: Warning: Device type is CPU. This is probably not the device you want.\n"); + } +} - PFN_vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV - _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV = - (PFN_vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV) - vk_instance.instance.getProcAddr("vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV"); +static DispatchLoaderDynamic ggml_vk_default_dispatcher_instance; - _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV(device->physical_device, &count, nullptr); +DispatchLoaderDynamic & ggml_vk_default_dispatcher() { + return ggml_vk_default_dispatcher_instance; +} - VkCooperativeMatrixFlexibleDimensionsPropertiesNV empty_prop {}; - empty_prop.sType = VK_STRUCTURE_TYPE_COOPERATIVE_MATRIX_FLEXIBLE_DIMENSIONS_PROPERTIES_NV; - flexible_dimensions.resize(count, empty_prop); +void ggml_vk_instance_init() { + if (vk_instance_initialized) { + return; + } + VK_LOG_DEBUG("ggml_vk_instance_init()"); - _vkGetPhysicalDeviceCooperativeMatrixFlexibleDimensionsPropertiesNV(device->physical_device, &count, flexible_dimensions.data()); + // See https://github.com/KhronosGroup/Vulkan-Hpp?tab=readme-ov-file#extensions--per-device-function-pointers- + ggml_vk_default_dispatcher_instance.init(vkGetInstanceProcAddr); - bool found_fp16_128 = false, - found_fp16_256 = false, - found_fp32_128 = false, - found_fp32_256 = false; - bool found_bf16_128 = false, - found_bf16_256 = false; - // need to support fp16*fp16 with fp16/fp32 accumulator, for workgroupsize 128 - // with 32x16x16 and 256 with 32x32x16. - for (auto &prop : flexible_dimensions) { - if (prop.saturatingAccumulation == VK_FALSE && - prop.scope == VK_SCOPE_WORKGROUP_KHR) { + uint32_t api_version = vk::enumerateInstanceVersion(); - if (prop.AType == VK_COMPONENT_TYPE_FLOAT16_KHR && - prop.BType == VK_COMPONENT_TYPE_FLOAT16_KHR) { + if (api_version < VK_API_VERSION_1_2) { + std::cerr << "ggml_vulkan: Error: Vulkan 1.2 required." << std::endl; + throw vk::SystemError(vk::Result::eErrorFeatureNotPresent, "Vulkan 1.2 required"); + } - if (prop.workgroupInvocations == 128 && - prop.MGranularity <= 32 && - prop.NGranularity <= 16 && - prop.KGranularity <= 16) { - if (prop.CType == VK_COMPONENT_TYPE_FLOAT16_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT16_KHR) { - found_fp16_128 = true; - } - if (prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { - found_fp32_128 = true; - } - } - if (prop.workgroupInvocations == 256 && - prop.MGranularity <= 32 && - prop.NGranularity <= 32 && - prop.KGranularity <= 16) { - if (prop.CType == VK_COMPONENT_TYPE_FLOAT16_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT16_KHR) { - found_fp16_256 = true; - } - if (prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { - found_fp32_256 = true; - } - } - } + vk::ApplicationInfo app_info{ "ggml-vulkan", 1, nullptr, 0, api_version }; -#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (prop.AType == VK_COMPONENT_TYPE_BFLOAT16_KHR && - prop.BType == VK_COMPONENT_TYPE_BFLOAT16_KHR && - prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR) { + const std::vector instance_extensions = vk::enumerateInstanceExtensionProperties(); + const bool layer_settings = ggml_vk_instance_layer_settings_available(); +#ifdef __APPLE__ + const bool portability_enumeration_ext = ggml_vk_instance_portability_enumeration_ext_available(instance_extensions); +#endif + const bool debug_utils_ext = ggml_vk_instance_debug_utils_ext_available(instance_extensions) && getenv("GGML_VK_DEBUG_MARKERS") != nullptr; + std::vector layers; - if (prop.workgroupInvocations == 128 && - prop.MGranularity <= 32 && - prop.NGranularity <= 16 && - prop.KGranularity <= 16) { - found_bf16_128 = true; - } - if (prop.workgroupInvocations == 256 && - prop.MGranularity <= 32 && - prop.NGranularity <= 32 && - prop.KGranularity <= 16) { - found_bf16_256 = true; - } - } + if (layer_settings) { + layers.push_back("VK_LAYER_KHRONOS_validation"); + } + std::vector extensions; + if (layer_settings) { + extensions.push_back("VK_EXT_layer_settings"); + } +#ifdef __APPLE__ + if (portability_enumeration_ext) { + extensions.push_back("VK_KHR_portability_enumeration"); + } #endif - } - } - if (found_fp16_128 && found_fp16_256 && - found_fp32_128 && found_fp32_256 && - coopmat2_props.cooperativeMatrixFlexibleDimensionsMaxDimension >= 512) { - device->coopmat2 = true; - device->coopmat2_bf16_support = found_bf16_128 && found_bf16_256; - device->coopmat2_decode_vector = coopmat2_decode_vector_support && coopmat2_decode_vector_features.cooperativeMatrixDecodeVector; - } - } + if (debug_utils_ext) { + extensions.push_back("VK_EXT_debug_utils"); + } + VkBool32 enable_best_practice = layer_settings; + std::vector settings = { + { + "VK_LAYER_KHRONOS_validation", + "validate_best_practices", + vk::LayerSettingTypeEXT::eBool32, + 1, + &enable_best_practice + }, + }; + vk::LayerSettingsCreateInfoEXT layer_setting_info(settings); + vk::InstanceCreateInfo instance_create_info(vk::InstanceCreateFlags{}, &app_info, layers, extensions, &layer_setting_info); +#ifdef __APPLE__ + if (portability_enumeration_ext) { + instance_create_info.flags |= vk::InstanceCreateFlagBits::eEnumeratePortabilityKHR; + } #endif - } - if (!vk11_features.storageBuffer16BitAccess) { - std::cerr << "ggml_vulkan: device " << GGML_VK_NAME << idx << " does not support 16-bit storage." << std::endl; - throw std::runtime_error("Unsupported device"); - } + vk_instance.instance = vk::createInstance(instance_create_info); + vk_instance_initialized = true; - device_extensions.push_back("VK_KHR_16bit_storage"); + if (debug_utils_ext) { + vk_instance.debug_utils_support = true; + vk_instance.pfn_vkSetDebugUtilsObjectNameEXT = (PFN_vkSetDebugUtilsObjectNameEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkSetDebugUtilsObjectNameEXT"); + vk_instance.pfn_vkQueueBeginDebugUtilsLabelEXT = (PFN_vkQueueBeginDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkQueueBeginDebugUtilsLabelEXT"); + vk_instance.pfn_vkQueueEndDebugUtilsLabelEXT = (PFN_vkQueueEndDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkQueueEndDebugUtilsLabelEXT"); + vk_instance.pfn_vkCmdBeginDebugUtilsLabelEXT = (PFN_vkCmdBeginDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdBeginDebugUtilsLabelEXT"); + vk_instance.pfn_vkCmdEndDebugUtilsLabelEXT = (PFN_vkCmdEndDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdEndDebugUtilsLabelEXT"); + vk_instance.pfn_vkCmdInsertDebugUtilsLabelEXT = (PFN_vkCmdInsertDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdInsertDebugUtilsLabelEXT"); + } -#ifdef GGML_VULKAN_VALIDATE - device_extensions.push_back("VK_KHR_shader_non_semantic_info"); -#endif + vk_perf_logger_enabled = getenv("GGML_VK_PERF_LOGGER") != nullptr; + vk_perf_logger_concurrent = getenv("GGML_VK_PERF_LOGGER_CONCURRENT") != nullptr; + vk_enable_sync_logger = getenv("GGML_VK_SYNC_LOGGER") != nullptr; + vk_memory_logger_enabled = getenv("GGML_VK_MEMORY_LOGGER") != nullptr; + const char* GGML_VK_PIPELINE_STATS = getenv("GGML_VK_PIPELINE_STATS"); + if (GGML_VK_PIPELINE_STATS != nullptr) { + vk_pipeline_stats_filter = GGML_VK_PIPELINE_STATS; + } + const char* GGML_VK_PERF_LOGGER_FREQUENCY = getenv("GGML_VK_PERF_LOGGER_FREQUENCY"); - if (device->fp16) { - device_extensions.push_back("VK_KHR_shader_float16_int8"); - } + if (GGML_VK_PERF_LOGGER_FREQUENCY != nullptr) { + vk_perf_logger_frequency = std::stoul(GGML_VK_PERF_LOGGER_FREQUENCY); + } -#if defined(VK_KHR_cooperative_matrix) - if (device->coopmat_support) { - // Query supported shapes - std::vector cm_props; + // See https://github.com/KhronosGroup/Vulkan-Hpp?tab=readme-ov-file#extensions--per-device-function-pointers- + VULKAN_HPP_DEFAULT_DISPATCHER.init(vk_instance.instance); - PFN_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR = - (PFN_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR)vkGetInstanceProcAddr(vk_instance.instance, "vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR"); - - uint32_t cm_props_num; + std::vector devices = vk_instance.instance.enumeratePhysicalDevices(); - pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(device->physical_device, &cm_props_num, nullptr); + // Emulate behavior of CUDA_VISIBLE_DEVICES for Vulkan + char * devices_env = getenv("GGML_VK_VISIBLE_DEVICES"); + if (devices_env != nullptr) { + size_t num_available_devices = devices.size(); - cm_props.resize(cm_props_num); + std::string devices(devices_env); + std::replace(devices.begin(), devices.end(), ',', ' '); - for (auto& prop : cm_props) { - prop.sType = VK_STRUCTURE_TYPE_COOPERATIVE_MATRIX_PROPERTIES_KHR; + std::stringstream ss(devices); + size_t tmp; + while (ss >> tmp) { + if(tmp >= num_available_devices) { + std::cerr << "ggml_vulkan: Invalid device index " << tmp << " in GGML_VK_VISIBLE_DEVICES." << std::endl; + throw std::runtime_error("Invalid Vulkan device index"); } + vk_instance.device_indices.push_back(tmp); + } + } else { + // If no vulkan devices are found, return early + if (devices.empty()) { + GGML_LOG_INFO("ggml_vulkan: No devices found.\n"); + return; + } - pfn_vkGetPhysicalDeviceCooperativeMatrixPropertiesKHR(device->physical_device, &cm_props_num, cm_props.data()); + // Default to using all dedicated GPUs + for (size_t i = 0; i < devices.size(); i++) { + vk::PhysicalDeviceProperties2 new_props; + vk::PhysicalDeviceDriverProperties new_driver; + vk::PhysicalDeviceIDProperties new_id; + new_props.pNext = &new_driver; + new_driver.pNext = &new_id; + devices[i].getProperties2(&new_props); - VK_LOG_DEBUG("ggml_vulkan: Cooperative Matrix Shapes: " << cm_props.size()); + if ((new_props.properties.deviceType == vk::PhysicalDeviceType::eDiscreteGpu || new_props.properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu) && ggml_vk_device_is_supported(devices[i])) { + // Check if there are two physical devices corresponding to the same GPU + // This handles the case where the same GPU appears with different drivers (e.g., RADV + AMDVLK on Linux), + // see https://github.com/ggml-org/llama.cpp/pull/7582 for original deduplication. + // MoltenVK on macOS may report the same UUID for distinct GPUs on multi-GPU cards, + // see https://github.com/KhronosGroup/MoltenVK/issues/2683. Skip when both old/new + // driver is MoltenVK + auto old_device = std::find_if( + vk_instance.device_indices.begin(), + vk_instance.device_indices.end(), + [&devices, &new_id, &new_driver](const size_t k){ + vk::PhysicalDeviceProperties2 old_props; + vk::PhysicalDeviceDriverProperties old_driver; + vk::PhysicalDeviceIDProperties old_id; + old_props.pNext = &old_driver; + old_driver.pNext = &old_id; + devices[k].getProperties2(&old_props); - for (auto& prop : cm_props) { - VK_LOG_DEBUG("ggml_vulkan: M: " << prop.MSize << " N: " << prop.NSize << " K: " << prop.KSize << " A: " << vk::to_string((vk::ComponentTypeKHR)prop.AType) << " B: " << vk::to_string((vk::ComponentTypeKHR)prop.BType) << " C: " << vk::to_string((vk::ComponentTypeKHR)prop.CType) << " Result: " << vk::to_string((vk::ComponentTypeKHR)prop.ResultType) << " saturatingAccumulation: " << prop.saturatingAccumulation << " scope: " << vk::to_string((vk::ScopeKHR)prop.scope)); + bool same_uuid = std::equal(std::begin(old_id.deviceUUID), std::end(old_id.deviceUUID), std::begin(new_id.deviceUUID)); + same_uuid = same_uuid || ( + old_id.deviceLUIDValid && new_id.deviceLUIDValid && + std::equal(std::begin(old_id.deviceLUID), std::end(old_id.deviceLUID), std::begin(new_id.deviceLUID)) + ); + bool both_molten_vk = (new_driver.driverID == vk::DriverId::eMoltenvk && old_driver.driverID == vk::DriverId::eMoltenvk); - if ((vk::ComponentTypeKHR)prop.AType == vk::ComponentTypeKHR::eFloat16 && - (vk::ComponentTypeKHR)prop.BType == vk::ComponentTypeKHR::eFloat16 && - (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup - ) { - if ((vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eFloat32 && - (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eFloat32) { - // coopmat sizes not set yet - if (device->coopmat_m == 0) { - device->coopmat_acc_f32_support = true; - device->coopmat_m = prop.MSize; - device->coopmat_n = prop.NSize; - device->coopmat_k = prop.KSize; - } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { - // Only enable if shape is identical - device->coopmat_acc_f32_support = true; - } - if (prop.MSize == 16 && prop.NSize == 16 && prop.KSize == 16) { - device->coopmat_support_16x16x16_f32acc = true; - } - } else if ((vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eFloat16 && - (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eFloat16) { - // coopmat sizes not set yet - if (device->coopmat_m == 0) { - device->coopmat_acc_f16_support = true; - device->coopmat_m = prop.MSize; - device->coopmat_n = prop.NSize; - device->coopmat_k = prop.KSize; - } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { - // Only enable if shape is identical - device->coopmat_acc_f16_support = true; - } - if (prop.MSize == 16 && prop.NSize == 16 && prop.KSize == 16) { - device->coopmat_support_16x16x16_f16acc = true; - } + return same_uuid && !both_molten_vk; } - } else if ((vk::ComponentTypeKHR)prop.AType == vk::ComponentTypeKHR::eSint8 && - (vk::ComponentTypeKHR)prop.BType == vk::ComponentTypeKHR::eSint8 && - (vk::ComponentTypeKHR)prop.CType == vk::ComponentTypeKHR::eSint32 && - (vk::ComponentTypeKHR)prop.ResultType == vk::ComponentTypeKHR::eSint32 && - (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup && - device->coopmat_int_m == 0 - ) { - device->coopmat_int_support = true; - device->coopmat_int_m = prop.MSize; - device->coopmat_int_n = prop.NSize; - device->coopmat_int_k = prop.KSize; - } -#if defined(VK_KHR_shader_bfloat16) && defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - if (prop.AType == VK_COMPONENT_TYPE_BFLOAT16_KHR && - prop.BType == VK_COMPONENT_TYPE_BFLOAT16_KHR && - prop.CType == VK_COMPONENT_TYPE_FLOAT32_KHR && - prop.ResultType == VK_COMPONENT_TYPE_FLOAT32_KHR && - (vk::ScopeKHR)prop.scope == vk::ScopeKHR::eSubgroup - ) { - // coopmat sizes not set yet - if (device->coopmat_m == 0) { - device->coopmat_bf16_support = true; - device->coopmat_m = prop.MSize; - device->coopmat_n = prop.NSize; - device->coopmat_k = prop.KSize; - } else if (device->coopmat_m == prop.MSize && device->coopmat_n == prop.NSize && device->coopmat_k == prop.KSize) { - // Only enable if shape is identical - device->coopmat_bf16_support = true; + ); + if (old_device == vk_instance.device_indices.end()) { + vk_instance.device_indices.push_back(i); + } else { + // There can be two physical devices corresponding to the same GPU if there are 2 different drivers + // This can cause error when splitting layers aross the devices, need to keep only 1 + VK_LOG_DEBUG("Device " << i << " and device " << *old_device << " have the same deviceUUID"); + + vk::PhysicalDeviceProperties2 old_props; + vk::PhysicalDeviceDriverProperties old_driver; + old_props.pNext = &old_driver; + devices[*old_device].getProperties2(&old_props); + + std::map driver_priorities {}; + int old_priority = std::numeric_limits::max(); + int new_priority = std::numeric_limits::max(); + + // Check https://registry.khronos.org/vulkan/specs/1.3-extensions/man/html/VkDriverId.html for the list of driver id + // Smaller number -> higher priority + switch (old_props.properties.vendorID) { + case VK_VENDOR_ID_AMD: + driver_priorities[vk::DriverId::eMesaRadv] = 1; + driver_priorities[vk::DriverId::eAmdOpenSource] = 2; + driver_priorities[vk::DriverId::eAmdProprietary] = 3; + break; + case VK_VENDOR_ID_INTEL: + driver_priorities[vk::DriverId::eIntelOpenSourceMESA] = 1; + driver_priorities[vk::DriverId::eIntelProprietaryWindows] = 2; + break; + case VK_VENDOR_ID_NVIDIA: + driver_priorities[vk::DriverId::eNvidiaProprietary] = 1; +#if defined(VK_API_VERSION_1_3) && VK_HEADER_VERSION >= 235 + driver_priorities[vk::DriverId::eMesaNvk] = 2; +#endif + break; + case VK_VENDOR_ID_QUALCOMM: + driver_priorities[vk::DriverId::eQualcommProprietary] = 1; + driver_priorities[vk::DriverId::eMesaTurnip] = 2; + break; + } + driver_priorities[vk::DriverId::eMesaDozen] = 100; + + if (driver_priorities.count(old_driver.driverID)) { + old_priority = driver_priorities[old_driver.driverID]; + } + if (driver_priorities.count(new_driver.driverID)) { + new_priority = driver_priorities[new_driver.driverID]; + } + + if (new_priority < old_priority) { + auto r = std::remove(vk_instance.device_indices.begin(), vk_instance.device_indices.end(), *old_device); + vk_instance.device_indices.erase(r, vk_instance.device_indices.end()); + vk_instance.device_indices.push_back(i); + + VK_LOG_DEBUG("Prioritize device " << i << " driver " << new_driver.driverName << " over device " << *old_device << " driver " << old_driver.driverName); + } + else { + VK_LOG_DEBUG("Prioritize device " << *old_device << " driver " << old_driver.driverName << " over device " << i << " driver " << new_driver.driverName << std::endl); } } -#endif } + } - if (device->coopmat_m == 0 || !device->coopmat_acc_f32_support) { - // No suitable matmul mode found - GGML_LOG_DEBUG("ggml_vulkan: WARNING: No suitable matrix core mode found. Disabling matrix cores.\n"); - device->coopmat_support = false; - } - if (getenv("GGML_VK_DISABLE_BFLOAT16")) { - device->coopmat_bf16_support = false; + // If no GPUs found, fall back to the first non-CPU device. + // If only CPU devices are available, return without devices. + if (vk_instance.device_indices.empty()) { + for (size_t i = 0; i < devices.size(); i++) { + if (devices[i].getProperties().deviceType != vk::PhysicalDeviceType::eCpu) { + vk_instance.device_indices.push_back(i); + break; + } } } - if (device->coopmat_support) { - device_extensions.push_back("VK_KHR_cooperative_matrix"); - } -#if defined(VK_KHR_shader_bfloat16) - if (device->coopmat_bf16_support) { - device_extensions.push_back("VK_KHR_shader_bfloat16"); + if (vk_instance.device_indices.empty()) { + GGML_LOG_INFO("ggml_vulkan: No devices found.\n"); + return; } -#endif -#endif - device->name = GGML_VK_NAME + std::to_string(idx); + } + GGML_LOG_DEBUG("ggml_vulkan: Found %zu Vulkan devices:\n", vk_instance.device_indices.size()); - device_create_info - .setFlags(vk::DeviceCreateFlags()) - .setQueueCreateInfos(device_queue_create_infos) - .setPEnabledExtensionNames(device_extensions); - device_create_info.setPNext(&device_features2); - device->device = device->physical_device.createDevice(device_create_info); + for (size_t i = 0; i < vk_instance.device_indices.size(); i++) { + vk::PhysicalDevice vkdev = devices[vk_instance.device_indices[i]]; + std::vector extensionprops = vkdev.enumerateDeviceExtensionProperties(); - if (device->device_fault) { - device->pfn_vkGetDeviceFaultInfoEXT = (PFN_vkGetDeviceFaultInfoEXT) - vkGetDeviceProcAddr(device->device, "vkGetDeviceFaultInfoEXT"); + bool membudget_supported = false; + for (const auto & ext : extensionprops) { + if (strcmp(VK_EXT_MEMORY_BUDGET_EXTENSION_NAME, ext.extensionName) == 0) { + membudget_supported = true; + break; + } } - // Queues - device->compute_queue = ggml_vk_create_queue(device, compute_queue_family_index, 0, { vk::PipelineStageFlagBits::eComputeShader | vk::PipelineStageFlagBits::eTransfer }, false); + vk_instance.device_supports_membudget.push_back(membudget_supported); - // Shaders - // Disable matmul tile sizes early if performance low or not supported - for (uint32_t i = 0; i < GGML_TYPE_COUNT; ++i) { - switch (device->vendor_id) { -#ifndef GGML_VULKAN_RUN_TESTS - case VK_VENDOR_ID_AMD: - device->mul_mat_l[i] = device->coopmat_support && device->driver_id != vk::DriverId::eAmdProprietary; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = true; - device->mul_mat_id_l[i] = false; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = true; - break; - case VK_VENDOR_ID_INTEL: { - // Current Windows driver does not expose BF16 support. - // We only want to use l_warptile if coopmat is available - const bool use_l_warptile = (i == GGML_TYPE_BF16) ? (device->coopmat_bf16_support && device->coopmat_support) : device->coopmat_support; - device->mul_mat_l[i] = use_l_warptile; - device->mul_mat_id_l[i] = use_l_warptile; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = true; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = true; - break; - } - case VK_VENDOR_ID_APPLE: - device->mul_mat_l[i] = false; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = false; - device->mul_mat_id_l[i] = false; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = false; - break; - case VK_VENDOR_ID_QUALCOMM: - device->mul_mat_l[i] = false; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = true; - device->mul_mat_id_l[i] = false; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = true; - break; -#endif - default: - device->mul_mat_l[i] = true; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = true; - device->mul_mat_id_l[i] = true; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = true; - break; - } - -#if VK_HEADER_VERSION >= 287 - // Honeykrisp driver for Asahi Linux doesn't report VK_VENDOR_ID_APPLE. - // Check for Honeykrisp driver and force same configuration as the VK_VENDOR_ID_APPLE case. - if (device->driver_id == vk::DriverId::eMesaHoneykrisp) { - device->mul_mat_l[i] = false; - device->mul_mat_m[i] = true; - device->mul_mat_s[i] = false; - device->mul_mat_id_l[i] = false; - device->mul_mat_id_m[i] = true; - device->mul_mat_id_s[i] = false; - } -#endif - - device->mul_mat_l_int[i] = device->mul_mat_l[i]; - device->mul_mat_m_int[i] = device->mul_mat_m[i]; - device->mul_mat_s_int[i] = device->mul_mat_s[i]; - device->mul_mat_id_l_int[i] = device->mul_mat_id_l[i]; - device->mul_mat_id_m_int[i] = device->mul_mat_id_m[i]; - device->mul_mat_id_s_int[i] = device->mul_mat_id_s[i]; - } + ggml_vk_print_gpu_info(i); + } +} +void ggml_vk_init(ggml_backend_vk_context * ctx, size_t idx) { + VK_LOG_DEBUG("ggml_vk_init(" << ctx->name << ", " << idx << ")"); + ggml_vk_instance_init(); + GGML_ASSERT(idx < vk_instance.device_indices.size()); - std::vector dsl_binding; - std::vector dsl_binding_flags; - for (uint32_t i = 0; i < MAX_PARAMETER_COUNT; i++) { - dsl_binding.push_back({i, vk::DescriptorType::eStorageBuffer, 1, vk::ShaderStageFlagBits::eCompute}); - dsl_binding_flags.push_back({}); - } + ctx->name = GGML_VK_NAME + std::to_string(idx); - vk::DescriptorSetLayoutBindingFlagsCreateInfo dslbfci = { dsl_binding_flags }; + ctx->device = ggml_vk_get_device(idx); - vk::DescriptorSetLayoutCreateInfo descriptor_set_layout_create_info( - {}, - dsl_binding); - descriptor_set_layout_create_info.setPNext(&dslbfci); - device->dsl = device->device.createDescriptorSetLayout(descriptor_set_layout_create_info); + ctx->semaphore_idx = 0; + ctx->event_idx = 0; - ggml_vk_load_shaders(device); + ctx->prealloc_size_x = 0; + ctx->prealloc_size_y = 0; + ctx->prealloc_size_split_k = 0; + // Fixed size of 1KB, for deterministic behavior + ctx->prealloc_size_add_rms_partials = 1024; - // Prefer a dedicated transfer queue on AMD dGPUs (non-GCN) when graphics queue use is disabled. - const bool prefers_transfer_queue = - device->vendor_id == VK_VENDOR_ID_AMD && - device->architecture != AMD_GCN && - !device->uma && - !allow_graphics_queue; + ctx->fence = ctx->device->device.createFence({}); + ctx->almost_ready_fence = ctx->device->device.createFence({}); - if (!device->single_queue) { - const uint32_t transfer_queue_index = compute_queue_family_index == transfer_queue_family_index ? 1 : 0; - device->transfer_queue = ggml_vk_create_queue(device, transfer_queue_family_index, transfer_queue_index, { vk::PipelineStageFlagBits::eTransfer }, true); + ctx->compute_cmd_pool.init(ctx->device, ctx->device->compute_queue.get()); + if (ctx->device->async_use_transfer_queue) { + vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eTimeline, 0 }; + vk::SemaphoreCreateInfo ci{}; + ci.setPNext(&tci); + ctx->transfer_semaphore.s = ctx->device->device.createSemaphore(ci); + ctx->transfer_semaphore.value = 0; - device->async_use_transfer_queue = prefers_transfer_queue || (getenv("GGML_VK_ASYNC_USE_TRANSFER_QUEUE") != nullptr); - } else { - device->transfer_queue = ggml_vk_create_aliased_queue(device, device->compute_queue); + ctx->transfer_cmd_pool.init(ctx->device, ctx->device->transfer_queue.get()); + } - device->async_use_transfer_queue = false; - } + if (vk_perf_logger_enabled) { + ctx->perf_logger = std::unique_ptr(new vk_perf_logger()); + } - device->buffer_type = { - /* .iface = */ ggml_backend_vk_buffer_type_interface, - /* .device = */ ggml_backend_reg_dev_get(ggml_backend_vk_reg(), idx), - /* .context = */ new ggml_backend_vk_buffer_type_context{ device->name, device }, - }; +#ifdef GGML_VULKAN_CHECK_RESULTS + const char* skip_checks = getenv("GGML_VULKAN_SKIP_CHECKS"); + vk_skip_checks = (skip_checks == NULL ? 0 : atoi(skip_checks)); + const char* output_tensor = getenv("GGML_VULKAN_OUTPUT_TENSOR"); + vk_output_tensor = (output_tensor == NULL ? 0 : atoi(output_tensor)); +#endif +} - device->fence = device->device.createFence({}); +vk_pipeline ggml_vk_get_to_fp16(ggml_backend_vk_context * ctx, ggml_type type) { + VK_LOG_DEBUG("ggml_vk_get_to_fp16()"); + switch (type) { + case GGML_TYPE_F32: + case GGML_TYPE_Q1_0: + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ2_XXS: + case GGML_TYPE_IQ2_XS: + case GGML_TYPE_IQ2_S: + case GGML_TYPE_IQ3_XXS: + case GGML_TYPE_IQ3_S: + case GGML_TYPE_IQ4_XS: + case GGML_TYPE_IQ4_NL: + case GGML_TYPE_MXFP4: + case GGML_TYPE_NVFP4: + case GGML_TYPE_TQ1_0: + case GGML_TYPE_TQ2_0: + break; + default: + return nullptr; + } - device->idx = idx; + return ctx->device->pipeline_dequant[type]; +} - device->serialize_submissions = getenv("GGML_VK_SERIALIZE_SUBMISSIONS") != nullptr; +static vk_pipeline ggml_vk_get_dequantize_mul_mat_vec(ggml_backend_vk_context * ctx, ggml_type a_type, ggml_type b_type, uint32_t num_cols, uint32_t m, uint32_t k) { + VK_LOG_DEBUG("ggml_vk_get_dequantize_mul_mat_vec()"); + GGML_ASSERT(b_type == GGML_TYPE_F32 || b_type == GGML_TYPE_F16 || b_type == GGML_TYPE_Q8_1); + GGML_ASSERT(num_cols >= 1 && num_cols <= mul_mat_vec_max_cols); - device->disable_fusion = getenv("GGML_VK_DISABLE_FUSION") != nullptr; + if (b_type == GGML_TYPE_Q8_1) { + switch (a_type) { + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + case GGML_TYPE_MXFP4: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ4_XS: + break; + default: + return nullptr; + } + } - device->add_rms_fusion = !device->disable_fusion && - device->subgroup_arithmetic && - device->vendor_id != VK_VENDOR_ID_INTEL; - device->partials_binding_alignment = - std::max(4u, (uint32_t)device->properties.limits.minStorageBufferOffsetAlignment); + switch (a_type) { + case GGML_TYPE_F32: + case GGML_TYPE_F16: + case GGML_TYPE_BF16: + case GGML_TYPE_Q1_0: + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ2_XXS: + case GGML_TYPE_IQ2_XS: + case GGML_TYPE_IQ2_S: + case GGML_TYPE_IQ3_XXS: + case GGML_TYPE_IQ3_S: + case GGML_TYPE_IQ4_XS: + case GGML_TYPE_IQ4_NL: + case GGML_TYPE_MXFP4: + case GGML_TYPE_NVFP4: + case GGML_TYPE_TQ1_0: + case GGML_TYPE_TQ2_0: + break; + default: + return nullptr; + } - device->mmvq_mode = 0; - if (getenv("GGML_VK_DISABLE_MMVQ")) { - device->mmvq_mode = -1; - } else if (getenv("GGML_VK_FORCE_MMVQ")) { - device->mmvq_mode = 1; + // heuristic to choose workgroup size + uint32_t dmmv_wg = DMMV_WG_SIZE_SUBGROUP; + if ((ctx->device->vendor_id == VK_VENDOR_ID_NVIDIA && ctx->device->architecture != vk_device_architecture::NVIDIA_PRE_TURING) || ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { + // Prefer larger workgroups when M is small, to spread the work out more + // and keep more SMs busy. + // q6_k seems to prefer small workgroup size even for "medium" values of M. + if (a_type == GGML_TYPE_Q6_K) { + if (m < 4096 && k >= 1024) { + dmmv_wg = DMMV_WG_SIZE_LARGE; + } + } else { + if (m <= 8192 && k >= 1024) { + dmmv_wg = DMMV_WG_SIZE_LARGE; + } } + } - return device; + if (b_type == GGML_TYPE_Q8_1) { + if (ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { + dmmv_wg = DMMV_WG_SIZE_SUBGROUP; + } + return ctx->device->pipeline_dequant_mul_mat_vec_q8_1_f32[dmmv_wg][a_type][num_cols-1]; } - return vk_instance.devices[idx]; + return b_type == GGML_TYPE_F32 ? ctx->device->pipeline_dequant_mul_mat_vec_f32_f32[dmmv_wg][a_type][num_cols-1] : ctx->device->pipeline_dequant_mul_mat_vec_f16_f32[dmmv_wg][a_type][num_cols-1]; } -static void ggml_vk_print_gpu_info(size_t idx) { - GGML_ASSERT(idx < vk_instance.device_indices.size()); - size_t dev_num = vk_instance.device_indices[idx]; - VK_LOG_DEBUG("ggml_vk_print_gpu_info(" << dev_num << ")"); - GGML_ASSERT(vk_instance_initialized); - - std::vector devices = vk_instance.instance.enumeratePhysicalDevices(); +static vk_pipeline ggml_vk_get_dequantize_mul_mat_vec_id(ggml_backend_vk_context * ctx, ggml_type a_type, ggml_type b_type, uint32_t m, uint32_t k) { + VK_LOG_DEBUG("ggml_vk_get_dequantize_mul_mat_vec_id()"); + GGML_ASSERT(b_type == GGML_TYPE_F32 || b_type == GGML_TYPE_Q8_1); - if (dev_num >= devices.size()) { - std::cerr << "ggml_vulkan: Device with index " << dev_num << " does not exist." << std::endl; - throw std::runtime_error("Device not found"); + if (b_type == GGML_TYPE_Q8_1) { + switch (a_type) { + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + case GGML_TYPE_MXFP4: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ4_XS: + break; + default: + return nullptr; + } } - vk::PhysicalDevice physical_device = devices[dev_num]; - std::vector ext_props = physical_device.enumerateDeviceExtensionProperties(); + switch (a_type) { + case GGML_TYPE_F32: + case GGML_TYPE_F16: + case GGML_TYPE_BF16: + case GGML_TYPE_Q1_0: + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q3_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + case GGML_TYPE_Q6_K: + case GGML_TYPE_IQ1_S: + case GGML_TYPE_IQ1_M: + case GGML_TYPE_IQ2_XXS: + case GGML_TYPE_IQ2_XS: + case GGML_TYPE_IQ2_S: + case GGML_TYPE_IQ3_XXS: + case GGML_TYPE_IQ3_S: + case GGML_TYPE_IQ4_XS: + case GGML_TYPE_IQ4_NL: + case GGML_TYPE_MXFP4: + case GGML_TYPE_NVFP4: + case GGML_TYPE_TQ1_0: + case GGML_TYPE_TQ2_0: + break; + default: + return nullptr; + } - bool fp16_storage = false; - bool fp16_compute = false; - bool coopmat_support = false; - bool coopmat2_support = false; - bool coopmat2_decode_vector_support = false; - bool integer_dot_product = false; - bool bfloat16_support = false; - bool dot2_f16_support = false; - bool ocp_microscaling_extension = false; - bool shader_float8_extension = false; + // heuristic to choose workgroup size + uint32_t dmmv_wg = DMMV_WG_SIZE_SUBGROUP; + if ((ctx->device->vendor_id == VK_VENDOR_ID_NVIDIA && ctx->device->architecture != vk_device_architecture::NVIDIA_PRE_TURING) || ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { + // Prefer larger workgroups when M is small, to spread the work out more + // and keep more SMs busy. + // q6_k seems to prefer small workgroup size even for "medium" values of M. + if (a_type == GGML_TYPE_Q6_K) { + if (m < 4096 && k >= 1024) { + dmmv_wg = DMMV_WG_SIZE_LARGE; + } + } else { + if (m <= 8192 && k >= 1024) { + dmmv_wg = DMMV_WG_SIZE_LARGE; + } + } + } - for (auto properties : ext_props) { - if (strcmp("VK_KHR_16bit_storage", properties.extensionName) == 0) { - fp16_storage = true; - } else if (strcmp("VK_KHR_shader_float16_int8", properties.extensionName) == 0) { - fp16_compute = true; -#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_cooperative_matrix", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT")) { - coopmat_support = true; -#endif -#if defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - } else if (strcmp("VK_NV_cooperative_matrix2", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT2")) { - coopmat2_support = true; -#endif - } else if (strcmp(VK_NV_COOPERATIVE_MATRIX_DECODE_VECTOR_EXTENSION_NAME, properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_COOPMAT2_DECODE_VECTOR")) { - coopmat2_decode_vector_support = true; -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_shader_integer_dot_product", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_INTEGER_DOT_PRODUCT")) { - integer_dot_product = true; -#endif -#if defined(GGML_VULKAN_BFLOAT16_GLSLC_SUPPORT) - } else if (strcmp("VK_KHR_shader_bfloat16", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_BFLOAT16")) { - bfloat16_support = true; -#endif -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) - } else if (strcmp(VK_EXT_SHADER_OCP_MICROSCALING_TYPES_EXTENSION_NAME, properties.extensionName) == 0) { - ocp_microscaling_extension = true; -#endif -#if defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - } else if (strcmp(VK_EXT_SHADER_FLOAT8_EXTENSION_NAME, properties.extensionName) == 0) { - shader_float8_extension = true; -#endif - } else if (strcmp("VK_VALVE_shader_mixed_float_dot_product", properties.extensionName) == 0 && - !getenv("GGML_VK_DISABLE_DOT2")) { - dot2_f16_support = true; + if (b_type == GGML_TYPE_Q8_1) { + if (ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { + dmmv_wg = DMMV_WG_SIZE_SUBGROUP; } + return ctx->device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[dmmv_wg][a_type]; } - const vk_device_architecture device_architecture = get_device_architecture(physical_device); + return ctx->device->pipeline_dequant_mul_mat_vec_id_f32[dmmv_wg][a_type]; +} - const char* GGML_VK_DISABLE_F16 = getenv("GGML_VK_DISABLE_F16"); - bool force_disable_f16 = GGML_VK_DISABLE_F16 != nullptr; +vk_subbuffer ggml_vk_tensor_subbuffer( + const ggml_backend_vk_context * ctx, const ggml_tensor * tensor, bool allow_misalign) { - bool fp16 = !force_disable_f16 && fp16_storage && fp16_compute; + vk_buffer buffer = nullptr; + size_t offset = 0; + if (ctx->device->uma) { + ggml_vk_host_get(ctx->device, tensor->data, buffer, offset); + } + if (!buffer) { + auto buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; + buffer = buf_ctx->dev_buffer; + offset = vk_tensor_offset(tensor) + tensor->view_offs; + } + GGML_ASSERT(buffer != nullptr); - vk::PhysicalDeviceProperties2 props2; - vk::PhysicalDeviceMaintenance3Properties props3; - vk::PhysicalDeviceSubgroupProperties subgroup_props; - vk::PhysicalDeviceDriverProperties driver_props; - vk::PhysicalDeviceShaderIntegerDotProductPropertiesKHR shader_integer_dot_product_props; - props2.pNext = &props3; - props3.pNext = &subgroup_props; - subgroup_props.pNext = &driver_props; + size_t size = ggml_nbytes(tensor); - // Pointer to the last chain element - VkBaseOutStructure * last_struct = (VkBaseOutStructure *)&driver_props; + const size_t descriptor_offset = ggml_vk_descriptor_offset( + offset, ctx->device->properties.limits.minStorageBufferOffsetAlignment, ggml_type_size(tensor->type)); + const size_t misalign_bytes = offset - descriptor_offset; + // The shader must support misaligned offsets when indexing into the buffer + GGML_ASSERT(allow_misalign || misalign_bytes == 0); + offset = descriptor_offset; + size += misalign_bytes; - if (integer_dot_product) { - last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_props; - last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_props; + return vk_subbuffer{buffer, offset, size}; +} + +static vk_command_buffer* ggml_vk_get_or_create_cmd_buffer(vk_device& device, vk_command_pool& pool) { + for (auto& cmd_buffer : pool.cmd_buffers) { + if (!cmd_buffer.in_use) { + cmd_buffer.use_counter++; + cmd_buffer.in_use = true; + return &cmd_buffer; + } } + return ggml_vk_create_cmd_buffer(device, pool); +} - physical_device.getProperties2(&props2); +static vk_submission ggml_vk_begin_submission(vk_device& device, vk_command_pool& p, bool one_time = true) { + vk_submission s; + s.buffer = ggml_vk_get_or_create_cmd_buffer(device, p); + if (one_time) { + s.buffer->buf.begin({ vk::CommandBufferUsageFlagBits::eOneTimeSubmit }); + } else { + s.buffer->buf.begin({ vk::CommandBufferUsageFlags{} }); + } - VkPhysicalDeviceFeatures2 device_features2; - device_features2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2; - device_features2.pNext = nullptr; + return s; +} - VkPhysicalDeviceVulkan11Features vk11_features; - vk11_features.pNext = nullptr; - vk11_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_FEATURES; - device_features2.pNext = &vk11_features; +void ggml_vk_cmd_label_begin(vk::CommandBuffer buf, const char * name) { + vk::DebugUtilsLabelEXT label = {}; + label.pLabelName = name; + label.color = std::array{1.0f, 1.0f, 1.0f, 1.0f}; + vk_instance.pfn_vkCmdBeginDebugUtilsLabelEXT(buf, reinterpret_cast(&label)); +} - VkPhysicalDeviceVulkan12Features vk12_features; - vk12_features.pNext = nullptr; - vk12_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES; - vk11_features.pNext = &vk12_features; +void ggml_vk_ctx_end(vk_context& ctx) { + VK_LOG_DEBUG("ggml_vk_ctx_end(" << ctx << ", " << ctx->seqs.size() << ")"); + if (ctx->s == nullptr) { + return; + } - // Pointer to the last chain element - last_struct = (VkBaseOutStructure *)&vk12_features; + // close open labels so this buffer is balanced; reopened in ggml_vk_ctx_begin + if (vk_instance.debug_utils_support) { + for (size_t i = 0; i < ctx->debug_labels.size(); i++) { + vk_instance.pfn_vkCmdEndDebugUtilsLabelEXT(ctx->s->buffer->buf); + } + // the enclosing per-command-buffer region + vk_instance.pfn_vkCmdEndDebugUtilsLabelEXT(ctx->s->buffer->buf); + } -#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - VkPhysicalDeviceCooperativeMatrixFeaturesKHR coopmat_features; - coopmat_features.pNext = nullptr; - coopmat_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_FEATURES_KHR; - coopmat_features.cooperativeMatrix = VK_FALSE; + ctx->s->buffer->buf.end(); + ctx->s = nullptr; +} - if (coopmat_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat_features; - last_struct = (VkBaseOutStructure *)&coopmat_features; +void ggml_vk_ctx_begin(vk_device& device, vk_context& subctx) { + VK_LOG_DEBUG("ggml_vk_ctx_begin(" << device->name << ")"); + if (subctx->s != nullptr) { + ggml_vk_ctx_end(subctx); } -#endif - VkPhysicalDeviceShaderIntegerDotProductFeaturesKHR shader_integer_dot_product_features {}; - shader_integer_dot_product_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_INTEGER_DOT_PRODUCT_FEATURES_KHR; - if (integer_dot_product) { - last_struct->pNext = (VkBaseOutStructure *)&shader_integer_dot_product_features; - last_struct = (VkBaseOutStructure *)&shader_integer_dot_product_features; - } + subctx->seqs.push_back({ ggml_vk_begin_submission(device, *subctx->p) }); + subctx->s = subctx->seqs[subctx->seqs.size() - 1].data(); -#if defined(VK_KHR_shader_bfloat16) - VkPhysicalDeviceShaderBfloat16FeaturesKHR bfloat16_features {}; - bfloat16_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_BFLOAT16_FEATURES_KHR; - if (bfloat16_support) { - last_struct->pNext = (VkBaseOutStructure *)&bfloat16_features; - last_struct = (VkBaseOutStructure *)&bfloat16_features; - } -#endif + if (vk_instance.debug_utils_support) { + // outermost region, one per command buffer, so the gaps between submits stand out + const std::string name = "submit " + std::to_string(device->debug_cmdbuf_idx++); + ggml_vk_cmd_label_begin(subctx->s->buffer->buf, name.c_str()); -#if defined(VK_NV_cooperative_matrix2) - VkPhysicalDeviceCooperativeMatrix2FeaturesNV coopmat2_features {}; - coopmat2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_2_FEATURES_NV; - if (coopmat2_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat2_features; - last_struct = (VkBaseOutStructure *)&coopmat2_features; + // reopen labels left open when the previous command buffer was submitted + for (const std::string & label : subctx->debug_labels) { + ggml_vk_cmd_label_begin(subctx->s->buffer->buf, label.c_str()); + } } -#endif +} - VkPhysicalDeviceCooperativeMatrixDecodeVectorFeaturesNV coopmat2_decode_vector_features {}; - coopmat2_decode_vector_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_COOPERATIVE_MATRIX_DECODE_VECTOR_FEATURES_NV; - if (coopmat2_decode_vector_support) { - last_struct->pNext = (VkBaseOutStructure *)&coopmat2_decode_vector_features; - last_struct = (VkBaseOutStructure *)&coopmat2_decode_vector_features; +vk_context ggml_vk_get_compute_ctx(ggml_backend_vk_context * ctx) { + vk_context result; + if (!ctx->compute_ctx.expired()) { + result = ctx->compute_ctx.lock(); + } else { + result = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); + + ctx->compute_ctx = result; + ggml_vk_ctx_begin(ctx->device, result); } - VkPhysicalDeviceShaderMixedFloatDotProductFeaturesVALVE dot2_features {}; - dot2_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_MIXED_FLOAT_DOT_PRODUCT_FEATURES_VALVE; - if (dot2_f16_support) { - last_struct->pNext = (VkBaseOutStructure *)&dot2_features; - last_struct = (VkBaseOutStructure *)&dot2_features; + if (ctx->device->async_use_transfer_queue && ctx->transfer_semaphore_last_submitted < ctx->transfer_semaphore.value) { + result->s->wait_semaphores.push_back(ctx->transfer_semaphore); + ctx->transfer_semaphore_last_submitted = ctx->transfer_semaphore.value; } -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - VkPhysicalDeviceShaderOCPMicroscalingTypesFeaturesEXT ocp_microscaling_features {}; - ocp_microscaling_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_OCP_MICROSCALING_TYPES_FEATURES_EXT; - VkPhysicalDeviceShaderFloat8FeaturesEXT shader_float8_features {}; - shader_float8_features.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SHADER_FLOAT8_FEATURES_EXT; - if (ocp_microscaling_extension) { - last_struct->pNext = (VkBaseOutStructure *)&ocp_microscaling_features; - last_struct = (VkBaseOutStructure *)&ocp_microscaling_features; + return result; +} + +vk_context ggml_vk_get_transfer_ctx(ggml_backend_vk_context * ctx) { + vk_context result; + if (!ctx->transfer_ctx.expired()) { + result = ctx->transfer_ctx.lock(); + } else { + result = ggml_vk_create_context(ctx, ctx->transfer_cmd_pool); + + ctx->transfer_ctx = result; + ggml_vk_ctx_begin(ctx->device, result); } - if (shader_float8_extension) { - last_struct->pNext = (VkBaseOutStructure *)&shader_float8_features; - last_struct = (VkBaseOutStructure *)&shader_float8_features; + + return result; +} + +bool ggml_vk_submit_transfer_ctx(ggml_backend_vk_context * ctx) { + if (!ctx->device->async_use_transfer_queue || ctx->transfer_ctx.expired()) { + return false; } -#endif - vkGetPhysicalDeviceFeatures2(physical_device, &device_features2); + vk_context cpy_ctx = ctx->transfer_ctx.lock(); + ggml_vk_ctx_end(cpy_ctx); - fp16 = fp16 && vk12_features.shaderFloat16; + for (auto& cpy : cpy_ctx->in_memcpys) { + memcpy(cpy.dst, cpy.src, cpy.n); + } -#if defined(VK_KHR_shader_bfloat16) - bool bf16 = bfloat16_support && bfloat16_features.shaderBFloat16Type; -#else - bool bf16 = false; -#endif + ctx->transfer_semaphore.value++; + cpy_ctx->seqs.back().back().signal_semaphores.push_back(ctx->transfer_semaphore); - uint32_t default_subgroup_size = get_subgroup_size("", device_architecture); - const size_t subgroup_size = (default_subgroup_size != 0) ? default_subgroup_size : subgroup_props.subgroupSize; - const bool uma = props2.properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu; + ggml_vk_submit(cpy_ctx, {}); + ctx->transfer_ctx.reset(); + return true; +} - integer_dot_product = integer_dot_product - && shader_integer_dot_product_props.integerDotProduct4x8BitPackedSignedAccelerated - && shader_integer_dot_product_features.shaderIntegerDotProduct; +size_t ggml_vk_align_size(size_t width, size_t align) { + VK_LOG_DEBUG("ggml_vk_align_size(" << width << ", " << align << ")"); + return CEIL_DIV(width, align) * align; +} - coopmat_support = coopmat_support -#if defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - && coopmat_features.cooperativeMatrix -#endif - && ggml_vk_khr_cooperative_matrix_support(props2.properties, driver_props, device_architecture); +void deferred_memcpy(void * dst, const void * src, size_t size, std::vector* memcpys) { + if (memcpys == nullptr) { + memcpy(dst, src, size); + } else { + memcpys->emplace_back(dst, src, size); + } +} -#if defined(VK_NV_cooperative_matrix2) && defined(GGML_VULKAN_COOPMAT2_GLSLC_SUPPORT) - coopmat2_support = coopmat2_support && - coopmat2_features.cooperativeMatrixWorkgroupScope && - coopmat2_features.cooperativeMatrixFlexibleDimensions && - coopmat2_features.cooperativeMatrixReductions && - coopmat2_features.cooperativeMatrixConversions && - coopmat2_features.cooperativeMatrixPerElementOperations && - coopmat2_features.cooperativeMatrixTensorAddressing && - coopmat2_features.cooperativeMatrixBlockLoads; -#else - coopmat2_support = false; -#endif +void deferred_memset(void * dst, uint32_t val, size_t size, std::vector* memsets) { + if (memsets == nullptr) { + memset(dst, val, size); + } else { + memsets->emplace_back(dst, val, size); + } +} - coopmat2_decode_vector_support = coopmat2_decode_vector_support && coopmat2_decode_vector_features.cooperativeMatrixDecodeVector; -#if !defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) - coopmat2_decode_vector_support = false; -#endif +static uint32_t ggml_vk_guess_split_k(ggml_backend_vk_context * ctx, uint32_t m, uint32_t n, uint32_t k, bool disable_split_k, const vk_pipeline& pipeline) { + VK_LOG_DEBUG("ggml_vk_guess_split_k(" << m << ", " << n << ", " << k << ", " << disable_split_k << ")"); - std::string matrix_cores = coopmat2_support ? (coopmat2_decode_vector_support ? "NV_coopmat2v" : "NV_coopmat2") - : coopmat_support ? "KHR_coopmat" - : "none"; + if (disable_split_k) { + return 1; + } - bool dot2_f16 = dot2_f16_support && dot2_features.shaderMixedFloatDotProductFloat16AccFloat32; - const char *fp16_str = fp16 ? (dot2_f16 ? "dot2" : "1") : "0"; -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - const bool fp4 = ocp_microscaling_extension && ocp_microscaling_features.shaderFloat4 && - shader_float8_extension && shader_float8_features.shaderFloat8 && - !getenv("GGML_VK_DISABLE_OCP_FP4"); -#else - GGML_UNUSED(ocp_microscaling_extension); - GGML_UNUSED(shader_float8_extension); - const bool fp4 = false; -#endif + uint32_t split_k = 1; + if (ctx->device->shader_core_count != 0 && n >= pipeline->wg_denoms[1]) { + // If k is 'large' and the SMs will fill less than halfway, use split_k. + uint32_t m_tiles = CEIL_DIV(m, pipeline->wg_denoms[0]); + uint32_t n_tiles = CEIL_DIV(n, pipeline->wg_denoms[1]); - std::string device_name = props2.properties.deviceName.data(); - GGML_LOG_DEBUG("ggml_vulkan: %zu = %s (%s) | uma: %d | fp16: %s | bf16: %d | fp4: %d | warp size: %zu | shared memory: %d | int dot: %d | matrix cores: %s\n", - idx, device_name.c_str(), driver_props.driverName.data(), uma, fp16_str, bf16, fp4, subgroup_size, - props2.properties.limits.maxComputeSharedMemorySize, integer_dot_product, matrix_cores.c_str()); + if (k >= 2048) { + if (m_tiles * n_tiles <= ctx->device->shader_core_count / 2) { + split_k = ctx->device->shader_core_count / (m_tiles * n_tiles); + } else if (m_tiles * n_tiles <= ctx->device->shader_core_count * 2 / 3) { + split_k = 3; + } + // Cap the split at 8x. Unless k is huge this is a lot of overhead. + split_k = std::min(split_k, 8u); - if (props2.properties.deviceType == vk::PhysicalDeviceType::eCpu) { - GGML_LOG_DEBUG("ggml_vulkan: Warning: Device type is CPU. This is probably not the device you want.\n"); + // ggml_vk_matmul will align the splits to be a multiple of 256. + // If this rounded up size would cause the last split to be empty, + // then reduce the split count. + while (true) { + if (split_k == 1) { + break; + } + uint32_t k_split = CEIL_DIV(k, split_k); + k_split = ROUNDUP_POW2(k_split, 256); + if (k_split * (split_k - 1) < k) { + break; + } + split_k--; + } + } } + + return split_k; } -static bool ggml_vk_instance_layer_settings_available(); -static bool ggml_vk_instance_portability_enumeration_ext_available(const std::vector& instance_extensions); -static bool ggml_vk_instance_debug_utils_ext_available(const std::vector & instance_extensions); -static bool ggml_vk_device_is_supported(const vk::PhysicalDevice & vkdev); +void ggml_vk_matmul( + ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline& pipeline, + vk_subbuffer&& a, vk_subbuffer&& b, vk_subbuffer&& d, vk_subbuffer&& split_k_buffer, + uint32_t m, uint32_t n, uint32_t k, uint32_t stride_a, uint32_t stride_b, uint32_t stride_d, + uint32_t batch_stride_a, uint32_t batch_stride_b, uint32_t batch_stride_d, + uint32_t split_k, uint32_t batch, uint32_t ne02, uint32_t ne12, uint32_t broadcast2, uint32_t broadcast3, + uint32_t padded_n) { + VK_LOG_DEBUG("ggml_vk_matmul(a: (" << a.buffer->buffer << ", " << a.offset << ", " << a.size << "), b: (" << b.buffer->buffer << ", " << b.offset << ", " << b.size << "), d: (" << d.buffer->buffer << ", " << d.offset << ", " << d.size << "), split_k: (" << (split_k_buffer.buffer != nullptr ? split_k_buffer.buffer->buffer : VK_NULL_HANDLE) << ", " << split_k_buffer.offset << ", " << split_k_buffer.size << "), m: " << m << ", n: " << n << ", k: " << k << ", stride_a: " << stride_a << ", stride_b: " << stride_b << ", stride_d: " << stride_d << ", batch_stride_a: " << batch_stride_a << ", batch_stride_b: " << batch_stride_b << ", batch_stride_d: " << batch_stride_d << ", split_k: " << split_k << ", batch: " << batch << ", ne02: " << ne02 << ", ne12: " << ne12 << ", broadcast2: " << broadcast2 << ", broadcast3: " << broadcast3 << ", padded_n: " << padded_n << ")"); + if (split_k == 1) { + ggml_pipeline_request_descriptor_sets(ctx, pipeline, CEIL_DIV(batch, ctx->device->properties.limits.maxComputeWorkGroupCount[2])); -static DispatchLoaderDynamic ggml_vk_default_dispatcher_instance; -DispatchLoaderDynamic & ggml_vk_default_dispatcher() { - return ggml_vk_default_dispatcher_instance; -} + uint32_t base_work_group_z = 0; + while (base_work_group_z < batch) { + uint32_t groups_z = std::min(batch - base_work_group_z, ctx->device->properties.limits.maxComputeWorkGroupCount[2]); -static void ggml_vk_instance_init() { - if (vk_instance_initialized) { + const vk_mat_mat_push_constants pc = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, base_work_group_z, batch, k, ne02, ne12, broadcast2, broadcast3, padded_n }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, d }, pc, { m, n, groups_z }); + base_work_group_z += groups_z; + } return; } - VK_LOG_DEBUG("ggml_vk_instance_init()"); - // See https://github.com/KhronosGroup/Vulkan-Hpp?tab=readme-ov-file#extensions--per-device-function-pointers- - ggml_vk_default_dispatcher_instance.init(vkGetInstanceProcAddr); + if (ctx->prealloc_split_k_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - uint32_t api_version = vk::enumerateInstanceVersion(); + GGML_ASSERT(batch_stride_d == m * n); - if (api_version < VK_API_VERSION_1_2) { - std::cerr << "ggml_vulkan: Error: Vulkan 1.2 required." << std::endl; - throw vk::SystemError(vk::Result::eErrorFeatureNotPresent, "Vulkan 1.2 required"); - } + // Round the split size up to a multiple of 256 (k-quant alignment) + uint32_t k_split = CEIL_DIV(k, split_k); + k_split = ROUNDUP_POW2(k_split, 256); - vk::ApplicationInfo app_info{ "ggml-vulkan", 1, nullptr, 0, api_version }; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, CEIL_DIV(batch, ctx->device->properties.limits.maxComputeWorkGroupCount[2])); - const std::vector instance_extensions = vk::enumerateInstanceExtensionProperties(); - const bool layer_settings = ggml_vk_instance_layer_settings_available(); -#ifdef __APPLE__ - const bool portability_enumeration_ext = ggml_vk_instance_portability_enumeration_ext_available(instance_extensions); -#endif - const bool debug_utils_ext = ggml_vk_instance_debug_utils_ext_available(instance_extensions) && getenv("GGML_VK_DEBUG_MARKERS") != nullptr; - std::vector layers; + uint32_t base_work_group_z = 0; + while (base_work_group_z < batch) { + uint32_t groups_z = std::min(batch - base_work_group_z, ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - if (layer_settings) { - layers.push_back("VK_LAYER_KHRONOS_validation"); + const vk_mat_mat_push_constants pc1 = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, base_work_group_z, batch, k_split, ne02, ne12, broadcast2, broadcast3, padded_n }; + // Make sure enough workgroups get assigned for split k to work + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, split_k_buffer }, pc1, { (CEIL_DIV(m, pipeline->wg_denoms[0]) * pipeline->wg_denoms[0]) * split_k, n, groups_z }); + base_work_group_z += groups_z; } - std::vector extensions; - if (layer_settings) { - extensions.push_back("VK_EXT_layer_settings"); + ggml_vk_sync_buffers(ctx, subctx); + const std::array pc2 = { (uint32_t)(m * n * batch), split_k }; + ggml_vk_dispatch_pipeline(ctx, subctx, ctx->device->pipeline_matmul_split_k_reduce, { split_k_buffer, d }, pc2, { m * n * batch, 1, 1 }); + ctx->prealloc_split_k_need_sync = true; +} + +static bool ggml_vk_get_mul_mat_mat_f16acc(ggml_backend_vk_context * ctx, ggml_type src0_type, ggml_type src1_type, ggml_prec prec) { + if (src0_type == GGML_TYPE_F32 || src0_type == GGML_TYPE_BF16) return false; + if (src1_type == GGML_TYPE_Q8_1) return false; + if (src0_type == GGML_TYPE_F16) { + return prec == GGML_PREC_DEFAULT && ctx->device->fp16 && !(ctx->device->coopmat_support && !ctx->device->coopmat_acc_f16_support); } -#ifdef __APPLE__ - if (portability_enumeration_ext) { - extensions.push_back("VK_KHR_portability_enumeration"); + // quant types + if (ctx->device->coopmat2) { + return prec == GGML_PREC_DEFAULT; } -#endif - if (debug_utils_ext) { - extensions.push_back("VK_EXT_debug_utils"); + if (ctx->device->coopmat_support) { + return ctx->device->fp16 && ctx->device->coopmat_acc_f16_support && prec == GGML_PREC_DEFAULT; } - VkBool32 enable_best_practice = layer_settings; - std::vector settings = { - { - "VK_LAYER_KHRONOS_validation", - "validate_best_practices", - vk::LayerSettingTypeEXT::eBool32, - 1, - &enable_best_practice - }, - }; - vk::LayerSettingsCreateInfoEXT layer_setting_info(settings); - vk::InstanceCreateInfo instance_create_info(vk::InstanceCreateFlags{}, &app_info, layers, extensions, &layer_setting_info); -#ifdef __APPLE__ - if (portability_enumeration_ext) { - instance_create_info.flags |= vk::InstanceCreateFlagBits::eEnumeratePortabilityKHR; + return ctx->device->fp16 && prec == GGML_PREC_DEFAULT; +} + +static const std::vector* ggml_vk_get_mul_mat_mat_pipeline_map( + ggml_backend_vk_context * ctx, ggml_type src0_type, ggml_type src1_type, ggml_prec prec, bool mul_mat_id = false) { + bool f16acc = ggml_vk_get_mul_mat_mat_f16acc(ctx, src0_type, src1_type, prec); + vk_matmul_pipeline_key key{src0_type, src1_type, mul_mat_id, f16acc}; + auto it = ctx->device->pipeline_matmul.find(key); + if (it == ctx->device->pipeline_matmul.end() || it->second.empty()) { + // Try without f16acc + if (f16acc) { + key.f16acc = false; + it = ctx->device->pipeline_matmul.find(key); + if (it != ctx->device->pipeline_matmul.end() && !it->second.empty()) return &it->second; + } + return nullptr; } -#endif + return &it->second; +} - vk_instance.instance = vk::createInstance(instance_create_info); - vk_instance_initialized = true; +static vk_pipeline ggml_vk_guess_matmul_pipeline_map(ggml_backend_vk_context * ctx, + const std::vector& configs, + uint32_t m, uint32_t n, bool aligned, bool mul_mat_id) { + auto& selector = mul_mat_id ? ctx->device->matmul_id_tile_selector : ctx->device->matmul_tile_selector; + uint32_t idx = selector(m, n, 0, ctx->device->shader_core_count, configs); + if (idx >= configs.size()) idx = (uint32_t)configs.size() - 1; + return (aligned && configs[idx].aligned) ? configs[idx].aligned : configs[idx].unaligned; +} - if (debug_utils_ext) { - vk_instance.debug_utils_support = true; - vk_instance.pfn_vkSetDebugUtilsObjectNameEXT = (PFN_vkSetDebugUtilsObjectNameEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkSetDebugUtilsObjectNameEXT"); - vk_instance.pfn_vkQueueBeginDebugUtilsLabelEXT = (PFN_vkQueueBeginDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkQueueBeginDebugUtilsLabelEXT"); - vk_instance.pfn_vkQueueEndDebugUtilsLabelEXT = (PFN_vkQueueEndDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkQueueEndDebugUtilsLabelEXT"); - vk_instance.pfn_vkCmdBeginDebugUtilsLabelEXT = (PFN_vkCmdBeginDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdBeginDebugUtilsLabelEXT"); - vk_instance.pfn_vkCmdEndDebugUtilsLabelEXT = (PFN_vkCmdEndDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdEndDebugUtilsLabelEXT"); - vk_instance.pfn_vkCmdInsertDebugUtilsLabelEXT = (PFN_vkCmdInsertDebugUtilsLabelEXT) vkGetInstanceProcAddr(vk_instance.instance, "vkCmdInsertDebugUtilsLabelEXT"); - } +static uint32_t ggml_vk_guess_matmul_pipeline_align_map(ggml_backend_vk_context * ctx, + const std::vector& configs, + uint32_t m, uint32_t n, bool mul_mat_id) { + auto& selector = mul_mat_id ? ctx->device->matmul_id_tile_selector : ctx->device->matmul_tile_selector; + uint32_t idx = selector(m, n, 0, ctx->device->shader_core_count, configs); + if (idx >= configs.size()) idx = (uint32_t)configs.size() - 1; + return configs[idx].align; +} - vk_perf_logger_enabled = getenv("GGML_VK_PERF_LOGGER") != nullptr; - vk_perf_logger_concurrent = getenv("GGML_VK_PERF_LOGGER_CONCURRENT") != nullptr; - vk_enable_sync_logger = getenv("GGML_VK_SYNC_LOGGER") != nullptr; - vk_memory_logger_enabled = getenv("GGML_VK_MEMORY_LOGGER") != nullptr; - const char* GGML_VK_PIPELINE_STATS = getenv("GGML_VK_PIPELINE_STATS"); - if (GGML_VK_PIPELINE_STATS != nullptr) { - vk_pipeline_stats_filter = GGML_VK_PIPELINE_STATS; - } - const char* GGML_VK_PERF_LOGGER_FREQUENCY = getenv("GGML_VK_PERF_LOGGER_FREQUENCY"); - - if (GGML_VK_PERF_LOGGER_FREQUENCY != nullptr) { - vk_perf_logger_frequency = std::stoul(GGML_VK_PERF_LOGGER_FREQUENCY); - } +static void ggml_vk_matmul_id( + ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline& pipeline, + vk_subbuffer&& a, vk_subbuffer&& b, vk_subbuffer&& d, vk_subbuffer&& ids, const vk_subbuffer & expert_count_buf, + uint32_t m, uint32_t n, uint32_t k, uint32_t stride_a, uint32_t stride_b, uint32_t stride_d, + uint32_t batch_stride_a, uint32_t batch_stride_b, uint32_t batch_stride_d, + uint32_t n_as, uint32_t nei0, uint32_t nei1, uint32_t nbi1, uint32_t ne11, + bool hoist_row_ids) { + VK_LOG_DEBUG("ggml_vk_matmul_id(a: (" << a.buffer->buffer << ", " << a.offset << ", " << a.size << "), b: (" << b.buffer->buffer << ", " << b.offset << ", " << b.size << "), d: (" << d.buffer->buffer << ", " << d.offset << ", " << d.size << "), ids: (" << ids.buffer->buffer << ", " << ids.offset << ", " << ids.size << "), expert_count: (" << expert_count_buf.buffer->buffer << ", " << expert_count_buf.offset << ", " << expert_count_buf.size << "), " << + "m: " << m << ", n: " << n << ", k: " << k << ", stride_a: " << stride_a << ", stride_b: " << stride_b << ", stride_d: " << stride_d << ", " << + "batch_stride_a: " << batch_stride_a << ", batch_stride_b: " << batch_stride_b << ", batch_stride_d: " << batch_stride_d << ", " << + "n_as: " << n_as << ", nei0: " << nei0 << ", nei1: " << nei1 << ", nbi1: " << nbi1 << ", ne11: " << ne11 << ")"); + const vk_mat_mat_id_push_constants pc = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, + nei0, nei1, nbi1, ne11, n_as, uint32_t(hoist_row_ids) }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, d, ids, expert_count_buf }, pc, { m, nei1, n_as }); +} - // See https://github.com/KhronosGroup/Vulkan-Hpp?tab=readme-ov-file#extensions--per-device-function-pointers- - VULKAN_HPP_DEFAULT_DISPATCHER.init(vk_instance.instance); +bool ggml_vk_dim01_contiguous(const ggml_tensor * tensor) { + return + tensor->nb[0] == ggml_type_size(tensor->type) && + tensor->nb[1] == (tensor->nb[0]*tensor->ne[0])/ggml_blck_size(tensor->type) && + (tensor->ne[3] == 1 || tensor->nb[3] == tensor->nb[2]*tensor->ne[2]); +} - std::vector devices = vk_instance.instance.enumeratePhysicalDevices(); +vk_pipeline ggml_vk_get_cpy_pipeline(ggml_backend_vk_context * ctx, const ggml_tensor * src, const ggml_tensor * dst, ggml_type to) { - // Emulate behavior of CUDA_VISIBLE_DEVICES for Vulkan - char * devices_env = getenv("GGML_VK_VISIBLE_DEVICES"); - if (devices_env != nullptr) { - size_t num_available_devices = devices.size(); + // Choose "contiguous copy" shader if src/dst are contiguous + bool contig = ggml_is_contiguous(src) && (!dst || ggml_is_contiguous(dst)); - std::string devices(devices_env); - std::replace(devices.begin(), devices.end(), ',', ' '); + // Use optimized "transpose" shader if src dim1 is the innermost dimension. + bool transpose = dst && src->nb[1] == ggml_type_size(to) && ggml_are_same_shape(dst, src); - std::stringstream ss(devices); - size_t tmp; - while (ss >> tmp) { - if(tmp >= num_available_devices) { - std::cerr << "ggml_vulkan: Invalid device index " << tmp << " in GGML_VK_VISIBLE_DEVICES." << std::endl; - throw std::runtime_error("Invalid Vulkan device index"); - } - vk_instance.device_indices.push_back(tmp); - } - } else { - // If no vulkan devices are found, return early - if (devices.empty()) { - GGML_LOG_INFO("ggml_vulkan: No devices found.\n"); - return; + if (transpose && src->type == to) { + if (ggml_type_size(to) == 4) { + return ctx->device->pipeline_cpy_transpose_32; + } else if (ggml_type_size(to) == 2) { + return ctx->device->pipeline_cpy_transpose_16; } + } - // Default to using all dedicated GPUs - for (size_t i = 0; i < devices.size(); i++) { - vk::PhysicalDeviceProperties2 new_props; - vk::PhysicalDeviceDriverProperties new_driver; - vk::PhysicalDeviceIDProperties new_id; - new_props.pNext = &new_driver; - new_driver.pNext = &new_id; - devices[i].getProperties2(&new_props); - - if ((new_props.properties.deviceType == vk::PhysicalDeviceType::eDiscreteGpu || new_props.properties.deviceType == vk::PhysicalDeviceType::eIntegratedGpu) && ggml_vk_device_is_supported(devices[i])) { - // Check if there are two physical devices corresponding to the same GPU - // This handles the case where the same GPU appears with different drivers (e.g., RADV + AMDVLK on Linux), - // see https://github.com/ggml-org/llama.cpp/pull/7582 for original deduplication. - // MoltenVK on macOS may report the same UUID for distinct GPUs on multi-GPU cards, - // see https://github.com/KhronosGroup/MoltenVK/issues/2683. Skip when both old/new - // driver is MoltenVK - auto old_device = std::find_if( - vk_instance.device_indices.begin(), - vk_instance.device_indices.end(), - [&devices, &new_id, &new_driver](const size_t k){ - vk::PhysicalDeviceProperties2 old_props; - vk::PhysicalDeviceDriverProperties old_driver; - vk::PhysicalDeviceIDProperties old_id; - old_props.pNext = &old_driver; - old_driver.pNext = &old_id; - devices[k].getProperties2(&old_props); - - bool same_uuid = std::equal(std::begin(old_id.deviceUUID), std::end(old_id.deviceUUID), std::begin(new_id.deviceUUID)); - same_uuid = same_uuid || ( - old_id.deviceLUIDValid && new_id.deviceLUIDValid && - std::equal(std::begin(old_id.deviceLUID), std::end(old_id.deviceLUID), std::begin(new_id.deviceLUID)) - ); - bool both_molten_vk = (new_driver.driverID == vk::DriverId::eMoltenvk && old_driver.driverID == vk::DriverId::eMoltenvk); - - return same_uuid && !both_molten_vk; - } - ); - if (old_device == vk_instance.device_indices.end()) { - vk_instance.device_indices.push_back(i); - } else { - // There can be two physical devices corresponding to the same GPU if there are 2 different drivers - // This can cause error when splitting layers aross the devices, need to keep only 1 - VK_LOG_DEBUG("Device " << i << " and device " << *old_device << " have the same deviceUUID"); - - vk::PhysicalDeviceProperties2 old_props; - vk::PhysicalDeviceDriverProperties old_driver; - old_props.pNext = &old_driver; - devices[*old_device].getProperties2(&old_props); - - std::map driver_priorities {}; - int old_priority = std::numeric_limits::max(); - int new_priority = std::numeric_limits::max(); - - // Check https://registry.khronos.org/vulkan/specs/1.3-extensions/man/html/VkDriverId.html for the list of driver id - // Smaller number -> higher priority - switch (old_props.properties.vendorID) { - case VK_VENDOR_ID_AMD: - driver_priorities[vk::DriverId::eMesaRadv] = 1; - driver_priorities[vk::DriverId::eAmdOpenSource] = 2; - driver_priorities[vk::DriverId::eAmdProprietary] = 3; - break; - case VK_VENDOR_ID_INTEL: - driver_priorities[vk::DriverId::eIntelOpenSourceMESA] = 1; - driver_priorities[vk::DriverId::eIntelProprietaryWindows] = 2; - break; - case VK_VENDOR_ID_NVIDIA: - driver_priorities[vk::DriverId::eNvidiaProprietary] = 1; -#if defined(VK_API_VERSION_1_3) && VK_HEADER_VERSION >= 235 - driver_priorities[vk::DriverId::eMesaNvk] = 2; -#endif - break; - case VK_VENDOR_ID_QUALCOMM: - driver_priorities[vk::DriverId::eQualcommProprietary] = 1; - driver_priorities[vk::DriverId::eMesaTurnip] = 2; - break; - } - driver_priorities[vk::DriverId::eMesaDozen] = 100; - - if (driver_priorities.count(old_driver.driverID)) { - old_priority = driver_priorities[old_driver.driverID]; - } - if (driver_priorities.count(new_driver.driverID)) { - new_priority = driver_priorities[new_driver.driverID]; - } - - if (new_priority < old_priority) { - auto r = std::remove(vk_instance.device_indices.begin(), vk_instance.device_indices.end(), *old_device); - vk_instance.device_indices.erase(r, vk_instance.device_indices.end()); - vk_instance.device_indices.push_back(i); + // Same, for a 0<->2 swap: src dim2 is the innermost dimension. + bool transpose02 = dst && !contig && src->nb[2] == ggml_type_size(to) && + ggml_is_contiguous(dst) && ggml_are_same_shape(dst, src); - VK_LOG_DEBUG("Prioritize device " << i << " driver " << new_driver.driverName << " over device " << *old_device << " driver " << old_driver.driverName); - } - else { - VK_LOG_DEBUG("Prioritize device " << *old_device << " driver " << old_driver.driverName << " over device " << i << " driver " << new_driver.driverName << std::endl); - } - } - } + if (transpose02 && src->type == to) { + if (ggml_type_size(to) == 4) { + return ctx->device->pipeline_cpy_transpose_02_32; + } else if (ggml_type_size(to) == 2) { + return ctx->device->pipeline_cpy_transpose_02_16; } + } - // If no GPUs found, fall back to the first non-CPU device. - // If only CPU devices are available, return without devices. - if (vk_instance.device_indices.empty()) { - for (size_t i = 0; i < devices.size(); i++) { - if (devices[i].getProperties().deviceType != vk::PhysicalDeviceType::eCpu) { - vk_instance.device_indices.push_back(i); - break; - } - } + if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_F32) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f32_f32; + } else { + return ctx->device->pipeline_cpy_f32_f32; } - - if (vk_instance.device_indices.empty()) { - GGML_LOG_INFO("ggml_vulkan: No devices found.\n"); - return; + } + if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_F16) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f32_f16; + } else { + return ctx->device->pipeline_cpy_f32_f16; } } - GGML_LOG_DEBUG("ggml_vulkan: Found %zu Vulkan devices:\n", vk_instance.device_indices.size()); - - for (size_t i = 0; i < vk_instance.device_indices.size(); i++) { - vk::PhysicalDevice vkdev = devices[vk_instance.device_indices[i]]; - std::vector extensionprops = vkdev.enumerateDeviceExtensionProperties(); - - bool membudget_supported = false; - for (const auto & ext : extensionprops) { - if (strcmp(VK_EXT_MEMORY_BUDGET_EXTENSION_NAME, ext.extensionName) == 0) { - membudget_supported = true; - break; - } + if (src->type == GGML_TYPE_F16 && to == GGML_TYPE_F16) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f16_f16; + } else { + return ctx->device->pipeline_cpy_f16_f16; } - - vk_instance.device_supports_membudget.push_back(membudget_supported); - - ggml_vk_print_gpu_info(i); } -} - -static void ggml_vk_init(ggml_backend_vk_context * ctx, size_t idx) { - VK_LOG_DEBUG("ggml_vk_init(" << ctx->name << ", " << idx << ")"); - ggml_vk_instance_init(); - GGML_ASSERT(idx < vk_instance.device_indices.size()); - - ctx->name = GGML_VK_NAME + std::to_string(idx); - - ctx->device = ggml_vk_get_device(idx); - - ctx->semaphore_idx = 0; - ctx->event_idx = 0; - - ctx->prealloc_size_x = 0; - ctx->prealloc_size_y = 0; - ctx->prealloc_size_split_k = 0; - // Fixed size of 1KB, for deterministic behavior - ctx->prealloc_size_add_rms_partials = 1024; - - ctx->fence = ctx->device->device.createFence({}); - ctx->almost_ready_fence = ctx->device->device.createFence({}); - - ctx->compute_cmd_pool.init(ctx->device, ctx->device->compute_queue.get()); - if (ctx->device->async_use_transfer_queue) { - vk::SemaphoreTypeCreateInfo tci{ vk::SemaphoreType::eTimeline, 0 }; - vk::SemaphoreCreateInfo ci{}; - ci.setPNext(&tci); - ctx->transfer_semaphore.s = ctx->device->device.createSemaphore(ci); - ctx->transfer_semaphore.value = 0; - - ctx->transfer_cmd_pool.init(ctx->device, ctx->device->transfer_queue.get()); + if (src->type == GGML_TYPE_F16 && to == GGML_TYPE_F32) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f16_f32; + } else { + return ctx->device->pipeline_cpy_f16_f32; + } } - - if (vk_perf_logger_enabled) { - ctx->perf_logger = std::unique_ptr(new vk_perf_logger()); + if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_BF16) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f32_bf16; + } else { + return ctx->device->pipeline_cpy_f32_bf16; + } } - -#ifdef GGML_VULKAN_CHECK_RESULTS - const char* skip_checks = getenv("GGML_VULKAN_SKIP_CHECKS"); - vk_skip_checks = (skip_checks == NULL ? 0 : atoi(skip_checks)); - const char* output_tensor = getenv("GGML_VULKAN_OUTPUT_TENSOR"); - vk_output_tensor = (output_tensor == NULL ? 0 : atoi(output_tensor)); -#endif -} - -static vk_pipeline ggml_vk_get_to_fp16(ggml_backend_vk_context * ctx, ggml_type type) { - VK_LOG_DEBUG("ggml_vk_get_to_fp16()"); - switch (type) { - case GGML_TYPE_F32: - case GGML_TYPE_Q1_0: - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - case GGML_TYPE_IQ2_XXS: - case GGML_TYPE_IQ2_XS: - case GGML_TYPE_IQ2_S: - case GGML_TYPE_IQ3_XXS: - case GGML_TYPE_IQ3_S: - case GGML_TYPE_IQ4_XS: - case GGML_TYPE_IQ4_NL: - case GGML_TYPE_MXFP4: - case GGML_TYPE_NVFP4: - case GGML_TYPE_TQ2_0: - break; - default: - return nullptr; - } - - return ctx->device->pipeline_dequant[type]; -} - -static vk_matmul_pipeline ggml_vk_get_mul_mat_mat_pipeline(ggml_backend_vk_context * ctx, ggml_type src0_type, ggml_type src1_type, ggml_prec prec) { - VK_LOG_DEBUG("ggml_vk_get_mul_mat_mat_pipeline(" << ggml_type_name(src0_type) << ", " << ggml_type_name(src1_type) << ", " << prec << ")"); - if (src0_type == GGML_TYPE_F32 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_f32; - } - if (src0_type == GGML_TYPE_F32 && src1_type == GGML_TYPE_F16) { - return ctx->device->pipeline_matmul_f32_f16; - } - if (src0_type == GGML_TYPE_BF16 && src1_type == GGML_TYPE_BF16) { - return ctx->device->pipeline_matmul_bf16; - } - if (prec == GGML_PREC_DEFAULT && ctx->device->fp16 && !(ctx->device->coopmat_support && !ctx->device->coopmat_acc_f16_support)) { - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_f16_f32.f16acc; - } - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F16) { - return ctx->device->pipeline_matmul_f16.f16acc; - } - } else { - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_f16_f32.f32acc; - } - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F16) { - return ctx->device->pipeline_matmul_f16.f32acc; + if (src->type == GGML_TYPE_BF16 && to == GGML_TYPE_F32) { + if (contig) { + return ctx->device->pipeline_contig_cpy_bf16_f32; + } else { + return ctx->device->pipeline_cpy_bf16_f32; } } - - // MMQ - if (src1_type == GGML_TYPE_Q8_1) { - vk_matmul_pipeline pipelines = ctx->device->pipeline_dequant_mul_mat_mat_q8_1[src0_type].f32acc; - - if (pipelines->is_empty()) { - return nullptr; + if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_I32) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f32_i32; + } else { + return ctx->device->pipeline_cpy_f32_i32; } - - return pipelines; } - - if (src1_type != GGML_TYPE_F32 && !ctx->device->coopmat2) { - return nullptr; + if (src->type == GGML_TYPE_I32 && to == GGML_TYPE_F32) { + if (contig) { + return ctx->device->pipeline_contig_cpy_i32_f32; + } else { + return ctx->device->pipeline_cpy_i32_f32; + } } - - switch (src0_type) { + if (src->type == GGML_TYPE_F32) { + switch (to) { case GGML_TYPE_Q1_0: case GGML_TYPE_Q2_0: case GGML_TYPE_Q4_0: @@ -7721,69 +5902,15 @@ static vk_matmul_pipeline ggml_vk_get_mul_mat_mat_pipeline(ggml_backend_vk_conte case GGML_TYPE_Q5_0: case GGML_TYPE_Q5_1: case GGML_TYPE_Q8_0: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - case GGML_TYPE_IQ2_XXS: - case GGML_TYPE_IQ2_XS: - case GGML_TYPE_IQ2_S: - case GGML_TYPE_IQ3_XXS: - case GGML_TYPE_IQ3_S: - case GGML_TYPE_IQ4_XS: case GGML_TYPE_IQ4_NL: - case GGML_TYPE_MXFP4: - case GGML_TYPE_NVFP4: - case GGML_TYPE_TQ2_0: - break; + return ctx->device->pipeline_cpy_f32_quant[to]; default: - return nullptr; - } - - if (ctx->device->coopmat2) { - assert(src1_type == GGML_TYPE_F16); - return prec == GGML_PREC_DEFAULT ? ctx->device->pipeline_dequant_mul_mat_mat_f16[src0_type].f16acc : ctx->device->pipeline_dequant_mul_mat_mat_f16[src0_type].f32acc; - } - if (ctx->device->coopmat_support) { - return (ctx->device->fp16 && ctx->device->coopmat_acc_f16_support && prec == GGML_PREC_DEFAULT) ? ctx->device->pipeline_dequant_mul_mat_mat[src0_type].f16acc : ctx->device->pipeline_dequant_mul_mat_mat[src0_type].f32acc; - } - return (ctx->device->fp16 && prec == GGML_PREC_DEFAULT) ? ctx->device->pipeline_dequant_mul_mat_mat[src0_type].f16acc : ctx->device->pipeline_dequant_mul_mat_mat[src0_type].f32acc; -} - -static vk_pipeline ggml_vk_get_dequantize_mul_mat_vec(ggml_backend_vk_context * ctx, ggml_type a_type, ggml_type b_type, uint32_t num_cols, uint32_t m, uint32_t k) { - VK_LOG_DEBUG("ggml_vk_get_dequantize_mul_mat_vec()"); - GGML_ASSERT(b_type == GGML_TYPE_F32 || b_type == GGML_TYPE_F16 || b_type == GGML_TYPE_Q8_1); - GGML_ASSERT(num_cols >= 1 && num_cols <= mul_mat_vec_max_cols); - - if (b_type == GGML_TYPE_Q8_1) { - switch (a_type) { - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_MXFP4: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - break; - default: - return nullptr; + break; } } - switch (a_type) { - case GGML_TYPE_F32: - case GGML_TYPE_F16: - case GGML_TYPE_BF16: + if (to == GGML_TYPE_F32) { + switch (src->type) { case GGML_TYPE_Q1_0: case GGML_TYPE_Q2_0: case GGML_TYPE_Q4_0: @@ -7791,1472 +5918,1323 @@ static vk_pipeline ggml_vk_get_dequantize_mul_mat_vec(ggml_backend_vk_context * case GGML_TYPE_Q5_0: case GGML_TYPE_Q5_1: case GGML_TYPE_Q8_0: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - case GGML_TYPE_IQ2_XXS: - case GGML_TYPE_IQ2_XS: - case GGML_TYPE_IQ2_S: - case GGML_TYPE_IQ3_XXS: - case GGML_TYPE_IQ3_S: - case GGML_TYPE_IQ4_XS: case GGML_TYPE_IQ4_NL: - case GGML_TYPE_MXFP4: - case GGML_TYPE_NVFP4: - case GGML_TYPE_TQ2_0: - break; + return ctx->device->pipeline_cpy_quant_f32[src->type]; default: - return nullptr; + break; + } } - // heuristic to choose workgroup size - uint32_t dmmv_wg = DMMV_WG_SIZE_SUBGROUP; - if ((ctx->device->vendor_id == VK_VENDOR_ID_NVIDIA && ctx->device->architecture != vk_device_architecture::NVIDIA_PRE_TURING) || ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { - // Prefer larger workgroups when M is small, to spread the work out more - // and keep more SMs busy. - // q6_k seems to prefer small workgroup size even for "medium" values of M. - if (a_type == GGML_TYPE_Q6_K) { - if (m < 4096 && k >= 1024) { - dmmv_wg = DMMV_WG_SIZE_LARGE; + if (src->type == to) { + // Copy two or four bytes at a time, depending on block size. + // For quantized types, we scale by block size/type size. But + // this path is also used for bf16->bf16 for example, where the + // type size must be exactly 2 or 4. + GGML_ASSERT(ggml_is_quantized(to) || ggml_type_size(src->type) == 2 || ggml_type_size(src->type) == 4); + if ((ggml_type_size(src->type) % 4) == 0) { + if (contig) { + return ctx->device->pipeline_contig_cpy_f32_f32; + } else { + return ctx->device->pipeline_cpy_f32_f32; } } else { - if (m <= 8192 && k >= 1024) { - dmmv_wg = DMMV_WG_SIZE_LARGE; + if (contig) { + return ctx->device->pipeline_contig_cpy_f16_f16; + } else { + return ctx->device->pipeline_cpy_f16_f16; } } } - if (b_type == GGML_TYPE_Q8_1) { - if (ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { - dmmv_wg = DMMV_WG_SIZE_SUBGROUP; - } - return ctx->device->pipeline_dequant_mul_mat_vec_q8_1_f32[dmmv_wg][a_type][num_cols-1]; + std::cerr << "Missing CPY op for types: " << ggml_type_name(src->type) << " " << ggml_type_name(to) << std::endl; + GGML_ABORT("fatal error"); +} + +void ggml_vk_cpy_to_contiguous(ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline pipeline, const ggml_tensor * tensor, const vk_subbuffer & in, const vk_subbuffer & out) { + VK_LOG_DEBUG("ggml_vk_cpy_to_contiguous((" << tensor << ", type=" << tensor->type << ", ne0=" << tensor->ne[0] << ", ne1=" << tensor->ne[1] << ", ne2=" << tensor->ne[2] << ", ne3=" << tensor->ne[3] << ", nb0=" << tensor->nb[0] << ", nb1=" << tensor->nb[1] << ", nb2=" << tensor->nb[2] << ", nb3=" << tensor->nb[3] << "), "; + std::cerr << "buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ")"); + + const uint32_t ne = ggml_nelements(tensor); + std::array elements; + + if (ne > 262144) { + elements = { 512, 512, CEIL_DIV(ne, 262144) }; + } else if (ne > 512) { + elements = { 512, CEIL_DIV(ne, 512), 1 }; + } else { + elements = { ne, 1, 1 }; } - return b_type == GGML_TYPE_F32 ? ctx->device->pipeline_dequant_mul_mat_vec_f32_f32[dmmv_wg][a_type][num_cols-1] : ctx->device->pipeline_dequant_mul_mat_vec_f16_f32[dmmv_wg][a_type][num_cols-1]; + vk_op_unary_push_constants pc = vk_op_unary_push_constants_init(tensor, tensor, ne); + pc.nb10 = 1; + pc.nb11 = (uint32_t)tensor->ne[0]; + pc.nb12 = (uint32_t)(tensor->ne[0] * tensor->ne[1]); + pc.nb13 = (uint32_t)(tensor->ne[0] * tensor->ne[1] * tensor->ne[2]); + init_pushconst_fastdiv(pc); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, elements); + ggml_vk_sync_buffers(ctx, subctx); } -static vk_matmul_pipeline ggml_vk_get_mul_mat_mat_id_pipeline(ggml_backend_vk_context * ctx, ggml_type src0_type, ggml_type src1_type, ggml_prec prec) { - VK_LOG_DEBUG("ggml_vk_get_mul_mat_mat_id_pipeline()"); - if (src0_type == GGML_TYPE_F32 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_id_f32; - } - if (src0_type == GGML_TYPE_BF16 && src1_type == GGML_TYPE_BF16) { - return ctx->device->pipeline_matmul_id_bf16; - } - if (prec == GGML_PREC_DEFAULT && ctx->device->fp16 && !(ctx->device->coopmat_support && !ctx->device->coopmat_acc_f16_support)) { - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_id_f16_f32.f16acc; - } - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F16) { - return ctx->device->pipeline_matmul_id_f16.f16acc; - } +static void ggml_vk_cpy_to_strided( + ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline pipeline, const ggml_tensor * tensor, + const vk_subbuffer & in, const vk_subbuffer & out, + uint32_t nb10, uint32_t nb11, uint32_t nb12, uint32_t nb13) { + VK_LOG_DEBUG("ggml_vk_cpy_to_strided((" << tensor << ", type=" << tensor->type << ", ne0=" << tensor->ne[0] << ", ne1=" << tensor->ne[1] << ", ne2=" << tensor->ne[2] << ", ne3=" << tensor->ne[3] << ", nb0=" << tensor->nb[0] << ", nb1=" << tensor->nb[1] << ", nb2=" << tensor->nb[2] << ", nb3=" << tensor->nb[3] << "), "; + std::cerr << "dst_nb=(" << nb10 << ", " << nb11 << ", " << nb12 << ", " << nb13 << "), buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ")"); + + const uint32_t ne = ggml_nelements(tensor); + std::array elements; + + if (ne > 262144) { + elements = { 512, 512, CEIL_DIV(ne, 262144) }; + } else if (ne > 512) { + elements = { 512, CEIL_DIV(ne, 512), 1 }; } else { - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F32) { - return ctx->device->pipeline_matmul_id_f16_f32.f32acc; - } - if (src0_type == GGML_TYPE_F16 && src1_type == GGML_TYPE_F16) { - return ctx->device->pipeline_matmul_id_f16.f32acc; - } + elements = { ne, 1, 1 }; } - // MMQ - if (src1_type == GGML_TYPE_Q8_1) { - vk_matmul_pipeline pipelines = ctx->device->pipeline_dequant_mul_mat_mat_id_q8_1[src0_type].f32acc; - - if (pipelines->is_empty()) { - return nullptr; - } + vk_op_unary_push_constants pc = vk_op_unary_push_constants_init(tensor, tensor, ne); + pc.nb10 = nb10; + pc.nb11 = nb11; + pc.nb12 = nb12; + pc.nb13 = nb13; + init_pushconst_fastdiv(pc); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, elements); + ggml_vk_sync_buffers(ctx, subctx); +} - return pipelines; +vk_pipeline ggml_vk_get_quantize_pipeline(ggml_backend_vk_context * ctx, ggml_type type) { + switch(type) { + case GGML_TYPE_Q8_1: + return ctx->device->pipeline_quantize_q8_1_x4; + default: + std::cerr << "Missing quantize pipeline for type: " << ggml_type_name(type) << std::endl; + GGML_ABORT("fatal error"); } +} - GGML_ASSERT(src1_type == GGML_TYPE_F32 || (ctx->device->coopmat2 && src1_type == GGML_TYPE_F16)); +void ggml_vk_quantize_q8_1(ggml_backend_vk_context * ctx, vk_context& subctx, const vk_subbuffer & in, const vk_subbuffer & out, uint32_t ne) { + VK_LOG_DEBUG("ggml_vk_quantize_q8_1(" << "buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ", " << ne << ")"); - switch (src0_type) { - case GGML_TYPE_Q1_0: - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - case GGML_TYPE_IQ2_XXS: - case GGML_TYPE_IQ2_XS: - case GGML_TYPE_IQ2_S: - case GGML_TYPE_IQ3_XXS: - case GGML_TYPE_IQ3_S: - case GGML_TYPE_IQ4_XS: - case GGML_TYPE_IQ4_NL: - case GGML_TYPE_MXFP4: - case GGML_TYPE_NVFP4: - case GGML_TYPE_TQ2_0: - break; - default: - return nullptr; - } - - vk_matmul_pipeline2& mmp = ctx->device->pipeline_dequant_mul_mat_mat_id[src0_type]; - // XXX TODO 'prec' is not actually allowed in mul_mat_id. - bool prefer_fp16acc = ctx->device->fp16 /*&& prec == GGML_PREC_DEFAULT*/; - bool support_fp16acc = !mmp.f16acc->is_empty(); - bool support_fp32acc = !mmp.f32acc->is_empty(); - - if (support_fp16acc && (prefer_fp16acc || !support_fp32acc)) { - return mmp.f16acc; - } else { - GGML_ASSERT(support_fp32acc); - return mmp.f32acc; - } -} - -static vk_pipeline ggml_vk_get_dequantize_mul_mat_vec_id(ggml_backend_vk_context * ctx, ggml_type a_type, ggml_type b_type, uint32_t m, uint32_t k) { - VK_LOG_DEBUG("ggml_vk_get_dequantize_mul_mat_vec_id()"); - GGML_ASSERT(b_type == GGML_TYPE_F32 || b_type == GGML_TYPE_Q8_1); + vk_pipeline pipeline = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); - if (b_type == GGML_TYPE_Q8_1) { - switch (a_type) { - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_MXFP4: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - break; - default: - return nullptr; - } - } + const uint32_t num_blocks = CEIL_DIV(ne, pipeline->wg_denoms[0]); + // clamp the number of elements to the max workgroup count. The shader will iterate over the total number of blocks. + const uint64_t max_elements = std::min(uint64_t{ctx->device->properties.limits.maxComputeWorkGroupCount[0]} * pipeline->wg_denoms[0], std::numeric_limits::max()); + const uint32_t elements = std::min(ne, static_cast(max_elements)); - switch (a_type) { - case GGML_TYPE_F32: - case GGML_TYPE_F16: - case GGML_TYPE_BF16: - case GGML_TYPE_Q1_0: - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_Q2_K: - case GGML_TYPE_Q3_K: - case GGML_TYPE_Q4_K: - case GGML_TYPE_Q5_K: - case GGML_TYPE_Q6_K: - case GGML_TYPE_IQ1_S: - case GGML_TYPE_IQ1_M: - case GGML_TYPE_IQ2_XXS: - case GGML_TYPE_IQ2_XS: - case GGML_TYPE_IQ2_S: - case GGML_TYPE_IQ3_XXS: - case GGML_TYPE_IQ3_S: - case GGML_TYPE_IQ4_XS: - case GGML_TYPE_IQ4_NL: - case GGML_TYPE_MXFP4: - case GGML_TYPE_NVFP4: - case GGML_TYPE_TQ2_0: - break; - default: - return nullptr; - } + const vk_quantize_q8_1_push_constants pc = { + ne, + num_blocks, + }; - // heuristic to choose workgroup size - uint32_t dmmv_wg = DMMV_WG_SIZE_SUBGROUP; - if ((ctx->device->vendor_id == VK_VENDOR_ID_NVIDIA && ctx->device->architecture != vk_device_architecture::NVIDIA_PRE_TURING) || ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { - // Prefer larger workgroups when M is small, to spread the work out more - // and keep more SMs busy. - // q6_k seems to prefer small workgroup size even for "medium" values of M. - if (a_type == GGML_TYPE_Q6_K) { - if (m < 4096 && k >= 1024) { - dmmv_wg = DMMV_WG_SIZE_LARGE; - } - } else { - if (m <= 8192 && k >= 1024) { - dmmv_wg = DMMV_WG_SIZE_LARGE; - } - } - } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, { elements, 1, 1 }); + ggml_vk_sync_buffers(ctx, subctx); +} - if (b_type == GGML_TYPE_Q8_1) { - if (ctx->device->vendor_id == VK_VENDOR_ID_INTEL) { - dmmv_wg = DMMV_WG_SIZE_SUBGROUP; +static vk_pipeline ggml_vk_get_64b_indexing_pipeline(ggml_backend_vk_context * ctx, vk_pipeline &pipeline) { + GGML_UNUSED(ctx); +#if defined(VK_EXT_shader_64bit_indexing) + vk_pipeline *ptr = &pipeline; + while (*ptr) { + if ((*ptr)->is_64b_indexing) { + return *ptr; } - return ctx->device->pipeline_dequant_mul_mat_vec_id_q8_1_f32[dmmv_wg][a_type]; + ptr = &(*ptr)->next; } - - return ctx->device->pipeline_dequant_mul_mat_vec_id_f32[dmmv_wg][a_type]; +#endif + return pipeline; } -static void * ggml_vk_host_malloc(vk_device& device, size_t size) { - VK_LOG_MEMORY("ggml_vk_host_malloc(" << size << ")"); - vk_buffer buf = ggml_vk_create_buffer(device, size, - {vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent}); - - if(!(buf->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible)) { - fprintf(stderr, "WARNING: failed to allocate %.2f MB of pinned memory\n", - size/1024.0/1024.0); - device->device.freeMemory(buf->device_memory); - device->device.destroyBuffer(buf->buffer); - return nullptr; - } +static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst, bool disable_split_k) { + VK_LOG_DEBUG("ggml_vk_mul_mat_q_f16((" << src0 << ", name=" << src0->name << ", type=" << ggml_type_name(src0->type) << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << ggml_type_name(src1->type) << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << ggml_type_name(dst->type) << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << "))"); + GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT + GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT - std::lock_guard guard(device->pinned_memory_mutex); - device->pinned_memory.push_back(std::make_tuple(buf->ptr, size, buf)); + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + const uint64_t ne03 = src0->ne[3]; - return buf->ptr; -} + const uint64_t ne10 = src1->ne[0]; + const uint64_t ne11 = src1->ne[1]; + const uint64_t ne12 = src1->ne[2]; + const uint64_t ne13 = src1->ne[3]; -static void ggml_vk_host_free(vk_device& device, void* ptr) { - if (ptr == nullptr) { - return; - } - VK_LOG_MEMORY("ggml_vk_host_free(" << ptr << ")"); - std::lock_guard guard(device->pinned_memory_mutex); - - vk_buffer buf; - size_t index; - for (size_t i = 0; i < device->pinned_memory.size(); i++) { - const uint8_t* addr = (const uint8_t*) std::get<0>(device->pinned_memory[i]); - const uint8_t* endr = addr + std::get<1>(device->pinned_memory[i]); - if (ptr >= addr && ptr < endr) { - buf = std::get<2>(device->pinned_memory[i]); - index = i; - break; - } - } - if (buf == nullptr) { - fprintf(stderr, "WARNING: failed to free pinned memory: memory not in map\n"); - return; - } + const uint64_t ne21 = dst->ne[1]; + const uint32_t stride_d = dst->nb[1] / ggml_type_size(dst->type); + const uint32_t stride_batch_d = stride_d*ne21; - ggml_vk_destroy_buffer(buf); + const uint64_t r2 = ne12 / ne02; + const uint64_t r3 = ne13 / ne03; - device->pinned_memory.erase(device->pinned_memory.begin() + index); -} + ggml_backend_vk_buffer_context * dst_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; + ggml_backend_vk_buffer_context * src0_buf_ctx = (ggml_backend_vk_buffer_context *)src0->buffer->context; + ggml_backend_vk_buffer_context * src1_buf_ctx = (ggml_backend_vk_buffer_context *)src1->buffer->context; -static void ggml_vk_host_get(const vk_device& device, const void * ptr, vk_buffer& buf, size_t& buf_offset) { - std::shared_lock guard(device->pinned_memory_mutex); - buf = nullptr; - buf_offset = 0; - for (size_t i = 0; i < device->pinned_memory.size(); i++) { - const uint8_t* addr = (const uint8_t*) std::get<0>(device->pinned_memory[i]); - const uint8_t* endr = addr + std::get<1>(device->pinned_memory[i]); - if (ptr >= addr && ptr < endr) { - buf = std::get<2>(device->pinned_memory[i]); - buf_offset = ((const uint8_t *)ptr) - addr; - break; - } - } -} + vk_buffer d_Qx = nullptr; + size_t qx_buf_offset = 0; + vk_buffer d_Qy = nullptr; + size_t qy_buf_offset = 0; -static vk_subbuffer ggml_vk_tensor_subbuffer( - const ggml_backend_vk_context * ctx, const ggml_tensor * tensor, bool allow_misalign = false) { + bool src0_uma = false; + bool src1_uma = false; - vk_buffer buffer = nullptr; - size_t offset = 0; if (ctx->device->uma) { - ggml_vk_host_get(ctx->device, tensor->data, buffer, offset); - } - if (!buffer) { - auto buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; - buffer = buf_ctx->dev_buffer; - offset = vk_tensor_offset(tensor) + tensor->view_offs; + ggml_vk_host_get(ctx->device, src0->data, d_Qx, qx_buf_offset); + ggml_vk_host_get(ctx->device, src1->data, d_Qy, qy_buf_offset); + src0_uma = d_Qx != nullptr; + src1_uma = d_Qy != nullptr; } - GGML_ASSERT(buffer != nullptr); - size_t size = ggml_nbytes(tensor); - - size_t misalign_bytes = offset & (ctx->device->properties.limits.minStorageBufferOffsetAlignment - 1); - // The shader must support misaligned offsets when indexing into the buffer - GGML_ASSERT(allow_misalign || misalign_bytes == 0); - offset &= ~misalign_bytes; - size += misalign_bytes; + // TODO: Clean up this logic to pick src1 type by capability + // Reformat and convert to fp16 if non-contiguous, or for coopmat2 for better perf + const bool x_non_contig = (ctx->device->coopmat2 && src0->type == GGML_TYPE_F32) || + !ggml_vk_dim01_contiguous(src0); + const bool y_non_contig = (ctx->device->coopmat2 && src1->type == GGML_TYPE_F32) || + // coopmat1: force f32->f16 conversion so the f16 B-type quant pipeline is used. + (ctx->device->coopmat_support && !ctx->device->coopmat2 && + ggml_is_quantized(src0->type) && src1->type == GGML_TYPE_F32) || + (src0->type == GGML_TYPE_BF16 && src1->type != GGML_TYPE_BF16) || + !ggml_vk_dim01_contiguous(src1); - return vk_subbuffer{buffer, offset, size}; -} + // If src0 is BF16, try to use a BF16 x BF16 multiply + ggml_type f16_type = src0->type == GGML_TYPE_BF16 ? GGML_TYPE_BF16 : GGML_TYPE_F16; -// Get a command buffer from pool. Create a new one if no reusable buffer is available -static vk_command_buffer* ggml_vk_get_or_create_cmd_buffer(vk_device& device, vk_command_pool& pool) { - for (auto& cmd_buffer : pool.cmd_buffers) { - if (!cmd_buffer.in_use) { - cmd_buffer.use_counter++; - cmd_buffer.in_use = true; - return &cmd_buffer; - } - } - return ggml_vk_create_cmd_buffer(device, pool); -} + const bool y_f32_kernel = src1->type == GGML_TYPE_F32 && !y_non_contig; -static vk_submission ggml_vk_begin_submission(vk_device& device, vk_command_pool& p, bool one_time = true) { - vk_submission s; - s.buffer = ggml_vk_get_or_create_cmd_buffer(device, p); - if (one_time) { - s.buffer->buf.begin({ vk::CommandBufferUsageFlagBits::eOneTimeSubmit }); - } else { - s.buffer->buf.begin({ vk::CommandBufferUsageFlags{} }); - } + bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0; - return s; -} + // Check for mmq first + const std::vector* mmp_map = quantize_y ? ggml_vk_get_mul_mat_mat_pipeline_map(ctx, src0->type, GGML_TYPE_Q8_1, (ggml_prec)dst->op_params[0]) : nullptr; -template size_t push_constant_size(const T &t) { - static_assert(std::is_class::value, "T must be a struct/class"); - GGML_UNUSED(t); - return sizeof(T); -} -template size_t push_constant_size(const std::vector &t) { - GGML_UNUSED(t); - return sizeof(T) * t.size(); -} -template size_t push_constant_size(const std::array &t) { - GGML_UNUSED(t); - return sizeof(T) * N; -} + if (mmp_map == nullptr) { + // Fall back to f16 dequant mul mat + mmp_map = ggml_vk_get_mul_mat_mat_pipeline_map(ctx, src0->type, y_non_contig ? f16_type : src1->type, (ggml_prec)dst->op_params[0]); + quantize_y = false; + } -template const T *push_constant_data(const T &t) { - static_assert(std::is_class::value, "T must be a struct/class"); - return &t; -} -template const T *push_constant_data(const std::vector &t) { - return t.data(); -} -template const T *push_constant_data(const std::array &t) { - return t.data(); -} + const bool qx_needs_dequant = mmp_map == nullptr || x_non_contig; + const bool qy_needs_dequant = !quantize_y && ((src1->type != f16_type && !y_f32_kernel) || y_non_contig); -template -static void ggml_vk_dispatch_pipeline(ggml_backend_vk_context* ctx, vk_context& subctx, vk_pipeline& pipeline, std::initializer_list const& descriptor_buffer_infos, const T &push_constants, std::array elements) { - const uint32_t wg0 = CEIL_DIV(elements[0], pipeline->wg_denoms[0]); - const uint32_t wg1 = CEIL_DIV(elements[1], pipeline->wg_denoms[1]); - const uint32_t wg2 = CEIL_DIV(elements[2], pipeline->wg_denoms[2]); - VK_LOG_DEBUG("ggml_vk_dispatch_pipeline(" << pipeline->name << ", {"; - for (auto& buffer : descriptor_buffer_infos) { - std::cerr << "(" << buffer.buffer << ", " << buffer.offset << ", " << buffer.range << "), "; + if (qx_needs_dequant) { + // Fall back to dequant + f16 mulmat + mmp_map = ggml_vk_get_mul_mat_mat_pipeline_map(ctx, f16_type, y_f32_kernel ? GGML_TYPE_F32 : f16_type, (ggml_prec)dst->op_params[0]); } - std::cerr << "}, (" << wg0 << "," << wg1 << "," << wg2 << "))"); - GGML_ASSERT(wg0 <= ctx->device->properties.limits.maxComputeWorkGroupCount[0] && - wg1 <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && - wg2 <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - GGML_ASSERT(ctx->descriptor_set_idx < ctx->descriptor_sets.size()); - GGML_ASSERT(descriptor_buffer_infos.size() <= MAX_PARAMETER_COUNT); - GGML_ASSERT(pipeline->parameter_count == descriptor_buffer_infos.size()); - GGML_ASSERT(pipeline->push_constant_size == push_constant_size(push_constants)); - vk::DescriptorSet& descriptor_set = ctx->descriptor_sets[ctx->descriptor_set_idx++]; - vk::WriteDescriptorSet write_descriptor_set{ descriptor_set, 0, 0, pipeline->parameter_count, vk::DescriptorType::eStorageBuffer, nullptr, descriptor_buffer_infos.begin() }; - ctx->device->device.updateDescriptorSets({ write_descriptor_set }, {}); + // Not implemented + GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT - subctx->s->buffer->buf.pushConstants(pipeline->layout, vk::ShaderStageFlagBits::eCompute, 0, push_constant_size(push_constants), push_constant_data(push_constants)); - subctx->s->buffer->buf.bindPipeline(vk::PipelineBindPoint::eCompute, pipeline->pipeline); - subctx->s->buffer->buf.bindDescriptorSets(vk::PipelineBindPoint::eCompute, - pipeline->layout, - 0, - { descriptor_set }, - {}); - subctx->s->buffer->buf.dispatch(wg0, wg1, wg2); -} + GGML_ASSERT(mmp_map != nullptr); -static void ggml_vk_ctx_end(vk_context& ctx) { - VK_LOG_DEBUG("ggml_vk_ctx_end(" << ctx << ", " << ctx->seqs.size() << ")"); - if (ctx->s == nullptr) { - return; - } + const uint32_t kpad = quantize_y ? 0 : ggml_vk_align_size(ne10, ggml_vk_guess_matmul_pipeline_align_map(ctx, *mmp_map, ne01, ne11, false)); + const bool aligned = !quantize_y && ne10 == kpad && ne01 > 8 && ne11 > 8; - ctx->s->buffer->buf.end(); - ctx->s = nullptr; -} + vk_pipeline pipeline = ggml_vk_guess_matmul_pipeline_map(ctx, *mmp_map, ne01, ne11, aligned, false); -static void ggml_vk_ctx_begin(vk_device& device, vk_context& subctx) { - VK_LOG_DEBUG("ggml_vk_ctx_begin(" << device->name << ")"); - if (subctx->s != nullptr) { - ggml_vk_ctx_end(subctx); + if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { + pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); } - subctx->seqs.push_back({ ggml_vk_begin_submission(device, *subctx->p) }); - subctx->s = subctx->seqs[subctx->seqs.size() - 1].data(); -} - -static vk_context ggml_vk_get_compute_ctx(ggml_backend_vk_context * ctx) { - vk_context result; - if (!ctx->compute_ctx.expired()) { - result = ctx->compute_ctx.lock(); - } else { - result = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); + // Reserve extra storage in the N dimension for the Y matrix, so we can avoid bounds-checking + uint32_t padded_n = qy_needs_dequant ? ROUNDUP_POW2(ne11, pipeline->wg_denoms[1]) : ne11; + const uint64_t x_ne = ggml_nelements(src0); + // 128 elements per Q8_1 x4 block + const uint64_t y_ne = padded_n * ne10 * ne12 * ne13; + const uint64_t d_ne = ggml_nelements(dst); - ctx->compute_ctx = result; - ggml_vk_ctx_begin(ctx->device, result); - } + const uint32_t split_k = ggml_vk_guess_split_k(ctx, ne01, ne11, ne10, disable_split_k, pipeline); - if (ctx->device->async_use_transfer_queue && ctx->transfer_semaphore_last_submitted < ctx->transfer_semaphore.value) { - result->s->wait_semaphores.push_back(ctx->transfer_semaphore); - ctx->transfer_semaphore_last_submitted = ctx->transfer_semaphore.value; - } + const uint64_t qx_sz = ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type); + const uint64_t qy_sz = ggml_type_size(src1->type) * y_ne / ggml_blck_size(src1->type); + const uint64_t x_sz = !qx_needs_dequant ? qx_sz : sizeof(ggml_fp16_t) * x_ne; + const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : (y_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); + const uint64_t d_sz = sizeof(float) * d_ne; - return result; -} + vk_pipeline to_fp16_vk_0 = nullptr; + vk_pipeline to_fp16_vk_1 = nullptr; + vk_pipeline to_q8_1 = nullptr; -static vk_context ggml_vk_get_transfer_ctx(ggml_backend_vk_context * ctx) { - vk_context result; - if (!ctx->transfer_ctx.expired()) { - result = ctx->transfer_ctx.lock(); + if (x_non_contig) { + to_fp16_vk_0 = ggml_vk_get_cpy_pipeline(ctx, src0, nullptr, f16_type); } else { - result = ggml_vk_create_context(ctx, ctx->transfer_cmd_pool); - - ctx->transfer_ctx = result; - ggml_vk_ctx_begin(ctx->device, result); + to_fp16_vk_0 = ggml_vk_get_to_fp16(ctx, src0->type); } - - return result; -} - -// Submit any pending transfer queue work and signal the transfer semaphore. -// The next compute context created via ggml_vk_get_compute_ctx will wait on this semaphore. -// Returns true if work was submitted. -static bool ggml_vk_submit_transfer_ctx(ggml_backend_vk_context * ctx) { - if (!ctx->device->async_use_transfer_queue || ctx->transfer_ctx.expired()) { - return false; + if (y_non_contig) { + to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, nullptr, f16_type); + } else { + to_fp16_vk_1 = ggml_vk_get_to_fp16(ctx, src1->type); } + GGML_ASSERT(!qx_needs_dequant || to_fp16_vk_0 != nullptr); // NOLINT + GGML_ASSERT(!qy_needs_dequant || to_fp16_vk_1 != nullptr); // NOLINT - vk_context cpy_ctx = ctx->transfer_ctx.lock(); - ggml_vk_ctx_end(cpy_ctx); - - for (auto& cpy : cpy_ctx->in_memcpys) { - memcpy(cpy.dst, cpy.src, cpy.n); + if (quantize_y) { + to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); } - ctx->transfer_semaphore.value++; - cpy_ctx->seqs.back().back().signal_semaphores.push_back(ctx->transfer_semaphore); - - ggml_vk_submit(cpy_ctx, {}); - ctx->transfer_ctx.reset(); - return true; -} + { + const uint64_t split_k_size = split_k > 1 ? d_sz * split_k : 0; + if ( + (qx_needs_dequant && x_sz > ctx->device->properties.limits.maxStorageBufferRange) || + (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange) || + (split_k > 1 && split_k_size > ctx->device->properties.limits.maxStorageBufferRange)) { + GGML_ABORT("Requested preallocation size is too large"); + } + if (qx_needs_dequant && ctx->prealloc_size_x < x_sz) { + ctx->prealloc_size_x = x_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if ((qy_needs_dequant || quantize_y) && ctx->prealloc_size_y < y_sz) { + ctx->prealloc_size_y = y_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if (split_k > 1 && ctx->prealloc_size_split_k < split_k_size) { + ctx->prealloc_size_split_k = split_k_size; + ggml_vk_preallocate_buffers(ctx, subctx); + } -static size_t ggml_vk_align_size(size_t width, size_t align) { - VK_LOG_DEBUG("ggml_vk_align_size(" << width << ", " << align << ")"); - return CEIL_DIV(width, align) * align; -} + // Request descriptor sets + if (qx_needs_dequant) { + ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); + } + if (qy_needs_dequant) { + ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_1, 1); + } + if (quantize_y) { + ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); + } + if (split_k > 1) { + ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, 1); + } + } -static void deferred_memcpy(void * dst, const void * src, size_t size, std::vector* memcpys = nullptr) { - if (memcpys == nullptr) { - memcpy(dst, src, size); + vk_buffer d_D = dst_buf_ctx->dev_buffer; + const uint64_t d_buf_offset = vk_tensor_offset(dst) + dst->view_offs; + GGML_ASSERT(d_D != nullptr); + GGML_ASSERT(d_D->size >= d_buf_offset + d_sz); + vk_buffer d_X; + uint64_t x_buf_offset = 0; + vk_buffer d_Y; + uint64_t y_buf_offset = 0; + if (!src0_uma) { + d_Qx = src0_buf_ctx->dev_buffer; + qx_buf_offset = vk_tensor_offset(src0) + src0->view_offs; + GGML_ASSERT(d_Qx != nullptr); + } + if (!src1_uma) { + d_Qy = src1_buf_ctx->dev_buffer; + qy_buf_offset = vk_tensor_offset(src1) + src1->view_offs; + GGML_ASSERT(d_Qy != nullptr); + } + if (qx_needs_dequant) { + d_X = ctx->prealloc_x; + GGML_ASSERT(d_X->size >= x_sz); } else { - memcpys->emplace_back(dst, src, size); + d_X = d_Qx; + x_buf_offset = qx_buf_offset; + GGML_ASSERT(qx_sz == x_sz); } -} - -static void deferred_memset(void * dst, uint32_t val, size_t size, std::vector* memsets = nullptr) { - if (memsets == nullptr) { - memset(dst, val, size); + if (qy_needs_dequant) { + d_Y = ctx->prealloc_y; + GGML_ASSERT(d_Y->size >= y_sz); + } else if (quantize_y) { + d_Y = ctx->prealloc_y; + GGML_ASSERT(d_Y->size >= CEIL_DIV(y_sz, 144) * 144); } else { - memsets->emplace_back(dst, val, size); + d_Y = d_Qy; + y_buf_offset = qy_buf_offset; + GGML_ASSERT(qy_sz == y_sz); } -} -static void ggml_vk_ensure_sync_staging_buffer(vk_device& device, size_t size) { - if (device->sync_staging == nullptr || device->sync_staging->size < size) { - VK_LOG_MEMORY("ggml_vk_ensure_sync_staging_buffer(" << size << ")"); - ggml_vk_destroy_buffer(device->sync_staging); - device->sync_staging = ggml_vk_create_buffer_check(device, size, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent); + if (x_non_contig || qx_needs_dequant) { + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } } -} -static void ggml_vk_ensure_sync_staging_buffer(ggml_backend_vk_context * ctx, size_t size) { - if (ctx->sync_staging == nullptr || ctx->sync_staging->size < size) { - VK_LOG_MEMORY("ggml_vk_ensure_sync_staging_buffer(" << size << ")"); - ggml_vk_destroy_buffer(ctx->sync_staging); - ctx->sync_staging = ggml_vk_create_buffer_check(ctx->device, size, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached, - vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent); + if (x_non_contig) { + ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, ggml_vk_subbuffer(ctx, d_Qx, qx_buf_offset), ggml_vk_subbuffer(ctx, d_X, 0)); + } else if (qx_needs_dequant) { + const std::vector pc = { (uint32_t)ne01, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)(ggml_nelements(src0)) }; + ggml_vk_dispatch_pipeline(ctx, subctx, to_fp16_vk_0, { vk_subbuffer{ d_Qx, qx_buf_offset, qx_sz }, vk_subbuffer{ d_X, 0, x_sz } }, pc, { (uint32_t)(x_ne), 1, 1}); + ggml_vk_sync_buffers(ctx, subctx); } -} - -static void ggml_vk_buffer_write_nc_async(ggml_backend_vk_context * ctx, vk_context& subctx, vk_buffer& dst, size_t offset, const ggml_tensor * tensor, bool sync_staging = false) { - VK_LOG_DEBUG("ggml_vk_buffer_write_nc_async(" << tensor << ")"); - GGML_ASSERT(!ggml_is_contiguous(tensor)); - // Buffer is already mapped - if(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { - std::cerr << "ggml_vulkan: buffer_write_nc_async dst buffer is host_visible. Use synchronous write." << std::endl; - GGML_ABORT("fatal error"); + if (y_non_contig) { + if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || + ctx->prealloc_y_last_tensor_used != src1 || + ctx->prealloc_y_last_k_padded) { + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0)); + ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); + ctx->prealloc_y_last_tensor_used = src1; + ctx->prealloc_y_last_k_padded = false; + } } - // Check if src is pinned memory - vk_buffer buf = nullptr; - size_t buf_offset = 0; - ggml_vk_host_get(ctx->device, tensor->data, buf, buf_offset); - - const uint64_t ne0 = tensor->ne[0]; - const uint64_t ne1 = tensor->ne[1]; - const uint64_t ne2 = tensor->ne[2]; - const uint64_t ne3 = tensor->ne[3]; - const uint64_t nb0 = tensor->nb[0]; - const uint64_t nb1 = tensor->nb[1]; - const uint64_t nb2 = tensor->nb[2]; - const uint64_t nb3 = tensor->nb[3]; - const ggml_type type = tensor->type; - const uint64_t ts = ggml_type_size(type); - const uint64_t bs = ggml_blck_size(type); - - const uint64_t dstnb0 = ts; - const uint64_t dstnb1 = dstnb0*(ne0/bs); - const uint64_t dstnb2 = dstnb1*ne1; - const uint64_t dstnb3 = dstnb2*ne2; - - const uint64_t ne = ggml_nelements(tensor); - - if (buf != nullptr) { - // Memory is pinned, use as staging buffer - std::vector slices; - - for (uint64_t i3 = 0; i3 < ne3; i3++) { - for (uint64_t i2 = 0; i2 < ne2; i2++) { - // Find longest contiguous slice - if (ne1*nb1 == dstnb2) { - slices.push_back({ buf_offset + i3*nb3 + i2*nb2, offset + i3*dstnb3 + i2*dstnb2, dstnb2 }); - } else { - for (uint64_t i1 = 0; i1 < ne1; i1++) { - if (ne0*nb0/bs == dstnb1) { - slices.push_back({ buf_offset + i3*nb3 + i2*nb2 + i1*nb1, offset + i3*dstnb3 + i2*dstnb2 + i1*dstnb1, dstnb1 }); - } else { - const uint64_t s_off = buf_offset + i3*nb3 + i2*nb2 + i1*nb1; - const uint64_t d_off = offset + i3*dstnb3 + i2*dstnb2 + i1*dstnb1; - for (uint64_t i0 = 0; i0 < ne0; i0++) { - slices.push_back({ s_off + i0*nb0, d_off + i0*dstnb0, dstnb0 }); - } - } - } - } + if (quantize_y) { + if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || + ctx->prealloc_y_last_tensor_used != src1 || + ctx->prealloc_y_last_k_padded) { + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); } + ggml_vk_quantize_q8_1(ctx, subctx, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0), y_ne); + ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); + ctx->prealloc_y_last_tensor_used = src1; + ctx->prealloc_y_last_k_padded = false; } - - ggml_vk_sync_buffers(ctx, subctx); - subctx->s->buffer->buf.copyBuffer(buf->buffer, dst->buffer, slices); - return; } - if (!sync_staging) { - GGML_ABORT("Asynchronous write to non-pinned memory not supported"); + uint32_t stride_batch_x = ne00*ne01; + uint32_t stride_batch_y = ne10*ne11; + + if (!ggml_vk_dim01_contiguous(src0) && !qx_needs_dequant) { + stride_batch_x = src0->nb[0] / ggml_type_size(src0->type); } - // Staging buffer required - vk_buffer& staging = ctx->device->sync_staging; - const uint64_t copy_size = ts*ne/bs; - ggml_vk_ensure_sync_staging_buffer(ctx->device, copy_size); - VkBufferCopy buf_copy{ 0, offset, copy_size }; + if (!ggml_vk_dim01_contiguous(src1) && !qy_needs_dequant && !quantize_y) { + stride_batch_y = src1->nb[0] / ggml_type_size(src1->type); + } - ggml_vk_sync_buffers(ctx, subctx); - vkCmdCopyBuffer(subctx->s->buffer->buf, (VkBuffer)staging->buffer, (VkBuffer)dst->buffer, 1, &buf_copy); + // compute + ggml_vk_matmul( + ctx, subctx, pipeline, + { d_X, x_buf_offset, x_sz }, { d_Y, y_buf_offset, y_sz }, + ggml_vk_subbuffer(ctx, d_D, d_buf_offset), { ctx->prealloc_split_k, 0, d_sz * split_k }, + ne01, ne11, ne10, + ne10, ne10, stride_d, stride_batch_x, stride_batch_y, stride_batch_d, + split_k, ne12*ne13, ne02, ne12, r2, r3, padded_n + ); // NOLINT - for (uint64_t i3 = 0; i3 < ne3; i3++) { - for (uint64_t i2 = 0; i2 < ne2; i2++) { - // Find longest contiguous slice - if (ne1*nb1 == dstnb2) { - deferred_memcpy((uint8_t *)staging->ptr + i3*dstnb3 + i2*dstnb2, (const uint8_t *) tensor->data + buf_offset + i3*nb3 + i2*nb2, dstnb2, &subctx->in_memcpys); - } else { - for (uint64_t i1 = 0; i1 < ne1; i1++) { - if (ne0*nb0/bs == dstnb1) { - deferred_memcpy((uint8_t *)staging->ptr + i3*dstnb3 + i2*dstnb2 + i1*dstnb1, (const uint8_t *) tensor->data + buf_offset + i3*nb3 + i2*nb2 + i1*nb1, dstnb1, &subctx->in_memcpys); - } else { - const uint64_t s_off = buf_offset + i3*nb3 + i2*nb2 + i1*nb1; - const uint64_t d_off = i3*dstnb3 + i2*dstnb2 + i1*dstnb1; - for (uint64_t i0 = 0; i0 < ne0; i0++) { - deferred_memcpy((uint8_t *)staging->ptr + d_off + i0*dstnb0, (const uint8_t *) tensor->data + s_off + i0*nb0, dstnb0, &subctx->in_memcpys); - } - } - } - } - } + if (x_non_contig || qx_needs_dequant) { + ctx->prealloc_x_need_sync = true; + } + if (y_non_contig || quantize_y) { + ctx->prealloc_y_need_sync = true; } } -static bool ggml_vk_buffer_write_2d_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging = false) { - VK_LOG_DEBUG("ggml_vk_buffer_write_2d_async(" << width << ", " << height << ")"); - // Check if src is pinned memory - vk_buffer buf = nullptr; - size_t buf_offset = 0; - ggml_vk_host_get(dst->device, src, buf, buf_offset); - - if (buf != nullptr) { - // Memory is pinned, use as staging buffer - std::vector slices(1); - if (width == spitch && width == dpitch) { - // Only do single write if stride is equal - slices[0].srcOffset = buf_offset; - slices[0].dstOffset = offset; - slices[0].size = width * height; - } else { - slices.resize(height); - for (size_t i = 0; i < height; i++) { - slices[i].srcOffset = buf_offset + i * spitch; - slices[i].dstOffset = offset + i * dpitch; - slices[i].size = width; - } - } - - ggml_vk_sync_buffers(nullptr, subctx); - subctx->s->buffer->buf.copyBuffer(buf->buffer, dst->buffer, slices); +static bool ggml_vk_should_use_mmvq(const vk_device& device, uint32_t m, uint32_t n, uint32_t k, ggml_type src0_type) { + if (device->mmvq_mode == 1) { return true; + } else if (device->mmvq_mode == -1) { + return false; } - VK_LOG_DEBUG("STAGING"); - if (!sync_staging) { - // copy was not handled caller needs to fall back + // q6_k only has 2-byte alignment which makes it somewhat problematic, + // using MMVQ is only a win on Intel. + bool mmvq_q6 = device->vendor_id == VK_VENDOR_ID_INTEL; + if (src0_type == GGML_TYPE_Q6_K && !mmvq_q6) { return false; } - // Staging buffer required - const size_t staging_size = width * height; - ggml_vk_ensure_sync_staging_buffer(dst->device, staging_size); - - vk_buffer& staging_buffer = dst->device->sync_staging; - - std::vector slices(1); - if (width == dpitch) { - slices[0].srcOffset = 0; - slices[0].dstOffset = offset; - slices[0].size = staging_size; - } else { - slices.resize(height); - for (size_t i = 0; i < height; i++) { - slices[i].srcOffset = i * width; - slices[i].dstOffset = offset + i * dpitch; - slices[i].size = width; - } + // MMVQ is generally good for batches + if (n > 1) { + return true; } - ggml_vk_sync_buffers(nullptr, subctx); - subctx->s->buffer->buf.copyBuffer((VkBuffer)staging_buffer->buffer, (VkBuffer)dst->buffer, slices); - - if (width == spitch) { - deferred_memcpy((uint8_t *)staging_buffer->ptr, src, staging_size, &subctx->in_memcpys); - } else { - for (size_t i = 0; i < height; i++) { - deferred_memcpy((uint8_t *)staging_buffer->ptr + i * width, (const uint8_t *) src + i * spitch, width, &subctx->in_memcpys); + // Quantization overhead is not worth it for small k + switch (device->vendor_id) { + case VK_VENDOR_ID_NVIDIA: + if (src0_type == GGML_TYPE_Q2_0 || src0_type == GGML_TYPE_Q2_K || src0_type == GGML_TYPE_Q3_K || src0_type == GGML_TYPE_IQ1_S || src0_type == GGML_TYPE_IQ1_M) { + return true; } - } - return true; -} - -static bool ggml_vk_buffer_write_async(vk_context subctx, vk_buffer& dst, size_t offset, const void * src, size_t size, bool sync_staging = false) { - VK_LOG_DEBUG("ggml_vk_buffer_write_async(" << size << ")"); - return ggml_vk_buffer_write_2d_async(subctx, dst, offset, src, size, size, size, 1, sync_staging); -} -static void ggml_vk_buffer_write_2d(vk_buffer& dst, size_t offset, const void * src, size_t spitch, size_t dpitch, size_t width, size_t height) { - VK_LOG_DEBUG("ggml_vk_buffer_write_2d(" << width << ", " << height << ")"); - // Buffer is already mapped - if(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible) { - GGML_ASSERT(dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent); - - if (width == spitch && width == dpitch) { - memcpy((uint8_t *)dst->ptr + offset, src, width * height); - } else { - for (size_t i = 0; i < height; i++) { - memcpy((uint8_t *)dst->ptr + offset + i * dpitch, (const uint8_t *) src + i * spitch, width); - } + if (k <= 4096) { + return false; } - } else { - std::lock_guard guard(dst->device->mutex); - - vk_context subctx = ggml_vk_create_temporary_context(dst->device->transfer_queue->cmd_pool); - ggml_vk_ctx_begin(dst->device, subctx); - bool ret = ggml_vk_buffer_write_2d_async(subctx, dst, offset, src, spitch, dpitch, width, height, true); - GGML_ASSERT(ret); - ggml_vk_ctx_end(subctx); - for (auto& cpy : subctx->in_memcpys) { - memcpy(cpy.dst, cpy.src, cpy.n); + switch (src0_type) { + case GGML_TYPE_MXFP4: + case GGML_TYPE_Q8_0: + return device->architecture == vk_device_architecture::NVIDIA_PRE_TURING; + default: + return true; + } + case VK_VENDOR_ID_AMD: + if (k < 2048) { + return false; } - for (auto& mset : subctx->memsets) { - memset(mset.dst, mset.val, mset.n); + switch (src0_type) { + case GGML_TYPE_Q8_0: + return device->architecture == vk_device_architecture::AMD_GCN; + default: + return true; + } + case VK_VENDOR_ID_INTEL: + if (device->architecture == vk_device_architecture::INTEL_XE2) { + if (src0_type == GGML_TYPE_Q2_0 || src0_type == GGML_TYPE_Q2_K || src0_type == GGML_TYPE_Q3_K || src0_type == GGML_TYPE_Q6_K) { + return true; + } + } + + if (device->driver_id == vk::DriverId::eIntelProprietaryWindows) { + // Intel Windows proprietary driver MMVQ performance for !Q2/Q3/Q6 is worse than fp16, + // see https://github.com/ggml-org/llama.cpp/issues/17628 and + // https://github.com/ggml-org/llama.cpp/pull/23056 + return false; } - ggml_vk_submit(subctx, dst->device->fence); - VK_CHECK(dst->device->device.waitForFences({ dst->device->fence }, true, UINT64_MAX), "vk_buffer_write_2d waitForFences", dst->device); - dst->device->device.resetFences({ dst->device->fence }); - ggml_vk_queue_command_pools_cleanup(dst->device); + if (k < 2048) { + return false; + } + + switch (src0_type) { + // From tests on A770 Linux, may need more tuning + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_IQ4_XS: + return false; + default: + return true; + } + default: + return true; } -} -static void ggml_vk_buffer_write(vk_buffer& dst, size_t offset, const void * src, size_t size) { - VK_LOG_DEBUG("ggml_vk_buffer_write(" << size << ")"); - ggml_vk_buffer_write_2d(dst, offset, src, size, size, size, 1); + GGML_UNUSED(m); } -static bool ggml_vk_buffer_read_2d_async(vk_context subctx, vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height, bool sync_staging = false) { - VK_LOG_DEBUG("ggml_vk_buffer_read_2d_async(offset=" << offset << ", width=" << width << ", height=" << height << ")"); - GGML_ASSERT(width > 0); - GGML_ASSERT(height > 0); - GGML_ASSERT(src != nullptr); +static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx, bool swap_inputs = false) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = dst->src[swap_inputs ? 1 : 0]; + const ggml_tensor * src1 = dst->src[swap_inputs ? 0 : 1]; - // TODO: staging_offset is not used + VK_LOG_DEBUG("ggml_vk_mul_mat_vec_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << ")),)"); + GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT + GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT - // Check if dst is pinned memory - vk_buffer buf = nullptr; - size_t buf_offset = 0; - ggml_vk_host_get(src->device, dst, buf, buf_offset); + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + const uint64_t ne03 = src0->ne[3]; - std::vector slices(1); - if (width == spitch && width == dpitch) { - // Only do single write if stride is equal - slices[0].srcOffset = offset; - slices[0].dstOffset = buf_offset; - slices[0].size = width * height; - } else { - slices.resize(height); - for (size_t i = 0; i < height; i++) { - slices[i].srcOffset = offset + i * spitch; - slices[i].dstOffset = buf_offset + i * dpitch; - slices[i].size = width; - } - } + const uint64_t ne10 = src1->ne[0]; + const uint64_t ne11 = src1->ne[1]; + const uint64_t ne12 = src1->ne[2]; + const uint64_t ne13 = src1->ne[3]; - if (buf != nullptr) { - // Memory is pinned, use as staging buffer - ggml_vk_sync_buffers(nullptr, subctx); - subctx->s->buffer->buf.copyBuffer(src->buffer, buf->buffer, slices); + const uint64_t ne20 = dst->ne[swap_inputs ? 1 : 0]; + const uint64_t ne21 = dst->ne[swap_inputs ? 0 : 1]; + // const uint64_t ne22 = dst->ne[2]; + // const uint64_t ne23 = dst->ne[3]; - return true; - } - VK_LOG_DEBUG("STAGING"); + const uint64_t r2 = ne12 / ne02; + const uint64_t r3 = ne13 / ne03; - if (!sync_staging) { - // copy was not handled caller needs to fall back - return false; - } + // batch_n indicates that we need to compute a few vector results, and this assumes + // ne12 and ne13 are 1. It overloads the batch_strides to hold the row strides. + GGML_ASSERT(ne11 == 1 || ne12 * ne13 == 1); + bool batch_n = ne11 > 1; - // Fall back to staging buffer - const size_t staging_size = width * height; - ggml_vk_ensure_sync_staging_buffer(src->device, staging_size); + const bool x_non_contig = !ggml_vk_dim01_contiguous(src0); + const bool y_non_contig = !ggml_vk_dim01_contiguous(src1); - vk_buffer& staging_buffer = src->device->sync_staging; + const bool f16_f32_kernel = src1->type == GGML_TYPE_F32; + bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0 && ggml_vk_should_use_mmvq(ctx->device, ne01, ne11, ne10, src0->type); - std::vector staging_slices(1); - if (width == spitch) { - staging_slices[0].srcOffset = offset; - staging_slices[0].dstOffset = 0; - staging_slices[0].size = staging_size; + vk_pipeline to_fp16_vk_0 = nullptr; + vk_pipeline to_fp16_vk_1 = nullptr; + if (x_non_contig) { + to_fp16_vk_0 = ggml_vk_get_cpy_pipeline(ctx, src0, nullptr, src0->type); + } + if (y_non_contig) { + to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, nullptr, src1->type); } else { - staging_slices.resize(height); - for (size_t i = 0; i < height; i++) { - staging_slices[i].srcOffset = offset + i * spitch; - staging_slices[i].dstOffset = i * width; - staging_slices[i].size = width; - } + to_fp16_vk_1 = ggml_vk_get_to_fp16(ctx, src1->type); } - ggml_vk_sync_buffers(nullptr, subctx); - subctx->s->buffer->buf.copyBuffer(src->buffer, staging_buffer->buffer, staging_slices); + // Check for mmq first + vk_pipeline dmmv = quantize_y ? ggml_vk_get_dequantize_mul_mat_vec(ctx, src0->type, GGML_TYPE_Q8_1, ne11, ne20, ne00) : nullptr; + vk_pipeline to_q8_1 = nullptr; - if (width == dpitch) { - deferred_memcpy(dst, staging_buffer->ptr, staging_size, &subctx->out_memcpys); - } else { - for (size_t i = 0; i < height; i++) { - deferred_memcpy((uint8_t *) dst + i * dpitch, (const uint8_t *) staging_buffer->ptr + i * width, width, &subctx->out_memcpys); - } + if (dmmv == nullptr) { + // Fall back to f16 dequant mul mat + dmmv = ggml_vk_get_dequantize_mul_mat_vec(ctx, src0->type, src1->type, ne11, ne20, ne00); + quantize_y = false; } - return true; -} -static bool ggml_vk_buffer_read_async(vk_context subctx, vk_buffer& src, size_t offset, void * dst, size_t size, bool sync_staging = false) { - return ggml_vk_buffer_read_2d_async(subctx, src, offset, dst, size, size, size, 1, sync_staging); -} + if (quantize_y) { + to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); + } -static void ggml_vk_buffer_read_2d(vk_buffer& src, size_t offset, void * dst, size_t spitch, size_t dpitch, size_t width, size_t height) { - VK_LOG_DEBUG("ggml_vk_buffer_read_2d(" << src->buffer << ", " << offset << ", " << width << ", " << height << ")"); + if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { + dmmv = ggml_vk_get_64b_indexing_pipeline(ctx, dmmv); + } - // If the device is not an UMA device the memory is host-accessible through rebar. While writing - // through PCIe is sufficient fast reading back data from PCIe is slower than going through - // the HW device to host copy path. - if(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && src->device->uma) { - GGML_ASSERT(src->memory_property_flags & vk::MemoryPropertyFlagBits::eHostCoherent); + const bool qx_needs_dequant = x_non_contig; + const bool qy_needs_dequant = !quantize_y && ((src1->type != GGML_TYPE_F16 && !f16_f32_kernel) || y_non_contig); - std::lock_guard guard(src->device->mutex); - vk_context subctx = ggml_vk_create_temporary_context(src->device->compute_queue->cmd_pool); - ggml_vk_ctx_begin(src->device, subctx); - subctx->s->buffer->buf.pipelineBarrier( - vk::PipelineStageFlagBits::eComputeShader | vk::PipelineStageFlagBits::eTransfer, - vk::PipelineStageFlagBits::eHost, - {}, - { { vk::AccessFlagBits::eShaderWrite | vk::AccessFlagBits::eTransferWrite, - vk::AccessFlagBits::eHostRead } }, - {}, {}); - ggml_vk_ctx_end(subctx); - ggml_vk_submit(subctx, src->device->fence); - VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), - "vk_buffer_read_2d uma waitForFences", src->device); - src->device->device.resetFences({ src->device->fence }); - ggml_vk_queue_command_pools_cleanup(src->device); - - if (width == spitch && width == dpitch) { - memcpy(dst, (const uint8_t *) src->ptr + offset, width * height); - } else { - for (size_t i = 0; i < height; i++) { - memcpy((uint8_t *) dst + i * dpitch, (const uint8_t *) src->ptr + offset + i * spitch, width); - } - } - } else { - std::lock_guard guard(src->device->mutex); + // Not implemented + GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT - vk_context subctx = ggml_vk_create_temporary_context(src->device->transfer_queue->cmd_pool); - ggml_vk_ctx_begin(src->device, subctx); - bool ret = ggml_vk_buffer_read_2d_async(subctx, src, offset, dst, spitch, dpitch, width, height, true); - GGML_ASSERT(ret); - ggml_vk_ctx_end(subctx); + GGML_ASSERT(!qx_needs_dequant || to_fp16_vk_0 != nullptr); // NOLINT + GGML_ASSERT(!qy_needs_dequant || to_fp16_vk_1 != nullptr); // NOLINT + GGML_ASSERT(dmmv != nullptr); - ggml_vk_submit(subctx, src->device->fence); - VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), "vk_buffer_read_2d waitForFences", src->device); - src->device->device.resetFences({ src->device->fence }); - ggml_vk_queue_command_pools_cleanup(src->device); + const uint64_t x_ne = ggml_nelements(src0); + const uint64_t y_ne = ggml_nelements(src1); - for (auto& cpy : subctx->out_memcpys) { - memcpy(cpy.dst, cpy.src, cpy.n); + const uint64_t qx_sz = ggml_vk_align_size(ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type), ctx->device->properties.limits.minStorageBufferOffsetAlignment); + const uint64_t x_sz = x_non_contig ? ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment) : qx_sz; + const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : + (f16_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); + + { + if ( + (qx_needs_dequant && x_sz > ctx->device->properties.limits.maxStorageBufferRange) || + (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange)) { + GGML_ABORT("Requested preallocation size is too large"); + } + if (qx_needs_dequant && ctx->prealloc_size_x < x_sz) { + ctx->prealloc_size_x = x_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if ((qy_needs_dequant || quantize_y) && ctx->prealloc_size_y < y_sz) { + ctx->prealloc_size_y = y_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + + // Request descriptor sets + if (qx_needs_dequant) { + ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); + } + if (qy_needs_dequant) { + ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_1, 1); + } + if (quantize_y) { + ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); } } -} -static void ggml_vk_buffer_read(vk_buffer& src, size_t offset, void * dst, size_t size) { - VK_LOG_DEBUG("ggml_vk_buffer_read(" << src->buffer << ", " << offset << ", " << size << ")"); - ggml_vk_buffer_read_2d(src, offset, dst, size, size, size, 1); -} + vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); + vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1); + vk_subbuffer d_X, d_Y; -static void ggml_vk_buffer_copy_async(vk_context& ctx, vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size) { - VK_LOG_DEBUG("ggml_vk_buffer_copy_async(" << size << ")"); - // Make sure both buffers are on same device - GGML_ASSERT(src->device == dst->device); + if (qx_needs_dequant) { + d_X = { ctx->prealloc_x, 0, ctx->prealloc_x->size }; + } else { + d_X = d_Qx; + GGML_ASSERT(qx_sz == x_sz); + } + if (qy_needs_dequant || quantize_y) { + d_Y = { ctx->prealloc_y, 0, ctx->prealloc_y->size }; + } else { + d_Y = d_Qy; + } - VkBufferCopy bc{ src_offset, dst_offset, size }; + if (x_non_contig) { + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - vkCmdCopyBuffer(ctx->s->buffer->buf, (VkBuffer)src->buffer, (VkBuffer)dst->buffer, 1, &bc); -} + GGML_ASSERT(x_sz == ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment)); + ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, d_Qx, d_X); + } + if (y_non_contig) { + GGML_ASSERT(y_sz == ggml_type_size(src1->type) * y_ne); + if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || + ctx->prealloc_y_last_tensor_used != src1 || + ctx->prealloc_y_last_k_padded) { + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, d_Qy, d_Y); + ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); + ctx->prealloc_y_last_tensor_used = src1; + ctx->prealloc_y_last_k_padded = false; + } + } + if (quantize_y) { + if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || + ctx->prealloc_y_last_tensor_used != src1 || + ctx->prealloc_y_last_k_padded) { + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + ggml_vk_quantize_q8_1(ctx, subctx, d_Qy, d_Y, y_ne); + ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); + ctx->prealloc_y_last_tensor_used = src1; + ctx->prealloc_y_last_k_padded = false; + } + } -static void ggml_vk_buffer_copy(vk_buffer& dst, size_t dst_offset, vk_buffer& src, size_t src_offset, size_t size) { - if (src->device == dst->device) { - std::lock_guard guard(src->device->mutex); - VK_LOG_DEBUG("ggml_vk_buffer_copy(SINGLE_DEVICE, " << size << ")"); - // Copy within the device - vk_context subctx = ggml_vk_create_temporary_context(src->device->transfer_queue->cmd_pool); - ggml_vk_ctx_begin(src->device, subctx); - ggml_vk_buffer_copy_async(subctx, dst, dst_offset, src, src_offset, size); - ggml_vk_ctx_end(subctx); - ggml_vk_submit(subctx, src->device->fence); - VK_CHECK(src->device->device.waitForFences({ src->device->fence }, true, UINT64_MAX), "vk_buffer_copy waitForFences", src->device); - src->device->device.resetFences({ src->device->fence }); - ggml_vk_queue_command_pools_cleanup(src->device); - } else { - VK_LOG_DEBUG("ggml_vk_buffer_copy(MULTI_DEVICE, " << size << ")"); - // Copy device to device - ggml_vk_ensure_sync_staging_buffer(src->device, size); + // For batch_n, the A matrix is the same for each batch, and B/D use the row stride as the batch stride + uint32_t stride_batch_x = batch_n ? 0 : ne00*ne01; + uint32_t stride_batch_y = batch_n ? ne10 : (ne10*ne11); + uint32_t stride_batch_d = batch_n ? ne20 : (ne20*ne21); - // Copy to src staging buffer - ggml_vk_buffer_copy(src->device->sync_staging, 0, src, src_offset, size); - // Copy to dst buffer - ggml_vk_buffer_write(dst, dst_offset, src->device->sync_staging->ptr, size); + if (!ggml_vk_dim01_contiguous(src0) && !qx_needs_dequant) { + stride_batch_x = src0->nb[0] / ggml_type_size(src0->type); } -} -static void ggml_vk_buffer_memset_async(vk_context& ctx, vk_buffer& dst, size_t offset, uint32_t c, size_t size) { - VK_LOG_DEBUG("ggml_vk_buffer_memset_async(" << offset << ", " << c << ", " << size << ")"); - - if (dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && - dst->device->uma) { - deferred_memset((uint8_t*)dst->ptr + offset, c, size, &ctx->memsets); - return; + if (!ggml_vk_dim01_contiguous(src1) && !qy_needs_dequant) { + stride_batch_y = src1->nb[0] / ggml_type_size(src1->type); } - // Fall back to GPU fillBuffer for non-UMA or non-host-visible buffers - ctx->s->buffer->buf.fillBuffer(dst->buffer, offset, size, c); -} + const uint32_t max_groups_x = ctx->device->properties.limits.maxComputeWorkGroupCount[0]; -static void ggml_vk_buffer_memset(vk_buffer& dst, size_t offset, uint32_t c, size_t size) { - VK_LOG_DEBUG("ggml_vk_buffer_memset(" << offset << ", " << c << ", " << size << ")"); + uint32_t groups_x = ne01; + uint32_t groups_z = 1; - if (dst->memory_property_flags & vk::MemoryPropertyFlagBits::eHostVisible && - dst->device->uma) { - memset((uint8_t*)dst->ptr + offset, c, size); - return; + if (ne01 > max_groups_x) { + groups_z = 64; + groups_x = CEIL_DIV(groups_x, groups_z); } - std::lock_guard guard(dst->device->mutex); - vk_context subctx = ggml_vk_create_temporary_context(dst->device->transfer_queue->cmd_pool); - ggml_vk_ctx_begin(dst->device, subctx); - subctx->s->buffer->buf.fillBuffer(dst->buffer, offset, size, c); - ggml_vk_ctx_end(subctx); + uint32_t fusion_flags = 0; - ggml_vk_submit(subctx, dst->device->fence); - VK_CHECK(dst->device->device.waitForFences({ dst->device->fence }, true, UINT64_MAX), "vk_memset waitForFences", dst->device); - dst->device->device.resetFences({ dst->device->fence }); - ggml_vk_queue_command_pools_cleanup(dst->device); -} + vk_subbuffer d_F0 = d_D; + if (ctx->num_additional_fused_ops > 0) { + const ggml_tensor * add = cgraph->nodes[node_idx + 1]; + const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; -static uint32_t ggml_vk_guess_split_k(ggml_backend_vk_context * ctx, uint32_t m, uint32_t n, uint32_t k, bool disable_split_k, const vk_pipeline& pipeline) { - VK_LOG_DEBUG("ggml_vk_guess_split_k(" << m << ", " << n << ", " << k << ", " << disable_split_k << ")"); + d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; + } - if (disable_split_k) { - return 1; + vk_subbuffer d_F1 = d_D; + if (ctx->num_additional_fused_ops == 2) { + const ggml_tensor * add = cgraph->nodes[node_idx + 2]; + const ggml_tensor * bias = add->src[0] == cgraph->nodes[node_idx + 1] ? add->src[1] : add->src[0]; + + d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; } - uint32_t split_k = 1; - if (ctx->device->shader_core_count != 0 && m >= pipeline->wg_denoms[0] && n >= pipeline->wg_denoms[1]) { - // If k is 'large' and the SMs will fill less than halfway, use split_k. - uint32_t m_tiles = CEIL_DIV(m, pipeline->wg_denoms[0]); - uint32_t n_tiles = CEIL_DIV(n, pipeline->wg_denoms[1]); + ggml_pipeline_request_descriptor_sets(ctx, dmmv, CEIL_DIV(ne12 * ne13, ctx->device->properties.limits.maxComputeWorkGroupCount[1])); - if (k >= 2048) { - if (m_tiles * n_tiles <= ctx->device->shader_core_count / 2) { - split_k = ctx->device->shader_core_count / (m_tiles * n_tiles); - } else if (m_tiles * n_tiles <= ctx->device->shader_core_count * 2 / 3) { - split_k = 3; - } - // Cap the split at 8x. Unless k is huge this is a lot of overhead. - split_k = std::min(split_k, 8u); + uint32_t base_work_group_y = 0; + while (base_work_group_y < ne12 * ne13) { - // ggml_vk_matmul will align the splits to be a multiple of 256. - // If this rounded up size would cause the last split to be empty, - // then reduce the split count. - while (true) { - if (split_k == 1) { - break; - } - uint32_t k_split = CEIL_DIV(k, split_k); - k_split = ROUNDUP_POW2(k_split, 256); - if (k_split * (split_k - 1) < k) { - break; - } - split_k--; - } - } + uint32_t groups_y = std::min((uint32_t)(ne12 * ne13) - base_work_group_y, ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + const vk_mat_vec_push_constants pc = { + (uint32_t)ne00, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne01, + stride_batch_x, stride_batch_y, stride_batch_d, + fusion_flags, base_work_group_y, + (uint32_t)ne02, (uint32_t)ne12, (uint32_t)r2, (uint32_t)r3, + }; + ggml_vk_dispatch_pipeline(ctx, subctx, dmmv, + { + d_X, + d_Y, + d_D, + d_F0, + d_F1, + }, + pc, { groups_x, groups_y, groups_z }); + base_work_group_y += groups_y; } - return split_k; + if (x_non_contig) { + ctx->prealloc_x_need_sync = true; + } + if (y_non_contig || quantize_y) { + ctx->prealloc_y_need_sync = true; + } } -static vk_pipeline ggml_vk_guess_matmul_pipeline(ggml_backend_vk_context * ctx, vk_matmul_pipeline& mmp, uint32_t m, uint32_t n, bool aligned, ggml_type src0_type, ggml_type src1_type) { - VK_LOG_DEBUG("ggml_vk_guess_matmul_pipeline(" << m << ", " << n << ", " << aligned << ", " << ggml_type_name(src0_type) << ", " << ggml_type_name(src1_type) << ")"); - - // The q8_1 (integer dot) mmq path uses a different shader with its own - // shared-memory layout, so use the int-specific availability flags. - const bool is_q8_1 = (src1_type == GGML_TYPE_Q8_1); - const bool mm_l = is_q8_1 ? ctx->device->mul_mat_l_int[src0_type] : ctx->device->mul_mat_l[src0_type]; - const bool mm_m = is_q8_1 ? ctx->device->mul_mat_m_int[src0_type] : ctx->device->mul_mat_m[src0_type]; - const bool mm_s = is_q8_1 ? ctx->device->mul_mat_s_int[src0_type] : ctx->device->mul_mat_s[src0_type]; +static void ggml_vk_mul_mat_vec_p021_f16_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + VK_LOG_DEBUG("ggml_vk_mul_mat_p021_f16_f32(" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << "))"); + GGML_ASSERT(ggml_is_permuted(src0) && ggml_is_permuted(src1)); + GGML_ASSERT(src0->nb[0] <= src0->nb[1] && src0->nb[2] <= src0->nb[3]); // NOLINT + GGML_ASSERT(src1->nb[0] <= src1->nb[1] && src1->nb[2] <= src1->nb[3]); // NOLINT + GGML_ASSERT(src0->type == GGML_TYPE_F16); + GGML_ASSERT(src1->type == GGML_TYPE_F32); - if (ctx->device->coopmat2) { - const uint32_t shader_core_count = ctx->device->shader_core_count; - const uint32_t tiles_l = CEIL_DIV(m, mmp->a_l->wg_denoms[0]) * CEIL_DIV(n, mmp->a_l->wg_denoms[1]); - const uint32_t tiles_m = CEIL_DIV(m, mmp->a_m->wg_denoms[0]) * CEIL_DIV(n, mmp->a_m->wg_denoms[1]); + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + // const uint64_t ne03 = src0->ne[3]; - // Use large shader when the N dimension is greater than the medium shader's tile size - uint32_t crossover_large = mmp->m->wg_denoms[1]; + //const uint64_t ne10 = src1->ne[0]; + const uint64_t ne11 = src1->ne[1]; + const uint64_t ne12 = src1->ne[2]; + // const uint64_t ne13 = src1->ne[3]; - // Prefer large over medium if either: - // - medium or large tiles would overfill the GPU - // - large tiles with a split_k==3 fits in the GPU and medium tiles with split_k==2 does not - // (medium with split_k==2 is probably better if it fits - more workgroups running and less split_k overhead) - bool prefer_large = tiles_m > shader_core_count || tiles_l > shader_core_count || - // split_k==3 with large tiles likely better than medium tiles with no split_k. - (tiles_l <= shader_core_count / 3 && tiles_m > shader_core_count / 2); + GGML_ASSERT(ne11 == 1); - if ((mm_l && (n > crossover_large && prefer_large)) || (!mm_m && !mm_s)) { - return aligned ? mmp->a_l : mmp->l; - } - // Use medium shader when the N dimension is greater than the small shader's tile size - uint32_t crossover_medium = mmp->s->wg_denoms[1]; - if ((mm_m && (n > crossover_medium)) || !mm_s) { - return aligned ? mmp->a_m : mmp->m; - } - return aligned ? mmp->a_s : mmp->s; + // With grouped query attention there are > 1 Q matrices per K, V matrix. + uint32_t gqa_ratio = (uint32_t)ne12 / (uint32_t)ne02; + if (gqa_ratio > 8 || gqa_ratio == 0 || ne12 != ne02 * gqa_ratio) { + gqa_ratio = 1; } - if ((mm_s && (m <= 32 || n <= 32)) || (!mm_m && !mm_l)) { - return aligned ? mmp->a_s : mmp->s; + vk_pipeline pipeline = ctx->device->pipeline_mul_mat_vec_p021_f16_f32[gqa_ratio - 1]; + + if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { + pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); } - if ((mm_m && (m <= 64 || n <= 64)) || !mm_l) { - return aligned ? mmp->a_m : mmp->m; + + { + // Request descriptor sets + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); } - return aligned ? mmp->a_l : mmp->l; -} -static uint32_t ggml_vk_guess_matmul_pipeline_align(ggml_backend_vk_context * ctx, vk_matmul_pipeline& mmp, int m, int n, ggml_type src0_type, ggml_type src1_type) { - VK_LOG_DEBUG("ggml_vk_guess_matmul_pipeline_align(" << m << ", " << n << ", " << ggml_type_name(src0_type) << ", " << ggml_type_name(src1_type) << ")"); - return ggml_vk_guess_matmul_pipeline(ctx, mmp, m, n, true, src0_type, src1_type)->align; -} + vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops], true); + vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1, true); -static void ggml_vk_matmul( - ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline& pipeline, - vk_subbuffer&& a, vk_subbuffer&& b, vk_subbuffer&& d, vk_subbuffer&& split_k_buffer, - uint32_t m, uint32_t n, uint32_t k, uint32_t stride_a, uint32_t stride_b, uint32_t stride_d, - uint32_t batch_stride_a, uint32_t batch_stride_b, uint32_t batch_stride_d, - uint32_t split_k, uint32_t batch, uint32_t ne02, uint32_t ne12, uint32_t broadcast2, uint32_t broadcast3, - uint32_t padded_n) { - VK_LOG_DEBUG("ggml_vk_matmul(a: (" << a.buffer->buffer << ", " << a.offset << ", " << a.size << "), b: (" << b.buffer->buffer << ", " << b.offset << ", " << b.size << "), d: (" << d.buffer->buffer << ", " << d.offset << ", " << d.size << "), split_k: (" << (split_k_buffer.buffer != nullptr ? split_k_buffer.buffer->buffer : VK_NULL_HANDLE) << ", " << split_k_buffer.offset << ", " << split_k_buffer.size << "), m: " << m << ", n: " << n << ", k: " << k << ", stride_a: " << stride_a << ", stride_b: " << stride_b << ", stride_d: " << stride_d << ", batch_stride_a: " << batch_stride_a << ", batch_stride_b: " << batch_stride_b << ", batch_stride_d: " << batch_stride_d << ", split_k: " << split_k << ", batch: " << batch << ", ne02: " << ne02 << ", ne12: " << ne12 << ", broadcast2: " << broadcast2 << ", broadcast3: " << broadcast3 << ", padded_n: " << padded_n << ")"); - if (split_k == 1) { - ggml_pipeline_request_descriptor_sets(ctx, pipeline, CEIL_DIV(batch, ctx->device->properties.limits.maxComputeWorkGroupCount[2])); + vk_subbuffer d_F0 = d_D; - uint32_t base_work_group_z = 0; - while (base_work_group_z < batch) { - uint32_t groups_z = std::min(batch - base_work_group_z, ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + uint32_t fusion_flags = 0; - const vk_mat_mat_push_constants pc = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, base_work_group_z, batch, k, ne02, ne12, broadcast2, broadcast3, padded_n }; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, d }, pc, { m, n, groups_z }); - base_work_group_z += groups_z; - } - return; - } + if (ctx->num_additional_fused_ops > 0) { + const ggml_tensor * add = cgraph->nodes[node_idx + 1]; + const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; - if (ctx->prealloc_split_k_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; } - GGML_ASSERT(batch_stride_d == m * n); + vk_subbuffer d_F1 = d_D; + if (ctx->num_additional_fused_ops > 1) { + const ggml_tensor * bias = cgraph->nodes[node_idx + 2]->src[1]; - // Round the split size up to a multiple of 256 (k-quant alignment) - uint32_t k_split = CEIL_DIV(k, split_k); - k_split = ROUNDUP_POW2(k_split, 256); + d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; + } - ggml_pipeline_request_descriptor_sets(ctx, pipeline, CEIL_DIV(batch, ctx->device->properties.limits.maxComputeWorkGroupCount[2])); + // compute - uint32_t base_work_group_z = 0; - while (base_work_group_z < batch) { - uint32_t groups_z = std::min(batch - base_work_group_z, ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + vk_mat_vec_p021_push_constants pc = { + (uint32_t)ne00, (uint32_t)ne01, (uint32_t)ne02, (uint32_t)ne12, + 0, 0, fusion_flags + }; - const vk_mat_mat_push_constants pc1 = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, base_work_group_z, batch, k_split, ne02, ne12, broadcast2, broadcast3, padded_n }; - // Make sure enough workgroups get assigned for split k to work - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, split_k_buffer }, pc1, { (CEIL_DIV(m, pipeline->wg_denoms[0]) * pipeline->wg_denoms[0]) * split_k, n, groups_z }); - base_work_group_z += groups_z; + init_pushconst_tensor_offsets(ctx, pc, src0, src1, nullptr, nullptr, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); + + uint32_t workgroups_z = (uint32_t)ne12; + // When gqa_ratio > 1, each invocation does multiple rows and we can launch fewer workgroups + if (gqa_ratio > 1) { + workgroups_z /= gqa_ratio; } - ggml_vk_sync_buffers(ctx, subctx); - const std::array pc2 = { (uint32_t)(m * n * batch), split_k }; - ggml_vk_dispatch_pipeline(ctx, subctx, ctx->device->pipeline_matmul_split_k_reduce, { split_k_buffer, d }, pc2, { m * n * batch, 1, 1 }); - ctx->prealloc_split_k_need_sync = true; + + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + d_Qx, + d_Qy, + d_D, + d_F0, + d_F1, + }, pc, { 1, (uint32_t)ne01, workgroups_z }); } -static vk_pipeline ggml_vk_guess_matmul_id_pipeline(ggml_backend_vk_context * ctx, vk_matmul_pipeline& mmp, uint32_t m, uint32_t n, bool aligned, ggml_type src0_type, ggml_type src1_type) { - VK_LOG_DEBUG("ggml_vk_guess_matmul_id_pipeline(" << m << ", " << n << ", " << aligned << ", " << ggml_type_name(src0_type) << ", " << ggml_type_name(src1_type) << ")"); +static void ggml_vk_mul_mat_vec_nc_f16_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + VK_LOG_DEBUG("ggml_vk_mul_mat_nc_f16_f32((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << "))"); + GGML_ASSERT(!ggml_is_transposed(src0)); + GGML_ASSERT(!ggml_is_transposed(src1)); + GGML_ASSERT(!ggml_is_permuted(src0)); + GGML_ASSERT(src0->type == GGML_TYPE_F16); + GGML_ASSERT(src1->type == GGML_TYPE_F32); + + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + const uint64_t ne03 = src0->ne[3]; + + const uint64_t nb01 = src0->nb[1]; + const uint64_t nb02 = src0->nb[2]; + + const uint64_t nb12 = src1->nb[2]; - // The q8_1 (integer dot) mmq path uses a different shader with its own - // shared-memory layout, so use the int-specific availability flags. - const bool is_q8_1 = (src1_type == GGML_TYPE_Q8_1); - const bool mm_l = is_q8_1 ? ctx->device->mul_mat_id_l_int[src0_type] : ctx->device->mul_mat_id_l[src0_type]; - const bool mm_m = is_q8_1 ? ctx->device->mul_mat_id_m_int[src0_type] : ctx->device->mul_mat_id_m[src0_type]; - const bool mm_s = is_q8_1 ? ctx->device->mul_mat_id_s_int[src0_type] : ctx->device->mul_mat_id_s[src0_type]; + // const uint64_t ne10 = src1->ne[0]; + const uint64_t ne11 = src1->ne[1]; + const uint64_t ne12 = src1->ne[2]; + // const uint64_t ne13 = src1->ne[3]; - if (ctx->device->coopmat2) { - // Use large shader when the N dimension is greater than the medium shader's tile size - uint32_t crossover_large = mmp->m->wg_denoms[1]; - if ((mm_l && (n > crossover_large)) || (!mm_m && !mm_s)) { - return aligned ? mmp->a_l : mmp->l; - } - // Use medium shader when the N dimension is greater than the small shader's tile size - uint32_t crossover_medium = mmp->s->wg_denoms[1]; - if ((mm_m && (n > crossover_medium)) || !mm_s) { - return aligned ? mmp->a_m : mmp->m; - } - return aligned ? mmp->a_s : mmp->s; - } + const uint32_t nb03 = (uint32_t)(src0->nb[3] / sizeof(ggml_fp16_t)); + const uint32_t nb13 = (uint32_t)(src1->nb[3] / sizeof(float)); + const uint32_t nb23 = (uint32_t)(dst->nb[3] / sizeof(float)); - if ((mm_s && (m <= 32 || n <= 32)) || (!mm_m && !mm_l)) { - return aligned ? mmp->a_s : mmp->s; - } - if ((mm_m && (m <= 64 || n <= 64)) || !mm_l) { - return aligned ? mmp->a_m : mmp->m; + GGML_ASSERT(ne11 == 1); + GGML_ASSERT(src0->ne[3] == src1->ne[3]); // checked in supports_op + + const uint32_t row_stride_x = nb01 / sizeof(ggml_fp16_t); + const uint32_t channel_stride_x = nb02 / sizeof(ggml_fp16_t); + const uint32_t channel_stride_y = nb12 / sizeof(float); + + vk_pipeline pipeline = ctx->device->pipeline_mul_mat_vec_nc_f16_f32; + if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { + pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); } - return aligned ? mmp->a_l : mmp->l; -} -static uint32_t ggml_vk_guess_matmul_id_pipeline_align(ggml_backend_vk_context * ctx, vk_matmul_pipeline& mmp, int m, int n, ggml_type src0_type, ggml_type src1_type) { - VK_LOG_DEBUG("ggml_vk_guess_matmul_pipeline_align(" << m << ", " << n << ", " << ggml_type_name(src0_type) << ", " << ggml_type_name(src1_type) << ")"); - return ggml_vk_guess_matmul_id_pipeline(ctx, mmp, m, n, true, src0_type, src1_type)->align; -} + { + // Request descriptor sets + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + } -static void ggml_vk_matmul_id( - ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline& pipeline, - vk_subbuffer&& a, vk_subbuffer&& b, vk_subbuffer&& d, vk_subbuffer&& ids, const vk_subbuffer & expert_count_buf, - uint32_t m, uint32_t n, uint32_t k, uint32_t stride_a, uint32_t stride_b, uint32_t stride_d, - uint32_t batch_stride_a, uint32_t batch_stride_b, uint32_t batch_stride_d, - uint32_t n_as, uint32_t nei0, uint32_t nei1, uint32_t nbi1, uint32_t ne11, - uint32_t padded_n) { - VK_LOG_DEBUG("ggml_vk_matmul_id(a: (" << a.buffer->buffer << ", " << a.offset << ", " << a.size << "), b: (" << b.buffer->buffer << ", " << b.offset << ", " << b.size << "), d: (" << d.buffer->buffer << ", " << d.offset << ", " << d.size << "), ids: (" << ids.buffer->buffer << ", " << ids.offset << ", " << ids.size << "), expert_count: (" << expert_count_buf.buffer->buffer << ", " << expert_count_buf.offset << ", " << expert_count_buf.size << "), " << - "m: " << m << ", n: " << n << ", k: " << k << ", stride_a: " << stride_a << ", stride_b: " << stride_b << ", stride_d: " << stride_d << ", " << - "batch_stride_a: " << batch_stride_a << ", batch_stride_b: " << batch_stride_b << ", batch_stride_d: " << batch_stride_d << ", " << - "n_as: " << n_as << ", nei0: " << nei0 << ", nei1: " << nei1 << ", nbi1: " << nbi1 << ", ne11: " << ne11 << ")"); - const vk_mat_mat_id_push_constants pc = { m, n, k, stride_a, stride_b, stride_d, batch_stride_a, batch_stride_b, batch_stride_d, - nei0, nei1, nbi1, ne11, padded_n }; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { a, b, d, ids, expert_count_buf }, pc, { m, nei1, n_as }); -} + vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops], true); + vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1, true); + vk_subbuffer d_F0 = d_D; -static bool ggml_vk_dim01_contiguous(const ggml_tensor * tensor) { - return - tensor->nb[0] == ggml_type_size(tensor->type) && - tensor->nb[1] == (tensor->nb[0]*tensor->ne[0])/ggml_blck_size(tensor->type) && - (tensor->ne[3] == 1 || tensor->nb[3] == tensor->nb[2]*tensor->ne[2]); -} + uint32_t fusion_flags = 0; -static vk_pipeline ggml_vk_get_cpy_pipeline(ggml_backend_vk_context * ctx, const ggml_tensor * src, const ggml_tensor * dst, ggml_type to) { + if (ctx->num_additional_fused_ops > 0) { + const ggml_tensor * add = cgraph->nodes[node_idx + 1]; + const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; - // Choose "contiguous copy" shader if src/dst are contiguous - bool contig = ggml_is_contiguous(src) && (!dst || ggml_is_contiguous(dst)); + d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; + } - // Use optimized "transpose" shader if src dim1 is the innermost dimension. - bool transpose = dst && src->nb[1] == ggml_type_size(to) && ggml_are_same_shape(dst, src); + vk_subbuffer d_F1 = d_D; + if (ctx->num_additional_fused_ops > 1) { + const ggml_tensor * bias = cgraph->nodes[node_idx + 2]->src[1]; - if (transpose && src->type == to) { - if (ggml_type_size(to) == 4) { - return ctx->device->pipeline_cpy_transpose_32; - } else if (ggml_type_size(to) == 2) { - return ctx->device->pipeline_cpy_transpose_16; - } + d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; } - if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_F32) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f32_f32; - } else { - return ctx->device->pipeline_cpy_f32_f32; - } - } - if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_F16) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f32_f16; - } else { - return ctx->device->pipeline_cpy_f32_f16; - } - } - if (src->type == GGML_TYPE_F16 && to == GGML_TYPE_F16) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f16_f16; - } else { - return ctx->device->pipeline_cpy_f16_f16; - } - } - if (src->type == GGML_TYPE_F16 && to == GGML_TYPE_F32) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f16_f32; - } else { - return ctx->device->pipeline_cpy_f16_f32; - } - } - if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_BF16) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f32_bf16; - } else { - return ctx->device->pipeline_cpy_f32_bf16; - } - } - if (src->type == GGML_TYPE_BF16 && to == GGML_TYPE_F32) { - if (contig) { - return ctx->device->pipeline_contig_cpy_bf16_f32; - } else { - return ctx->device->pipeline_cpy_bf16_f32; - } + // compute + vk_mat_vec_nc_push_constants pc = { + (uint32_t)ne00, (uint32_t)ne01, + row_stride_x, channel_stride_x, channel_stride_y, + (uint32_t)(ne12 / ne02), (uint32_t)ne12, + 0, 0, + nb03, nb13, nb23, fusion_flags + }; + + init_pushconst_tensor_offsets(ctx, pc, src0, src1, nullptr, nullptr, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); + + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + d_Qx, + d_Qy, + d_D, + d_F0, + d_F1, + }, pc, { (uint32_t)ne03, (uint32_t)ne01, (uint32_t)ne12 }); +} + +static int ggml_vk_fwht_pipeline_idx(int64_t n) { + switch (n) { + case 64: return 0; + case 128: return 1; + case 256: return 2; + case 512: return 3; + default: return -1; } - if (src->type == GGML_TYPE_F32 && to == GGML_TYPE_I32) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f32_i32; - } else { - return ctx->device->pipeline_cpy_f32_i32; - } +} + +bool ggml_vk_can_use_fwht(const ggml_backend_vk_context * ctx, const ggml_tensor * src1, const ggml_tensor * dst) { + if (ctx->num_additional_fused_ops != 0) { + return false; } - if (src->type == GGML_TYPE_I32 && to == GGML_TYPE_F32) { - if (contig) { - return ctx->device->pipeline_contig_cpy_i32_f32; - } else { - return ctx->device->pipeline_cpy_i32_f32; - } + + if (ggml_get_op_params_i32(dst, 1) != GGML_HINT_SRC0_IS_HADAMARD) { + return false; } - if (src->type == GGML_TYPE_F32) { - switch (to) { - case GGML_TYPE_Q1_0: - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_IQ4_NL: - return ctx->device->pipeline_cpy_f32_quant[to]; - default: - break; - } + + const int idx = ggml_vk_fwht_pipeline_idx(src1->ne[0]); + if (idx < 0 || ctx->device->pipeline_fwht_f32[idx] == nullptr) { + return false; } - if (to == GGML_TYPE_F32) { - switch (src->type) { - case GGML_TYPE_Q1_0: - case GGML_TYPE_Q2_0: - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q5_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q8_0: - case GGML_TYPE_IQ4_NL: - return ctx->device->pipeline_cpy_quant_f32[src->type]; - default: - break; - } + if (src1->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { + return false; } - if (src->type == to) { - // Copy two or four bytes at a time, depending on block size. - // For quantized types, we scale by block size/type size. But - // this path is also used for bf16->bf16 for example, where the - // type size must be exactly 2 or 4. - GGML_ASSERT(ggml_is_quantized(to) || ggml_type_size(src->type) == 2 || ggml_type_size(src->type) == 4); - if ((ggml_type_size(src->type) % 4) == 0) { - if (contig) { - return ctx->device->pipeline_contig_cpy_f32_f32; - } else { - return ctx->device->pipeline_cpy_f32_f32; - } - } else { - if (contig) { - return ctx->device->pipeline_contig_cpy_f16_f16; - } else { - return ctx->device->pipeline_cpy_f16_f16; - } - } + if (!ggml_is_contiguous(src1)) { + return false; } + GGML_ASSERT(ggml_is_contiguous(dst)); - std::cerr << "Missing CPY op for types: " << ggml_type_name(src->type) << " " << ggml_type_name(to) << std::endl; - GGML_ABORT("fatal error"); + return true; } -static void ggml_vk_cpy_to_contiguous(ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline pipeline, const ggml_tensor * tensor, const vk_subbuffer & in, const vk_subbuffer & out) { - VK_LOG_DEBUG("ggml_vk_cpy_to_contiguous((" << tensor << ", type=" << tensor->type << ", ne0=" << tensor->ne[0] << ", ne1=" << tensor->ne[1] << ", ne2=" << tensor->ne[2] << ", ne3=" << tensor->ne[3] << ", nb0=" << tensor->nb[0] << ", nb1=" << tensor->nb[1] << ", nb2=" << tensor->nb[2] << ", nb3=" << tensor->nb[3] << "), "; - std::cerr << "buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ")"); - - const uint32_t ne = ggml_nelements(tensor); - std::array elements; - - if (ne > 262144) { - elements = { 512, 512, CEIL_DIV(ne, 262144) }; - } else if (ne > 512) { - elements = { 512, CEIL_DIV(ne, 512), 1 }; - } else { - elements = { ne, 1, 1 }; - } +void ggml_vk_fwht(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src, ggml_tensor * dst) { + const int idx = ggml_vk_fwht_pipeline_idx(src->ne[0]); + vk_pipeline pipeline = ctx->device->pipeline_fwht_f32[idx]; - vk_op_unary_push_constants pc = vk_op_unary_push_constants_init(tensor, tensor, ne); - pc.nb10 = 1; - pc.nb11 = (uint32_t)tensor->ne[0]; - pc.nb12 = (uint32_t)(tensor->ne[0] * tensor->ne[1]); - pc.nb13 = (uint32_t)(tensor->ne[0] * tensor->ne[1] * tensor->ne[2]); - init_pushconst_fastdiv(pc); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, elements); - ggml_vk_sync_buffers(ctx, subctx); -} + const uint32_t rows_per_workgroup = 4; + const uint32_t n_rows = (uint32_t)ggml_nrows(src); + const uint32_t max_workgroups_x = ctx->device->properties.limits.maxComputeWorkGroupCount[0]; -// Copy/convert tensor into a caller-defined dense layout. Destination strides -// are in output elements, not bytes. -static void ggml_vk_cpy_to_strided( - ggml_backend_vk_context * ctx, vk_context& subctx, vk_pipeline pipeline, const ggml_tensor * tensor, - const vk_subbuffer & in, const vk_subbuffer & out, - uint32_t nb10, uint32_t nb11, uint32_t nb12, uint32_t nb13) { - VK_LOG_DEBUG("ggml_vk_cpy_to_strided((" << tensor << ", type=" << tensor->type << ", ne0=" << tensor->ne[0] << ", ne1=" << tensor->ne[1] << ", ne2=" << tensor->ne[2] << ", ne3=" << tensor->ne[3] << ", nb0=" << tensor->nb[0] << ", nb1=" << tensor->nb[1] << ", nb2=" << tensor->nb[2] << ", nb3=" << tensor->nb[3] << "), "; - std::cerr << "dst_nb=(" << nb10 << ", " << nb11 << ", " << nb12 << ", " << nb13 << "), buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ")"); + const uint32_t total_workgroups = CEIL_DIV(n_rows, rows_per_workgroup); + const uint32_t workgroups_x = std::min(total_workgroups, max_workgroups_x); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - const uint32_t ne = ggml_nelements(tensor); - std::array elements; + const vk_subbuffer src_buf = ggml_vk_tensor_subbuffer(ctx, src, true); + const vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); - if (ne > 262144) { - elements = { 512, 512, CEIL_DIV(ne, 262144) }; - } else if (ne > 512) { - elements = { 512, CEIL_DIV(ne, 512), 1 }; - } else { - elements = { ne, 1, 1 }; - } + vk_op_fwht_push_constants pc = { + n_rows, + 0, + 0, + 1.0f / std::sqrt((float)src->ne[0]), + }; + init_pushconst_tensor_offsets(ctx, pc, src, nullptr, nullptr, nullptr, dst); - vk_op_unary_push_constants pc = vk_op_unary_push_constants_init(tensor, tensor, ne); - pc.nb10 = nb10; - pc.nb11 = nb11; - pc.nb12 = nb12; - pc.nb13 = nb13; - init_pushconst_fastdiv(pc); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, elements); - ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src_buf, dst_buf }, pc, { workgroups_x, 1, 1 }); } -static vk_pipeline ggml_vk_get_quantize_pipeline(ggml_backend_vk_context * ctx, ggml_type type) { - switch(type) { - case GGML_TYPE_Q8_1: - return ctx->device->pipeline_quantize_q8_1_x4; - default: - std::cerr << "Missing quantize pipeline for type: " << ggml_type_name(type) << std::endl; - GGML_ABORT("fatal error"); - } +static uint32_t ggml_vk_nb_elem(const ggml_tensor * t, int i) { + return (uint32_t)(t->nb[i] / ggml_type_size(t->type)); } -static void ggml_vk_quantize_q8_1(ggml_backend_vk_context * ctx, vk_context& subctx, const vk_subbuffer & in, const vk_subbuffer & out, uint32_t ne) { - VK_LOG_DEBUG("ggml_vk_quantize_q8_1(" << "buffer in size=" << in.buffer->size << ", buffer out size=" << out.buffer->size << ", " << ne << ")"); +void ggml_vk_dsv4_hc_comb(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * mixes, const ggml_tensor * scale, const ggml_tensor * base, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_dsv4_hc_comb(" << mixes << ", " << scale << ", " << base << ", " << dst << ")"); - vk_pipeline pipeline = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); + vk_pipeline pipeline = ctx->device->pipeline_dsv4_hc_comb_f32; + GGML_ASSERT(pipeline != nullptr); - const uint32_t num_blocks = CEIL_DIV(ne, pipeline->wg_denoms[0]); - // clamp the number of elements to the max workgroup count. The shader will iterate over the total number of blocks. - const uint64_t max_elements = std::min(uint64_t{ctx->device->properties.limits.maxComputeWorkGroupCount[0]} * pipeline->wg_denoms[0], std::numeric_limits::max()); - const uint32_t elements = std::min(ne, static_cast(max_elements)); + const uint32_t n_tokens = (uint32_t)mixes->ne[1]; - const vk_quantize_q8_1_push_constants pc = { - ne, - num_blocks, + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + const vk_subbuffer mixes_buf = ggml_vk_tensor_subbuffer(ctx, mixes, true); + const vk_subbuffer scale_buf = ggml_vk_tensor_subbuffer(ctx, scale, true); + const vk_subbuffer base_buf = ggml_vk_tensor_subbuffer(ctx, base, true); + const vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); + + vk_op_dsv4_hc_comb_push_constants pc = { + n_tokens, + ggml_vk_nb_elem(mixes, 0), ggml_vk_nb_elem(mixes, 1), + ggml_vk_nb_elem(scale, 0), + ggml_vk_nb_elem(base, 0), + ggml_vk_nb_elem(dst, 0), ggml_vk_nb_elem(dst, 1), ggml_vk_nb_elem(dst, 2), + 0, 0, 0, 0, + ggml_get_op_params_f32(dst, 0), + (uint32_t)ggml_get_op_params_i32(dst, 1), }; + init_pushconst_tensor_offsets(ctx, pc, mixes, scale, base, nullptr, dst); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { in, out }, pc, { elements, 1, 1 }); - ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { mixes_buf, scale_buf, base_buf, dst_buf }, pc, { n_tokens, 1, 1 }); } -static vk_pipeline ggml_vk_get_64b_indexing_pipeline(ggml_backend_vk_context * ctx, vk_pipeline &pipeline) { - GGML_UNUSED(ctx); -#if defined(VK_EXT_shader_64bit_indexing) - vk_pipeline *ptr = &pipeline; - while (*ptr) { - if ((*ptr)->is_64b_indexing) { - return *ptr; - } - ptr = &(*ptr)->next; - } -#endif - return pipeline; -} +void ggml_vk_dsv4_hc_pre(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * x, const ggml_tensor * weights, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_dsv4_hc_pre(" << x << ", " << weights << ", " << dst << ")"); -static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst, bool disable_split_k) { - VK_LOG_DEBUG("ggml_vk_mul_mat_q_f16((" << src0 << ", name=" << src0->name << ", type=" << ggml_type_name(src0->type) << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << ggml_type_name(src1->type) << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << ggml_type_name(dst->type) << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << "))"); - GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT - GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT + const float scale = ggml_get_op_params_f32(dst, 0); + const bool gated = ggml_get_op_params_i32(dst, 1) != 0; - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - const uint64_t ne03 = src0->ne[3]; + vk_pipeline pipeline = gated ? ctx->device->pipeline_dsv4_hc_pre_gated_f32 : ctx->device->pipeline_dsv4_hc_pre_f32; + GGML_ASSERT(pipeline != nullptr); - const uint64_t ne10 = src1->ne[0]; - const uint64_t ne11 = src1->ne[1]; - const uint64_t ne12 = src1->ne[2]; + const uint32_t n_embd = (uint32_t)x->ne[0]; + const uint32_t n_tokens = (uint32_t)x->ne[2]; + + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + const vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x, true); + const vk_subbuffer w_buf = ggml_vk_tensor_subbuffer(ctx, weights, true); + const vk_subbuffer d_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); + + vk_op_dsv4_hc_pre_push_constants pc = { + n_embd, n_tokens, + ggml_vk_nb_elem(x, 0), ggml_vk_nb_elem(x, 1), ggml_vk_nb_elem(x, 2), + ggml_vk_nb_elem(weights, 0), ggml_vk_nb_elem(weights, 1), ggml_vk_nb_elem(weights, 2), + ggml_vk_nb_elem(dst, 0), ggml_vk_nb_elem(dst, 1), + 0, 0, 0, + scale, + }; + init_pushconst_tensor_offsets(ctx, pc, x, weights, nullptr, nullptr, dst); + + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { x_buf, w_buf, d_buf }, pc, { n_embd, n_tokens, 1 }); +} + +void ggml_vk_dsv4_hc_post(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * x, const ggml_tensor * residual, const ggml_tensor * post, const ggml_tensor * comb, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_dsv4_hc_post(" << x << ", " << residual << ", " << post << ", " << comb << ", " << dst << ")"); + + vk_pipeline pipeline = comb ? ctx->device->pipeline_dsv4_hc_post_f32 : ctx->device->pipeline_dsv4_hc_post_nocomb_f32; + GGML_ASSERT(pipeline != nullptr); + + const uint32_t n_embd = (uint32_t)x->ne[0]; + const uint32_t n_tokens = (uint32_t)x->ne[1]; + + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + const vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x, true); + const vk_subbuffer r_buf = ggml_vk_tensor_subbuffer(ctx, residual, true); + const vk_subbuffer p_buf = ggml_vk_tensor_subbuffer(ctx, post, true); + const vk_subbuffer c_buf = comb ? ggml_vk_tensor_subbuffer(ctx, comb, true) : x_buf; + const vk_subbuffer d_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); + + vk_op_dsv4_hc_post_push_constants pc = { + n_embd, n_tokens, + ggml_vk_nb_elem(x, 0), ggml_vk_nb_elem(x, 1), + ggml_vk_nb_elem(residual, 0), ggml_vk_nb_elem(residual, 1), ggml_vk_nb_elem(residual, 2), + ggml_vk_nb_elem(post, 0), ggml_vk_nb_elem(post, 1), + comb ? ggml_vk_nb_elem(comb, 0) : 0, comb ? ggml_vk_nb_elem(comb, 1) : 0, comb ? ggml_vk_nb_elem(comb, 2) : 0, + ggml_vk_nb_elem(dst, 0), ggml_vk_nb_elem(dst, 1), ggml_vk_nb_elem(dst, 2), + 0, 0, 0, 0, 0, + }; + init_pushconst_tensor_offsets(ctx, pc, x, residual, post, comb, dst); + + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { x_buf, r_buf, p_buf, c_buf, d_buf }, pc, { n_embd, n_tokens, 1 }); +} + +void ggml_vk_mul_mat(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + ggml_tensor * src0 = dst->src[0]; + ggml_tensor * src1 = dst->src[1]; + VK_LOG_DEBUG("ggml_vk_mul_mat(" << src0 << ", " << src1 << ", " << dst << ")"); + + // Handle huge A matrix by splitting the M dimensions. This works well for convolution use cases + // where the M dimension is very large. + // Split_k doesn't work with M splitting. + // This only supports batchsize == 1. + const size_t nbytes = ggml_nbytes(src0); + const bool needs_split = dst->ne[2] == 1 && dst->ne[3] == 1 && nbytes > ctx->device->properties.limits.maxStorageBufferRange; + if (needs_split) { + // Choose the number of rows that can fit (and divide by two, to allow for any additional offsets) + const uint32_t M_split = ctx->device->properties.limits.maxStorageBufferRange / (2 * src0->nb[1]); + uint32_t m_offset = 0; + while (m_offset < dst->ne[0]) { + const uint32_t cur_M_size = std::min(M_split, (uint32_t)(dst->ne[0] - m_offset)); + ggml_tensor dst2 = *dst; + ggml_tensor src02 = *src0; + + dst2.view_src = dst->view_src ? dst->view_src : dst; + src02.view_src = src0->view_src ? src0->view_src : src0; + + dst2.view_offs += m_offset * dst->nb[0]; + src02.view_offs += m_offset * src0->nb[1]; + dst2.ne[0] = cur_M_size; + src02.ne[1] = cur_M_size; + + ggml_vk_mul_mat_q_f16(ctx, subctx, &src02, src1, &dst2, true); + + m_offset += cur_M_size; + } + } else if (ggml_vk_can_use_fwht(ctx, src1, dst)) { + ggml_vk_fwht(ctx, subctx, src1, dst); + } else if (src0->type == GGML_TYPE_F16 && ggml_is_permuted(src0) && ggml_is_permuted(src1) && dst->ne[1] == 1 && + // detect 0213 permutation, and batch size of 1 + src0->nb[0] <= src0->nb[2] && + src0->nb[2] <= src0->nb[1] && + src0->nb[1] <= src0->nb[3] && + src1->nb[0] <= src1->nb[2] && + src1->nb[2] <= src1->nb[1] && + src1->nb[1] <= src1->nb[3] && + src0->ne[3] == 1 && + src1->ne[3] == 1 && + src0->ne[1] <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && + src1->ne[2] <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]) { + ggml_vk_mul_mat_vec_p021_f16_f32(ctx, subctx, cgraph, node_idx); + } else if (src0->type == GGML_TYPE_F16 && !ggml_is_contiguous(src0) && !ggml_is_transposed(src1) && dst->ne[1] == 1 && + !ggml_is_permuted(src0) && !ggml_is_permuted(src1) && + src0->ne[3] <= ctx->device->properties.limits.maxComputeWorkGroupCount[0] && + src0->ne[1] <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && + src1->ne[2] <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]) { + ggml_vk_mul_mat_vec_nc_f16_f32(ctx, subctx, cgraph, node_idx); + // With one output row, B^T*A has the same flat output as A^T*B. + } else if (ctx->num_additional_fused_ops == 0 && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16) && + (src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16 || src1->type == GGML_TYPE_BF16 || ggml_is_quantized(src1->type)) && + dst->ne[0] == 1 && dst->ne[1] > mul_mat_vec_max_cols && + src0->ne[2] == 1 && src0->ne[3] == 1 && + src1->ne[2] == 1 && src1->ne[3] == 1 && + ggml_is_contiguous(src0) && ggml_is_contiguous(src1) && ggml_is_contiguous(dst) && + get_misalign_bytes(ctx, src0) == 0 && get_misalign_bytes(ctx, src1) == 0 && get_misalign_bytes(ctx, dst) == 0) { + ggml_vk_mul_mat_vec_q_f16(ctx, subctx, cgraph, node_idx, true); + // mul_mat_vec supports batching ne12*ne13 when ne11==1, or treating ne11 as the batch size (up to four) + // when ne12 and ne13 are one. + } else if ((dst->ne[1] == 1 || (dst->ne[1] <= mul_mat_vec_max_cols && src1->ne[2] * src1->ne[3] == 1)) && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16 || ggml_is_quantized(src0->type))) { + ggml_vk_mul_mat_vec_q_f16(ctx, subctx, cgraph, node_idx); + } else { + ggml_vk_mul_mat_q_f16(ctx, subctx, src0, src1, dst, false); + } +} + +static void ggml_vk_mul_mat_id_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * ids, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_mul_mat_id_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << ids << ", name=" << ids->name << ", type=" << ids->type << ", ne0=" << ids->ne[0] << ", ne1=" << ids->ne[1] << ", ne2=" << ids->ne[2] << ", ne3=" << ids->ne[3] << ", nb0=" << ids->nb[0] << ", nb1=" << ids->nb[1] << ", nb2=" << ids->nb[2] << ", nb3=" << ids->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3] << "),)"); + GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT + GGML_ASSERT(ids->type == GGML_TYPE_I32); + + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + // const uint64_t ne03 = src0->ne[3]; + + const uint64_t ne10 = src1->ne[0]; + const uint64_t ne11 = src1->ne[1]; + const uint64_t ne12 = src1->ne[2]; const uint64_t ne13 = src1->ne[3]; + const uint64_t nei0 = ids->ne[0]; + const uint64_t nei1 = ids->ne[1]; + + const uint32_t nbi0 = ids->nb[0]; + const uint32_t nbi1 = ids->nb[1]; + const uint32_t nbi2 = ids->nb[2]; + + const uint64_t ne20 = dst->ne[0]; const uint64_t ne21 = dst->ne[1]; - const uint32_t stride_d = dst->nb[1] / ggml_type_size(dst->type); - const uint32_t stride_batch_d = stride_d*ne21; + // const uint64_t ne22 = dst->ne[2]; + // const uint64_t ne23 = dst->ne[3]; - const uint64_t r2 = ne12 / ne02; - const uint64_t r3 = ne13 / ne03; + const uint64_t n_as = ne02; + // n_as counts, n_as offsets, one total, then one packed row id per (expert, token). + // Hoisting requires 16-bit indices for the packing and a table that fits one binding. + const uint64_t hoisted_row_id_words = 2 * n_as + 1 + nei0 * nei1; + // 1024 matches MAX_EXPERTS in count_experts.comp and LLAMA_MAX_EXPERTS. It costs + // 3 * 1024 * 4 = 12 KiB of shared memory, within the 16 KiB Vulkan guarantees. + const bool hoist_row_ids = n_as <= 1024 && nei0 <= 0xffff && nei1 <= 0xffff && + hoisted_row_id_words * sizeof(uint32_t) <= + ctx->device->properties.limits.maxStorageBufferRange; ggml_backend_vk_buffer_context * dst_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; ggml_backend_vk_buffer_context * src0_buf_ctx = (ggml_backend_vk_buffer_context *)src0->buffer->context; ggml_backend_vk_buffer_context * src1_buf_ctx = (ggml_backend_vk_buffer_context *)src1->buffer->context; + ggml_backend_vk_buffer_context * ids_buf_ctx = (ggml_backend_vk_buffer_context *)ids->buffer->context; vk_buffer d_Qx = nullptr; size_t qx_buf_offset = 0; vk_buffer d_Qy = nullptr; size_t qy_buf_offset = 0; + vk_buffer d_ids = nullptr; + size_t ids_buf_offset = 0; bool src0_uma = false; bool src1_uma = false; + bool ids_uma = false; if (ctx->device->uma) { ggml_vk_host_get(ctx->device, src0->data, d_Qx, qx_buf_offset); ggml_vk_host_get(ctx->device, src1->data, d_Qy, qy_buf_offset); + ggml_vk_host_get(ctx->device, ids->data, d_ids, ids_buf_offset); src0_uma = d_Qx != nullptr; src1_uma = d_Qy != nullptr; + ids_uma = d_ids != nullptr; } // Reformat and convert to fp16 if non-contiguous, or for coopmat2 for better perf const bool x_non_contig = (ctx->device->coopmat2 && src0->type == GGML_TYPE_F32) || !ggml_vk_dim01_contiguous(src0); - const bool y_non_contig = (ctx->device->coopmat2 && src1->type == GGML_TYPE_F32) || - (src0->type == GGML_TYPE_BF16 && src1->type != GGML_TYPE_BF16) || - !ggml_vk_dim01_contiguous(src1); - // If src0 is BF16, try to use a BF16 x BF16 multiply ggml_type f16_type = src0->type == GGML_TYPE_BF16 ? GGML_TYPE_BF16 : GGML_TYPE_F16; +#if defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) + // B must already be, or be convertible to, the matmul B type used by this path. + const bool y_decode_vector_supported = ctx->device->coopmat2_decode_vector && + (f16_type != GGML_TYPE_BF16 || ctx->device->coopmat2_bf16_support) && + (src1->type == GGML_TYPE_F32 || src1->type == f16_type); + // If B is copied to prealloc_y, we can choose a 4-element-aligned row stride. + const bool y_decode_vector_uses_prealloc = !ggml_vk_dim01_contiguous(src1) || src1->type != f16_type; + // Direct B reads are safe only if row starts and the original buffer offset are 4-element aligned. + const bool y_decode_vector_aligned = + (ne10 % 4 == 0) && + (y_decode_vector_uses_prealloc || get_misalign_bytes(ctx, src1) % (4 * ggml_type_size(src1->type)) == 0); + // Stage B only when decode-vector is available and direct B reads would be misaligned. + const bool y_decode_vector_staging = y_decode_vector_supported && !y_decode_vector_aligned; +#else + const bool y_decode_vector_staging = false; +#endif + const bool y_non_contig = y_decode_vector_staging || + (ctx->device->coopmat2 && src1->type == GGML_TYPE_F32) || + // Intel coopmat1: force f32->f16 conversion so the f16 B-type quant pipeline is used. + (ctx->device->coopmat_support && !ctx->device->coopmat2 && + ctx->device->vendor_id == VK_VENDOR_ID_INTEL && + ggml_is_quantized(src0->type) && src1->type == GGML_TYPE_F32) || + (src0->type == GGML_TYPE_BF16 && src1->type != GGML_TYPE_BF16) || + !ggml_vk_dim01_contiguous(src1); const bool y_f32_kernel = src1->type == GGML_TYPE_F32 && !y_non_contig; bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0; // Check for mmq first - vk_matmul_pipeline mmp = quantize_y ? ggml_vk_get_mul_mat_mat_pipeline(ctx, src0->type, GGML_TYPE_Q8_1, (ggml_prec)dst->op_params[0]) : nullptr; + const std::vector* mmp_map = quantize_y ? ggml_vk_get_mul_mat_mat_pipeline_map(ctx, src0->type, GGML_TYPE_Q8_1, (ggml_prec)dst->op_params[0], true) : nullptr; - if (mmp == nullptr) { + if (mmp_map == nullptr) { // Fall back to f16 dequant mul mat - mmp = ggml_vk_get_mul_mat_mat_pipeline(ctx, src0->type, y_non_contig ? f16_type : src1->type, (ggml_prec)dst->op_params[0]); + mmp_map = ggml_vk_get_mul_mat_mat_pipeline_map(ctx, src0->type, y_non_contig ? f16_type : src1->type, (ggml_prec)dst->op_params[0], true); quantize_y = false; } - const bool qx_needs_dequant = mmp == nullptr || x_non_contig; - const bool qy_needs_dequant = !quantize_y && ((src1->type != f16_type && !y_f32_kernel) || y_non_contig); + const bool qx_needs_dequant = mmp_map == nullptr || x_non_contig; + bool qy_needs_dequant = !quantize_y && ((src1->type != f16_type && !y_f32_kernel) || y_non_contig); if (qx_needs_dequant) { // Fall back to dequant + f16 mulmat - mmp = ggml_vk_get_mul_mat_mat_pipeline(ctx, f16_type, y_f32_kernel ? GGML_TYPE_F32 : f16_type, (ggml_prec)dst->op_params[0]); + mmp_map = ggml_vk_get_mul_mat_mat_pipeline_map(ctx, f16_type, y_f32_kernel ? GGML_TYPE_F32 : f16_type, (ggml_prec)dst->op_params[0], true); } + // Coopmat2 MUL_MAT_ID BK specialization constants in ggml_vk_load_shaders are at most 64. + const uint32_t y_staged_row_stride = ctx->device->coopmat2 && !quantize_y ? ggml_vk_align_size(ne10, 64) : ne10; + const bool y_needs_k_padding = ne10 != y_staged_row_stride; + const bool y_needs_reformat = y_non_contig || y_needs_k_padding; + qy_needs_dequant = qy_needs_dequant || y_needs_k_padding; + // Not implemented - GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT + GGML_ASSERT(y_needs_reformat || !qy_needs_dequant); // NOLINT - const ggml_type effective_src1_type = quantize_y ? GGML_TYPE_Q8_1 : (y_f32_kernel ? GGML_TYPE_F32 : src1->type); + GGML_ASSERT(mmp_map != nullptr); - const uint32_t kpad = quantize_y ? 0 : ggml_vk_align_size(ne10, ggml_vk_guess_matmul_pipeline_align(ctx, mmp, ne01, ne11, qx_needs_dequant ? f16_type : src0->type, effective_src1_type)); - const bool aligned = !quantize_y && ne10 == kpad && ne01 > 8 && ne11 > 8; + const uint32_t kpad = quantize_y ? 0 : ggml_vk_align_size(ne10, ggml_vk_guess_matmul_pipeline_align_map(ctx, *mmp_map, ne01, nei1, true)); + const bool aligned = !quantize_y && ne10 == kpad && ne01 > 8 && nei1 > 8; - vk_pipeline pipeline = ggml_vk_guess_matmul_pipeline(ctx, mmp, ne01, ne11, aligned, qx_needs_dequant ? f16_type : src0->type, effective_src1_type); + vk_pipeline pipeline = ggml_vk_guess_matmul_pipeline_map(ctx, *mmp_map, ne01, nei1, aligned, true); if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); } - - // Reserve extra storage in the N dimension for the Y matrix, so we can avoid bounds-checking - uint32_t padded_n = qy_needs_dequant ? ROUNDUP_POW2(ne11, pipeline->wg_denoms[1]) : ne11; const uint64_t x_ne = ggml_nelements(src0); - // 128 elements per Q8_1 x4 block - const uint64_t y_ne = padded_n * ne10 * ne12 * ne13; + const uint64_t y_ne = (uint64_t)y_staged_row_stride * ne11 * ne12 * ne13; const uint64_t d_ne = ggml_nelements(dst); - const uint32_t split_k = ggml_vk_guess_split_k(ctx, ne01, ne11, ne10, disable_split_k, pipeline); - const uint64_t qx_sz = ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type); - const uint64_t qy_sz = ggml_type_size(src1->type) * y_ne / ggml_blck_size(src1->type); + const uint64_t qy_sz = ggml_type_size(src1->type) * ggml_nelements(src1) / ggml_blck_size(src1->type); const uint64_t x_sz = !qx_needs_dequant ? qx_sz : sizeof(ggml_fp16_t) * x_ne; const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : (y_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); + const uint64_t ids_sz = nbi2; const uint64_t d_sz = sizeof(float) * d_ne; vk_pipeline to_fp16_vk_0 = nullptr; vk_pipeline to_fp16_vk_1 = nullptr; vk_pipeline to_q8_1 = nullptr; + auto make_y_staged_dst = [&]() { + ggml_tensor y_staged_dst = *src1; + y_staged_dst.type = f16_type; + y_staged_dst.nb[0] = ggml_type_size(f16_type); + y_staged_dst.nb[1] = y_staged_dst.nb[0] * y_staged_row_stride; + y_staged_dst.nb[2] = y_staged_dst.nb[1] * ne11; + y_staged_dst.nb[3] = y_staged_dst.nb[2] * y_staged_dst.ne[2]; + return y_staged_dst; + }; + if (x_non_contig) { to_fp16_vk_0 = ggml_vk_get_cpy_pipeline(ctx, src0, nullptr, f16_type); } else { to_fp16_vk_0 = ggml_vk_get_to_fp16(ctx, src0->type); } - if (y_non_contig) { - to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, nullptr, f16_type); + if (y_needs_reformat) { + ggml_tensor y_staged_dst; + const ggml_tensor * y_staged_dst_ptr = nullptr; + if (y_needs_k_padding) { + y_staged_dst = make_y_staged_dst(); + y_staged_dst_ptr = &y_staged_dst; + } + + to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, y_staged_dst_ptr, f16_type); } else { to_fp16_vk_1 = ggml_vk_get_to_fp16(ctx, src1->type); } @@ -9266,13 +7244,15 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub if (quantize_y) { to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); } + vk_pipeline count_experts = ctx->device->pipeline_count_experts; + + const size_t expert_data_size = sizeof(uint32_t) * + (hoist_row_ids ? hoisted_row_id_words : n_as); { - const uint64_t split_k_size = split_k > 1 ? d_sz * split_k : 0; if ( (qx_needs_dequant && x_sz > ctx->device->properties.limits.maxStorageBufferRange) || - (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange) || - (split_k > 1 && split_k_size > ctx->device->properties.limits.maxStorageBufferRange)) { + (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange)) { GGML_ABORT("Requested preallocation size is too large"); } if (qx_needs_dequant && ctx->prealloc_size_x < x_sz) { @@ -9283,12 +7263,13 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub ctx->prealloc_size_y = y_sz; ggml_vk_preallocate_buffers(ctx, subctx); } - if (split_k > 1 && ctx->prealloc_size_split_k < split_k_size) { - ctx->prealloc_size_split_k = split_k_size; + if (ctx->prealloc_size_split_k < expert_data_size) { + ctx->prealloc_size_split_k = expert_data_size; ggml_vk_preallocate_buffers(ctx, subctx); } // Request descriptor sets + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); if (qx_needs_dequant) { ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); } @@ -9298,15 +7279,12 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub if (quantize_y) { ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); } - if (split_k > 1) { - ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, 1); - } + ggml_pipeline_request_descriptor_sets(ctx, count_experts, 1); } vk_buffer d_D = dst_buf_ctx->dev_buffer; const uint64_t d_buf_offset = vk_tensor_offset(dst) + dst->view_offs; GGML_ASSERT(d_D != nullptr); - GGML_ASSERT(d_D->size >= d_buf_offset + d_sz); vk_buffer d_X; uint64_t x_buf_offset = 0; vk_buffer d_Y; @@ -9321,6 +7299,11 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub qy_buf_offset = vk_tensor_offset(src1) + src1->view_offs; GGML_ASSERT(d_Qy != nullptr); } + if (!ids_uma) { + d_ids = ids_buf_ctx->dev_buffer; + ids_buf_offset = vk_tensor_offset(ids) + ids->view_offs; + GGML_ASSERT(d_ids != nullptr); + } if (qx_needs_dequant) { d_X = ctx->prealloc_x; GGML_ASSERT(d_X->size >= x_sz); @@ -9346,43 +7329,80 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub ggml_vk_sync_buffers(ctx, subctx); } } + // Count how many times each expert is used + vk_subbuffer expert_count_buf = { ctx->prealloc_split_k, 0, expert_data_size }; + if (ctx->prealloc_split_k_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + { + vk_op_count_experts_push_constants pc = { (uint32_t)nei0, + (uint32_t)nei1, + (uint32_t)(nbi0 / ggml_type_size(ids->type)), + (uint32_t)(nbi1 / ggml_type_size(ids->type)), + (uint32_t)(get_misalign_bytes(ctx, ids) / ggml_type_size(ids->type)), + (uint32_t)n_as, + uint32_t(hoist_row_ids), + 0, 0 }; + init_pushconst_fastdiv(pc); + ggml_vk_dispatch_pipeline(ctx, subctx, count_experts, + { vk_subbuffer{ d_ids, ids_buf_offset, ids_sz }, expert_count_buf }, pc, + { hoist_row_ids ? 1u : (uint32_t)n_as, 1, 1}); + } if (x_non_contig) { ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, ggml_vk_subbuffer(ctx, d_Qx, qx_buf_offset), ggml_vk_subbuffer(ctx, d_X, 0)); } else if (qx_needs_dequant) { const std::vector pc = { (uint32_t)ne01, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)(ggml_nelements(src0)) }; - ggml_vk_dispatch_pipeline(ctx, subctx, to_fp16_vk_0, { vk_subbuffer{ d_Qx, qx_buf_offset, qx_sz }, vk_subbuffer{ d_X, 0, x_sz } }, pc, { (uint32_t)(x_ne), 1, 1}); - ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, to_fp16_vk_0, + { vk_subbuffer{ d_Qx, qx_buf_offset, qx_sz }, vk_subbuffer{ d_X, 0, x_sz } }, pc, { (uint32_t)x_ne, 1, 1}); } - if (y_non_contig) { + if (y_needs_reformat) { if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { + ctx->prealloc_y_last_k_padded != y_needs_k_padding) { if (ctx->prealloc_y_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } - ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0)); + if (y_needs_k_padding) { + GGML_ASSERT(y_sz % 4 == 0); + // Zero B padding because clamping only A can produce 0 * Inf or NaN. + subctx->s->buffer->buf.fillBuffer(d_Y->buffer, 0, y_sz, 0); + ggml_vk_sync_buffers(ctx, subctx); + const ggml_tensor y_staged_dst = make_y_staged_dst(); + const uint32_t y_staged_dst_type_size = ggml_type_size(y_staged_dst.type); + ggml_vk_cpy_to_strided( + ctx, subctx, to_fp16_vk_1, src1, + ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0), + (uint32_t)(y_staged_dst.nb[0] / y_staged_dst_type_size), + (uint32_t)(y_staged_dst.nb[1] / y_staged_dst_type_size), + (uint32_t)(y_staged_dst.nb[2] / y_staged_dst_type_size), + (uint32_t)(y_staged_dst.nb[3] / y_staged_dst_type_size)); + } else { + ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0)); + } ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = y_needs_k_padding; } } if (quantize_y) { if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { + ctx->prealloc_y_last_k_padded) { if (ctx->prealloc_y_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } ggml_vk_quantize_q8_1(ctx, subctx, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0), y_ne); ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; } } + ggml_vk_sync_buffers(ctx, subctx); uint32_t stride_batch_x = ne00*ne01; - uint32_t stride_batch_y = ne10*ne11; + uint32_t stride_b_y = y_needs_k_padding ? y_staged_row_stride : ne10; + uint32_t stride_batch_y = y_needs_k_padding ? y_staged_row_stride * ne11 : ne10*ne11; if (!ggml_vk_dim01_contiguous(src0) && !qx_needs_dequant) { stride_batch_x = src0->nb[0] / ggml_type_size(src0->type); @@ -9393,145 +7413,62 @@ static void ggml_vk_mul_mat_q_f16(ggml_backend_vk_context * ctx, vk_context& sub } // compute - ggml_vk_matmul( + ggml_vk_matmul_id( ctx, subctx, pipeline, { d_X, x_buf_offset, x_sz }, { d_Y, y_buf_offset, y_sz }, - ggml_vk_subbuffer(ctx, d_D, d_buf_offset), { ctx->prealloc_split_k, 0, d_sz * split_k }, - ne01, ne11, ne10, - ne10, ne10, stride_d, stride_batch_x, stride_batch_y, stride_batch_d, - split_k, ne12*ne13, ne02, ne12, r2, r3, padded_n + { d_D, d_buf_offset, d_sz }, { d_ids, ids_buf_offset, ids_sz }, expert_count_buf, + ne01, ne21, ne10, ne10, stride_b_y, ne01, + stride_batch_x, stride_batch_y, ne20*ne21, + n_as, nei0, nei1, nbi1 / ggml_type_size(ids->type), ne11, hoist_row_ids ); // NOLINT if (x_non_contig || qx_needs_dequant) { ctx->prealloc_x_need_sync = true; } - if (y_non_contig || quantize_y) { + if (y_needs_reformat || quantize_y) { ctx->prealloc_y_need_sync = true; } + ctx->prealloc_split_k_need_sync = true; } -// Device tuning -static bool ggml_vk_should_use_mmvq(const vk_device& device, uint32_t m, uint32_t n, uint32_t k, ggml_type src0_type) { - if (device->mmvq_mode == 1) { - return true; - } else if (device->mmvq_mode == -1) { - return false; - } - - // q6_k only has 2-byte alignment which makes it somewhat problematic, - // using MMVQ is only a win on Intel. - bool mmvq_q6 = device->vendor_id == VK_VENDOR_ID_INTEL; - if (src0_type == GGML_TYPE_Q6_K && !mmvq_q6) { - return false; - } - - // MMVQ is generally good for batches - if (n > 1) { - return true; - } - - // Quantization overhead is not worth it for small k - switch (device->vendor_id) { - case VK_VENDOR_ID_NVIDIA: - if (src0_type == GGML_TYPE_Q2_0 || src0_type == GGML_TYPE_Q2_K || src0_type == GGML_TYPE_Q3_K || src0_type == GGML_TYPE_IQ1_S || src0_type == GGML_TYPE_IQ1_M) { - return true; - } - - if (k <= 4096) { - return false; - } - - switch (src0_type) { - case GGML_TYPE_MXFP4: - case GGML_TYPE_Q8_0: - return device->architecture == vk_device_architecture::NVIDIA_PRE_TURING; - default: - return true; - } - case VK_VENDOR_ID_AMD: - if (k < 2048) { - return false; - } - - switch (src0_type) { - case GGML_TYPE_Q8_0: - return device->architecture == vk_device_architecture::AMD_GCN; - default: - return true; - } - case VK_VENDOR_ID_INTEL: - if (device->architecture == vk_device_architecture::INTEL_XE2) { - if (src0_type == GGML_TYPE_Q2_0 || src0_type == GGML_TYPE_Q2_K || src0_type == GGML_TYPE_Q3_K || src0_type == GGML_TYPE_Q6_K) { - return true; - } - } - - if (device->driver_id == vk::DriverId::eIntelProprietaryWindows) { - // Intel Windows proprietary driver MMVQ performance for !Q2/Q3/Q6 is worse than fp16, - // see https://github.com/ggml-org/llama.cpp/issues/17628 and - // https://github.com/ggml-org/llama.cpp/pull/23056 - return false; - } - - if (k < 2048) { - return false; - } - - switch (src0_type) { - // From tests on A770 Linux, may need more tuning - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q5_1: - return false; - default: - return true; - } - default: - return true; - } - - GGML_UNUSED(m); -} - -static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - - VK_LOG_DEBUG("ggml_vk_mul_mat_vec_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << ")),)"); - GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT - GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT +static void ggml_vk_mul_mat_vec_id_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + ggml_tensor * src0 = dst->src[0]; + ggml_tensor * src1 = dst->src[1]; + ggml_tensor * ids = dst->src[2]; + VK_LOG_DEBUG("ggml_vk_mul_mat_vec_id_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + std::cerr << "), (" << ids << ", name=" << ids->name << ", type=" << ids->type << ", ne0=" << ids->ne[0] << ", ne1=" << ids->ne[1] << ", ne2=" << ids->ne[2] << ", ne3=" << ids->ne[3] << ", nb0=" << ids->nb[0] << ", nb1=" << ids->nb[1] << ", nb2=" << ids->nb[2] << ", nb3=" << ids->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << "))"); + GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT + GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT + GGML_ASSERT(ids->type == GGML_TYPE_I32); const uint64_t ne00 = src0->ne[0]; const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - const uint64_t ne03 = src0->ne[3]; + // const uint64_t ne02 = src0->ne[2]; + // const uint64_t ne03 = src0->ne[3]; const uint64_t ne10 = src1->ne[0]; const uint64_t ne11 = src1->ne[1]; const uint64_t ne12 = src1->ne[2]; - const uint64_t ne13 = src1->ne[3]; + // const uint64_t ne13 = src1->ne[3]; + + const uint64_t nei0 = ids->ne[0]; + const uint64_t nei1 = ids->ne[1]; + const uint32_t nbi1 = (uint32_t)(ids->nb[1] / sizeof(int)); const uint64_t ne20 = dst->ne[0]; const uint64_t ne21 = dst->ne[1]; // const uint64_t ne22 = dst->ne[2]; // const uint64_t ne23 = dst->ne[3]; - const uint64_t r2 = ne12 / ne02; - const uint64_t r3 = ne13 / ne03; - - // batch_n indicates that we need to compute a few vector results, and this assumes - // ne12 and ne13 are 1. It overloads the batch_strides to hold the row strides. - GGML_ASSERT(ne11 == 1 || ne12 * ne13 == 1); - bool batch_n = ne11 > 1; - const bool x_non_contig = !ggml_vk_dim01_contiguous(src0); const bool y_non_contig = !ggml_vk_dim01_contiguous(src1); const bool f16_f32_kernel = src1->type == GGML_TYPE_F32; - bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0 && ggml_vk_should_use_mmvq(ctx->device, ne01, ne11, ne10, src0->type); + bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0 && ggml_vk_should_use_mmvq(ctx->device, ne01, ne12, ne10, src0->type); vk_pipeline to_fp16_vk_0 = nullptr; vk_pipeline to_fp16_vk_1 = nullptr; @@ -9545,12 +7482,12 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& } // Check for mmq first - vk_pipeline dmmv = quantize_y ? ggml_vk_get_dequantize_mul_mat_vec(ctx, src0->type, GGML_TYPE_Q8_1, ne11, ne20, ne00) : nullptr; + vk_pipeline dmmv = quantize_y ? ggml_vk_get_dequantize_mul_mat_vec_id(ctx, src0->type, GGML_TYPE_Q8_1, ne20, ne00) : nullptr; vk_pipeline to_q8_1 = nullptr; if (dmmv == nullptr) { // Fall back to f16 dequant mul mat - dmmv = ggml_vk_get_dequantize_mul_mat_vec(ctx, src0->type, src1->type, ne11, ne20, ne00); + dmmv = ggml_vk_get_dequantize_mul_mat_vec_id(ctx, src0->type, src1->type, ne20, ne00); quantize_y = false; } @@ -9558,16 +7495,15 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); } + const bool qx_needs_dequant = x_non_contig; + const bool qy_needs_dequant = !quantize_y && ((src1->type != GGML_TYPE_F16 && !f16_f32_kernel) || y_non_contig); + if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { dmmv = ggml_vk_get_64b_indexing_pipeline(ctx, dmmv); } - const bool qx_needs_dequant = x_non_contig; - const bool qy_needs_dequant = !quantize_y && ((src1->type != GGML_TYPE_F16 && !f16_f32_kernel) || y_non_contig); - // Not implemented GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT - GGML_ASSERT(!qx_needs_dequant || to_fp16_vk_0 != nullptr); // NOLINT GGML_ASSERT(!qy_needs_dequant || to_fp16_vk_1 != nullptr); // NOLINT GGML_ASSERT(dmmv != nullptr); @@ -9578,7 +7514,7 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& const uint64_t qx_sz = ggml_vk_align_size(ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type), ctx->device->properties.limits.minStorageBufferOffsetAlignment); const uint64_t x_sz = x_non_contig ? ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment) : qx_sz; const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : - (f16_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); + (f16_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); { if ( @@ -9605,18 +7541,20 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& if (quantize_y) { ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); } + ggml_pipeline_request_descriptor_sets(ctx, dmmv, nei1); } vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1); + vk_subbuffer d_ids = ggml_vk_tensor_subbuffer(ctx, ids); + vk_subbuffer d_F0 = d_D; vk_subbuffer d_X, d_Y; if (qx_needs_dequant) { d_X = { ctx->prealloc_x, 0, ctx->prealloc_x->size }; } else { d_X = d_Qx; - GGML_ASSERT(qx_sz == x_sz); } if (qy_needs_dequant || quantize_y) { d_Y = { ctx->prealloc_y, 0, ctx->prealloc_y->size }; @@ -9628,7 +7566,9 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& if (ctx->prealloc_x_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } + } + if (x_non_contig) { GGML_ASSERT(x_sz == ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment)); ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, d_Qx, d_X); } @@ -9636,41 +7576,34 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& GGML_ASSERT(y_sz == ggml_type_size(src1->type) * y_ne); if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { + ctx->prealloc_y_last_k_padded) { if (ctx->prealloc_y_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, d_Qy, d_Y); ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; } } if (quantize_y) { if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { + ctx->prealloc_y_last_k_padded) { if (ctx->prealloc_y_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } ggml_vk_quantize_q8_1(ctx, subctx, d_Qy, d_Y, y_ne); ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; } } - // For batch_n, the A matrix is the same for each batch, and B/D use the row stride as the batch stride - uint32_t stride_batch_x = batch_n ? 0 : ne00*ne01; - uint32_t stride_batch_y = batch_n ? ne10 : (ne10*ne11); - uint32_t stride_batch_d = batch_n ? ne20 : (ne20*ne21); - - if (!ggml_vk_dim01_contiguous(src0) && !qx_needs_dequant) { - stride_batch_x = src0->nb[0] / ggml_type_size(src0->type); - } + uint32_t stride_batch_y = ne10*ne11; if (!ggml_vk_dim01_contiguous(src1) && !qy_needs_dequant) { - stride_batch_y = src1->nb[0] / ggml_type_size(src1->type); + stride_batch_y = src1->nb[2] / ggml_type_size(src1->type); } const uint32_t max_groups_x = ctx->device->properties.limits.maxComputeWorkGroupCount[0]; @@ -9685,46 +7618,45 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& uint32_t fusion_flags = 0; - vk_subbuffer d_F0 = d_D; if (ctx->num_additional_fused_ops > 0) { - const ggml_tensor * add = cgraph->nodes[node_idx + 1]; - const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; + const ggml_tensor * bias = cgraph->nodes[node_idx + 1]->src[1]; d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; + + if (cgraph->nodes[node_idx + 1]->op == GGML_OP_MUL) { + fusion_flags |= MAT_VEC_FUSION_FLAGS_SCALE0; + } else { + GGML_ASSERT(cgraph->nodes[node_idx + 1]->op == GGML_OP_ADD_ID); + fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; + } } vk_subbuffer d_F1 = d_D; - if (ctx->num_additional_fused_ops == 2) { - const ggml_tensor * add = cgraph->nodes[node_idx + 2]; - const ggml_tensor * bias = add->src[0] == cgraph->nodes[node_idx + 1] ? add->src[1] : add->src[0]; + if (ctx->num_additional_fused_ops > 1) { + const ggml_tensor * scale = cgraph->nodes[node_idx + 2]->src[1]; - d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; + d_F1 = ggml_vk_tensor_subbuffer(ctx, scale); + fusion_flags |= MAT_VEC_FUSION_FLAGS_SCALE1; } - ggml_pipeline_request_descriptor_sets(ctx, dmmv, CEIL_DIV(ne12 * ne13, ctx->device->properties.limits.maxComputeWorkGroupCount[1])); - - uint32_t base_work_group_y = 0; - while (base_work_group_y < ne12 * ne13) { - - uint32_t groups_y = std::min((uint32_t)(ne12 * ne13) - base_work_group_y, ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - const vk_mat_vec_push_constants pc = { + // Loop over the batch dimension + for (uint32_t expert_i1 = 0; expert_i1 < nei1; ++expert_i1) { + const vk_mat_vec_id_push_constants pc = { (uint32_t)ne00, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne01, - stride_batch_x, stride_batch_y, stride_batch_d, - fusion_flags, base_work_group_y, - (uint32_t)ne02, (uint32_t)ne12, (uint32_t)r2, (uint32_t)r3, + (uint32_t)(ne00 * ne01), stride_batch_y, (uint32_t)(ne20 * ne21), + fusion_flags, + (uint32_t)nei0, (uint32_t)ne11, expert_i1, nbi1 }; ggml_vk_dispatch_pipeline(ctx, subctx, dmmv, - { - d_X, - d_Y, - d_D, - d_F0, - d_F1, - }, - pc, { groups_x, groups_y, groups_z }); - base_work_group_y += groups_y; + { + d_X, + d_Y, + d_D, + d_F0, + d_F1, + d_ids, + }, + pc, { groups_x, (uint32_t)nei0, groups_z }); } if (x_non_contig) { @@ -9735,3343 +7667,2100 @@ static void ggml_vk_mul_mat_vec_q_f16(ggml_backend_vk_context * ctx, vk_context& } } -static void ggml_vk_mul_mat_vec_p021_f16_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { +bool ggml_vk_use_mul_mat_vec_id(const struct ggml_cgraph * cgraph, int node_idx) { ggml_tensor * dst = cgraph->nodes[node_idx]; - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - VK_LOG_DEBUG("ggml_vk_mul_mat_p021_f16_f32(" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << "))"); - GGML_ASSERT(ggml_is_permuted(src0) && ggml_is_permuted(src1)); - GGML_ASSERT(src0->nb[0] <= src0->nb[1] && src0->nb[2] <= src0->nb[3]); // NOLINT - GGML_ASSERT(src1->nb[0] <= src1->nb[1] && src1->nb[2] <= src1->nb[3]); // NOLINT - GGML_ASSERT(src0->type == GGML_TYPE_F16); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - // const uint64_t ne03 = src0->ne[3]; - - //const uint64_t ne10 = src1->ne[0]; - const uint64_t ne11 = src1->ne[1]; - const uint64_t ne12 = src1->ne[2]; - // const uint64_t ne13 = src1->ne[3]; - - GGML_ASSERT(ne11 == 1); + ggml_tensor * src0 = dst->src[0]; + ggml_tensor * src2 = dst->src[2]; + return (src2->ne[1] <= 8) && (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || ggml_is_quantized(src0->type)); +} - // With grouped query attention there are > 1 Q matrices per K, V matrix. - uint32_t gqa_ratio = (uint32_t)ne12 / (uint32_t)ne02; - if (gqa_ratio > 8 || gqa_ratio == 0 || ne12 != ne02 * gqa_ratio) { - gqa_ratio = 1; +void ggml_vk_mul_mat_id(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + ggml_tensor * src0 = dst->src[0]; + ggml_tensor * src1 = dst->src[1]; + ggml_tensor * src2 = dst->src[2]; + VK_LOG_DEBUG("ggml_vk_mul_mat_id(" << src0 << ", " << src1 << ", " << src2 << ", " << dst << ")"); + if (ggml_vk_use_mul_mat_vec_id(cgraph, node_idx)) { + ggml_vk_mul_mat_vec_id_q_f16(ctx, subctx, cgraph, node_idx); + } else { + ggml_vk_mul_mat_id_q_f16(ctx, subctx, src0, src1, src2, dst); } +} - vk_pipeline pipeline = ctx->device->pipeline_mul_mat_vec_p021_f16_f32[gqa_ratio - 1]; - - if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { - pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); - } +bool ggml_vk_flash_attn_scalar_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type) { + GGML_UNUSED(f32acc); + // Needs to be kept up to date on shader changes + const uint32_t wg_size = params.workgroup_size; + const uint32_t Br = params.block_rows; + const uint32_t Bc = params.block_cols; - { - // Request descriptor sets - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - } + // BF16 uses the fp32 shader (FLOAT_TYPE=float) + const uint32_t float_type_size = (device->fp16 && k_type != GGML_TYPE_BF16) ? sizeof(ggml_fp16_t) : sizeof(float); - vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops], true); - vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1, true); + const bool mmq = ggml_vk_fa_scalar_uses_mmq(device, k_type, v_type); - vk_subbuffer d_F0 = d_D; + // tmpsh is overestimated slightly + const uint32_t tmpsh = wg_size * sizeof(float); + const uint32_t tmpshv4 = wg_size * 4 * float_type_size; - uint32_t fusion_flags = 0; - - if (ctx->num_additional_fused_ops > 0) { - const ggml_tensor * add = cgraph->nodes[node_idx + 1]; - const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; - - d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; - } - - vk_subbuffer d_F1 = d_D; - if (ctx->num_additional_fused_ops > 1) { - const ggml_tensor * bias = cgraph->nodes[node_idx + 2]->src[1]; + const uint32_t masksh = Bc * (Br + 1) * float_type_size; + // DATA_A_IQ4_NL is compiled into the FA shaders unconditionally, so its shared table is always allocated. + const uint32_t iq_shmem = 16 * float_type_size; - d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; - } + uint32_t Qf, kvsh, kblocksh_size; + if (mmq) { + // block_b_cache: int32_t qs[8] + FLOAT_TYPEV2 ds + const uint32_t block_b_size = 8 * sizeof(int32_t) + 2 * float_type_size; + Qf = Br * (hsk / 32) * block_b_size; - // compute + // kvsh uses D = HSV (K goes through kblocksh instead) + kvsh = params.shmem_staging ? Bc * (hsv / 4 + 1) * 4 * float_type_size : 4 * float_type_size; - vk_mat_vec_p021_push_constants pc = { - (uint32_t)ne00, (uint32_t)ne01, (uint32_t)ne02, (uint32_t)ne12, - 0, 0, fusion_flags - }; + // The mixed MMQ shader uses a superset block_a_cache that fits every + // FA-supported quant: int32_t qs[8] + uint32_t qh + FLOAT_TYPEV2 dm. + // Single-scale types leave dm.y unused; non-Q5_* leave qh unused. + const uint32_t block_a_size = 8 * sizeof(int32_t) + sizeof(uint32_t) + 2 * float_type_size; + kblocksh_size = params.shmem_staging ? Bc * (hsk / 32) * block_a_size : block_a_size; + } else { + Qf = Br * (hsk / 4 + 1) * 4 * float_type_size; - init_pushconst_tensor_offsets(ctx, pc, src0, src1, nullptr, nullptr, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); + const uint32_t D = std::max(hsk, hsv); + kvsh = params.shmem_staging ? Bc * (D / 4 + 1) * 4 * float_type_size : 4 * float_type_size; - uint32_t workgroups_z = (uint32_t)ne12; - // When gqa_ratio > 1, each invocation does multiple rows and we can launch fewer workgroups - if (gqa_ratio > 1) { - workgroups_z /= gqa_ratio; + kblocksh_size = 0; } - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - { - d_Qx, - d_Qy, - d_D, - d_F0, - d_F1, - }, pc, { 1, (uint32_t)ne01, workgroups_z }); -} - -static void ggml_vk_mul_mat_vec_nc_f16_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - VK_LOG_DEBUG("ggml_vk_mul_mat_nc_f16_f32((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << "))"); - GGML_ASSERT(!ggml_is_transposed(src0)); - GGML_ASSERT(!ggml_is_transposed(src1)); - GGML_ASSERT(!ggml_is_permuted(src0)); - GGML_ASSERT(src0->type == GGML_TYPE_F16); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - const uint64_t ne03 = src0->ne[3]; + const uint32_t total_size = tmpsh + tmpshv4 + masksh + iq_shmem + Qf + kvsh + kblocksh_size; + const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - const uint64_t nb01 = src0->nb[1]; - const uint64_t nb02 = src0->nb[2]; + VK_LOG_DEBUG("ggml_vk_flash_attn_scalar_shmem_support(HSK=" << hsk << ", HSV=" << hsv << ", mmq=" << mmq << ", total_size=" << total_size << ", supported=" << supported); - const uint64_t nb12 = src1->nb[2]; + return supported; +} - // const uint64_t ne10 = src1->ne[0]; - const uint64_t ne11 = src1->ne[1]; - const uint64_t ne12 = src1->ne[2]; - // const uint64_t ne13 = src1->ne[3]; +bool ggml_vk_flash_attn_coopmat_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type) { + GGML_UNUSED(v_type); + // Needs to be kept up to date on shader changes + const uint32_t Br = params.block_rows; + const uint32_t Bc = params.block_cols; - const uint32_t nb03 = (uint32_t)(src0->nb[3] / sizeof(ggml_fp16_t)); - const uint32_t nb13 = (uint32_t)(src1->nb[3] / sizeof(float)); - const uint32_t nb23 = (uint32_t)(dst->nb[3] / sizeof(float)); + const uint32_t MatBr = 16, MatBc = 16; - GGML_ASSERT(ne11 == 1); - GGML_ASSERT(src0->ne[3] == src1->ne[3]); // checked in supports_op + const uint32_t row_split = Bc / MatBc; - const uint32_t row_stride_x = nb01 / sizeof(ggml_fp16_t); - const uint32_t channel_stride_x = nb02 / sizeof(ggml_fp16_t); - const uint32_t channel_stride_y = nb12 / sizeof(float); + const uint32_t hsk_pad = ROUNDUP_POW2(hsk, 16); + const uint32_t hsv_pad = ROUNDUP_POW2(hsv, 16); - vk_pipeline pipeline = ctx->device->pipeline_mul_mat_vec_nc_f16_f32; - if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { - pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); - } + const uint32_t acctype = f32acc ? 4 : 2; + const uint32_t f16vec4 = 8; - { - // Request descriptor sets - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - } + const uint32_t tmpsh = (Bc / MatBc) * sizeof(float); + // DATA_A_IQ4_NL is compiled into the FA shaders unconditionally, so its shared table is always allocated. + const uint32_t iq_shmem = 16 * sizeof(ggml_fp16_t); - vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops], true); - vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1, true); - vk_subbuffer d_F0 = d_D; + const uint32_t qstride = hsk_pad / 4 + 2; + const uint32_t Qf = Br * qstride * f16vec4; - uint32_t fusion_flags = 0; + const uint32_t psh_stride = Br / 4 + 2; + const uint32_t Psh = Bc * psh_stride * f16vec4; - if (ctx->num_additional_fused_ops > 0) { - const ggml_tensor * add = cgraph->nodes[node_idx + 1]; - const ggml_tensor * bias = add->src[0] == dst ? add->src[1] : add->src[0]; + const uint32_t sfshstride = (hsk <= 128) ? (Br + 8) : Br; + const uint32_t sfsh = Bc * sfshstride * acctype; - d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; - } + const uint32_t kvshstride = (params.shmem_staging ? std::max(hsk_pad, hsv_pad) : MatBr) / 4 + 2; + const uint32_t vsh_stride = MatBc / 4 * row_split; + const uint32_t ksh = ((kvshstride >= vsh_stride) ? (Bc * kvshstride) : (Bc * vsh_stride)) * f16vec4; - vk_subbuffer d_F1 = d_D; - if (ctx->num_additional_fused_ops > 1) { - const ggml_tensor * bias = cgraph->nodes[node_idx + 2]->src[1]; + // BF16 PVMat accumulator is f32 (no bf16 accumulator support), so pvsh is vec4 (16 bytes) + const uint32_t pvsh_elem_size = (k_type == GGML_TYPE_BF16) ? 16u : f16vec4; + const uint32_t osh_stride = params.row_split * MatBr / 4; + const uint32_t pvsh = MatBc * osh_stride * pvsh_elem_size; - d_F1 = ggml_vk_tensor_subbuffer(ctx, bias); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS1; - } + const uint32_t slope = Br * acctype; - // compute - vk_mat_vec_nc_push_constants pc = { - (uint32_t)ne00, (uint32_t)ne01, - row_stride_x, channel_stride_x, channel_stride_y, - (uint32_t)(ne12 / ne02), (uint32_t)ne12, - 0, 0, - nb03, nb13, nb23, fusion_flags - }; + const uint32_t total_size = tmpsh + iq_shmem + Qf + Psh + sfsh + ksh + pvsh + slope; + const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - init_pushconst_tensor_offsets(ctx, pc, src0, src1, nullptr, nullptr, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); + VK_LOG_DEBUG("ggml_vk_flash_attn_coopmat_shmem_support(HSK=" << hsk << ", HSV=" << hsv << ", f32acc=" << f32acc << ", total_size=" << total_size << ", supported=" << supported); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - { - d_Qx, - d_Qy, - d_D, - d_F0, - d_F1, - }, pc, { (uint32_t)ne03, (uint32_t)ne01, (uint32_t)ne12 }); + return supported; } -static int ggml_vk_fwht_pipeline_idx(int64_t n) { - switch (n) { - case 64: return 0; - case 128: return 1; - case 256: return 2; - case 512: return 3; - default: return -1; +void ggml_vk_flash_attn(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * q, const ggml_tensor * k, const ggml_tensor * v, const ggml_tensor * mask, const ggml_tensor * sinks, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_flash_attn((" << q << ", name=" << q->name << ", type=" << q->type << ", ne0=" << q->ne[0] << ", ne1=" << q->ne[1] << ", ne2=" << q->ne[2] << ", ne3=" << q->ne[3] << ", nb0=" << q->nb[0] << ", nb1=" << q->nb[1] << ", nb2=" << q->nb[2] << ", nb3=" << q->nb[3]; + std::cerr << "), (" << k << ", name=" << k->name << ", type=" << k->type << ", ne0=" << k->ne[0] << ", ne1=" << k->ne[1] << ", ne2=" << k->ne[2] << ", ne3=" << k->ne[3] << ", nb0=" << k->nb[0] << ", nb1=" << k->nb[1] << ", nb2=" << k->nb[2] << ", nb3=" << k->nb[3]; + std::cerr << "), (" << v << ", name=" << v->name << ", type=" << v->type << ", ne0=" << v->ne[0] << ", ne1=" << v->ne[1] << ", ne2=" << v->ne[2] << ", ne3=" << v->ne[3] << ", nb0=" << v->nb[0] << ", nb1=" << v->nb[1] << ", nb2=" << v->nb[2] << ", nb3=" << v->nb[3]; + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + if (sinks) { + std::cerr << "), (" << sinks << ", name=" << sinks->name << ", type=" << sinks->type << ", ne0=" << sinks->ne[0] << ", ne1=" << sinks->ne[1] << ", ne2=" << sinks->ne[2] << ", ne3=" << sinks->ne[3] << ", nb0=" << sinks->nb[0] << ", nb1=" << sinks->nb[1] << ", nb2=" << sinks->nb[2] << ", nb3=" << sinks->nb[3]; } -} + std::cerr << "))"); -static bool ggml_vk_can_use_fwht(const ggml_backend_vk_context * ctx, const ggml_tensor * src1, const ggml_tensor * dst) { - if (ctx->num_additional_fused_ops != 0) { - return false; - } + GGML_TENSOR_LOCALS(int64_t, neq, q, ne) + GGML_TENSOR_LOCALS(size_t, nbq, q, nb) + GGML_TENSOR_LOCALS(int64_t, nek, k, ne) + GGML_TENSOR_LOCALS(size_t, nbk, k, nb) + GGML_TENSOR_LOCALS(int64_t, nev, v, ne) + GGML_TENSOR_LOCALS(size_t, nbv, v, nb) + GGML_TENSOR_LOCALS(int64_t, ne, dst, ne) + GGML_TENSOR_LOCALS(size_t, nb, dst, nb) - if (ggml_get_op_params_i32(dst, 1) != GGML_HINT_SRC0_IS_HADAMARD) { - return false; - } + const uint32_t nem0 = mask ? mask->ne[0] : 0; + const uint32_t nem1 = mask ? mask->ne[1] : 0; + const uint32_t nem2 = mask ? mask->ne[2] : 0; + const uint32_t nem3 = mask ? mask->ne[3] : 0; - const int idx = ggml_vk_fwht_pipeline_idx(src1->ne[0]); - if (idx < 0 || ctx->device->pipeline_fwht_f32[idx] == nullptr) { - return false; - } + const uint32_t HSK = nek0; + const uint32_t HSV = nev0; + uint32_t N = neq1; + const uint32_t KV = nek1; - if (src1->type != GGML_TYPE_F32 || dst->type != GGML_TYPE_F32) { - return false; - } + GGML_ASSERT(ne0 == HSV); + GGML_ASSERT(ne2 == N); - if (!ggml_is_contiguous(src1)) { - return false; - } - GGML_ASSERT(ggml_is_contiguous(dst)); + // input tensor rows must be contiguous + GGML_ASSERT(nbq0 == ggml_type_size(q->type)); + GGML_ASSERT(nbk0 == ggml_type_size(k->type)); + GGML_ASSERT(nbv0 == ggml_type_size(v->type)); - return true; -} + GGML_ASSERT(neq0 == HSK); -static void ggml_vk_fwht(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src, ggml_tensor * dst) { - const int idx = ggml_vk_fwht_pipeline_idx(src->ne[0]); - vk_pipeline pipeline = ctx->device->pipeline_fwht_f32[idx]; + GGML_ASSERT(neq1 == N); - const uint32_t rows_per_workgroup = 4; - const uint32_t n_rows = (uint32_t)ggml_nrows(src); - const uint32_t max_workgroups_x = ctx->device->properties.limits.maxComputeWorkGroupCount[0]; + GGML_ASSERT(nev1 == nek1); - const uint32_t total_workgroups = CEIL_DIV(n_rows, rows_per_workgroup); - const uint32_t workgroups_x = std::min(total_workgroups, max_workgroups_x); - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + // dst cannot be transposed or permuted + GGML_ASSERT(nb0 == sizeof(float)); + GGML_ASSERT(nb0 <= nb1); + GGML_ASSERT(nb1 <= nb2); + GGML_ASSERT(nb2 <= nb3); - const vk_subbuffer src_buf = ggml_vk_tensor_subbuffer(ctx, src, true); - const vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); + assert(dst->type == GGML_TYPE_F32); + assert(q->type == GGML_TYPE_F32); + uint32_t gqa_ratio = 1; + uint32_t qk_ratio = neq2 / nek2; + uint32_t workgroups_x = (uint32_t)neq1; + uint32_t workgroups_y = (uint32_t)neq2; + uint32_t workgroups_z = (uint32_t)neq3; - vk_op_fwht_push_constants pc = { - n_rows, - 0, - 0, - 1.0f / std::sqrt((float)src->ne[0]), + const bool f32acc = !ctx->device->fp16 || dst->op_params[3] == GGML_PREC_F32 || k->type == GGML_TYPE_BF16; + + // dequant K/V once into an f16 scratch, reordered KV layout so FA can read without a stride + auto is_dense_kv_cache = [](const ggml_tensor * t) { + return t->nb[0] == ggml_type_size(t->type) && + t->nb[2] == ggml_row_size(t->type, t->ne[0]) && + t->nb[1] == t->nb[2] * t->ne[2] && + (t->ne[3] == 1 || t->nb[3] == t->nb[1] * t->ne[1]); }; - init_pushconst_tensor_offsets(ctx, pc, src, nullptr, nullptr, nullptr, dst); + const bool k_quant = k->type != GGML_TYPE_F16 && k->type != GGML_TYPE_BF16 && k->type != GGML_TYPE_F32; + const bool v_quant = v->type != GGML_TYPE_F16 && v->type != GGML_TYPE_BF16 && v->type != GGML_TYPE_F32; + const bool use_dequant_kv = k_quant && v_quant && neq1 >= 64 && + is_dense_kv_cache(k) && is_dense_kv_cache(v) && + (uint64_t)ggml_nelements(k) * sizeof(ggml_fp16_t) <= ctx->device->properties.limits.maxStorageBufferRange && + (uint64_t)ggml_nelements(v) * sizeof(ggml_fp16_t) <= ctx->device->properties.limits.maxStorageBufferRange && + ctx->device->pipeline_dequant_transpose[k->type] != nullptr && + ctx->device->pipeline_dequant_transpose[v->type] != nullptr && + // coopmat2 path does not benefit from the f16 scratch + !ctx->device->coopmat2 && + // Intel Xe1 regresses, see PR 25494 + (ctx->device->vendor_id != VK_VENDOR_ID_INTEL || + (ctx->device->coopmat_support && ctx->device->architecture != vk_device_architecture::INTEL_XE1)); + const ggml_type k_type_eff = use_dequant_kv ? GGML_TYPE_F16 : k->type; + const ggml_type v_type_eff = use_dequant_kv ? GGML_TYPE_F16 : v->type; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src_buf, dst_buf }, pc, { workgroups_x, 1, 1 }); -} - -static void ggml_vk_mul_mat(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - ggml_tensor * src0 = dst->src[0]; - ggml_tensor * src1 = dst->src[1]; - VK_LOG_DEBUG("ggml_vk_mul_mat(" << src0 << ", " << src1 << ", " << dst << ")"); + // For scalar/coopmat1 FA, we can use the "large" size to accommodate qga. + // For coopmat2 FA, we always use the small size (which is still pretty large for gqa). + vk_fa_tuning_params tuning_params = get_fa_tuning_params(ctx->device, HSK, HSV, 512, KV, k_type_eff, v_type_eff, f32acc); + const uint32_t max_gqa = std::min(tuning_params.block_rows, 32u); - // Handle huge A matrix by splitting the M dimensions. This works well for convolution use cases - // where the M dimension is very large. - // Split_k doesn't work with M splitting. - // This only supports batchsize == 1. - const size_t nbytes = ggml_nbytes(src0); - const bool needs_split = dst->ne[2] == 1 && dst->ne[3] == 1 && nbytes > ctx->device->properties.limits.maxStorageBufferRange; - if (needs_split) { - // Choose the number of rows that can fit (and divide by two, to allow for any additional offsets) - const uint32_t M_split = ctx->device->properties.limits.maxStorageBufferRange / (2 * src0->nb[1]); - uint32_t m_offset = 0; - while (m_offset < dst->ne[0]) { - const uint32_t cur_M_size = std::min(M_split, (uint32_t)(dst->ne[0] - m_offset)); - ggml_tensor dst2 = *dst; - ggml_tensor src02 = *src0; + if (N <= 8 && qk_ratio > 1 && qk_ratio <= max_gqa && + qk_ratio * nek2 == neq2 && nek2 == nev2 && nem2 <= 1) { + // grouped query attention - make the N dimension equal to gqa_ratio, reduce + // workgroups proportionally in y dimension. The shader will detect gqa_ratio > 1 + // and change addressing calculations to index Q's dimension 2. + gqa_ratio = qk_ratio; + N = gqa_ratio; + workgroups_y /= gqa_ratio; + } - dst2.view_src = dst->view_src ? dst->view_src : dst; - src02.view_src = src0->view_src ? src0->view_src : src0; + tuning_params = get_fa_tuning_params(ctx->device, HSK, HSV, N, KV, k_type_eff, v_type_eff, f32acc); - dst2.view_offs += m_offset * dst->nb[0]; - src02.view_offs += m_offset * src0->nb[1]; - dst2.ne[0] = cur_M_size; - src02.ne[1] = cur_M_size; + float scale = 1.0f; + float max_bias = 0.0f; + float logit_softcap = 0.0f; - ggml_vk_mul_mat_q_f16(ctx, subctx, &src02, src1, &dst2, true); + memcpy(&scale, (const float *) dst->op_params + 0, sizeof(float)); + memcpy(&max_bias, (const float *) dst->op_params + 1, sizeof(float)); + memcpy(&logit_softcap, (const float *) dst->op_params + 2, sizeof(float)); - m_offset += cur_M_size; - } - } else if (ggml_vk_can_use_fwht(ctx, src1, dst)) { - ggml_vk_fwht(ctx, subctx, src1, dst); - } else if (src0->type == GGML_TYPE_F16 && ggml_is_permuted(src0) && ggml_is_permuted(src1) && dst->ne[1] == 1 && - // detect 0213 permutation, and batch size of 1 - src0->nb[0] <= src0->nb[2] && - src0->nb[2] <= src0->nb[1] && - src0->nb[1] <= src0->nb[3] && - src1->nb[0] <= src1->nb[2] && - src1->nb[2] <= src1->nb[1] && - src1->nb[1] <= src1->nb[3] && - src0->ne[3] == 1 && - src1->ne[3] == 1 && - src0->ne[1] <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && - src1->ne[2] <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]) { - ggml_vk_mul_mat_vec_p021_f16_f32(ctx, subctx, cgraph, node_idx); - } else if (src0->type == GGML_TYPE_F16 && !ggml_is_contiguous(src0) && !ggml_is_transposed(src1) && dst->ne[1] == 1 && - !ggml_is_permuted(src0) && !ggml_is_permuted(src1) && - src0->ne[3] <= ctx->device->properties.limits.maxComputeWorkGroupCount[0] && - src0->ne[1] <= ctx->device->properties.limits.maxComputeWorkGroupCount[1] && - src1->ne[2] <= ctx->device->properties.limits.maxComputeWorkGroupCount[2]) { - ggml_vk_mul_mat_vec_nc_f16_f32(ctx, subctx, cgraph, node_idx); - // mul_mat_vec supports batching ne12*ne13 when ne11==1, or treating ne11 as the batch size (up to four) - // when ne12 and ne13 are one. - } else if ((dst->ne[1] == 1 || (dst->ne[1] <= mul_mat_vec_max_cols && src1->ne[2] * src1->ne[3] == 1)) && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16 || ggml_is_quantized(src0->type))) { - ggml_vk_mul_mat_vec_q_f16(ctx, subctx, cgraph, node_idx); - } else { - ggml_vk_mul_mat_q_f16(ctx, subctx, src0, src1, dst, false); + if (logit_softcap != 0) { + scale /= logit_softcap; } -} - -static void ggml_vk_mul_mat_id_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * ids, ggml_tensor * dst) { - VK_LOG_DEBUG("ggml_vk_mul_mat_id_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << ids << ", name=" << ids->name << ", type=" << ids->type << ", ne0=" << ids->ne[0] << ", ne1=" << ids->ne[1] << ", ne2=" << ids->ne[2] << ", ne3=" << ids->ne[3] << ", nb0=" << ids->nb[0] << ", nb1=" << ids->nb[1] << ", nb2=" << ids->nb[2] << ", nb3=" << ids->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3] << "),)"); - GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT - GGML_ASSERT(ids->type == GGML_TYPE_I32); - - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - // const uint64_t ne03 = src0->ne[3]; - const uint64_t ne10 = src1->ne[0]; - const uint64_t ne11 = src1->ne[1]; - const uint64_t ne12 = src1->ne[2]; - const uint64_t ne13 = src1->ne[3]; + // Sparse mask hint (op_params[4]): compact the <= n_kv_max finite positions and gather only those. + const int32_t n_kv_max = mask ? ggml_get_op_params_i32(dst, 4) : 0; + static const bool disable_sparse = getenv("GGML_VK_FA_SPARSE_DISABLE") != nullptr; + // cm2 dense is fast, so it needs a larger reduction to win. + const int64_t min_ratio = tuning_params.path == FA_COOPMAT2 ? 4 : 2; + const bool use_sparse = !disable_sparse && n_kv_max > 0 && mask && + max_bias == 0.0f && logit_softcap == 0.0f && + k_type_eff == GGML_TYPE_F16 && v_type_eff == GGML_TYPE_F16 && + nem0 == KV && + (int64_t)KV >= std::max(4096, min_ratio * (int64_t)n_kv_max) && + (gqa_ratio > 1 || (tuning_params.path == FA_SCALAR && N == 1)); - const uint64_t nei0 = ids->ne[0]; - const uint64_t nei1 = ids->ne[1]; + const uint32_t q_stride = (uint32_t)(nbq1 / ggml_type_size(q->type)); + uint32_t k_stride = (uint32_t)(nbk1 / ggml_type_size(k->type)); + uint32_t v_stride = (uint32_t)(nbv1 / ggml_type_size(v->type)); - const uint32_t nbi0 = ids->nb[0]; - const uint32_t nbi1 = ids->nb[1]; - const uint32_t nbi2 = ids->nb[2]; + // For F32, the shader treats it as a block of size 4 (for vec4 loads) + if (k->type == GGML_TYPE_F32) { + k_stride /= 4; + } + if (v->type == GGML_TYPE_F32) { + v_stride /= 4; + } - const uint64_t ne20 = dst->ne[0]; - const uint64_t ne21 = dst->ne[1]; - // const uint64_t ne22 = dst->ne[2]; - // const uint64_t ne23 = dst->ne[3]; + uint32_t nbk2_eff = (uint32_t)nbk2, nbk3_eff = (uint32_t)nbk3; + uint32_t nbv2_eff = (uint32_t)nbv2, nbv3_eff = (uint32_t)nbv3; + if (use_dequant_kv) { + k_stride = HSK; + v_stride = HSV; + nbk2_eff = (uint32_t)((uint64_t)HSK * KV * sizeof(ggml_fp16_t)); + nbk3_eff = (uint32_t)((uint64_t)HSK * KV * nek2 * sizeof(ggml_fp16_t)); + nbv2_eff = (uint32_t)((uint64_t)HSV * KV * sizeof(ggml_fp16_t)); + nbv3_eff = (uint32_t)((uint64_t)HSV * KV * nev2 * sizeof(ggml_fp16_t)); + } + const uint32_t alignment = tuning_params.block_cols; + bool aligned = (KV % alignment) == 0 && + // the "aligned" shader variant will forcibly align strides, for performance + (q_stride & 7) == 0 && (k_stride & 7) == 0 && (v_stride & 7) == 0; - const uint64_t n_as = ne02; + // Need to use the coopmat2 variant that clamps loads when HSK/HSV aren't sufficiently aligned. + if (((HSK | HSV) % 16) != 0 && tuning_params.path == FA_COOPMAT2) { + aligned = false; + } - ggml_backend_vk_buffer_context * dst_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; - ggml_backend_vk_buffer_context * src0_buf_ctx = (ggml_backend_vk_buffer_context *)src0->buffer->context; - ggml_backend_vk_buffer_context * src1_buf_ctx = (ggml_backend_vk_buffer_context *)src1->buffer->context; - ggml_backend_vk_buffer_context * ids_buf_ctx = (ggml_backend_vk_buffer_context *)ids->buffer->context; + // Only use mask opt when the mask is fairly large. This hasn't been tuned extensively. + bool use_mask_opt = mask && !use_sparse && nem1 >= 32 && nem0 * nem1 > 32768 && nem0 >= tuning_params.block_cols * 16 + && (ctx->device->architecture != vk_device_architecture::AMD_GCN || HSK > 256 || HSV > 256); + vk_fa_pipeline_state fa_pipeline_state = get_fa_pipeline_state(ctx->device, tuning_params, HSK, HSV, aligned, f32acc, + mask != nullptr, use_mask_opt, logit_softcap != 0, use_sparse, k_type_eff, v_type_eff); - vk_buffer d_Qx = nullptr; - size_t qx_buf_offset = 0; - vk_buffer d_Qy = nullptr; - size_t qy_buf_offset = 0; - vk_buffer d_ids = nullptr; - size_t ids_buf_offset = 0; + vk_pipeline pipeline = nullptr; - bool src0_uma = false; - bool src1_uma = false; - bool ids_uma = false; + bool xe_fa_opt = false; + bool fa_copy_qstate = false; + bool xe_fa_supported_platform = + (ctx->device.get()->architecture == INTEL_XE2 && ctx->device.get()->properties.deviceID != 0xFD80 && ctx->device.get()->properties.deviceID != 0xFD81) || + (ctx->device.get()->architecture == INTEL_XE1 && ctx->device.get()->coopmat_support && ctx->device.get()->uma); + bool xe_fa_supported_usage = neq0 % 32 == 0 && nev0 % 16 == 0 && q->nb[1] > q->nb[2] && k->nb[1] > k->nb[2] && v->nb[1] > v->nb[2] && mask != nullptr; + bool xe_fa_supported_dtype = q->type == GGML_TYPE_F32 && k->type == GGML_TYPE_F16 && v->type == GGML_TYPE_F16 && (mask != nullptr && mask->type == GGML_TYPE_F16); + std::pair xe_fa_pipeline_dual_phases = { nullptr , nullptr }; + vk_pipeline xe_fa_pipeline = nullptr; + size_t size_p = 0; + size_t size_group_max = 0; - if (ctx->device->uma) { - ggml_vk_host_get(ctx->device, src0->data, d_Qx, qx_buf_offset); - ggml_vk_host_get(ctx->device, src1->data, d_Qy, qy_buf_offset); - ggml_vk_host_get(ctx->device, ids->data, d_ids, ids_buf_offset); - src0_uma = d_Qx != nullptr; - src1_uma = d_Qy != nullptr; - ids_uma = d_ids != nullptr; + { + std::lock_guard guard(ctx->device->compile_mutex); + auto &pipelines = ctx->device->pipeline_flash_attn_f32_f16; + auto it = pipelines.find(fa_pipeline_state); + if (it != pipelines.end()) { + pipeline = it->second; + } else { + pipelines[fa_pipeline_state] = pipeline = std::make_shared(); + } } - // Reformat and convert to fp16 if non-contiguous, or for coopmat2 for better perf - const bool x_non_contig = (ctx->device->coopmat2 && src0->type == GGML_TYPE_F32) || - !ggml_vk_dim01_contiguous(src0); - // If src0 is BF16, try to use a BF16 x BF16 multiply - ggml_type f16_type = src0->type == GGML_TYPE_BF16 ? GGML_TYPE_BF16 : GGML_TYPE_F16; -#if defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR_GLSLC_SUPPORT) - // B must already be, or be convertible to, the matmul B type used by this path. - const bool y_decode_vector_supported = ctx->device->coopmat2_decode_vector && - (f16_type != GGML_TYPE_BF16 || ctx->device->coopmat2_bf16_support) && - (src1->type == GGML_TYPE_F32 || src1->type == f16_type); - // If B is copied to prealloc_y, we can choose a 4-element-aligned row stride. - const bool y_decode_vector_uses_prealloc = !ggml_vk_dim01_contiguous(src1) || src1->type != f16_type; - // Direct B reads are safe only if row starts and the original buffer offset are 4-element aligned. - const bool y_decode_vector_aligned = - (ne10 % 4 == 0) && - (y_decode_vector_uses_prealloc || get_misalign_bytes(ctx, src1) % (4 * ggml_type_size(src1->type)) == 0); - // Stage B only when decode-vector is available and direct B reads would be misaligned. - const bool y_decode_vector_staging = y_decode_vector_supported && !y_decode_vector_aligned; -#else - const bool y_decode_vector_staging = false; -#endif - const bool y_non_contig = y_decode_vector_staging || - (ctx->device->coopmat2 && src1->type == GGML_TYPE_F32) || - (src0->type == GGML_TYPE_BF16 && src1->type != GGML_TYPE_BF16) || - !ggml_vk_dim01_contiguous(src1); - - const uint32_t y_staged_row_stride = y_decode_vector_staging ? (uint32_t)ggml_vk_align_size(ne10, 4) : (uint32_t)ne10; + assert(pipeline); + // Compile early to initialize wg_denoms. + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - const bool y_f32_kernel = src1->type == GGML_TYPE_F32 && !y_non_contig; + uint32_t split_kv = KV; + uint32_t split_k = 1; - bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0; + // Intel Alchemist prefers more workgroups + const uint32_t shader_core_count_multiplier = (ctx->device->vendor_id == VK_VENDOR_ID_INTEL && ctx->device->architecture != INTEL_XE2) ? 2 : 1; - // Check for mmq first - vk_matmul_pipeline mmp = quantize_y ? ggml_vk_get_mul_mat_mat_id_pipeline(ctx, src0->type, GGML_TYPE_Q8_1, (ggml_prec)dst->op_params[0]) : nullptr; + // Use a placeholder core count if one isn't available. split_k is a big help for perf. + const uint32_t shader_core_count = ctx->device->shader_core_count ? ctx->device->shader_core_count * shader_core_count_multiplier : 16; - if (mmp == nullptr) { - // Fall back to f16 dequant mul mat - mmp = ggml_vk_get_mul_mat_mat_id_pipeline(ctx, src0->type, y_non_contig ? f16_type : src1->type, (ggml_prec)dst->op_params[0]); - quantize_y = false; - } + const uint32_t Br = fa_pipeline_state.Br; + const uint32_t Bc = fa_pipeline_state.Bc; - const bool qx_needs_dequant = mmp == nullptr || x_non_contig; - const bool qy_needs_dequant = !quantize_y && ((src1->type != f16_type && !y_f32_kernel) || y_non_contig); + GGML_ASSERT(Br == pipeline->wg_denoms[0]); + const uint32_t Tr = CEIL_DIV(N, Br); - if (qx_needs_dequant) { - // Fall back to dequant + f16 mulmat - mmp = ggml_vk_get_mul_mat_mat_id_pipeline(ctx, f16_type, y_f32_kernel ? GGML_TYPE_F32 : f16_type, (ggml_prec)dst->op_params[0]); + // Try to use split_k when KV is large enough to be worth the overhead. + // Sparse: split_kv carries n_kv_max, split_k partitions its blocks for occupancy. + if (use_sparse) { + split_kv = (uint32_t)n_kv_max; + const uint32_t total_blocks = CEIL_DIV((uint32_t)n_kv_max, Bc); + const uint32_t base_wgs = (gqa_ratio > 1 ? workgroups_x : Tr) * workgroups_y * workgroups_z; + if (base_wgs < shader_core_count * 2) { + split_k = shader_core_count * 2 / base_wgs; + } + split_k = std::max(1u, std::min(split_k, total_blocks)); + // Match the shader's per-split block count so no split is empty. + const uint32_t per_blocks = CEIL_DIV(total_blocks, split_k); + split_k = CEIL_DIV(total_blocks, per_blocks); + } else if (gqa_ratio > 1 && workgroups_x <= Br) { + split_k = shader_core_count * 2 / (workgroups_x * workgroups_y * workgroups_z); + } else if (gqa_ratio <= 1) { + uint32_t total_wgs_no_split = Tr * workgroups_y * workgroups_z; + if (total_wgs_no_split < shader_core_count * 2) { + split_k = shader_core_count * 2 / total_wgs_no_split; + } } - // Not implemented - GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT + if (!use_sparse && split_k > 1) { + // Try to evenly split KV into split_k chunks, but it needs to be a multiple + // of "align", so recompute split_k based on that. + split_kv = ROUNDUP_POW2(std::max(1u, KV / split_k), alignment); + split_k = CEIL_DIV(KV, split_kv); + xe_fa_opt = xe_fa_supported_platform && xe_fa_supported_usage && xe_fa_supported_dtype; + if (xe_fa_opt) { + std::lock_guard guard(ctx->device->compile_mutex); + const uint32_t split_p_size = 32; + const size_t max_dim = (nek1 + split_p_size - 1) / split_p_size; + const size_t p_dim = max_dim * split_p_size; + auto& pipelines = ctx->device->pipeline_xe_fa_decode_dual_phases; + auto it = pipelines.find({ (uint32_t)neq0, (uint32_t)nev0, qk_ratio, (uint32_t)neq1 }); + if (it != pipelines.end()) { + xe_fa_pipeline_dual_phases = it->second; + } else { + pipelines[{(uint32_t)neq0, (uint32_t)nev0, qk_ratio, (uint32_t)neq1}] = xe_fa_pipeline_dual_phases = std::make_pair(std::make_shared(), std::make_shared()); + } - const ggml_type effective_src1_type = quantize_y ? GGML_TYPE_Q8_1 : (y_f32_kernel ? GGML_TYPE_F32 : src1->type); + size_p = neq1 * neq2 * p_dim * neq3 * sizeof(ggml_fp16_t); + size_group_max = neq1 * neq2 * max_dim * neq3 * sizeof(float); + size_t temp_size = ggml_nelements(q) * sizeof(ggml_fp16_t) + size_p + size_group_max; + fa_copy_qstate = true; + if (ctx->prealloc_size_x < temp_size) { + ctx->prealloc_size_x = temp_size; + ggml_vk_preallocate_buffers(ctx, subctx); + } - const uint32_t kpad = quantize_y ? 0 : ggml_vk_align_size(ne10, ggml_vk_guess_matmul_id_pipeline_align(ctx, mmp, ne01, nei1, qx_needs_dequant ? f16_type : src0->type, effective_src1_type)); - const bool aligned = !quantize_y && ne10 == kpad && ne01 > 8 && nei1 > 8; + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + } + } - vk_pipeline pipeline = ggml_vk_guess_matmul_id_pipeline(ctx, mmp, ne01, nei1, aligned, qx_needs_dequant ? f16_type : src0->type, effective_src1_type); + if (xe_fa_opt == true) { + use_mask_opt = false; + } - if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { - pipeline = ggml_vk_get_64b_indexing_pipeline(ctx, pipeline); + // Reserve space for split_k temporaries. For each split x batch, we need to store the O matrix (D x ne1) + // and the per-row m and L values (ne1 rows). We store all the matrices first, followed by the rows. + // For matrices, the order is (inner to outer) [HSV, ne1, k, ne2, ne3]. + // For L/M, the order is (inner to outer) [ne1, k, ne2, ne3]. + const uint64_t split_k_size = split_k > 1 ? (HSV * ne1 * sizeof(float) + ne1 * sizeof(float) * 2) * split_k * ne2 * ne3 : 0; + if (split_k_size > ctx->device->properties.limits.maxStorageBufferRange) { + GGML_ABORT("Requested preallocation size is too large"); + } + if (ctx->prealloc_size_split_k < split_k_size) { + ctx->prealloc_size_split_k = split_k_size; + ggml_vk_preallocate_buffers(ctx, subctx); } - // Reserve extra storage in the N dimension for the Y matrix, so we can avoid bounds-checking - uint32_t padded_n = qy_needs_dequant ? ROUNDUP_POW2(ne11, pipeline->wg_denoms[1]) :ne11; - const uint64_t x_ne = ggml_nelements(src0); - const uint64_t y_ne = (uint64_t)y_staged_row_stride * padded_n * ne12 * ne13; - const uint64_t d_ne = ggml_nelements(dst); - const uint64_t qx_sz = ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type); - const uint64_t qy_sz = ggml_type_size(src1->type) * ggml_nelements(src1) / ggml_blck_size(src1->type); - const uint64_t x_sz = !qx_needs_dequant ? qx_sz : sizeof(ggml_fp16_t) * x_ne; - const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : (y_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); - const uint64_t ids_sz = nbi2; - const uint64_t d_sz = sizeof(float) * d_ne; + const uint32_t mask_opt_num_dwords = CEIL_DIV(nem0, 16 * Bc); + const uint64_t mask_opt_size = sizeof(uint32_t) * mask_opt_num_dwords * CEIL_DIV(nem1, Br) * nem2 * nem3; - vk_pipeline to_fp16_vk_0 = nullptr; - vk_pipeline to_fp16_vk_1 = nullptr; - vk_pipeline to_q8_1 = nullptr; - - auto make_y_staged_dst = [&]() { - ggml_tensor y_staged_dst = *src1; - y_staged_dst.type = f16_type; - y_staged_dst.nb[0] = ggml_type_size(f16_type); - y_staged_dst.nb[1] = y_staged_dst.nb[0] * y_staged_row_stride; - y_staged_dst.nb[2] = y_staged_dst.nb[1] * padded_n; - y_staged_dst.nb[3] = y_staged_dst.nb[2] * y_staged_dst.ne[2]; - return y_staged_dst; - }; - - if (x_non_contig) { - to_fp16_vk_0 = ggml_vk_get_cpy_pipeline(ctx, src0, nullptr, f16_type); - } else { - to_fp16_vk_0 = ggml_vk_get_to_fp16(ctx, src0->type); - } - if (y_non_contig) { - ggml_tensor y_staged_dst; - const ggml_tensor * y_staged_dst_ptr = nullptr; - if (y_decode_vector_staging) { - y_staged_dst = make_y_staged_dst(); - y_staged_dst_ptr = &y_staged_dst; + vk_pipeline pipeline_fa_mask_opt = nullptr; + if (use_mask_opt) { + { + std::lock_guard guard(ctx->device->compile_mutex); + auto &pipelines = ctx->device->pipeline_fa_mask_opt; + auto it = pipelines.find({Br, Bc}); + if (it != pipelines.end()) { + pipeline_fa_mask_opt = it->second; + } else { + pipelines[{Br, Bc}] = pipeline_fa_mask_opt = std::make_shared(); + } } + assert(pipeline_fa_mask_opt); + ggml_pipeline_request_descriptor_sets(ctx, pipeline_fa_mask_opt, 1); - to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, y_staged_dst_ptr, f16_type); - } else { - to_fp16_vk_1 = ggml_vk_get_to_fp16(ctx, src1->type); - } - GGML_ASSERT(!qx_needs_dequant || to_fp16_vk_0 != nullptr); // NOLINT - GGML_ASSERT(!qy_needs_dequant || to_fp16_vk_1 != nullptr); // NOLINT - - if (quantize_y) { - to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); - } - vk_pipeline count_experts = ctx->device->pipeline_count_experts; - - uint32_t expert_count_size = sizeof(uint32_t) * n_as; - - { - if ( - (qx_needs_dequant && x_sz > ctx->device->properties.limits.maxStorageBufferRange) || - (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange)) { - GGML_ABORT("Requested preallocation size is too large"); - } - if (qx_needs_dequant && ctx->prealloc_size_x < x_sz) { - ctx->prealloc_size_x = x_sz; - ggml_vk_preallocate_buffers(ctx, subctx); - } - if ((qy_needs_dequant || quantize_y) && ctx->prealloc_size_y < y_sz) { - ctx->prealloc_size_y = y_sz; + if (ctx->prealloc_size_y < mask_opt_size) { + ctx->prealloc_size_y = mask_opt_size; ggml_vk_preallocate_buffers(ctx, subctx); } - if (ctx->prealloc_size_split_k < expert_count_size) { - ctx->prealloc_size_split_k = expert_count_size; - ggml_vk_preallocate_buffers(ctx, subctx); + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); } + } - // Request descriptor sets - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - if (qx_needs_dequant) { - ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); - } - if (qy_needs_dequant) { - ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_1, 1); + // Sparse index scratch reuses prealloc_y (mutually exclusive with mask opt). + const uint64_t sparse_idx_size = use_sparse + ? sizeof(int32_t) * (uint64_t)n_kv_max * nem1 * nem2 * nem3 + : 0; + vk_pipeline sparse_compact_pipeline = ctx->device->fa_sparse_compact_use_subgroups + ? ctx->device->pipeline_fa_sparse_compact_subgroup + : ctx->device->pipeline_fa_sparse_compact; + if (use_sparse) { + ggml_pipeline_request_descriptor_sets(ctx, sparse_compact_pipeline, 1); + if (ctx->prealloc_size_y < sparse_idx_size) { + ctx->prealloc_size_y = sparse_idx_size; + ggml_vk_preallocate_buffers(ctx, subctx); } - if (quantize_y) { - ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); + if (ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); } - ggml_pipeline_request_descriptor_sets(ctx, count_experts, 1); } - vk_buffer d_D = dst_buf_ctx->dev_buffer; - const uint64_t d_buf_offset = vk_tensor_offset(dst) + dst->view_offs; - GGML_ASSERT(d_D != nullptr); - vk_buffer d_X; - uint64_t x_buf_offset = 0; - vk_buffer d_Y; - uint64_t y_buf_offset = 0; - if (!src0_uma) { - d_Qx = src0_buf_ctx->dev_buffer; - qx_buf_offset = vk_tensor_offset(src0) + src0->view_offs; - GGML_ASSERT(d_Qx != nullptr); - } - if (!src1_uma) { - d_Qy = src1_buf_ctx->dev_buffer; - qy_buf_offset = vk_tensor_offset(src1) + src1->view_offs; - GGML_ASSERT(d_Qy != nullptr); - } - if (!ids_uma) { - d_ids = ids_buf_ctx->dev_buffer; - ids_buf_offset = vk_tensor_offset(ids) + ids->view_offs; - GGML_ASSERT(d_ids != nullptr); - } - if (qx_needs_dequant) { - d_X = ctx->prealloc_x; - GGML_ASSERT(d_X->size >= x_sz); - } else { - d_X = d_Qx; - x_buf_offset = qx_buf_offset; - GGML_ASSERT(qx_sz == x_sz); - } - if (qy_needs_dequant) { - d_Y = ctx->prealloc_y; - GGML_ASSERT(d_Y->size >= y_sz); - } else if (quantize_y) { - d_Y = ctx->prealloc_y; - GGML_ASSERT(d_Y->size >= CEIL_DIV(y_sz, 144) * 144); - } else { - d_Y = d_Qy; - y_buf_offset = qy_buf_offset; - GGML_ASSERT(qy_sz == y_sz); - } + const uint32_t n_head_kv = neq2; + const uint32_t n_head_log2 = 1u << (uint32_t) floorf(log2f((float) n_head_kv)); + const float m0 = powf(2.0f, -(max_bias ) / n_head_log2); + const float m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2); - if (x_non_contig || qx_needs_dequant) { + vk_subbuffer q_buf = ggml_vk_tensor_subbuffer(ctx, q); + vk_subbuffer k_buf = ggml_vk_tensor_subbuffer(ctx, k); + vk_subbuffer v_buf = ggml_vk_tensor_subbuffer(ctx, v); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer mask_buf = mask ? ggml_vk_tensor_subbuffer(ctx, mask) : q_buf; + vk_subbuffer sinks_buf = sinks ? ggml_vk_tensor_subbuffer(ctx, sinks) : q_buf; + vk_subbuffer mask_opt_buf = use_mask_opt ? ggml_vk_subbuffer(ctx, ctx->prealloc_y, 0) : q_buf; + vk_subbuffer sparse_buf = use_sparse ? ggml_vk_subbuffer(ctx, ctx->prealloc_y, 0) : q_buf; + + if (use_dequant_kv) { + const uint64_t fp = sizeof(ggml_fp16_t); + const uint64_t k_f16_sz = (uint64_t)ggml_nelements(k) * fp; + const uint64_t v_f16_sz = (uint64_t)ggml_nelements(v) * fp; + if (ctx->prealloc_size_x < k_f16_sz + v_f16_sz) { + ctx->prealloc_size_x = k_f16_sz + v_f16_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + vk_pipeline tr_k = ctx->device->pipeline_dequant_transpose[k->type]; + vk_pipeline tr_v = ctx->device->pipeline_dequant_transpose[v->type]; + ggml_pipeline_request_descriptor_sets(ctx, tr_k, 1); + ggml_pipeline_request_descriptor_sets(ctx, tr_v, 1); if (ctx->prealloc_x_need_sync) { ggml_vk_sync_buffers(ctx, subctx); } - } - // Count how many times each expert is used - vk_subbuffer expert_count_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_split_k, 0); - if (ctx->prealloc_split_k_need_sync) { + vk_subbuffer k_dst = vk_subbuffer{ ctx->prealloc_x, 0, k_f16_sz }; + vk_subbuffer v_dst = vk_subbuffer{ ctx->prealloc_x, k_f16_sz, v_f16_sz }; + const uint32_t k_nel = (uint32_t)ggml_nelements(k); + const uint32_t v_nel = (uint32_t)ggml_nelements(v); + { const std::vector pc = { (uint32_t)HSK, (uint32_t)nek2, (uint32_t)KV, 0, k_nel }; + ggml_vk_dispatch_pipeline(ctx, subctx, tr_k, { k_buf, k_dst }, pc, { k_nel, 1, 1 }); } + { const std::vector pc = { (uint32_t)HSV, (uint32_t)nev2, (uint32_t)KV, 0, v_nel }; + ggml_vk_dispatch_pipeline(ctx, subctx, tr_v, { v_buf, v_dst }, pc, { v_nel, 1, 1 }); } ggml_vk_sync_buffers(ctx, subctx); + k_buf = k_dst; + v_buf = v_dst; } + + uint32_t mask_n_head_log2 = ((sinks != nullptr) << 24) | n_head_log2; + + if (use_mask_opt) { - const std::vector pc = { (uint32_t)nei0, - (uint32_t)nei1, - (uint32_t)(nbi0 / ggml_type_size(ids->type)), - (uint32_t)(nbi1 / ggml_type_size(ids->type)), - (uint32_t)(get_misalign_bytes(ctx, ids) / ggml_type_size(ids->type)) }; - ggml_vk_dispatch_pipeline(ctx, subctx, count_experts, - { vk_subbuffer{ d_ids, ids_buf_offset, ids_sz }, expert_count_buf }, pc, { (uint32_t)n_as, 1, 1}); - } + const vk_op_flash_attn_mask_opt_push_constants opt_pc = { + nem0, + nem1, + nem2, + (uint32_t)(mask->nb[1] / sizeof(ggml_fp16_t)), + (uint32_t)(mask->nb[2] / sizeof(ggml_fp16_t)), + (uint32_t)(mask->nb[3] / sizeof(ggml_fp16_t)), + mask_opt_num_dwords, + mask_opt_num_dwords * CEIL_DIV(nem1, Br), + mask_opt_num_dwords * CEIL_DIV(nem1, Br) * nem2, + }; - if (x_non_contig) { - ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, ggml_vk_subbuffer(ctx, d_Qx, qx_buf_offset), ggml_vk_subbuffer(ctx, d_X, 0)); - } else if (qx_needs_dequant) { - const std::vector pc = { (uint32_t)ne01, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)(ggml_nelements(src0)) }; - ggml_vk_dispatch_pipeline(ctx, subctx, to_fp16_vk_0, - { vk_subbuffer{ d_Qx, qx_buf_offset, qx_sz }, vk_subbuffer{ d_X, 0, x_sz } }, pc, { (uint32_t)x_ne, 1, 1}); - } - if (y_non_contig) { - if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || - ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging != y_decode_vector_staging) { - if (ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); - } - if (y_decode_vector_staging) { - const ggml_tensor y_staged_dst = make_y_staged_dst(); - const uint32_t y_staged_dst_type_size = ggml_type_size(y_staged_dst.type); - ggml_vk_cpy_to_strided( - ctx, subctx, to_fp16_vk_1, src1, - ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0), - (uint32_t)(y_staged_dst.nb[0] / y_staged_dst_type_size), - (uint32_t)(y_staged_dst.nb[1] / y_staged_dst_type_size), - (uint32_t)(y_staged_dst.nb[2] / y_staged_dst_type_size), - (uint32_t)(y_staged_dst.nb[3] / y_staged_dst_type_size)); - } else { - ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0)); - } - ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); - ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = y_decode_vector_staging; - } - } - if (quantize_y) { - if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || - ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { - if (ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); - } - ggml_vk_quantize_q8_1(ctx, subctx, ggml_vk_subbuffer(ctx, d_Qy, qy_buf_offset), ggml_vk_subbuffer(ctx, d_Y, 0), y_ne); - ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); - ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; - } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline_fa_mask_opt, + { mask_buf, mask_opt_buf }, opt_pc, + { mask_opt_num_dwords, CEIL_DIV(nem1, Br), nem2 * nem3 }); + ggml_vk_sync_buffers(ctx, subctx); } - ggml_vk_sync_buffers(ctx, subctx); - uint32_t stride_batch_x = ne00*ne01; - uint32_t stride_b_y = y_decode_vector_staging ? y_staged_row_stride : ne10; - uint32_t stride_batch_y = y_decode_vector_staging ? y_staged_row_stride * padded_n : ne10*ne11; + if (use_sparse) + { + const vk_op_flash_attn_sparse_compact_push_constants sc_pc = { + KV, + nem1, + nem2, + (uint32_t)(mask->nb[1] / sizeof(ggml_fp16_t)), + (uint32_t)(mask->nb[2] / sizeof(ggml_fp16_t)), + (uint32_t)(mask->nb[3] / sizeof(ggml_fp16_t)), + (uint32_t)n_kv_max, + }; - if (!ggml_vk_dim01_contiguous(src0) && !qx_needs_dequant) { - stride_batch_x = src0->nb[0] / ggml_type_size(src0->type); + ggml_vk_dispatch_pipeline(ctx, subctx, sparse_compact_pipeline, + { mask_buf, sparse_buf }, sc_pc, + { nem1, nem2, nem3 }); + ggml_vk_sync_buffers(ctx, subctx); } - if (!ggml_vk_dim01_contiguous(src1) && !qy_needs_dequant && !quantize_y) { - stride_batch_y = src1->nb[0] / ggml_type_size(src1->type); - } + const vk_flash_attn_push_constants pc = { N, KV, + (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, + (uint32_t)neq2, (uint32_t)neq3, + (uint32_t)nek2, (uint32_t)nek3, + (uint32_t)nev2, (uint32_t)nev3, + nem1, nem2, nem3, + q_stride, (uint32_t)nbq2, (uint32_t)nbq3, + k_stride, nbk2_eff, nbk3_eff, + v_stride, nbv2_eff, nbv3_eff, + scale, max_bias, logit_softcap, + mask_n_head_log2, m0, m1, + gqa_ratio, split_kv, split_k }; - // compute - ggml_vk_matmul_id( - ctx, subctx, pipeline, - { d_X, x_buf_offset, x_sz }, { d_Y, y_buf_offset, y_sz }, - { d_D, d_buf_offset, d_sz }, { d_ids, ids_buf_offset, ids_sz }, expert_count_buf, - ne01, ne21, ne10, ne10, stride_b_y, ne01, - stride_batch_x, stride_batch_y, ne20*ne21, - n_as, nei0, nei1, nbi1 / ggml_type_size(ids->type), ne11, padded_n - ); // NOLINT + if (xe_fa_opt && split_k > 1) { + auto upper_power_of_2 = [&](uint32_t in) { + GGML_ASSERT(in != 0); + if (in <= 1) return 1u; + uint32_t ret = in - 1; + ret |= ret >> 1; + ret |= ret >> 2; + ret |= ret >> 4; + ret |= ret >> 8; + ret |= ret >> 16; + return ret + 1; + }; + auto to_fp16_vk_0 = ggml_vk_get_to_fp16(ctx, q->type); + const uint32_t out_dim_per_wg = qk_ratio > 16 ? 8 : 16; + size_t x_ne = ggml_nelements(q); + size_t temp_buf_offset = 0; + uint32_t head_stride_k = uint32_t(nbk2 / ggml_type_size(k->type)); + uint32_t head_stride_v = uint32_t(nbv2 / ggml_type_size(v->type)); + uint32_t batch_stride_q = uint32_t(nbq3 / ggml_type_size(q->type)); + uint32_t batch_stride_k = uint32_t(nbk3 / ggml_type_size(k->type)); + uint32_t batch_stride_v = uint32_t(nbv3 / ggml_type_size(v->type)); + uint32_t batch_stride_m = mask ? uint32_t(mask->nb[3] / ggml_type_size(mask->type)) : 0u; + uint32_t batch_stride_o = uint32_t(nb3 / ggml_type_size(dst->type)); + vk_fa_xe_opt_push_constants pc_ph1 = { (uint32_t)nek1, (uint32_t)neq1, (uint32_t)neq2, (uint32_t)nek2, qk_ratio, 1, (sinks != nullptr) ? 1u : 0u, (uint32_t)k_stride, head_stride_k, + batch_stride_q, batch_stride_k, batch_stride_v, batch_stride_m, batch_stride_o, scale }; + vk_fa_xe_opt_push_constants pc_ph2 = pc_ph1; + pc_ph2.nbkv_tok = v_stride; + pc_ph2.nbkv_head = head_stride_v; + vk_subbuffer q_temp_buf = fa_copy_qstate ? ggml_vk_subbuffer(ctx, ctx->prealloc_x, temp_buf_offset) : q_buf; + temp_buf_offset += fa_copy_qstate ? x_ne * sizeof(ggml_fp16_t) : 0; + vk_subbuffer p_temp_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_x, temp_buf_offset); + temp_buf_offset += size_p; + vk_subbuffer max_temp_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_x, temp_buf_offset); + temp_buf_offset += size_group_max; + uint32_t xe_native_sub_group_size = ctx->device.get()->architecture == INTEL_XE1 ? 8 : 16; + uint32_t aligned_gqa_ratio = upper_power_of_2(qk_ratio); + uint32_t out_per_wg_ph1 = std::min(256u / xe_native_sub_group_size, (uint32_t)neq1); + uint32_t out_per_wg_ph2 = std::min(std::max(16u / aligned_gqa_ratio, 1u), (uint32_t)neq1); + uint32_t ph1_wg = ((neq1 + out_per_wg_ph1 - 1) / out_per_wg_ph1) * nek2; + uint32_t ph2_wg = ((neq1 + out_per_wg_ph2 - 1) / out_per_wg_ph2) * ne0 / out_dim_per_wg; + if (fa_copy_qstate) { + const std::vector pc_cpy_fp16 = + { (uint32_t)q->ne[0], (uint32_t)q->ne[1], (uint32_t)q->ne[2], (uint32_t)q->ne[3], (uint32_t)(x_ne) }; + ggml_vk_sync_buffers(ctx, subctx); + ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, to_fp16_vk_0, { q_buf, q_temp_buf }, pc_cpy_fp16, { (uint32_t)(x_ne), 1, 1 }); + } + + ggml_vk_sync_buffers(ctx, subctx); + ggml_pipeline_request_descriptor_sets(ctx, xe_fa_pipeline_dual_phases.first, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, xe_fa_pipeline_dual_phases.first, + { q_temp_buf, k_buf, mask_buf, p_temp_buf, max_temp_buf }, + pc_ph1, { (uint32_t)ph1_wg, (uint32_t)nek1, (uint32_t)neq3 }); + + ggml_vk_sync_buffers(ctx, subctx); + ggml_pipeline_request_descriptor_sets(ctx, xe_fa_pipeline_dual_phases.second, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, xe_fa_pipeline_dual_phases.second, + { p_temp_buf, v_buf, max_temp_buf, sinks_buf, dst_buf }, + pc_ph2, { (uint32_t)ph2_wg, (uint32_t)nev2, (uint32_t)neq3 }); - if (x_non_contig || qx_needs_dequant) { ctx->prealloc_x_need_sync = true; - } - if (y_non_contig || quantize_y) { - ctx->prealloc_y_need_sync = true; - } - ctx->prealloc_split_k_need_sync = true; -} + } else if (split_k > 1) { + ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_flash_attn_split_k_reduce, 1); -static void ggml_vk_mul_mat_vec_id_q_f16(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - ggml_tensor * src0 = dst->src[0]; - ggml_tensor * src1 = dst->src[1]; - ggml_tensor * ids = dst->src[2]; - VK_LOG_DEBUG("ggml_vk_mul_mat_vec_id_q_f16((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - std::cerr << "), (" << ids << ", name=" << ids->name << ", type=" << ids->type << ", ne0=" << ids->ne[0] << ", ne1=" << ids->ne[1] << ", ne2=" << ids->ne[2] << ", ne3=" << ids->ne[3] << ", nb0=" << ids->nb[0] << ", nb1=" << ids->nb[1] << ", nb2=" << ids->nb[2] << ", nb3=" << ids->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << "))"); - GGML_ASSERT(ggml_vk_dim01_contiguous(src0) || src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || src0->type == GGML_TYPE_BF16); // NOLINT - GGML_ASSERT(ggml_vk_dim01_contiguous(src1) || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); // NOLINT - GGML_ASSERT(ids->type == GGML_TYPE_I32); - - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - // const uint64_t ne02 = src0->ne[2]; - // const uint64_t ne03 = src0->ne[3]; - - const uint64_t ne10 = src1->ne[0]; - const uint64_t ne11 = src1->ne[1]; - const uint64_t ne12 = src1->ne[2]; - // const uint64_t ne13 = src1->ne[3]; - - const uint64_t nei0 = ids->ne[0]; - const uint64_t nei1 = ids->ne[1]; - const uint32_t nbi1 = (uint32_t)(ids->nb[1] / sizeof(int)); - - const uint64_t ne20 = dst->ne[0]; - const uint64_t ne21 = dst->ne[1]; - // const uint64_t ne22 = dst->ne[2]; - // const uint64_t ne23 = dst->ne[3]; + if (ctx->prealloc_split_k_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - const bool x_non_contig = !ggml_vk_dim01_contiguous(src0); - const bool y_non_contig = !ggml_vk_dim01_contiguous(src1); + // We reuse workgroups_x to mean the number of splits, so we need to + // cancel out the divide by wg_denoms[0]. + uint32_t dispatch_x; + if (gqa_ratio > 1) { + workgroups_x *= pipeline->wg_denoms[0]; + dispatch_x = split_k * workgroups_x; + } else { + dispatch_x = Tr * split_k * pipeline->wg_denoms[0]; + } - const bool f16_f32_kernel = src1->type == GGML_TYPE_F32; - bool quantize_y = ctx->device->integer_dot_product && src1->type == GGML_TYPE_F32 && ggml_is_contiguous(src1) && !y_non_contig && (ne11 * ne10) % 4 == 0 && ggml_vk_should_use_mmvq(ctx->device, ne01, ne12, ne10, src0->type); + vk_subbuffer split_k_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_split_k, 0); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {q_buf, k_buf, v_buf, mask_buf, sinks_buf, split_k_buf, mask_opt_buf, sparse_buf}, + pc, { dispatch_x, workgroups_y, workgroups_z }); - vk_pipeline to_fp16_vk_0 = nullptr; - vk_pipeline to_fp16_vk_1 = nullptr; - if (x_non_contig) { - to_fp16_vk_0 = ggml_vk_get_cpy_pipeline(ctx, src0, nullptr, src0->type); - } - if (y_non_contig) { - to_fp16_vk_1 = ggml_vk_get_cpy_pipeline(ctx, src1, nullptr, src1->type); + ggml_vk_sync_buffers(ctx, subctx); + const vk_op_flash_attn_split_k_reduce_push_constants pc2 = { HSV, (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, split_k, (sinks != nullptr) }; + ggml_vk_dispatch_pipeline(ctx, subctx, ctx->device->pipeline_flash_attn_split_k_reduce, + {split_k_buf, sinks_buf, dst_buf}, + pc2, { (uint32_t)ne1, HSV, (uint32_t)(ne2 * ne3) }); + ctx->prealloc_split_k_need_sync = true; } else { - to_fp16_vk_1 = ggml_vk_get_to_fp16(ctx, src1->type); + if (gqa_ratio > 1) { + // When using gqa, we want one actual workgroup per batch, so cancel out wg_denoms + workgroups_x *= pipeline->wg_denoms[0]; + } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {q_buf, k_buf, v_buf, mask_buf, sinks_buf, dst_buf, mask_opt_buf, sparse_buf}, + pc, { workgroups_x, workgroups_y, workgroups_z }); } - // Check for mmq first - vk_pipeline dmmv = quantize_y ? ggml_vk_get_dequantize_mul_mat_vec_id(ctx, src0->type, GGML_TYPE_Q8_1, ne20, ne00) : nullptr; - vk_pipeline to_q8_1 = nullptr; - - if (dmmv == nullptr) { - // Fall back to f16 dequant mul mat - dmmv = ggml_vk_get_dequantize_mul_mat_vec_id(ctx, src0->type, src1->type, ne20, ne00); - quantize_y = false; + if (use_dequant_kv) { + ctx->prealloc_x_need_sync = true; } - - if (quantize_y) { - to_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); + if (use_mask_opt || use_sparse) { + ctx->prealloc_y_need_sync = true; } +} - const bool qx_needs_dequant = x_non_contig; - const bool qy_needs_dequant = !quantize_y && ((src1->type != GGML_TYPE_F16 && !f16_f32_kernel) || y_non_contig); - - if (ggml_nbytes(src0) > ctx->device->properties.limits.maxStorageBufferRange) { - dmmv = ggml_vk_get_64b_indexing_pipeline(ctx, dmmv); - } +static vk_conv_shapes ggml_vk_conv_select_shape(ggml_backend_vk_context * ctx, uint32_t K, uint32_t NPQ) { + auto n_tiles = [&](vk_conv_shapes s) { + return CEIL_DIV(K, vk_conv_block_sizes[s].K) + * CEIL_DIV(NPQ, vk_conv_block_sizes[s].NPQ); + }; - // Not implemented - GGML_ASSERT(y_non_contig || !qy_needs_dequant); // NOLINT - GGML_ASSERT(!qx_needs_dequant || to_fp16_vk_0 != nullptr); // NOLINT - GGML_ASSERT(!qy_needs_dequant || to_fp16_vk_1 != nullptr); // NOLINT - GGML_ASSERT(dmmv != nullptr); + // We can't query number of shader cores on Intel, use 32 as a placeholder + // so small convolutions will still choose a smaller tile. + const uint32_t shader_core_count = ctx->device->shader_core_count > 0 ? ctx->device->shader_core_count : 32; - const uint64_t x_ne = ggml_nelements(src0); - const uint64_t y_ne = ggml_nelements(src1); + // 128x128 isn't used with cm1 due to shared memory size; fall through to a smaller tile. + bool allow_128x128 = true; +#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) + if (!ctx->device->coopmat2 && ctx->device->coopmat_support && ctx->device->coopmat_support_16x16x16_f16acc) { + allow_128x128 = false; + } +#endif - const uint64_t qx_sz = ggml_vk_align_size(ggml_type_size(src0->type) * x_ne / ggml_blck_size(src0->type), ctx->device->properties.limits.minStorageBufferOffsetAlignment); - const uint64_t x_sz = x_non_contig ? ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment) : qx_sz; - const uint64_t y_sz = quantize_y ? (ggml_vk_align_size(y_ne, 128) * ggml_type_size(GGML_TYPE_Q8_1) / ggml_blck_size(GGML_TYPE_Q8_1)) : - (f16_f32_kernel ? sizeof(float) * y_ne : sizeof(ggml_fp16_t) * y_ne); + if (allow_128x128 && K > 64 && n_tiles(CONV_SHAPE_128x128) >= shader_core_count * 2) { + return CONV_SHAPE_128x128; + } else if (K <= 32 && n_tiles(CONV_SHAPE_32x256) >= shader_core_count * 2) { + return CONV_SHAPE_32x256; + } else if (K <= 64 && n_tiles(CONV_SHAPE_64x128) >= shader_core_count * 2) { + return CONV_SHAPE_64x128; + } else if (!allow_128x128 && K > 64 && n_tiles(CONV_SHAPE_64x128) >= shader_core_count * 2) { + // cm1 fallback for large K when 128x128 isn't available + return CONV_SHAPE_64x128; + } else { + return CONV_SHAPE_64x32; + } +} - { - if ( - (qx_needs_dequant && x_sz > ctx->device->properties.limits.maxStorageBufferRange) || - (qy_needs_dequant && y_sz > ctx->device->properties.limits.maxStorageBufferRange)) { - GGML_ABORT("Requested preallocation size is too large"); +static vk_pipeline ggml_vk_op_get_pipeline(ggml_backend_vk_context * ctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * dst, ggml_op op) { + switch (op) { + case GGML_OP_GET_ROWS: + GGML_ASSERT(src1->type == GGML_TYPE_I32); + if (src0->type == GGML_TYPE_I32) { + // i32 src only supports i32 result + GGML_ASSERT(dst->type == GGML_TYPE_I32); + return ctx->device->pipeline_get_rows[src0->type]; } - if (qx_needs_dequant && ctx->prealloc_size_x < x_sz) { - ctx->prealloc_size_x = x_sz; - ggml_vk_preallocate_buffers(ctx, subctx); + if (dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_get_rows[src0->type]; } - if ((qy_needs_dequant || quantize_y) && ctx->prealloc_size_y < y_sz) { - ctx->prealloc_size_y = y_sz; - ggml_vk_preallocate_buffers(ctx, subctx); + if (dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_get_rows_f32[src0->type]; } - - // Request descriptor sets - if (qx_needs_dequant) { - ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_0, 1); + return nullptr; + case GGML_OP_GET_ROWS_BACK: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_get_rows_back_f32; } - if (qy_needs_dequant) { - ggml_pipeline_request_descriptor_sets(ctx, to_fp16_vk_1, 1); + return nullptr; + case GGML_OP_ACC: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_acc_f32; } - if (quantize_y) { - ggml_pipeline_request_descriptor_sets(ctx, to_q8_1, 1); + return nullptr; + case GGML_OP_SET: + if (src0->type == src1->type && src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_I32)) { + return ctx->device->pipeline_set_f32; } - ggml_pipeline_request_descriptor_sets(ctx, dmmv, nei1); - } - - vk_subbuffer d_D = ggml_vk_tensor_subbuffer(ctx, cgraph->nodes[node_idx + ctx->num_additional_fused_ops]); - vk_subbuffer d_Qx = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer d_Qy = ggml_vk_tensor_subbuffer(ctx, src1); - vk_subbuffer d_ids = ggml_vk_tensor_subbuffer(ctx, ids); - vk_subbuffer d_F0 = d_D; - vk_subbuffer d_X, d_Y; - - if (qx_needs_dequant) { - d_X = { ctx->prealloc_x, 0, ctx->prealloc_x->size }; - } else { - d_X = d_Qx; - } - if (qy_needs_dequant || quantize_y) { - d_Y = { ctx->prealloc_y, 0, ctx->prealloc_y->size }; - } else { - d_Y = d_Qy; - } - - if (x_non_contig) { - if (ctx->prealloc_x_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + return nullptr; + case GGML_OP_ADD: + case GGML_OP_SUB: + case GGML_OP_MUL: + case GGML_OP_DIV: + if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || + (src1->type != GGML_TYPE_F32 && src1->type != GGML_TYPE_F16) || + (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16)) { + return nullptr; } - } - - if (x_non_contig) { - GGML_ASSERT(x_sz == ggml_vk_align_size(ggml_type_size(src0->type) * x_ne, ctx->device->properties.limits.minStorageBufferOffsetAlignment)); - ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_0, src0, d_Qx, d_X); - } - if (y_non_contig) { - GGML_ASSERT(y_sz == ggml_type_size(src1->type) * y_ne); - if (ctx->prealloc_y_last_pipeline_used != to_fp16_vk_1.get() || - ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { - if (ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + switch (op) { + case GGML_OP_ADD: + { + if (ctx->num_additional_fused_ops > 0) { + if (ctx->do_add_rms_partials) { + return ctx->device->pipeline_multi_add_rms[ctx->num_additional_fused_ops]; + } else { + return ctx->device->pipeline_multi_add[ctx->num_additional_fused_ops]; + } } - ggml_vk_cpy_to_contiguous(ctx, subctx, to_fp16_vk_1, src1, d_Qy, d_Y); - ctx->prealloc_y_last_pipeline_used = to_fp16_vk_1.get(); - ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; - } - } - if (quantize_y) { - if (ctx->prealloc_y_last_pipeline_used != to_q8_1.get() || - ctx->prealloc_y_last_tensor_used != src1 || - ctx->prealloc_y_last_decode_vector_staging) { - if (ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + if (ctx->do_add_rms_partials) { + auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_add_rms_norepeat : ctx->device->pipeline_add_rms; + return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; + } else { + auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_add_norepeat : ctx->device->pipeline_add; + return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; } - ggml_vk_quantize_q8_1(ctx, subctx, d_Qy, d_Y, y_ne); - ctx->prealloc_y_last_pipeline_used = to_q8_1.get(); - ctx->prealloc_y_last_tensor_used = src1; - ctx->prealloc_y_last_decode_vector_staging = false; } - } - - uint32_t stride_batch_y = ne10*ne11; - - if (!ggml_vk_dim01_contiguous(src1) && !qy_needs_dequant) { - stride_batch_y = src1->nb[2] / ggml_type_size(src1->type); - } - - const uint32_t max_groups_x = ctx->device->properties.limits.maxComputeWorkGroupCount[0]; - - uint32_t groups_x = ne01; - uint32_t groups_z = 1; - - if (ne01 > max_groups_x) { - groups_z = 64; - groups_x = CEIL_DIV(groups_x, groups_z); - } - - uint32_t fusion_flags = 0; - - if (ctx->num_additional_fused_ops > 0) { - const ggml_tensor * bias = cgraph->nodes[node_idx + 1]->src[1]; - - d_F0 = ggml_vk_tensor_subbuffer(ctx, bias); - - if (cgraph->nodes[node_idx + 1]->op == GGML_OP_MUL) { - fusion_flags |= MAT_VEC_FUSION_FLAGS_SCALE0; - } else { - GGML_ASSERT(cgraph->nodes[node_idx + 1]->op == GGML_OP_ADD_ID); - fusion_flags |= MAT_VEC_FUSION_FLAGS_BIAS0; - } - } - - vk_subbuffer d_F1 = d_D; - if (ctx->num_additional_fused_ops > 1) { - const ggml_tensor * scale = cgraph->nodes[node_idx + 2]->src[1]; - - d_F1 = ggml_vk_tensor_subbuffer(ctx, scale); - fusion_flags |= MAT_VEC_FUSION_FLAGS_SCALE1; - } - - // Loop over the batch dimension - for (uint32_t expert_i1 = 0; expert_i1 < nei1; ++expert_i1) { - const vk_mat_vec_id_push_constants pc = { - (uint32_t)ne00, (uint32_t)ne10, (uint32_t)ne10, (uint32_t)ne01, - (uint32_t)(ne00 * ne01), stride_batch_y, (uint32_t)(ne20 * ne21), - fusion_flags, - (uint32_t)nei0, (uint32_t)ne11, expert_i1, nbi1 - }; - ggml_vk_dispatch_pipeline(ctx, subctx, dmmv, - { - d_X, - d_Y, - d_D, - d_F0, - d_F1, - d_ids, - }, - pc, { groups_x, (uint32_t)nei0, groups_z }); - } - - if (x_non_contig) { - ctx->prealloc_x_need_sync = true; - } - if (y_non_contig || quantize_y) { - ctx->prealloc_y_need_sync = true; - } -} - -static bool ggml_vk_use_mul_mat_vec_id(const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - ggml_tensor * src0 = dst->src[0]; - ggml_tensor * src2 = dst->src[2]; - return (src2->ne[1] <= 8) && (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16 || ggml_is_quantized(src0->type)); -} - -static void ggml_vk_mul_mat_id(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - ggml_tensor * src0 = dst->src[0]; - ggml_tensor * src1 = dst->src[1]; - ggml_tensor * src2 = dst->src[2]; - VK_LOG_DEBUG("ggml_vk_mul_mat_id(" << src0 << ", " << src1 << ", " << src2 << ", " << dst << ")"); - if (ggml_vk_use_mul_mat_vec_id(cgraph, node_idx)) { - ggml_vk_mul_mat_vec_id_q_f16(ctx, subctx, cgraph, node_idx); - } else { - ggml_vk_mul_mat_id_q_f16(ctx, subctx, src0, src1, src2, dst); - } -} - -static bool ggml_vk_flash_attn_scalar_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type) { - GGML_UNUSED(f32acc); - // Needs to be kept up to date on shader changes - const uint32_t wg_size = params.workgroup_size; - const uint32_t Br = params.block_rows; - const uint32_t Bc = params.block_cols; - - // BF16 uses the fp32 shader (FLOAT_TYPE=float) - const uint32_t float_type_size = (device->fp16 && k_type != GGML_TYPE_BF16) ? sizeof(ggml_fp16_t) : sizeof(float); - - const bool mmq = ggml_vk_fa_scalar_uses_mmq(device, k_type, v_type); - - // tmpsh is overestimated slightly - const uint32_t tmpsh = wg_size * sizeof(float); - const uint32_t tmpshv4 = wg_size * 4 * float_type_size; - - const uint32_t masksh = Bc * (Br + 1) * float_type_size; - // DATA_A_IQ4_NL is compiled into the FA shaders unconditionally, so its shared table is always allocated. - const uint32_t iq_shmem = 16 * float_type_size; - - uint32_t Qf, kvsh, kblocksh_size; - if (mmq) { - // block_b_cache: int32_t qs[8] + FLOAT_TYPEV2 ds - const uint32_t block_b_size = 8 * sizeof(int32_t) + 2 * float_type_size; - Qf = Br * (hsk / 32) * block_b_size; - - // kvsh uses D = HSV (K goes through kblocksh instead) - kvsh = params.shmem_staging ? Bc * (hsv / 4 + 1) * 4 * float_type_size : 4 * float_type_size; - - // The mixed MMQ shader uses a superset block_a_cache that fits every - // FA-supported quant: int32_t qs[8] + uint32_t qh + FLOAT_TYPEV2 dm. - // Single-scale types leave dm.y unused; non-Q5_* leave qh unused. - const uint32_t block_a_size = 8 * sizeof(int32_t) + sizeof(uint32_t) + 2 * float_type_size; - kblocksh_size = params.shmem_staging ? Bc * (hsk / 32) * block_a_size : block_a_size; - } else { - Qf = Br * (hsk / 4 + 1) * 4 * float_type_size; - - const uint32_t D = std::max(hsk, hsv); - kvsh = params.shmem_staging ? Bc * (D / 4 + 1) * 4 * float_type_size : 4 * float_type_size; - - kblocksh_size = 0; - } - - const uint32_t total_size = tmpsh + tmpshv4 + masksh + iq_shmem + Qf + kvsh + kblocksh_size; - const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - - VK_LOG_DEBUG("ggml_vk_flash_attn_scalar_shmem_support(HSK=" << hsk << ", HSV=" << hsv << ", mmq=" << mmq << ", total_size=" << total_size << ", supported=" << supported); - - return supported; -} - -static bool ggml_vk_flash_attn_coopmat_shmem_support(const vk_device& device, const vk_fa_tuning_params& params, uint32_t hsk, uint32_t hsv, bool f32acc, ggml_type k_type, ggml_type v_type) { - GGML_UNUSED(v_type); - // Needs to be kept up to date on shader changes - const uint32_t Br = params.block_rows; - const uint32_t Bc = params.block_cols; - - const uint32_t MatBr = 16, MatBc = 16; - - const uint32_t row_split = Bc / MatBc; - - const uint32_t hsk_pad = ROUNDUP_POW2(hsk, 16); - const uint32_t hsv_pad = ROUNDUP_POW2(hsv, 16); - - const uint32_t acctype = f32acc ? 4 : 2; - const uint32_t f16vec4 = 8; - - const uint32_t tmpsh = (Bc / MatBc) * sizeof(float); - // DATA_A_IQ4_NL is compiled into the FA shaders unconditionally, so its shared table is always allocated. - const uint32_t iq_shmem = 16 * sizeof(ggml_fp16_t); - - const uint32_t qstride = hsk_pad / 4 + 2; - const uint32_t Qf = Br * qstride * f16vec4; - - const uint32_t psh_stride = Br / 4 + 2; - const uint32_t Psh = Bc * psh_stride * f16vec4; - - const uint32_t sfshstride = (hsk <= 128) ? (Br + 8) : Br; - const uint32_t sfsh = Bc * sfshstride * acctype; - - const uint32_t kvshstride = (params.shmem_staging ? std::max(hsk_pad, hsv_pad) : MatBr) / 4 + 2; - const uint32_t vsh_stride = MatBc / 4 * row_split; - const uint32_t ksh = ((kvshstride >= vsh_stride) ? (Bc * kvshstride) : (Bc * vsh_stride)) * f16vec4; - - // BF16 PVMat accumulator is f32 (no bf16 accumulator support), so pvsh is vec4 (16 bytes) - const uint32_t pvsh_elem_size = (k_type == GGML_TYPE_BF16) ? 16u : f16vec4; - const uint32_t osh_stride = params.row_split * MatBr / 4; - const uint32_t pvsh = MatBc * osh_stride * pvsh_elem_size; - - const uint32_t slope = Br * acctype; - - const uint32_t total_size = tmpsh + iq_shmem + Qf + Psh + sfsh + ksh + pvsh + slope; - const bool supported = total_size <= device->properties.limits.maxComputeSharedMemorySize; - - VK_LOG_DEBUG("ggml_vk_flash_attn_coopmat_shmem_support(HSK=" << hsk << ", HSV=" << hsv << ", f32acc=" << f32acc << ", total_size=" << total_size << ", supported=" << supported); - - return supported; -} - -static void ggml_vk_flash_attn(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * q, const ggml_tensor * k, const ggml_tensor * v, const ggml_tensor * mask, const ggml_tensor * sinks, ggml_tensor * dst) { - VK_LOG_DEBUG("ggml_vk_flash_attn((" << q << ", name=" << q->name << ", type=" << q->type << ", ne0=" << q->ne[0] << ", ne1=" << q->ne[1] << ", ne2=" << q->ne[2] << ", ne3=" << q->ne[3] << ", nb0=" << q->nb[0] << ", nb1=" << q->nb[1] << ", nb2=" << q->nb[2] << ", nb3=" << q->nb[3]; - std::cerr << "), (" << k << ", name=" << k->name << ", type=" << k->type << ", ne0=" << k->ne[0] << ", ne1=" << k->ne[1] << ", ne2=" << k->ne[2] << ", ne3=" << k->ne[3] << ", nb0=" << k->nb[0] << ", nb1=" << k->nb[1] << ", nb2=" << k->nb[2] << ", nb3=" << k->nb[3]; - std::cerr << "), (" << v << ", name=" << v->name << ", type=" << v->type << ", ne0=" << v->ne[0] << ", ne1=" << v->ne[1] << ", ne2=" << v->ne[2] << ", ne3=" << v->ne[3] << ", nb0=" << v->nb[0] << ", nb1=" << v->nb[1] << ", nb2=" << v->nb[2] << ", nb3=" << v->nb[3]; - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - if (sinks) { - std::cerr << "), (" << sinks << ", name=" << sinks->name << ", type=" << sinks->type << ", ne0=" << sinks->ne[0] << ", ne1=" << sinks->ne[1] << ", ne2=" << sinks->ne[2] << ", ne3=" << sinks->ne[3] << ", nb0=" << sinks->nb[0] << ", nb1=" << sinks->nb[1] << ", nb2=" << sinks->nb[2] << ", nb3=" << sinks->nb[3]; - } - std::cerr << "))"); - - GGML_TENSOR_LOCALS(int64_t, neq, q, ne) - GGML_TENSOR_LOCALS(size_t, nbq, q, nb) - GGML_TENSOR_LOCALS(int64_t, nek, k, ne) - GGML_TENSOR_LOCALS(size_t, nbk, k, nb) - GGML_TENSOR_LOCALS(int64_t, nev, v, ne) - GGML_TENSOR_LOCALS(size_t, nbv, v, nb) - GGML_TENSOR_LOCALS(int64_t, ne, dst, ne) - GGML_TENSOR_LOCALS(size_t, nb, dst, nb) - - const uint32_t nem0 = mask ? mask->ne[0] : 0; - const uint32_t nem1 = mask ? mask->ne[1] : 0; - const uint32_t nem2 = mask ? mask->ne[2] : 0; - const uint32_t nem3 = mask ? mask->ne[3] : 0; - - const uint32_t HSK = nek0; - const uint32_t HSV = nev0; - uint32_t N = neq1; - const uint32_t KV = nek1; - - GGML_ASSERT(ne0 == HSV); - GGML_ASSERT(ne2 == N); - - // input tensor rows must be contiguous - GGML_ASSERT(nbq0 == ggml_type_size(q->type)); - GGML_ASSERT(nbk0 == ggml_type_size(k->type)); - GGML_ASSERT(nbv0 == ggml_type_size(v->type)); - - GGML_ASSERT(neq0 == HSK); - - GGML_ASSERT(neq1 == N); - - GGML_ASSERT(nev1 == nek1); - - // dst cannot be transposed or permuted - GGML_ASSERT(nb0 == sizeof(float)); - GGML_ASSERT(nb0 <= nb1); - GGML_ASSERT(nb1 <= nb2); - GGML_ASSERT(nb2 <= nb3); - - assert(dst->type == GGML_TYPE_F32); - assert(q->type == GGML_TYPE_F32); - uint32_t gqa_ratio = 1; - uint32_t qk_ratio = neq2 / nek2; - uint32_t workgroups_x = (uint32_t)neq1; - uint32_t workgroups_y = (uint32_t)neq2; - uint32_t workgroups_z = (uint32_t)neq3; - - const bool f32acc = !ctx->device->fp16 || dst->op_params[3] == GGML_PREC_F32 || k->type == GGML_TYPE_BF16; - - // For scalar/coopmat1 FA, we can use the "large" size to accommodate qga. - // For coopmat2 FA, we always use the small size (which is still pretty large for gqa). - vk_fa_tuning_params tuning_params = get_fa_tuning_params(ctx->device, HSK, HSV, 512, KV, k->type, v->type, f32acc); - const uint32_t max_gqa = std::min(tuning_params.block_rows, 32u); - - if (N <= 8 && qk_ratio > 1 && qk_ratio <= max_gqa && - qk_ratio * nek2 == neq2 && nek2 == nev2 && nem2 <= 1) { - // grouped query attention - make the N dimension equal to gqa_ratio, reduce - // workgroups proportionally in y dimension. The shader will detect gqa_ratio > 1 - // and change addressing calculations to index Q's dimension 2. - gqa_ratio = qk_ratio; - N = gqa_ratio; - workgroups_y /= gqa_ratio; - } - - tuning_params = get_fa_tuning_params(ctx->device, HSK, HSV, N, KV, k->type, v->type, f32acc); - - const uint32_t q_stride = (uint32_t)(nbq1 / ggml_type_size(q->type)); - uint32_t k_stride = (uint32_t)(nbk1 / ggml_type_size(k->type)); - uint32_t v_stride = (uint32_t)(nbv1 / ggml_type_size(v->type)); - - // For F32, the shader treats it as a block of size 4 (for vec4 loads) - if (k->type == GGML_TYPE_F32) { - k_stride /= 4; - } - if (v->type == GGML_TYPE_F32) { - v_stride /= 4; - } - - const uint32_t alignment = tuning_params.block_cols; - bool aligned = (KV % alignment) == 0 && - // the "aligned" shader variant will forcibly align strides, for performance - (q_stride & 7) == 0 && (k_stride & 7) == 0 && (v_stride & 7) == 0; - - // Need to use the coopmat2 variant that clamps loads when HSK/HSV aren't sufficiently aligned. - if (((HSK | HSV) % 16) != 0 && tuning_params.path == FA_COOPMAT2) { - aligned = false; - } - - float scale = 1.0f; - float max_bias = 0.0f; - float logit_softcap = 0.0f; - - memcpy(&scale, (const float *) dst->op_params + 0, sizeof(float)); - memcpy(&max_bias, (const float *) dst->op_params + 1, sizeof(float)); - memcpy(&logit_softcap, (const float *) dst->op_params + 2, sizeof(float)); - - if (logit_softcap != 0) { - scale /= logit_softcap; - } - - // Only use mask opt when the mask is fairly large. This hasn't been tuned extensively. - bool use_mask_opt = mask && nem1 >= 32 && nem0 * nem1 > 32768 && nem0 >= tuning_params.block_cols * 16 - && (ctx->device->architecture != vk_device_architecture::AMD_GCN || HSK > 256 || HSV > 256); - vk_fa_pipeline_state fa_pipeline_state = get_fa_pipeline_state(ctx->device, tuning_params, HSK, HSV, aligned, f32acc, - mask != nullptr, use_mask_opt, logit_softcap != 0, k->type, v->type); - - vk_pipeline pipeline = nullptr; - - { - std::lock_guard guard(ctx->device->compile_mutex); - auto &pipelines = ctx->device->pipeline_flash_attn_f32_f16; - auto it = pipelines.find(fa_pipeline_state); - if (it != pipelines.end()) { - pipeline = it->second; - } else { - pipelines[fa_pipeline_state] = pipeline = std::make_shared(); - } - } - - assert(pipeline); - // Compile early to initialize wg_denoms. - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - - uint32_t split_kv = KV; - uint32_t split_k = 1; - - // Intel Alchemist prefers more workgroups - const uint32_t shader_core_count_multiplier = (ctx->device->vendor_id == VK_VENDOR_ID_INTEL && ctx->device->architecture != INTEL_XE2) ? 2 : 1; - - // Use a placeholder core count if one isn't available. split_k is a big help for perf. - const uint32_t shader_core_count = ctx->device->shader_core_count ? ctx->device->shader_core_count * shader_core_count_multiplier : 16; - - const uint32_t Br = fa_pipeline_state.Br; - const uint32_t Bc = fa_pipeline_state.Bc; - - GGML_ASSERT(Br == pipeline->wg_denoms[0]); - const uint32_t Tr = CEIL_DIV(N, Br); - - // Try to use split_k when KV is large enough to be worth the overhead. - if (gqa_ratio > 1 && workgroups_x <= Br) { - split_k = shader_core_count * 2 / (workgroups_x * workgroups_y * workgroups_z); - } else if (gqa_ratio <= 1) { - uint32_t total_wgs_no_split = Tr * workgroups_y * workgroups_z; - if (total_wgs_no_split < shader_core_count * 2) { - split_k = shader_core_count * 2 / total_wgs_no_split; - } - } - - if (split_k > 1) { - // Try to evenly split KV into split_k chunks, but it needs to be a multiple - // of "align", so recompute split_k based on that. - split_kv = ROUNDUP_POW2(std::max(1u, KV / split_k), alignment); - split_k = CEIL_DIV(KV, split_kv); - } - - // Reserve space for split_k temporaries. For each split x batch, we need to store the O matrix (D x ne1) - // and the per-row m and L values (ne1 rows). We store all the matrices first, followed by the rows. - // For matrices, the order is (inner to outer) [HSV, ne1, k, ne2, ne3]. - // For L/M, the order is (inner to outer) [ne1, k, ne2, ne3]. - const uint64_t split_k_size = split_k > 1 ? (HSV * ne1 * sizeof(float) + ne1 * sizeof(float) * 2) * split_k * ne2 * ne3 : 0; - if (split_k_size > ctx->device->properties.limits.maxStorageBufferRange) { - GGML_ABORT("Requested preallocation size is too large"); - } - if (ctx->prealloc_size_split_k < split_k_size) { - ctx->prealloc_size_split_k = split_k_size; - ggml_vk_preallocate_buffers(ctx, subctx); - } - - const uint32_t mask_opt_num_dwords = CEIL_DIV(nem0, 16 * Bc); - const uint64_t mask_opt_size = sizeof(uint32_t) * mask_opt_num_dwords * CEIL_DIV(nem1, Br) * nem2 * nem3; - - vk_pipeline pipeline_fa_mask_opt = nullptr; - if (use_mask_opt) { - { - std::lock_guard guard(ctx->device->compile_mutex); - auto &pipelines = ctx->device->pipeline_fa_mask_opt; - auto it = pipelines.find({Br, Bc}); - if (it != pipelines.end()) { - pipeline_fa_mask_opt = it->second; - } else { - pipelines[{Br, Bc}] = pipeline_fa_mask_opt = std::make_shared(); - } - } - assert(pipeline_fa_mask_opt); - ggml_pipeline_request_descriptor_sets(ctx, pipeline_fa_mask_opt, 1); - - if (ctx->prealloc_size_y < mask_opt_size) { - ctx->prealloc_size_y = mask_opt_size; - ggml_vk_preallocate_buffers(ctx, subctx); - } - if (ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); - } - } - - const uint32_t n_head_kv = neq2; - const uint32_t n_head_log2 = 1u << (uint32_t) floorf(log2f((float) n_head_kv)); - const float m0 = powf(2.0f, -(max_bias ) / n_head_log2); - const float m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2); - - vk_subbuffer q_buf = ggml_vk_tensor_subbuffer(ctx, q); - vk_subbuffer k_buf = ggml_vk_tensor_subbuffer(ctx, k); - vk_subbuffer v_buf = ggml_vk_tensor_subbuffer(ctx, v); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer mask_buf = mask ? ggml_vk_tensor_subbuffer(ctx, mask) : q_buf; - vk_subbuffer sinks_buf = sinks ? ggml_vk_tensor_subbuffer(ctx, sinks) : q_buf; - vk_subbuffer mask_opt_buf = use_mask_opt ? ggml_vk_subbuffer(ctx, ctx->prealloc_y, 0) : q_buf; - - uint32_t mask_n_head_log2 = ((sinks != nullptr) << 24) | n_head_log2; - - if (use_mask_opt) - { - const vk_op_flash_attn_mask_opt_push_constants opt_pc = { - nem0, - nem1, - nem2, - (uint32_t)(mask->nb[1] / sizeof(ggml_fp16_t)), - (uint32_t)(mask->nb[2] / sizeof(ggml_fp16_t)), - (uint32_t)(mask->nb[3] / sizeof(ggml_fp16_t)), - mask_opt_num_dwords, - mask_opt_num_dwords * CEIL_DIV(nem1, Br), - mask_opt_num_dwords * CEIL_DIV(nem1, Br) * nem2, - }; - - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline_fa_mask_opt, - { mask_buf, mask_opt_buf }, opt_pc, - { mask_opt_num_dwords, CEIL_DIV(nem1, Br), nem2 * nem3 }); - ggml_vk_sync_buffers(ctx, subctx); - } - - const vk_flash_attn_push_constants pc = { N, KV, - (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, - (uint32_t)neq2, (uint32_t)neq3, - (uint32_t)nek2, (uint32_t)nek3, - (uint32_t)nev2, (uint32_t)nev3, - nem1, nem2, nem3, - q_stride, (uint32_t)nbq2, (uint32_t)nbq3, - k_stride, (uint32_t)nbk2, (uint32_t)nbk3, - v_stride, (uint32_t)nbv2, (uint32_t)nbv3, - scale, max_bias, logit_softcap, - mask_n_head_log2, m0, m1, - gqa_ratio, split_kv, split_k }; - - if (split_k > 1) { - ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_flash_attn_split_k_reduce, 1); - - if (ctx->prealloc_split_k_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); - } - - // We reuse workgroups_x to mean the number of splits, so we need to - // cancel out the divide by wg_denoms[0]. - uint32_t dispatch_x; - if (gqa_ratio > 1) { - workgroups_x *= pipeline->wg_denoms[0]; - dispatch_x = split_k * workgroups_x; - } else { - dispatch_x = Tr * split_k * pipeline->wg_denoms[0]; - } - - vk_subbuffer split_k_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_split_k, 0); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {q_buf, k_buf, v_buf, mask_buf, sinks_buf, split_k_buf, mask_opt_buf}, - pc, { dispatch_x, workgroups_y, workgroups_z }); - - ggml_vk_sync_buffers(ctx, subctx); - const vk_op_flash_attn_split_k_reduce_push_constants pc2 = { HSV, (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, split_k, (sinks != nullptr) }; - ggml_vk_dispatch_pipeline(ctx, subctx, ctx->device->pipeline_flash_attn_split_k_reduce, - {split_k_buf, sinks_buf, dst_buf}, - pc2, { (uint32_t)ne1, HSV, (uint32_t)(ne2 * ne3) }); - ctx->prealloc_split_k_need_sync = true; - } else { - if (gqa_ratio > 1) { - // When using gqa, we want one actual workgroup per batch, so cancel out wg_denoms - workgroups_x *= pipeline->wg_denoms[0]; - } - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {q_buf, k_buf, v_buf, mask_buf, sinks_buf, dst_buf, mask_opt_buf}, - pc, { workgroups_x, workgroups_y, workgroups_z }); - } -} - -static vk_conv_shapes ggml_vk_conv_select_shape(ggml_backend_vk_context * ctx, uint32_t K, uint32_t NPQ) { - auto n_tiles = [&](vk_conv_shapes s) { - return CEIL_DIV(K, vk_conv_block_sizes[s].K) - * CEIL_DIV(NPQ, vk_conv_block_sizes[s].NPQ); - }; - - // We can't query number of shader cores on Intel, use 32 as a placeholder - // so small convolutions will still choose a smaller tile. - const uint32_t shader_core_count = ctx->device->shader_core_count > 0 ? ctx->device->shader_core_count : 32; - - // 128x128 isn't used with cm1 due to shared memory size; fall through to a smaller tile. - bool allow_128x128 = true; -#if defined(VK_KHR_cooperative_matrix) && defined(GGML_VULKAN_COOPMAT_GLSLC_SUPPORT) - if (!ctx->device->coopmat2 && ctx->device->coopmat_support && ctx->device->coopmat_support_16x16x16_f16acc) { - allow_128x128 = false; - } -#endif - - if (allow_128x128 && K > 64 && n_tiles(CONV_SHAPE_128x128) >= shader_core_count * 2) { - return CONV_SHAPE_128x128; - } else if (K <= 32 && n_tiles(CONV_SHAPE_32x256) >= shader_core_count * 2) { - return CONV_SHAPE_32x256; - } else if (K <= 64 && n_tiles(CONV_SHAPE_64x128) >= shader_core_count * 2) { - return CONV_SHAPE_64x128; - } else if (!allow_128x128 && K > 64 && n_tiles(CONV_SHAPE_64x128) >= shader_core_count * 2) { - // cm1 fallback for large K when 128x128 isn't available - return CONV_SHAPE_64x128; - } else { - return CONV_SHAPE_64x32; - } -} - -static vk_pipeline ggml_vk_op_get_pipeline(ggml_backend_vk_context * ctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * dst, ggml_op op) { - switch (op) { - case GGML_OP_GET_ROWS: - GGML_ASSERT(src1->type == GGML_TYPE_I32); - if (src0->type == GGML_TYPE_I32) { - // i32 src only supports i32 result - GGML_ASSERT(dst->type == GGML_TYPE_I32); - return ctx->device->pipeline_get_rows[src0->type]; - } - if (dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_get_rows[src0->type]; - } - if (dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_get_rows_f32[src0->type]; - } - return nullptr; - case GGML_OP_GET_ROWS_BACK: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_get_rows_back_f32; - } - return nullptr; - case GGML_OP_ACC: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_acc_f32; - } - return nullptr; - case GGML_OP_SET: - if (src0->type == src1->type && src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_I32)) { - return ctx->device->pipeline_set_f32; - } - return nullptr; - case GGML_OP_ADD: - case GGML_OP_SUB: - case GGML_OP_MUL: - case GGML_OP_DIV: - if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || - (src1->type != GGML_TYPE_F32 && src1->type != GGML_TYPE_F16) || - (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16)) { - return nullptr; - } - switch (op) { - case GGML_OP_ADD: - { - if (ctx->num_additional_fused_ops > 0) { - if (ctx->do_add_rms_partials) { - return ctx->device->pipeline_multi_add_rms[ctx->num_additional_fused_ops]; - } else { - return ctx->device->pipeline_multi_add[ctx->num_additional_fused_ops]; - } - } - if (ctx->do_add_rms_partials) { - auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_add_rms_norepeat : ctx->device->pipeline_add_rms; - return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; - } else { - auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_add_norepeat : ctx->device->pipeline_add; - return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; - } - } - case GGML_OP_SUB: - { - auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_sub_norepeat : ctx->device->pipeline_sub; - return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; - } - case GGML_OP_MUL: - { - auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_mul_norepeat : ctx->device->pipeline_mul; - return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; - } - case GGML_OP_DIV: - { - auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_div_norepeat : ctx->device->pipeline_div; - return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; - } - default: - break; - } - return nullptr; - case GGML_OP_ADD_ID: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_add_id_f32; - } - return nullptr; - case GGML_OP_OUT_PROD: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_out_prod_f32; - } - return nullptr; - case GGML_OP_CONCAT: { - if (!ggml_vk_concat_supported(src0, src1, dst)) { - return nullptr; - } - switch (ggml_vk_concat_unit_size(src0->type)) { - case 1: - return ctx->device->pipeline_concat_i8; - case 2: - return ctx->device->pipeline_concat_i16; - case 4: - return ctx->device->pipeline_concat_i32; - case 8: - return ctx->device->pipeline_concat_i64; - default: - return nullptr; - } - } - case GGML_OP_UPSCALE: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - uint32_t mode = (ggml_get_op_params_i32(dst, 0) & (0xFF | GGML_SCALE_FLAG_ANTIALIAS)); - switch (mode) { - case GGML_SCALE_MODE_NEAREST: - return ctx->device->pipeline_upscale_nearest_f32; - case GGML_SCALE_MODE_BILINEAR: - return ctx->device->pipeline_upscale_bilinear_f32; - case GGML_SCALE_MODE_BICUBIC: - return ctx->device->pipeline_upscale_bicubic_f32; - case GGML_SCALE_MODE_BILINEAR | GGML_SCALE_FLAG_ANTIALIAS: - return ctx->device->pipeline_upscale_bilinear_antialias_f32; - default: - return nullptr; - } - } - return nullptr; - case GGML_OP_SCALE: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_scale_f32; - } - return nullptr; - case GGML_OP_SQR: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_sqr[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_SQRT: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_sqrt[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_SIN: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_sin[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_COS: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_cos[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_LOG: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_log[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_TRI: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_tri[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_DIAG: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_diag[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_CLAMP: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_clamp[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_PAD: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_pad_f32; - } - return nullptr; - case GGML_OP_ROLL: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_roll_f32; - } - return nullptr; - case GGML_OP_REPEAT: - if (ggml_type_size(src0->type) == sizeof(float) && ggml_type_size(dst->type) == sizeof(float)) { - return ctx->device->pipeline_repeat_i32; - } - if (ggml_type_size(src0->type) == 2 && ggml_type_size(dst->type) == 2) { - return ctx->device->pipeline_repeat_i16; - } - return nullptr; - case GGML_OP_REPEAT_BACK: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_repeat_back_f32; - } - return nullptr; - case GGML_OP_CPY: - case GGML_OP_CONT: - case GGML_OP_DUP: - return ggml_vk_get_cpy_pipeline(ctx, src0, dst, dst->type); - case GGML_OP_SET_ROWS: - { - if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) { - return nullptr; - } - const int src_idx = src0->type == GGML_TYPE_F16; - if (src1->type == GGML_TYPE_I64) { - return ctx->device->pipeline_set_rows_i64[src_idx][dst->type]; - } else if (src1->type == GGML_TYPE_I32) { - return ctx->device->pipeline_set_rows_i32[src_idx][dst->type]; - } - return nullptr; - } - case GGML_OP_SILU_BACK: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_silu_back_f32; - } - return nullptr; - case GGML_OP_NORM: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_norm_f32; - } - return nullptr; - case GGML_OP_GROUP_NORM: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_group_norm_f32; - } - return nullptr; - case GGML_OP_RMS_NORM: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - if (ctx->do_add_rms_partials) { - return ctx->num_additional_fused_ops > 0 ? ctx->device->pipeline_rms_norm_mul_partials_f32 : ctx->device->pipeline_rms_norm_partials_f32; - } else { - return ctx->num_additional_fused_ops > 0 ? ctx->device->pipeline_rms_norm_mul_f32 : ctx->device->pipeline_rms_norm_f32; - } - } - return nullptr; - case GGML_OP_RMS_NORM_BACK: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rms_norm_back_f32; - } - return nullptr; - case GGML_OP_L2_NORM: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_l2_norm_f32; - } - return nullptr; - case GGML_OP_UNARY: - if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || - (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16) || - (src0->type != dst->type)) { - return nullptr; - } - - switch (ggml_get_unary_op(dst)) { - case GGML_UNARY_OP_EXP: - return ctx->device->pipeline_exp[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_EXPM1: - return ctx->device->pipeline_expm1[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_ELU: - return ctx->device->pipeline_elu[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_SILU: - return ctx->device->pipeline_silu[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_GELU: - return ctx->device->pipeline_gelu[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_GELU_ERF: - return ctx->device->pipeline_gelu_erf[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_GELU_QUICK: - return ctx->device->pipeline_gelu_quick[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_RELU: - return ctx->device->pipeline_relu[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_XIELU: - return ctx->device->pipeline_xielu[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_NEG: - return ctx->device->pipeline_neg[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_TANH: - return ctx->device->pipeline_tanh[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_SIGMOID: - return ctx->device->pipeline_sigmoid[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_HARDSIGMOID: - return ctx->device->pipeline_hardsigmoid[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_HARDSWISH: - return ctx->device->pipeline_hardswish[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_ABS: - return ctx->device->pipeline_abs[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_SOFTPLUS: - return ctx->device->pipeline_softplus[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_STEP: - return ctx->device->pipeline_step[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_ROUND: - return ctx->device->pipeline_round[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_CEIL: - return ctx->device->pipeline_ceil[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_FLOOR: - return ctx->device->pipeline_floor[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_TRUNC: - return ctx->device->pipeline_trunc[dst->type == GGML_TYPE_F16]; - case GGML_UNARY_OP_SGN: - return ctx->device->pipeline_sgn[dst->type == GGML_TYPE_F16]; - default: - break; - } - return nullptr; - case GGML_OP_GLU: - if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || - (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16) || - (src0->type != dst->type)) { - return nullptr; - } - - switch (ggml_get_glu_op(dst)) { - case GGML_GLU_OP_GEGLU: - return ctx->device->pipeline_geglu[dst->type == GGML_TYPE_F16]; - case GGML_GLU_OP_REGLU: - return ctx->device->pipeline_reglu[dst->type == GGML_TYPE_F16]; - case GGML_GLU_OP_SWIGLU: - return ctx->device->pipeline_swiglu[dst->type == GGML_TYPE_F16]; - case GGML_GLU_OP_SWIGLU_OAI: - return ctx->device->pipeline_swiglu_oai[dst->type == GGML_TYPE_F16]; - case GGML_GLU_OP_GEGLU_ERF: - return ctx->device->pipeline_geglu_erf[dst->type == GGML_TYPE_F16]; - case GGML_GLU_OP_GEGLU_QUICK: - return ctx->device->pipeline_geglu_quick[dst->type == GGML_TYPE_F16]; - default: - break; - } - return nullptr; - case GGML_OP_DIAG_MASK_INF: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_diag_mask_inf_f32; - } - return nullptr; - case GGML_OP_SOFT_MAX: - GGML_ASSERT(!src1 || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); - GGML_ASSERT(!src2 || src2->type == GGML_TYPE_F32); - - if (ctx->num_additional_fused_ops) { - uint32_t idx = (uint32_t)ceilf(log2f(float(dst->ne[0]))); - GGML_ASSERT(idx < num_topk_moe_pipelines); - // use n_experts from push constant if it's not equal to the power of two spec constant - bool use_push = dst->ne[0] != (1u << idx); - return ctx->device->pipeline_topk_moe[idx][use_push]; - } - - if (src0->type == GGML_TYPE_F32 && (src1 == nullptr || src1->type == GGML_TYPE_F32) && dst->type == GGML_TYPE_F32) { - return src0->ne[0] > 1024 ? ctx->device->pipeline_soft_max_f32_wg512 : ctx->device->pipeline_soft_max_f32; - } - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F32) { - return src0->ne[0] > 1024 ? ctx->device->pipeline_soft_max_f32_f16_wg512 : ctx->device->pipeline_soft_max_f32_f16; - } - return nullptr; - case GGML_OP_SOFT_MAX_BACK: - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_soft_max_back_f32; - } - return nullptr; - case GGML_OP_ROPE: - case GGML_OP_ROPE_BACK: - { - const ggml_tensor *rope = ctx->num_additional_fused_ops == 2 ? dst->src[0]->src[0] : dst; - const int mode = ((const int32_t *) rope->op_params)[2]; - const bool is_neox = mode & GGML_ROPE_TYPE_NEOX; - const bool is_mrope = mode & GGML_ROPE_TYPE_MROPE; - const bool is_vision = mode == GGML_ROPE_TYPE_VISION; - - if (is_neox) { - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rope_neox_f32; - } - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_neox_f32_f16; - } - if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_neox_f16; - } - } else if (is_mrope && !is_vision) { - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rope_multi_f32; - } - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_multi_f32_f16; - } - if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_multi_f16; - } - } else if (is_vision) { - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rope_vision_f32; - } - if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_vision_f16; - } - } else { - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rope_norm_f32; - } - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_norm_f32_f16; - } - if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_rope_norm_f16; - } - } - return nullptr; - } - case GGML_OP_SUM: - case GGML_OP_SUM_ROWS: - case GGML_OP_MEAN: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_sum_rows_f32; - } - return nullptr; - case GGML_OP_CUMSUM: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - if (src0->ne[0] <= 512) { - return ctx->device->pipeline_cumsum_small_f32; - } else { - return ctx->device->pipeline_cumsum_f32; - } - } - return nullptr; - case GGML_OP_SOLVE_TRI: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - - vk_solve_tri_pipeline_state solve_tri_pipeline_state(src0->ne[0], src1->ne[0]); - - vk_pipeline pipeline = nullptr; - - { - std::lock_guard guard(ctx->device->compile_mutex); - auto it = ctx->device->pipeline_solve_tri_f32.find(solve_tri_pipeline_state); - if (it != ctx->device->pipeline_solve_tri_f32.end()) { - pipeline = it->second; - } else { - ctx->device->pipeline_solve_tri_f32[solve_tri_pipeline_state] = pipeline = std::make_shared(); - } - } - - return pipeline; - } - return nullptr; - case GGML_OP_ARGMAX: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_I32) { - return ctx->device->pipeline_argmax_f32; - } - return nullptr; - case GGML_OP_COUNT_EQUAL: - if (src0->type == GGML_TYPE_I32 && src1->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_I64) { - return ctx->device->pipeline_count_equal_i32; - } - return nullptr; - case GGML_OP_IM2COL: - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_im2col_f32; - } - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_im2col_f32_f16; - } - return nullptr; - case GGML_OP_IM2COL_3D: - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_im2col_3d_f32; - } - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_im2col_3d_f32_f16; - } - return nullptr; - case GGML_OP_TIMESTEP_EMBEDDING: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_timestep_embedding_f32; - } - return nullptr; - case GGML_OP_CONV_TRANSPOSE_1D: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_conv_transpose_1d_f32; - } - return nullptr; - case GGML_OP_COL2IM_1D: - switch (src0->type) { - case GGML_TYPE_F32: return ctx->device->pipeline_col2im_1d_f32; - case GGML_TYPE_F16: return ctx->device->pipeline_col2im_1d_f16; - case GGML_TYPE_BF16: return ctx->device->pipeline_col2im_1d_bf16; - default: return nullptr; - } - case GGML_OP_POOL_1D: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_pool1d_f32; - } - return nullptr; - case GGML_OP_POOL_2D: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_pool2d_f32; - } - return nullptr; - case GGML_OP_RWKV_WKV6: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rwkv_wkv6_f32; - } - return nullptr; - case GGML_OP_RWKV_WKV7: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_rwkv_wkv7_f32; - } - return nullptr; - case GGML_OP_GATED_LINEAR_ATTN: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_gated_linear_attn_f32; - } - return nullptr; - case GGML_OP_GATED_DELTA_NET: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - const uint32_t S_v = dst->src[2]->ne[0]; - const uint32_t kda = (dst->src[3]->ne[0] == (int64_t)S_v) ? 1 : 0; - uint32_t si; - switch (S_v) { - case 16: si = 0; break; - case 32: si = 1; break; - case 64: si = 2; break; - case 128: si = 3; break; - default: return nullptr; - } - return ctx->device->pipeline_gated_delta_net[si][kda]; - } - return nullptr; - case GGML_OP_SSM_SCAN: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - const uint32_t d_state = src0->ne[0]; - if (d_state == 128) { - return ctx->device->pipeline_ssm_scan_f32_d128; - } else if (d_state == 256) { - return ctx->device->pipeline_ssm_scan_f32_d256; - } - } - return nullptr; - case GGML_OP_SSM_CONV: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - switch (ctx->num_additional_fused_ops) { - case 0: return ctx->device->pipeline_ssm_conv_f32; - case 1: return ctx->device->pipeline_ssm_conv_silu_f32; - case 2: return ctx->device->pipeline_ssm_conv_bias_silu_f32; - default: return nullptr; - } - } - return nullptr; - case GGML_OP_OPT_STEP_ADAMW: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_opt_step_adamw_f32; - } - return nullptr; - case GGML_OP_OPT_STEP_SGD: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_opt_step_sgd_f32; - } - return nullptr; - case GGML_OP_LEAKY_RELU: - if (src0->type == dst->type && - (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { - return ctx->device->pipeline_leaky_relu[dst->type == GGML_TYPE_F16]; - } - return nullptr; - case GGML_OP_CONV_2D: - case GGML_OP_CONV_TRANSPOSE_2D: - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - uint32_t K = dst->ne[2]; // Cout - uint32_t NPQ = dst->ne[3] * dst->ne[1] * dst->ne[0]; // N * OH * OW - vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, K, NPQ); - - bool transpose = dst->op == GGML_OP_CONV_TRANSPOSE_2D; - uint32_t KW = (uint32_t)src0->ne[0]; - uint32_t KH = (uint32_t)src0->ne[1]; - uint32_t s0 = (uint32_t)(ggml_get_op_params_i32(dst, 0)); - uint32_t s1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 1) : s0; - uint32_t p0 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 2) : 0; - uint32_t p1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 3) : 0; - uint32_t d0 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 4) : 1; - uint32_t d1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 5) : 1; - - // tile-aligned shapes let the shader skip bounds checks - const uint32_t Cin = (uint32_t)src1->ne[2]; - const uint32_t CRS = Cin * KW * KH; - const uint32_t BS_K = vk_conv_block_sizes[shape].K; - const uint32_t BS_CRS = vk_conv_block_sizes[shape].CRS; - const uint32_t BS_NPQ = vk_conv_block_sizes[shape].NPQ; - const uint32_t aligned = ((K % BS_K == 0) && - (CRS % BS_CRS == 0) && - (NPQ % BS_NPQ == 0)) ? 1u : 0u; - - vk_conv2d_pipeline_state conv2d_pipeline_state(s0, s1, p0, p1, d0, d1, KW, KH, aligned); - - std::map *pipelines = nullptr; - if (op == GGML_OP_CONV_2D) { - if (src0->type == GGML_TYPE_F32) { - pipelines = &ctx->device->pipeline_conv2d_f32[shape]; - } else if (src0->type == GGML_TYPE_F16) { - pipelines = &ctx->device->pipeline_conv2d_f16_f32[shape]; - } - } else if (op == GGML_OP_CONV_TRANSPOSE_2D) { - if (src0->type == GGML_TYPE_F32) { - pipelines = &ctx->device->pipeline_conv_transpose_2d_f32[shape]; - } else if (src0->type == GGML_TYPE_F16) { - pipelines = &ctx->device->pipeline_conv_transpose_2d_f16_f32[shape]; - } - } - - vk_pipeline pipeline = nullptr; - - { - std::lock_guard guard(ctx->device->compile_mutex); - auto it = pipelines->find(conv2d_pipeline_state); - if (it != pipelines->end()) { - pipeline = it->second; - } else { - (*pipelines)[conv2d_pipeline_state] = pipeline = std::make_shared(); - } - } - - return pipeline; - } - return nullptr; - case GGML_OP_CONV_2D_DW: - if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - if (ggml_is_contiguous(src1)) { - return ctx->device->pipeline_conv2d_dw_whcn_f32; - } else if (ggml_is_contiguous_channels(src1)) { - return ctx->device->pipeline_conv2d_dw_cwhn_f32; - } - } else if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F32) { - if (ggml_is_contiguous(src1)) { - return ctx->device->pipeline_conv2d_dw_whcn_f16_f32; - } else if (ggml_is_contiguous_channels(src1)) { - return ctx->device->pipeline_conv2d_dw_cwhn_f16_f32; - } - } - return nullptr; - case GGML_OP_CONV_3D: - if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - const uint32_t OC = (uint32_t)ggml_get_op_params_i32(dst, 11); - const uint32_t IC = (uint32_t)ggml_get_op_params_i32(dst, 9); - const uint32_t N = (uint32_t)ggml_get_op_params_i32(dst, 10); - const uint32_t NPQ = N * dst->ne[2] * dst->ne[1] * dst->ne[0]; - const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, OC, NPQ); - - const uint32_t KW = (uint32_t)src0->ne[0]; - const uint32_t KH = (uint32_t)src0->ne[1]; - const uint32_t KD = (uint32_t)src0->ne[2]; - const uint32_t s0 = (uint32_t)ggml_get_op_params_i32(dst, 0); - const uint32_t s1 = (uint32_t)ggml_get_op_params_i32(dst, 1); - const uint32_t s2 = (uint32_t)ggml_get_op_params_i32(dst, 2); - const uint32_t p0 = (uint32_t)ggml_get_op_params_i32(dst, 3); - const uint32_t p1 = (uint32_t)ggml_get_op_params_i32(dst, 4); - const uint32_t p2 = (uint32_t)ggml_get_op_params_i32(dst, 5); - const uint32_t d0 = (uint32_t)ggml_get_op_params_i32(dst, 6); - const uint32_t d1 = (uint32_t)ggml_get_op_params_i32(dst, 7); - const uint32_t d2 = (uint32_t)ggml_get_op_params_i32(dst, 8); - - const uint32_t CRS = IC * KW * KH * KD; - const uint32_t BS_K = vk_conv_block_sizes[shape].K; - const uint32_t BS_CRS = vk_conv_block_sizes[shape].CRS; - const uint32_t BS_NPQ = vk_conv_block_sizes[shape].NPQ; - const uint32_t aligned = ((OC % BS_K == 0) && - (CRS % BS_CRS == 0) && - (NPQ % BS_NPQ == 0)) ? 1u : 0u; - - vk_conv3d_pipeline_state conv3d_pipeline_state(s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD, aligned); - - std::map *pipelines = nullptr; - if (src0->type == GGML_TYPE_F32) { - pipelines = &ctx->device->pipeline_conv3d_f32[shape]; - } else if (src0->type == GGML_TYPE_F16) { - pipelines = &ctx->device->pipeline_conv3d_f16_f32[shape]; - } else { - return nullptr; - } - - vk_pipeline pipeline = nullptr; - - { - std::lock_guard guard(ctx->device->compile_mutex); - auto it = pipelines->find(conv3d_pipeline_state); - if (it != pipelines->end()) { - pipeline = it->second; - } else { - (*pipelines)[conv3d_pipeline_state] = pipeline = std::make_shared(); - } - } - - return pipeline; - } - return nullptr; - case GGML_OP_ADD1: - if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_add1_f16_f16; - } - if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_add1_f16_f32; - } - if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_add1_f32_f32; - } - return nullptr; - case GGML_OP_ARANGE: - if (dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_arange_f32; - } - return nullptr; - case GGML_OP_FILL: - if (dst->type == GGML_TYPE_F32) { - return ctx->device->pipeline_fill_f32; - } - if (dst->type == GGML_TYPE_F16) { - return ctx->device->pipeline_fill_f16; - } - return nullptr; - default: - return nullptr; - } - - GGML_UNUSED(src2); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_unary_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - p.misalign_offsets = (a_offset << 16) | d_offset; - - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_glu_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t b_offset = src1 ? get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type) : a_offset; - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - GGML_ASSERT(a_offset < (1u << 8)); - GGML_ASSERT(b_offset < (1u << 8)); - GGML_ASSERT(d_offset < (1u << 8)); - - p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; - - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_sum_rows_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - p.misalign_offsets = (a_offset << 16) | d_offset; - - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_pad_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - p.misalign_offsets = (a_offset << 16) | d_offset; - - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_im2col_3d_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - p.misalign_offsets = (a_offset << 16) | d_offset; - - GGML_UNUSED(src0); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_binary_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t b_offset = get_misalign_bytes(ctx, src1) / ggml_type_size(src1->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - GGML_ASSERT(dst->op != GGML_OP_GET_ROWS || (a_offset == 0 && b_offset == 0 && d_offset == 0)); - - p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; - - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_concat_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t unit_size = ggml_vk_concat_unit_size(dst->type); - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / unit_size; - const uint32_t b_offset = get_misalign_bytes(ctx, src1) / unit_size; - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / unit_size; - - p.misalign_offsets = (a_offset << 16) | (b_offset << 8) | d_offset; - - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_upscale_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - const uint32_t a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - const uint32_t d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - p.a_offset = a_offset; - p.d_offset = d_offset; - - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template <> void init_pushconst_tensor_offsets(ggml_backend_vk_context * ctx, vk_op_rope_push_constants &p, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst) { - p.a_offset = get_misalign_bytes(ctx, src0) / ggml_type_size(src0->type); - p.d_offset = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - - GGML_UNUSED(src1); - GGML_UNUSED(src2); - GGML_UNUSED(src3); -} - -template -static void ggml_vk_op_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst, ggml_op op, PC&& pc) { - VK_LOG_DEBUG("ggml_vk_op_f32((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; - if (src1 != nullptr) { - std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; - } - if (src2 != nullptr) { - std::cerr << "), (" << src2 << ", name=" << src2->name << ", type=" << src2->type << ", ne0=" << src2->ne[0] << ", ne1=" << src2->ne[1] << ", ne2=" << src2->ne[2] << ", ne3=" << src2->ne[3] << ", nb0=" << src2->nb[0] << ", nb1=" << src2->nb[1] << ", nb2=" << src2->nb[2] << ", nb3=" << src2->nb[3]; - } - if (src3 != nullptr) { - std::cerr << "), (" << src3 << ", name=" << src3->name << ", type=" << src3->type << ", ne0=" << src3->ne[0] << ", ne1=" << src3->ne[1] << ", ne2=" << src3->ne[2] << ", ne3=" << src3->ne[3] << ", nb0=" << src3->nb[0] << ", nb1=" << src3->nb[1] << ", nb2=" << src3->nb[2] << ", nb3=" << src3->nb[3]; - } - std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; - std::cerr << "), " << ggml_op_name(op) << ")"); - GGML_ASSERT(op == GGML_OP_GET_ROWS || op == GGML_OP_CPY || op == GGML_OP_CONCAT || (!ggml_is_quantized(src0->type) && (src1 == nullptr || !ggml_is_quantized(src1->type)))); // NOLINT - GGML_ASSERT(dst->buffer != nullptr); - const uint64_t ne00 = src0->ne[0]; - const uint64_t ne01 = src0->ne[1]; - const uint64_t ne02 = src0->ne[2]; - const uint64_t ne03 = src0->ne[3]; - - const bool use_src1 = src1 != nullptr; - const uint64_t ne10 = use_src1 ? src1->ne[0] : 0; - const uint64_t ne11 = use_src1 ? src1->ne[1] : 0; - const uint64_t ne12 = use_src1 ? src1->ne[2] : 0; - const uint64_t ne13 = use_src1 ? src1->ne[3] : 0; - - const bool use_src2 = src2 != nullptr; - const bool use_src3 = src3 != nullptr; - - init_pushconst_fastdiv(pc); - - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, src0, src1, src2, dst, op); - - if (pipeline == nullptr) { - std::cerr << "ggml_vulkan: Error: Missing op: " << ggml_op_name(op) << " for " << ggml_type_name(src0->type); - if (src1 != nullptr) { - std::cerr << " and " << ggml_type_name(src1->type); - } - std::cerr << " to " << ggml_type_name(dst->type) << std::endl; - GGML_ABORT("fatal error"); - } - - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - - vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0, true); - vk_subbuffer src1_buf = use_src1 ? ggml_vk_tensor_subbuffer(ctx, src1, true) : vk_subbuffer{}; - vk_subbuffer src2_buf = use_src2 ? ggml_vk_tensor_subbuffer(ctx, src2, true) : vk_subbuffer{}; - vk_subbuffer src3_buf = use_src3 ? ggml_vk_tensor_subbuffer(ctx, src3, true) : vk_subbuffer{}; - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); - - // Compute misalignment offset for descriptors and store it in in push constants. - init_pushconst_tensor_offsets(ctx, pc, src0, src1, src2, src3, dst); - - std::array elements; - - switch (op) { - case GGML_OP_NORM: - case GGML_OP_RMS_NORM_BACK: - case GGML_OP_L2_NORM: - case GGML_OP_SOFT_MAX: - case GGML_OP_SOFT_MAX_BACK: - case GGML_OP_SUM_ROWS: - case GGML_OP_CUMSUM: - case GGML_OP_MEAN: - case GGML_OP_ARGMAX: - { - const uint32_t nr = ggml_nrows(src0); - if (nr > 262144) { - elements = { 512, 512, CEIL_DIV(nr, 262144) }; - } else if (nr > 512) { - elements = { 512, CEIL_DIV(nr, 512), 1 }; - } else { - elements = { nr, 1, 1 }; - } - } break; - case GGML_OP_SOLVE_TRI: - { - uint32_t nr = (uint32_t)(ne02 * ne03); - if (nr > 262144) { - elements = { 512, 512, CEIL_DIV(nr, 262144) }; - } else if (nr > 512) { - elements = { 512, CEIL_DIV(nr, 512), 1 }; - } else { - elements = { nr, 1, 1 }; - } - } - break; - case GGML_OP_RMS_NORM: - if (ctx->do_add_rms_partials) { - // Run one element per thread, 128 threads per workgroup - elements = { (uint32_t)CEIL_DIV(ne00, 128), 1, 1 }; - } else { - elements = { (uint32_t)ne01, (uint32_t)ne02, (uint32_t)ne03 }; - } - break; - - case GGML_OP_SUM: - // We use GGML_OP_SUM_ROWS with 1 row. - elements = { 1, 1, 1 }; - break; - case GGML_OP_GROUP_NORM: - { - const uint32_t num_groups = dst->op_params[0]; - elements = { num_groups * (uint32_t)src0->ne[3], 1, 1 }; - } break; - case GGML_OP_DIAG_MASK_INF: - elements = { (uint32_t)ggml_nrows(src0), (uint32_t)ne00, 1 }; - break; - case GGML_OP_ROPE: - case GGML_OP_ROPE_BACK: - { - uint32_t nrows = (uint32_t)ggml_nrows(src0); - uint32_t z = 1; - if (nrows > ctx->device->properties.limits.maxComputeWorkGroupCount[0]) { - z = CEIL_DIV(nrows, 32768); - nrows = 32768; - } - elements = { nrows, (uint32_t)ne00, z }; - - } break; - case GGML_OP_GET_ROWS: - elements = { (uint32_t)ne00, (uint32_t)ne10, (uint32_t)(ne11 * ne12) }; - elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - break; - case GGML_OP_GET_ROWS_BACK: - elements = { (uint32_t)dst->ne[0], (uint32_t)dst->ne[1], 1 }; - elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - break; - case GGML_OP_ARGSORT: - GGML_ASSERT(0); - break; - case GGML_OP_IM2COL: - { - const bool is_2D = dst->op_params[6] == 1; - - const uint32_t IC = src1->ne[is_2D ? 2 : 1]; - - const uint32_t KH = is_2D ? src0->ne[1] : 1; - const uint32_t KW = src0->ne[0]; - - const uint32_t OH = is_2D ? dst->ne[2] : 1; - const uint32_t OW = dst->ne[1]; - - const uint32_t batch = src1->ne[is_2D ? 3 : 2]; - - const uint32_t CHW = IC * KH * KW; - // Cap X workgroups to limit concurrent IC channel reads. - // The shader loops over X to cover the full CHW dimension. - // AMD prefers a lower limit - const uint32_t min_cap = ctx->device->vendor_id == VK_VENDOR_ID_AMD ? 512u : 4096u; - const uint32_t x_elements = std::min(CHW, std::max(min_cap, OW * KH * KW)); - elements = { x_elements, OW, OH * batch }; - elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - } break; - case GGML_OP_IM2COL_3D: - { - const uint32_t IC = ((const uint32_t *)(dst->op_params))[9]; - - const uint32_t N = ne13 / IC; - - const uint32_t KD = ne02; - const uint32_t KH = ne01; - const uint32_t KW = ne00; - - const uint32_t OD = dst->ne[3] / N; - const uint32_t OH = dst->ne[2]; - const uint32_t OW = dst->ne[1]; - - const uint32_t IC_KD_KH_KW = IC*KD*KH*KW; - const uint32_t N_OD_OH = N*OD*OH; - - elements = { IC_KD_KH_KW, OW, N_OD_OH }; - elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - } break; - case GGML_OP_TIMESTEP_EMBEDDING: - { - const uint32_t dim = dst->op_params[0]; - uint32_t half_ceil = (dim + 1) / 2; - elements = { half_ceil, (uint32_t)src0->ne[0], 1 }; - } break; - case GGML_OP_CONV_TRANSPOSE_1D: - { - elements = {uint32_t(src0->ne[1]), 1, 1}; // parallelize in {Cout, 1, 1} - } break; - case GGML_OP_COL2IM_1D: + case GGML_OP_SUB: { - elements = { uint32_t(dst->ne[0]), uint32_t(dst->ne[1]), 1 }; - } break; - case GGML_OP_POOL_1D: + auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_sub_norepeat : ctx->device->pipeline_sub; + return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; + } + case GGML_OP_MUL: { - const uint32_t N = dst->ne[3] * dst->ne[2]; - const uint32_t OC = dst->ne[1]; - const uint32_t OL = dst->ne[0]; - elements = { N * OC * OL, 1, 1}; - } break; - case GGML_OP_POOL_2D: + auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_mul_norepeat : ctx->device->pipeline_mul; + return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; + } + case GGML_OP_DIV: { - const uint32_t N = dst->ne[3]; - const uint32_t OC = dst->ne[2]; - const uint32_t OH = dst->ne[1]; - const uint32_t OW = dst->ne[0]; - elements = { N * OC * OH * OW, 1, 1}; - } break; - case GGML_OP_CONV_2D: - case GGML_OP_CONV_TRANSPOSE_2D: - if constexpr (std::is_same_v) { - const uint32_t NPQ = pc.N * pc.OH * pc.OW; - const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, pc.Cout, NPQ); - const uint32_t NPQ_blocks = CEIL_DIV(NPQ, vk_conv_block_sizes[shape].NPQ); - - elements = { pc.Cout, NPQ_blocks, 1 }; - if (elements[1] > 512) { - elements[2] = CEIL_DIV(elements[1], 512); - elements[1] = 512; - } - } else { - GGML_ABORT("invalid push constant type for CONV_2D"); + auto pipelines = ggml_are_same_shape(src0, src1) ? ctx->device->pipeline_div_norepeat : ctx->device->pipeline_div; + return pipelines[src0->type == GGML_TYPE_F16][src1->type == GGML_TYPE_F16][dst->type == GGML_TYPE_F16]; } - break; - case GGML_OP_CONV_3D: - if constexpr (std::is_same_v) { - const uint32_t NPQ = pc.N * pc.OD * pc.OH * pc.OW; - const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, pc.OC, NPQ); - const uint32_t NPQ_blocks = CEIL_DIV(NPQ, vk_conv_block_sizes[shape].NPQ); - - elements = { pc.OC, NPQ_blocks, 1 }; - if (elements[1] > 512) { - elements[2] = CEIL_DIV(elements[1], 512); - elements[1] = 512; - } - } else { - GGML_ABORT("invalid push constant type for CONV_3D"); + default: + break; } - break; - case GGML_OP_ADD: - case GGML_OP_SUB: - case GGML_OP_DIV: - case GGML_OP_MUL: - case GGML_OP_ADD1: - case GGML_OP_OUT_PROD: - case GGML_OP_ARANGE: - case GGML_OP_FILL: - case GGML_OP_SCALE: - case GGML_OP_SQR: - case GGML_OP_SQRT: - case GGML_OP_SIN: - case GGML_OP_COS: - case GGML_OP_LOG: - case GGML_OP_TRI: - case GGML_OP_DIAG: - case GGML_OP_CLAMP: - case GGML_OP_LEAKY_RELU: - case GGML_OP_PAD: - case GGML_OP_ROLL: - case GGML_OP_REPEAT: - case GGML_OP_REPEAT_BACK: - case GGML_OP_CPY: - case GGML_OP_CONCAT: - case GGML_OP_UPSCALE: - case GGML_OP_UNARY: - case GGML_OP_GLU: - case GGML_OP_CONV_2D_DW: - { - uint32_t ne = ggml_nelements(dst); - if (op == GGML_OP_CPY && ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { - // Convert from number of logical elements to 2- or 4-byte units. - ne /= ggml_blck_size(src0->type); - if ((ggml_type_size(src0->type) % 4) == 0) { - ne *= ggml_type_size(src0->type) / 4; - } else { - ne *= ggml_type_size(src0->type) / 2; - } - } - if (op == GGML_OP_CONCAT && ggml_is_quantized(dst->type)) { - ne = ne / ggml_blck_size(dst->type) * ggml_type_size(dst->type) / ggml_vk_concat_unit_size(dst->type); - } - // copy_to_quant has block size of 32, and each thread does QUANT_K elements. - // Splitting into 512x512xZ wouldn't work well since each workgroup does 1024 elements. - // So divide by block size here before splitting into 512x512 groups. - if (op == GGML_OP_CPY && !ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { - ne = CEIL_DIV(ne, ggml_blck_size(dst->type)); - } - if (ne > 262144) { - elements = { 512, 512, CEIL_DIV(ne, 262144) }; - } else if (ne > 512) { - elements = { 512, CEIL_DIV(ne, 512), 1 }; - } else { - elements = { ne, 1, 1 }; - } - - if (pipeline == ctx->device->pipeline_cpy_transpose_32 || - pipeline == ctx->device->pipeline_cpy_transpose_16) { - // 32x32 tiles - elements[0] = (uint32_t)CEIL_DIV(dst->ne[0], 32); - elements[1] = (uint32_t)CEIL_DIV(dst->ne[1], 32); - elements[2] = (uint32_t)(dst->ne[2]*dst->ne[3]); - elements[0] = std::min(elements[0], ctx->device->properties.limits.maxComputeWorkGroupCount[0]); - elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); - } - } break; + return nullptr; case GGML_OP_ADD_ID: - { - elements = { (uint32_t)ne01, (uint32_t)ne02, 1 }; - } break; - case GGML_OP_SET_ROWS: - { - uint32_t ne = ggml_nelements(src0); - if (ggml_is_quantized(dst->type)) { - // quants run 32 threads each doing QUANT_K elements - ne = CEIL_DIV(ne, 32 * ggml_blck_size(dst->type)); - } else { - // scalar types do one element per thread, running 512 threads - ne = CEIL_DIV(ne, 512); - } - if (ne > 262144) { - elements = { 512, 512, CEIL_DIV(ne, 262144) }; - } else if (ne > 512) { - elements = { 512, CEIL_DIV(ne, 512), 1 }; - } else { - elements = { ne, 1, 1 }; - } + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && src2->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_add_id_f32; + } + return nullptr; + case GGML_OP_OUT_PROD: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_out_prod_f32; + } + return nullptr; + case GGML_OP_CONCAT: { + if (!ggml_vk_concat_supported(src0, src1, dst)) { + return nullptr; } - break; - case GGML_OP_SSM_CONV: - { - const uint32_t nr = src0->ne[1]; - const uint32_t n_t = dst->ne[1]; - const uint32_t n_s = dst->ne[2]; - elements = { nr, n_t, n_s }; + switch (ggml_vk_concat_unit_size(src0->type)) { + case 1: + return ctx->device->pipeline_concat_i8; + case 2: + return ctx->device->pipeline_concat_i16; + case 4: + return ctx->device->pipeline_concat_i32; + case 8: + return ctx->device->pipeline_concat_i64; + default: + return nullptr; } - break; - default: - elements = { (uint32_t)ggml_nelements(src0), 1, 1 }; - break; } - - if (op == GGML_OP_ADD || op == GGML_OP_RMS_NORM) { - vk_subbuffer a_buf = src0_buf; - if (ctx->do_add_rms_partials) { - a_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_add_rms_partials, ctx->prealloc_size_add_rms_partials_offset); + case GGML_OP_UPSCALE: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + uint32_t mode = (ggml_get_op_params_i32(dst, 0) & (0xFF | GGML_SCALE_FLAG_ANTIALIAS)); + switch (mode) { + case GGML_SCALE_MODE_NEAREST: + return ctx->device->pipeline_upscale_nearest_f32; + case GGML_SCALE_MODE_BILINEAR: + return ctx->device->pipeline_upscale_bilinear_f32; + case GGML_SCALE_MODE_BICUBIC: + return ctx->device->pipeline_upscale_bicubic_f32; + case GGML_SCALE_MODE_BILINEAR | GGML_SCALE_FLAG_ANTIALIAS: + return ctx->device->pipeline_upscale_bilinear_antialias_f32; + default: + return nullptr; + } } - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - { src0_buf, src1_buf, dst_buf, a_buf }, pc, elements); - } else if (op == GGML_OP_GLU) { - // Empty src1 is possible in glu, but the shader needs a buffer - vk_subbuffer subbuf1 = use_src1 ? src1_buf : src0_buf; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc, elements); - } else if (op == GGML_OP_SOFT_MAX) { - // Empty src1 and src2 is possible in soft_max, but the shader needs a buffer - vk_subbuffer subbuf1 = use_src1 ? src1_buf : src0_buf; - vk_subbuffer subbuf2 = use_src2 ? src2_buf : src0_buf; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, subbuf2, dst_buf }, pc, elements); - } else if (op == GGML_OP_ROPE || op == GGML_OP_ROPE_BACK) { - // Empty src2 and src3 is possible in rope, but the shader needs a buffer - vk_subbuffer subbuf2 = use_src2 ? src2_buf : src0_buf; - vk_subbuffer subbuf3 = use_src3 ? src3_buf : src0_buf; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, subbuf2, dst_buf, subbuf3 }, pc, elements); - } else if (op == GGML_OP_IM2COL || op == GGML_OP_IM2COL_3D) { - if (ctx->device->shader_int64 && ctx->device->buffer_device_address) { - // buffer device address path doesn't use dst buffer - dst_buf.size = 1; + return nullptr; + case GGML_OP_SCALE: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_scale_f32; } - // im2col uses only src1 and dst buffers - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src1_buf, dst_buf }, pc, elements); - } else if (op == GGML_OP_COUNT_EQUAL) { - // count_equal assumes that destination buffer is initialized with zeroes - ggml_vk_buffer_memset_async(subctx, dst_buf.buffer, dst_buf.offset, 0, dst_buf.size); - ggml_vk_sync_buffers(ctx, subctx); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, dst_buf }, pc, elements); - } else if (op == GGML_OP_OPT_STEP_SGD) { - // OPT_STEP_SGD works on src0, it does not need dst - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf }, pc, elements); - } else if (use_src3) { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf, src3_buf, dst_buf }, pc, elements); - } else if (use_src2) { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf, dst_buf }, pc, elements); - } else if (use_src1) { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, dst_buf }, pc, elements); - } else { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, dst_buf }, pc, elements); - } -} - -static void ggml_vk_get_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GET_ROWS, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_get_rows_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GET_ROWS_BACK, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], (uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2], (uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2], (uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_acc(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - int nb1 = dst->op_params[0] / src0_type_size; // 4 bytes of float32 - int nb2 = dst->op_params[1] / src0_type_size; // 4 bytes of float32 - int nb3 = dst->op_params[2] / src0_type_size; // 4 bytes of float32 - int offset = dst->op_params[3] / src0_type_size; // offset in bytes - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, dst->op, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)nb1, (uint32_t)nb2, (uint32_t)nb3, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t)nb1, (uint32_t)nb2, (uint32_t)nb3, - 0, - 0.0f, 0.0f, offset, - }); -} - -static void ggml_vk_multi_add(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { - const ggml_tensor *first_node = cgraph->nodes[node_idx]; - const ggml_tensor *dst = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; - - // Make a list of all the tensors used by the op. - // Last element of the list is the dest tensor. - const ggml_tensor *tensors[MAX_PARAMETER_COUNT]; - uint32_t num_srcs = ctx->num_additional_fused_ops + 2; - uint32_t num_tensors = num_srcs + 1; - GGML_ASSERT(num_tensors + ctx->do_add_rms_partials <= MAX_PARAMETER_COUNT); - - tensors[0] = first_node->src[0]; - tensors[1] = first_node->src[1]; - for (int32_t i = 0; i < ctx->num_additional_fused_ops; ++i) { - // check whether the previous result is src[0] or src[1] - if (cgraph->nodes[node_idx + i] == cgraph->nodes[node_idx + i + 1]->src[0]) { - tensors[i+2] = cgraph->nodes[node_idx + i + 1]->src[1]; - } else { - tensors[i+2] = cgraph->nodes[node_idx + i + 1]->src[0]; + return nullptr; + case GGML_OP_SQR: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_sqr[dst->type == GGML_TYPE_F16]; } - } - tensors[num_srcs] = dst; - - vk_op_multi_add_push_constants pc; - pc.ne20 = (uint32_t)dst->ne[0]; - pc.ne21 = (uint32_t)dst->ne[1]; - pc.ne22 = (uint32_t)dst->ne[2]; - pc.ne23 = (uint32_t)dst->ne[3]; - - for (uint32_t i = 0; i < num_tensors; ++i) { - const ggml_tensor *t = tensors[i]; - pc.nb[i][0] = (uint32_t)t->nb[0] / sizeof(float); - pc.nb[i][1] = (uint32_t)t->nb[1] / sizeof(float); - pc.nb[i][2] = (uint32_t)t->nb[2] / sizeof(float); - pc.nb[i][3] = (uint32_t)t->nb[3] / sizeof(float); - } - pc.rms_partials = ctx->do_add_rms_partials; - - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, tensors[0], tensors[1], nullptr, dst, dst->op); - - if (pipeline == nullptr) { - std::cerr << "ggml_vulkan: Error: Missing multi_add"; - GGML_ABORT("fatal error"); - } - - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - - ggml_backend_vk_buffer_context * buf_ctx[MAX_PARAMETER_COUNT]; - vk_buffer buf[MAX_PARAMETER_COUNT]; - size_t offset[MAX_PARAMETER_COUNT]; - bool uma[MAX_PARAMETER_COUNT]; - - for (uint32_t i = 0; i < num_tensors; ++i) { - buf_ctx[i] = (ggml_backend_vk_buffer_context *)tensors[i]->buffer->context; - buf[i] = nullptr; - offset[i] = 0; - uma[i] = false; - - if (ctx->device->uma) { - ggml_vk_host_get(ctx->device, tensors[i]->data, buf[i], offset[i]); - uma[i] = buf[i] != nullptr; + return nullptr; + case GGML_OP_SQRT: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_sqrt[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_SIN: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_sin[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_COS: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_cos[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_LOG: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_log[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_TRI: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_tri[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_DIAG: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_diag[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_CLAMP: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_clamp[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_PAD: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_pad_f32; + } + return nullptr; + case GGML_OP_PAD_REFLECT_1D: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_pad_reflect_1d_f32; + } + return nullptr; + case GGML_OP_ROLL: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_roll_f32; + } + return nullptr; + case GGML_OP_REPEAT: + if (ggml_type_size(src0->type) == sizeof(float) && ggml_type_size(dst->type) == sizeof(float)) { + return ctx->device->pipeline_repeat_i32; } - if (!uma[i]) { - buf[i] = buf_ctx[i]->dev_buffer; - offset[i] = vk_tensor_offset(tensors[i]) + tensors[i]->view_offs; + if (ggml_type_size(src0->type) == 2 && ggml_type_size(dst->type) == 2) { + return ctx->device->pipeline_repeat_i16; } - GGML_ASSERT(buf[i] != nullptr); - } - // If any remaining descriptors are unused, just point them at src[0] - for (uint32_t i = num_tensors; i < MAX_PARAMETER_COUNT; ++i) { - buf[i] = buf[0]; - offset[i] = 0; - } - if (ctx->do_add_rms_partials) { - buf[num_tensors] = ctx->prealloc_add_rms_partials; - offset[num_tensors] = ctx->prealloc_size_add_rms_partials_offset; - } - - std::array elements; - - uint32_t ne = ggml_nelements(dst); - if (ne > 262144) { - elements = { 512, 512, CEIL_DIV(ne, 262144) }; - } else if (ne > 512) { - elements = { 512, CEIL_DIV(ne, 512), 1 }; - } else { - elements = { ne, 1, 1 }; - } - - static_assert(MAX_PARAMETER_COUNT == 12); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + return nullptr; + case GGML_OP_REPEAT_BACK: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_repeat_back_f32; + } + return nullptr; + case GGML_OP_CPY: + case GGML_OP_CONT: + case GGML_OP_DUP: + return ggml_vk_get_cpy_pipeline(ctx, src0, dst, dst->type); + case GGML_OP_SET_ROWS: { - ggml_vk_subbuffer(ctx, buf[0], offset[0]), - ggml_vk_subbuffer(ctx, buf[1], offset[1]), - ggml_vk_subbuffer(ctx, buf[2], offset[2]), - ggml_vk_subbuffer(ctx, buf[3], offset[3]), - ggml_vk_subbuffer(ctx, buf[4], offset[4]), - ggml_vk_subbuffer(ctx, buf[5], offset[5]), - ggml_vk_subbuffer(ctx, buf[6], offset[6]), - ggml_vk_subbuffer(ctx, buf[7], offset[7]), - ggml_vk_subbuffer(ctx, buf[8], offset[8]), - ggml_vk_subbuffer(ctx, buf[9], offset[9]), - ggml_vk_subbuffer(ctx, buf[10], offset[10]), - ggml_vk_subbuffer(ctx, buf[11], offset[11]), - }, pc, elements); -} - -static void ggml_vk_add(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_ADD, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, ctx->do_add_rms_partials, - }); -} - -static void ggml_vk_out_prod(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_OUT_PROD, { - (uint32_t)ggml_nelements(dst), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], - (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], - (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], - (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_sub(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SUB, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_mul(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_MUL, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_div(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_DIV, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); -} - -static void ggml_vk_add_id(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t src2_type_size = ggml_type_size(src2->type); - - ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_ADD_ID, { - (uint32_t)dst->ne[0], - (uint32_t)dst->ne[1], - (uint32_t)src0->nb[1] / src0_type_size, - (uint32_t)src0->nb[2] / src0_type_size, - (uint32_t)src1->nb[1] / src1_type_size, - (uint32_t)src2->nb[1] / src2_type_size, - }); -} - -static void ggml_vk_op_f32_wkv(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst, const vk_op_rwkv_wkv6_push_constants&& pc, int version) { - GGML_ASSERT(version == 6 || version == 7); - int num_srcs = version == 6 ? 6 : 7; - - for (int i = 0; i < num_srcs; i++) { - GGML_ASSERT(!ggml_is_quantized(dst->src[i]->type)); - } - - GGML_ASSERT(dst->buffer != nullptr); - - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); - GGML_ASSERT(pipeline != nullptr); - - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer src_buf[7] = {}; - for (int i = 0; i < num_srcs; i++) { - src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); - } - - std::array elements = { - (uint32_t)(pc.B * pc.H), - 1, - 1 - }; - - if (version == 6) { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], dst_buf}, - pc, elements); - } else if (version == 7) { - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], src_buf[6], dst_buf}, - pc, elements); - } else { - // shouldn't happen - GGML_ASSERT(false); - } -} + if (src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) { + return nullptr; + } + const int src_idx = src0->type == GGML_TYPE_F16; + if (src1->type == GGML_TYPE_I64) { + return ctx->device->pipeline_set_rows_i64[src_idx][dst->type]; + } else if (src1->type == GGML_TYPE_I32) { + return ctx->device->pipeline_set_rows_i32[src_idx][dst->type]; + } + return nullptr; + } + case GGML_OP_SILU_BACK: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_silu_back_f32; + } + return nullptr; + case GGML_OP_NORM: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_norm_f32; + } + return nullptr; + case GGML_OP_GROUP_NORM: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_group_norm_f32; + } + return nullptr; + case GGML_OP_RMS_NORM: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + if (ctx->do_add_rms_partials) { + return ctx->fused_rms_norm_mode == RMS_NORM_MUL ? ctx->device->pipeline_rms_norm_mul_partials_f32 : ctx->device->pipeline_rms_norm_partials_f32; + } + return ctx->fused_rms_norm_mode == RMS_NORM_MUL ? ctx->device->pipeline_rms_norm_mul_f32 : ctx->device->pipeline_rms_norm_f32; + } + return nullptr; + case GGML_OP_RMS_NORM_BACK: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rms_norm_back_f32; + } + return nullptr; + case GGML_OP_L2_NORM: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_l2_norm_f32; + } + return nullptr; + case GGML_OP_UNARY: + if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || + (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16) || + (src0->type != dst->type)) { + return nullptr; + } -static void ggml_vk_rwkv_wkv6(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const size_t seq_length = dst->src[0]->ne[2]; - const size_t n_embed = dst->ne[0]; - const size_t n_heads = dst->src[0]->ne[1]; - const size_t n_seqs = dst->src[5]->ne[1]; + switch (ggml_get_unary_op(dst)) { + case GGML_UNARY_OP_EXP: + return ctx->device->pipeline_exp[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_EXPM1: + return ctx->device->pipeline_expm1[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_ELU: + return ctx->device->pipeline_elu[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_SILU: + return ctx->device->pipeline_silu[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_GELU: + return ctx->device->pipeline_gelu[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_GELU_ERF: + return ctx->device->pipeline_gelu_erf[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_GELU_QUICK: + return ctx->device->pipeline_gelu_quick[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_RELU: + return ctx->device->pipeline_relu[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_XIELU: + return ctx->device->pipeline_xielu[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_NEG: + return ctx->device->pipeline_neg[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_TANH: + return ctx->device->pipeline_tanh[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_SIGMOID: + return ctx->device->pipeline_sigmoid[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_HARDSIGMOID: + return ctx->device->pipeline_hardsigmoid[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_HARDSWISH: + return ctx->device->pipeline_hardswish[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_ABS: + return ctx->device->pipeline_abs[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_SOFTPLUS: + return ctx->device->pipeline_softplus[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_STEP: + return ctx->device->pipeline_step[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_ROUND: + return ctx->device->pipeline_round[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_CEIL: + return ctx->device->pipeline_ceil[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_FLOOR: + return ctx->device->pipeline_floor[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_TRUNC: + return ctx->device->pipeline_trunc[dst->type == GGML_TYPE_F16]; + case GGML_UNARY_OP_SGN: + return ctx->device->pipeline_sgn[dst->type == GGML_TYPE_F16]; + default: + break; + } + return nullptr; + case GGML_OP_GLU: + if ((src0->type != GGML_TYPE_F32 && src0->type != GGML_TYPE_F16) || + (dst->type != GGML_TYPE_F32 && dst->type != GGML_TYPE_F16) || + (src0->type != dst->type)) { + return nullptr; + } - ggml_vk_op_f32_wkv( - ctx, subctx, dst, - { - (uint32_t)n_seqs, - (uint32_t)seq_length, - (uint32_t)n_embed, - (uint32_t)n_heads, - }, - 6 - ); -} + switch (ggml_get_glu_op(dst)) { + case GGML_GLU_OP_GEGLU: + return ctx->device->pipeline_geglu[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_REGLU: + return ctx->device->pipeline_reglu[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_SWIGLU: + return ctx->device->pipeline_swiglu[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_SWIGLU_OAI: + return ctx->device->pipeline_swiglu_oai[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_SWIGLU_CLAMP: + return ctx->device->pipeline_swiglu_clamp[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_GEGLU_ERF: + return ctx->device->pipeline_geglu_erf[dst->type == GGML_TYPE_F16]; + case GGML_GLU_OP_GEGLU_QUICK: + return ctx->device->pipeline_geglu_quick[dst->type == GGML_TYPE_F16]; + default: + break; + } + return nullptr; + case GGML_OP_DIAG_MASK_INF: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_diag_mask_inf_f32; + } + return nullptr; + case GGML_OP_SOFT_MAX: + GGML_ASSERT(!src1 || src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16); + GGML_ASSERT(!src2 || src2->type == GGML_TYPE_F32); -static void ggml_vk_rwkv_wkv7(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const size_t seq_length = dst->src[0]->ne[2]; - const size_t n_embed = dst->ne[0]; - const size_t n_heads = dst->src[0]->ne[1]; - const size_t n_seqs = dst->src[6]->ne[1]; + if (ctx->num_additional_fused_ops) { + uint32_t idx = (uint32_t)ceilf(log2f(float(dst->ne[0]))); + GGML_ASSERT(idx < num_topk_moe_pipelines); + // use n_experts from push constant if it's not equal to the power of two spec constant + bool use_push = dst->ne[0] != (1u << idx); + return ctx->device->pipeline_topk_moe[idx][use_push]; + } - ggml_vk_op_f32_wkv( - ctx, subctx, dst, + if (src0->type == GGML_TYPE_F32 && (src1 == nullptr || src1->type == GGML_TYPE_F32) && dst->type == GGML_TYPE_F32) { + return src0->ne[0] > 1024 ? ctx->device->pipeline_soft_max_f32_wg512 : ctx->device->pipeline_soft_max_f32; + } + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F32) { + return src0->ne[0] > 1024 ? ctx->device->pipeline_soft_max_f32_f16_wg512 : ctx->device->pipeline_soft_max_f32_f16; + } + return nullptr; + case GGML_OP_SOFT_MAX_BACK: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_soft_max_back_f32; + } + return nullptr; + case GGML_OP_ROPE: + case GGML_OP_ROPE_BACK: { - (uint32_t)n_seqs, - (uint32_t)seq_length, - (uint32_t)n_embed, - (uint32_t)n_heads, - }, - 7 - ); -} + const ggml_tensor *rope = ctx->num_additional_fused_ops == 2 ? dst->src[0]->src[0] : dst; + const int mode = ((const int32_t *) rope->op_params)[2]; + const bool is_neox = mode & GGML_ROPE_TYPE_NEOX; + const bool is_mrope = mode & GGML_ROPE_TYPE_MROPE; + const bool is_vision = mode == GGML_ROPE_TYPE_VISION; -static void ggml_vk_gated_linear_attn(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const size_t seq_length = dst->src[0]->ne[2]; - const size_t n_embed = dst->ne[0]; - const size_t n_heads = dst->src[0]->ne[1]; - const size_t n_seqs = dst->src[4]->ne[1]; + if (is_neox) { + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rope_neox_f32; + } + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_neox_f32_f16; + } + if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_neox_f16; + } + } else if (is_mrope && !is_vision) { + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rope_multi_f32; + } + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_multi_f32_f16; + } + if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_multi_f16; + } + } else if (is_vision) { + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rope_vision_f32; + } + if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_vision_f16; + } + } else { + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rope_norm_f32; + } + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_norm_f32_f16; + } + if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_rope_norm_f16; + } + } + return nullptr; + } + case GGML_OP_SUM: + case GGML_OP_SUM_ROWS: + case GGML_OP_MEAN: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_sum_rows_f32; + } + return nullptr; + case GGML_OP_CROSS_ENTROPY_LOSS: + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return src0->ne[0] > 1024 ? ctx->device->pipeline_cross_entropy_loss_f32_wg512 : ctx->device->pipeline_cross_entropy_loss_f32; + } + return nullptr; + case GGML_OP_CROSS_ENTROPY_LOSS_BACK: + // src0 is the scalar grad; src1 is logits + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && src2 && src2->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return src1->ne[0] > 1024 ? ctx->device->pipeline_cross_entropy_loss_back_f32_wg512 : ctx->device->pipeline_cross_entropy_loss_back_f32; + } + return nullptr; + case GGML_OP_CUMSUM: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + if (src0->ne[0] <= 512) { + return ctx->device->pipeline_cumsum_small_f32; + } else { + return ctx->device->pipeline_cumsum_f32; + } + } + return nullptr; + case GGML_OP_SOLVE_TRI: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { - float scale; - memcpy(&scale, dst->op_params, sizeof(float)); + vk_solve_tri_pipeline_state solve_tri_pipeline_state(src0->ne[0], src1->ne[0]); - GGML_ASSERT(dst->buffer != nullptr); + vk_pipeline pipeline = nullptr; - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); - GGML_ASSERT(pipeline != nullptr); + { + std::lock_guard guard(ctx->device->compile_mutex); + auto it = ctx->device->pipeline_solve_tri_f32.find(solve_tri_pipeline_state); + if (it != ctx->device->pipeline_solve_tri_f32.end()) { + pipeline = it->second; + } else { + ctx->device->pipeline_solve_tri_f32[solve_tri_pipeline_state] = pipeline = std::make_shared(); + } + } - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + return pipeline; + } + return nullptr; + case GGML_OP_ARGMAX: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_I32) { + return ctx->device->pipeline_argmax_f32; + } + return nullptr; + case GGML_OP_COUNT_EQUAL: + if (src0->type == GGML_TYPE_I32 && src1->type == GGML_TYPE_I32 && dst->type == GGML_TYPE_I64) { + return ctx->device->pipeline_count_equal_i32; + } + return nullptr; + case GGML_OP_IM2COL: + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_im2col_f32; + } + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_im2col_f32_f16; + } + return nullptr; + case GGML_OP_IM2COL_3D: + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_im2col_3d_f32; + } + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_im2col_3d_f32_f16; + } + return nullptr; + case GGML_OP_TIMESTEP_EMBEDDING: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_timestep_embedding_f32; + } + return nullptr; + case GGML_OP_CONV_TRANSPOSE_1D: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_conv_transpose_1d_f32; + } + return nullptr; + case GGML_OP_COL2IM_1D: + switch (src0->type) { + case GGML_TYPE_F32: return ctx->device->pipeline_col2im_1d_f32; + case GGML_TYPE_F16: return ctx->device->pipeline_col2im_1d_f16; + case GGML_TYPE_BF16: return ctx->device->pipeline_col2im_1d_bf16; + default: return nullptr; + } + case GGML_OP_POOL_1D: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_pool1d_f32; + } + return nullptr; + case GGML_OP_POOL_2D: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_pool2d_f32; + } + return nullptr; + case GGML_OP_RWKV_WKV6: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rwkv_wkv6_f32; + } + return nullptr; + case GGML_OP_RWKV_WKV7: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_rwkv_wkv7_f32; + } + return nullptr; + case GGML_OP_GATED_LINEAR_ATTN: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_gated_linear_attn_f32; + } + return nullptr; + case GGML_OP_LIGHTNING_INDEXER: + // only the k type selects a pipeline, the other types are fixed by ggml_lightning_indexer() + if (ggml_vk_lightning_indexer_k_type_supported(src1->type)) { + return ctx->device->pipeline_lightning_indexer_f32[src1->type]; + } + return nullptr; + case GGML_OP_GATED_DELTA_NET: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + const uint32_t S_v = dst->src[2]->ne[0]; + const uint32_t kda = (dst->src[3]->ne[0] == (int64_t)S_v) ? 1 : 0; + uint32_t si; + switch (S_v) { + case 16: si = 0; break; + case 32: si = 1; break; + case 64: si = 2; break; + case 128: si = 3; break; + default: return nullptr; + } + return ctx->device->pipeline_gated_delta_net[si][kda]; + } + return nullptr; + case GGML_OP_SSM_SCAN: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + const uint32_t d_state = src0->ne[0]; + if (d_state == 128) { + return ctx->device->pipeline_ssm_scan_f32_d128; + } else if (d_state == 256) { + return ctx->device->pipeline_ssm_scan_f32_d256; + } + } + return nullptr; + case GGML_OP_SSM_CONV: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + switch (ctx->num_additional_fused_ops) { + case 0: return ctx->device->pipeline_ssm_conv_f32; + case 1: return ctx->device->pipeline_ssm_conv_silu_f32; + case 2: return ctx->device->pipeline_ssm_conv_bias_silu_f32; + default: return nullptr; + } + } + return nullptr; + case GGML_OP_OPT_STEP_ADAMW: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_opt_step_adamw_f32; + } + return nullptr; + case GGML_OP_OPT_STEP_SGD: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_opt_step_sgd_f32; + } + return nullptr; + case GGML_OP_LEAKY_RELU: + if (src0->type == dst->type && + (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16)) { + return ctx->device->pipeline_leaky_relu[dst->type == GGML_TYPE_F16]; + } + return nullptr; + case GGML_OP_CONV_2D: + case GGML_OP_CONV_TRANSPOSE_2D: + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + uint32_t K = dst->ne[2]; // Cout + uint32_t NPQ = dst->ne[3] * dst->ne[1] * dst->ne[0]; // N * OH * OW + vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, K, NPQ); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer src_buf[5] = {}; - for (int i = 0; i < 5; i++) { - src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); - } + bool transpose = dst->op == GGML_OP_CONV_TRANSPOSE_2D; + uint32_t KW = (uint32_t)src0->ne[0]; + uint32_t KH = (uint32_t)src0->ne[1]; + uint32_t s0 = (uint32_t)(ggml_get_op_params_i32(dst, 0)); + uint32_t s1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 1) : s0; + uint32_t p0 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 2) : 0; + uint32_t p1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 3) : 0; + uint32_t d0 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 4) : 1; + uint32_t d1 = !transpose ? (uint32_t)ggml_get_op_params_i32(dst, 5) : 1; - const vk_op_gated_linear_attn_push_constants pc = { - (uint32_t)n_seqs, - (uint32_t)seq_length, - (uint32_t)n_embed, - (uint32_t)n_heads, - scale, - }; + // tile-aligned shapes let the shader skip bounds checks + const uint32_t Cin = (uint32_t)src1->ne[2]; + const uint32_t CRS = Cin * KW * KH; + const uint32_t BS_K = vk_conv_block_sizes[shape].K; + const uint32_t BS_CRS = vk_conv_block_sizes[shape].CRS; + const uint32_t BS_NPQ = vk_conv_block_sizes[shape].NPQ; + const uint32_t aligned = ((K % BS_K == 0) && + (CRS % BS_CRS == 0) && + (NPQ % BS_NPQ == 0)) ? 1u : 0u; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], dst_buf}, - pc, { (uint32_t)(n_seqs * n_heads), 1, 1 }); -} + vk_conv2d_pipeline_state conv2d_pipeline_state(s0, s1, p0, p1, d0, d1, KW, KH, aligned); -static void ggml_vk_gated_delta_net(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const ggml_tensor * src_q = dst->src[0]; - const ggml_tensor * src_v = dst->src[2]; - const ggml_tensor * src_beta = dst->src[4]; + std::map *pipelines = nullptr; + if (op == GGML_OP_CONV_2D) { + if (src0->type == GGML_TYPE_F32) { + pipelines = &ctx->device->pipeline_conv2d_f32[shape]; + } else if (src0->type == GGML_TYPE_F16) { + pipelines = &ctx->device->pipeline_conv2d_f16_f32[shape]; + } + } else if (op == GGML_OP_CONV_TRANSPOSE_2D) { + if (src0->type == GGML_TYPE_F32) { + pipelines = &ctx->device->pipeline_conv_transpose_2d_f32[shape]; + } else if (src0->type == GGML_TYPE_F16) { + pipelines = &ctx->device->pipeline_conv_transpose_2d_f16_f32[shape]; + } + } - GGML_ASSERT(dst->buffer != nullptr); + vk_pipeline pipeline = nullptr; - const uint32_t S_v = (uint32_t)src_v->ne[0]; - const uint32_t H = (uint32_t)src_v->ne[1]; - const uint32_t n_tokens = (uint32_t)src_v->ne[2]; - const uint32_t n_seqs = (uint32_t)src_v->ne[3]; + { + std::lock_guard guard(ctx->device->compile_mutex); + auto it = pipelines->find(conv2d_pipeline_state); + if (it != pipelines->end()) { + pipeline = it->second; + } else { + (*pipelines)[conv2d_pipeline_state] = pipeline = std::make_shared(); + } + } - // K (snapshot slot count) is an op param; state holds s0 only [S_v, S_v, H, n_seqs]. - const uint32_t K = (uint32_t)ggml_get_op_params_i32(dst, 0); + return pipeline; + } + return nullptr; + case GGML_OP_CONV_2D_DW: + if (src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + if (ggml_is_contiguous(src1)) { + return ctx->device->pipeline_conv2d_dw_whcn_f32; + } else if (ggml_is_contiguous_channels(src1)) { + return ctx->device->pipeline_conv2d_dw_cwhn_f32; + } + } else if (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F32) { + if (ggml_is_contiguous(src1)) { + return ctx->device->pipeline_conv2d_dw_whcn_f16_f32; + } else if (ggml_is_contiguous_channels(src1)) { + return ctx->device->pipeline_conv2d_dw_cwhn_f16_f32; + } + } + return nullptr; + case GGML_OP_CONV_3D: + if (src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + const uint32_t OC = (uint32_t)ggml_get_op_params_i32(dst, 11); + const uint32_t IC = (uint32_t)ggml_get_op_params_i32(dst, 9); + const uint32_t N = (uint32_t)ggml_get_op_params_i32(dst, 10); + const uint32_t NPQ = N * dst->ne[2] * dst->ne[1] * dst->ne[0]; + const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, OC, NPQ); - const uint32_t s_off = S_v * H * n_tokens * n_seqs; + const uint32_t KW = (uint32_t)src0->ne[0]; + const uint32_t KH = (uint32_t)src0->ne[1]; + const uint32_t KD = (uint32_t)src0->ne[2]; + const uint32_t s0 = (uint32_t)ggml_get_op_params_i32(dst, 0); + const uint32_t s1 = (uint32_t)ggml_get_op_params_i32(dst, 1); + const uint32_t s2 = (uint32_t)ggml_get_op_params_i32(dst, 2); + const uint32_t p0 = (uint32_t)ggml_get_op_params_i32(dst, 3); + const uint32_t p1 = (uint32_t)ggml_get_op_params_i32(dst, 4); + const uint32_t p2 = (uint32_t)ggml_get_op_params_i32(dst, 5); + const uint32_t d0 = (uint32_t)ggml_get_op_params_i32(dst, 6); + const uint32_t d1 = (uint32_t)ggml_get_op_params_i32(dst, 7); + const uint32_t d2 = (uint32_t)ggml_get_op_params_i32(dst, 8); - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); - GGML_ASSERT(pipeline != nullptr); + const uint32_t CRS = IC * KW * KH * KD; + const uint32_t BS_K = vk_conv_block_sizes[shape].K; + const uint32_t BS_CRS = vk_conv_block_sizes[shape].CRS; + const uint32_t BS_NPQ = vk_conv_block_sizes[shape].NPQ; + const uint32_t aligned = ((OC % BS_K == 0) && + (CRS % BS_CRS == 0) && + (NPQ % BS_NPQ == 0)) ? 1u : 0u; - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + vk_conv3d_pipeline_state conv3d_pipeline_state(s0, s1, s2, p0, p1, p2, d0, d1, d2, KW, KH, KD, aligned); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer src_buf[6] = {}; - for (int i = 0; i < 6; i++) { - src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); - } + std::map *pipelines = nullptr; + if (src0->type == GGML_TYPE_F32) { + pipelines = &ctx->device->pipeline_conv3d_f32[shape]; + } else if (src0->type == GGML_TYPE_F16) { + pipelines = &ctx->device->pipeline_conv3d_f16_f32[shape]; + } else { + return nullptr; + } - const uint32_t sq1 = (uint32_t)(src_q->nb[1] / sizeof(float)); - const uint32_t sq2 = (uint32_t)(src_q->nb[2] / sizeof(float)); - const uint32_t sq3 = (uint32_t)(src_q->nb[3] / sizeof(float)); - const uint32_t sv1 = (uint32_t)(src_v->nb[1] / sizeof(float)); - const uint32_t sv2 = (uint32_t)(src_v->nb[2] / sizeof(float)); - const uint32_t sv3 = (uint32_t)(src_v->nb[3] / sizeof(float)); - const uint32_t sb1 = (uint32_t)(src_beta->nb[1] / sizeof(float)); - const uint32_t sb2 = (uint32_t)(src_beta->nb[2] / sizeof(float)); - const uint32_t sb3 = (uint32_t)(src_beta->nb[3] / sizeof(float)); + vk_pipeline pipeline = nullptr; - const uint32_t neq1 = (uint32_t)src_q->ne[1]; - const uint32_t rq3 = (uint32_t)(src_v->ne[3] / src_q->ne[3]); + { + std::lock_guard guard(ctx->device->compile_mutex); + auto it = pipelines->find(conv3d_pipeline_state); + if (it != pipelines->end()) { + pipeline = it->second; + } else { + (*pipelines)[conv3d_pipeline_state] = pipeline = std::make_shared(); + } + } - const float scale = 1.0f / sqrtf((float)S_v); - const vk_op_gated_delta_net_push_constants pc = { - H, n_tokens, n_seqs, s_off, - sq1, sq2, sq3, - sv1, sv2, sv3, - sb1, sb2, sb3, - neq1, rq3, - scale, - K - }; + return pipeline; + } + return nullptr; + case GGML_OP_ADD1: + if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_add1_f16_f16; + } + if (src0->type == GGML_TYPE_F16 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_add1_f16_f32; + } + if (src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_add1_f32_f32; + } + return nullptr; + case GGML_OP_ARANGE: + if (dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_arange_f32; + } + return nullptr; + case GGML_OP_FILL: + if (dst->type == GGML_TYPE_F32) { + return ctx->device->pipeline_fill_f32; + } + if (dst->type == GGML_TYPE_F16) { + return ctx->device->pipeline_fill_f16; + } + return nullptr; + default: + return nullptr; + } - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], dst_buf}, - pc, { H, n_seqs, S_v }); + GGML_UNUSED(src2); } -static void ggml_vk_ssm_scan(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - const ggml_tensor * src2 = dst->src[2]; - const ggml_tensor * src3 = dst->src[3]; - const ggml_tensor * src4 = dst->src[4]; - const ggml_tensor * src5 = dst->src[5]; - +template +static void ggml_vk_op_f32(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, const ggml_tensor * src3, ggml_tensor * dst, ggml_op op, PC&& pc, vk_pipeline pipeline_override = nullptr) { + VK_LOG_DEBUG("ggml_vk_op_f32((" << src0 << ", name=" << src0->name << ", type=" << src0->type << ", ne0=" << src0->ne[0] << ", ne1=" << src0->ne[1] << ", ne2=" << src0->ne[2] << ", ne3=" << src0->ne[3] << ", nb0=" << src0->nb[0] << ", nb1=" << src0->nb[1] << ", nb2=" << src0->nb[2] << ", nb3=" << src0->nb[3]; + if (src1 != nullptr) { + std::cerr << "), (" << src1 << ", name=" << src1->name << ", type=" << src1->type << ", ne0=" << src1->ne[0] << ", ne1=" << src1->ne[1] << ", ne2=" << src1->ne[2] << ", ne3=" << src1->ne[3] << ", nb0=" << src1->nb[0] << ", nb1=" << src1->nb[1] << ", nb2=" << src1->nb[2] << ", nb3=" << src1->nb[3]; + } + if (src2 != nullptr) { + std::cerr << "), (" << src2 << ", name=" << src2->name << ", type=" << src2->type << ", ne0=" << src2->ne[0] << ", ne1=" << src2->ne[1] << ", ne2=" << src2->ne[2] << ", ne3=" << src2->ne[3] << ", nb0=" << src2->nb[0] << ", nb1=" << src2->nb[1] << ", nb2=" << src2->nb[2] << ", nb3=" << src2->nb[3]; + } + if (src3 != nullptr) { + std::cerr << "), (" << src3 << ", name=" << src3->name << ", type=" << src3->type << ", ne0=" << src3->ne[0] << ", ne1=" << src3->ne[1] << ", ne2=" << src3->ne[2] << ", ne3=" << src3->ne[3] << ", nb0=" << src3->nb[0] << ", nb1=" << src3->nb[1] << ", nb2=" << src3->nb[2] << ", nb3=" << src3->nb[3]; + } + std::cerr << "), (" << dst << ", name=" << dst->name << ", type=" << dst->type << ", ne0=" << dst->ne[0] << ", ne1=" << dst->ne[1] << ", ne2=" << dst->ne[2] << ", ne3=" << dst->ne[3] << ", nb0=" << dst->nb[0] << ", nb1=" << dst->nb[1] << ", nb2=" << dst->nb[2] << ", nb3=" << dst->nb[3]; + std::cerr << "), " << ggml_op_name(op) << ")"); + GGML_ASSERT(op == GGML_OP_GET_ROWS || op == GGML_OP_CPY || op == GGML_OP_CONCAT || (!ggml_is_quantized(src0->type) && (src1 == nullptr || !ggml_is_quantized(src1->type)))); // NOLINT GGML_ASSERT(dst->buffer != nullptr); + const uint64_t ne00 = src0->ne[0]; + const uint64_t ne01 = src0->ne[1]; + const uint64_t ne02 = src0->ne[2]; + const uint64_t ne03 = src0->ne[3]; - const uint32_t head_dim = src0->ne[1]; - const uint32_t n_head = src1->ne[1]; - const uint32_t n_group = src4->ne[1]; - const uint32_t n_tok = src1->ne[2]; - const uint32_t n_seq = src1->ne[3]; - - bool is_mamba2 = (src3->nb[1] == sizeof(float)); - GGML_ASSERT(is_mamba2); - - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, src0, src1, src2, dst, dst->op); - GGML_ASSERT(pipeline != nullptr); - - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + const bool use_src1 = src1 != nullptr; + const uint64_t ne10 = use_src1 ? src1->ne[0] : 0; + const uint64_t ne11 = use_src1 ? src1->ne[1] : 0; + const uint64_t ne12 = use_src1 ? src1->ne[2] : 0; + const uint64_t ne13 = use_src1 ? src1->ne[3] : 0; - const int64_t s_off = ggml_nelements(src1) * sizeof(float); + const bool use_src2 = src2 != nullptr; + const bool use_src3 = src3 != nullptr; - const vk_op_ssm_scan_push_constants pc = { - (uint32_t)src0->nb[2], (uint32_t)src0->nb[3], - (uint32_t)src1->nb[2], (uint32_t)src1->nb[3], - (uint32_t)src2->nb[1], (uint32_t)src2->nb[2], - (uint32_t)src3->nb[1], - (uint32_t)src4->nb[2], (uint32_t)src4->nb[3], - (uint32_t)src5->nb[2], (uint32_t)src5->nb[3], - (uint32_t)s_off, - n_head, head_dim, n_group, n_tok, - n_seq, (uint32_t) ggml_get_op_params_i32(dst, 0) - }; + init_pushconst_fastdiv(pc); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer src_buf[7] = {}; - for (int i = 0; i < 7 && dst->src[i] != nullptr; i++) { - src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); + vk_pipeline pipeline; + if (pipeline_override) { + pipeline = pipeline_override; + } else { + pipeline = ggml_vk_op_get_pipeline(ctx, src0, src1, src2, dst, op); } - std::array elements; - - const uint32_t d_state = src0->ne[0]; - uint32_t num_subgroups = d_state / ctx->device->subgroup_size; - const uint32_t num_workgroups_x = CEIL_DIV(n_head * head_dim, num_subgroups); - const uint32_t num_workgroups_y = n_seq; - elements = { num_workgroups_x, num_workgroups_y, 1 }; - - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], src_buf[6], dst_buf}, - pc, elements); -} - -static void ggml_vk_ssm_conv(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { - ggml_tensor * conv = cgraph->nodes[node_idx]; - const ggml_tensor * src0 = conv->src[0]; - const ggml_tensor * src1 = conv->src[1]; - - // Pick the destination tensor (last node in the fused chain) and the optional bias. - // Fusion modes: 0 = ssm_conv, 1 = ssm_conv+silu, 2 = ssm_conv+add(bias)+silu. - ggml_tensor * dst = conv; - const ggml_tensor * bias = nullptr; - - if (ctx->num_additional_fused_ops == 1) { - dst = cgraph->nodes[node_idx + 1]; // silu - } else if (ctx->num_additional_fused_ops == 2) { - ggml_tensor * add = cgraph->nodes[node_idx + 1]; - bias = (add->src[0] == conv) ? add->src[1] : add->src[0]; - dst = cgraph->nodes[node_idx + 2]; // silu + if (pipeline == nullptr) { + std::cerr << "ggml_vulkan: Error: Missing op: " << ggml_op_name(op) << " for " << ggml_type_name(src0->type); + if (src1 != nullptr) { + std::cerr << " and " << ggml_type_name(src1->type); + } + std::cerr << " to " << ggml_type_name(dst->type) << std::endl; + GGML_ABORT("fatal error"); } - // The shader always declares 4 bindings; bind src0 as a dummy when bias isn't fused. - const ggml_tensor * src2 = bias ? bias : src0; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_SSM_CONV, { - (uint32_t)src0->nb[1], (uint32_t)src0->nb[2], - (uint32_t)src1->nb[1], - (uint32_t)dst->nb[0], (uint32_t)dst->nb[1], (uint32_t)dst->nb[2], - (uint32_t)src1->ne[0], - (uint32_t)src0->ne[0], - (uint32_t)src0->ne[1], - (uint32_t)dst->ne[1], - (uint32_t)dst->ne[2], - }); -} + vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0, true); + vk_subbuffer src1_buf = use_src1 ? ggml_vk_tensor_subbuffer(ctx, src1, true) : vk_subbuffer{}; + vk_subbuffer src2_buf = use_src2 ? ggml_vk_tensor_subbuffer(ctx, src2, true) : vk_subbuffer{}; + vk_subbuffer src3_buf = use_src3 ? ggml_vk_tensor_subbuffer(ctx, src3, true) : vk_subbuffer{}; + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); -static void ggml_vk_op_f32_opt_step_adamw(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst, const vk_op_push_constants&& pc) { - const ggml_tensor * x = dst->src[0]; - const ggml_tensor * g = dst->src[1]; - const ggml_tensor * gm = dst->src[2]; - const ggml_tensor * gv = dst->src[3]; - const ggml_tensor * p = dst->src[4]; + // Compute misalignment offset for descriptors and store it in in push constants. + init_pushconst_tensor_offsets(ctx, pc, src0, src1, src2, src3, dst); - GGML_ASSERT(x->type == GGML_TYPE_F32); - GGML_ASSERT(g->type == GGML_TYPE_F32); - GGML_ASSERT(gm->type == GGML_TYPE_F32); - GGML_ASSERT(gv->type == GGML_TYPE_F32); - GGML_ASSERT(p->type == GGML_TYPE_F32); - GGML_ASSERT(dst->buffer != nullptr); - GGML_ASSERT(ggml_is_contiguous(x)); - GGML_ASSERT(ggml_is_contiguous(g)); - GGML_ASSERT(ggml_is_contiguous(gm)); - GGML_ASSERT(ggml_is_contiguous(gv)); - GGML_ASSERT(ggml_is_contiguous(p)); - GGML_ASSERT(ggml_are_same_shape(x, g)); - GGML_ASSERT(ggml_are_same_shape(x, gm)); - GGML_ASSERT(ggml_are_same_shape(x, gv)); - GGML_ASSERT(ggml_nelements(p) == 7); + std::array elements; - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, g, gm, gv, dst, GGML_OP_OPT_STEP_ADAMW); - GGML_ASSERT(pipeline != nullptr); + switch (op) { + case GGML_OP_NORM: + case GGML_OP_RMS_NORM_BACK: + case GGML_OP_L2_NORM: + case GGML_OP_SOFT_MAX: + case GGML_OP_SOFT_MAX_BACK: + case GGML_OP_SUM_ROWS: + case GGML_OP_CUMSUM: + case GGML_OP_MEAN: + case GGML_OP_ARGMAX: + { + const uint32_t nr = ggml_nrows(src0); + if (nr > 262144) { + elements = { 512, 512, CEIL_DIV(nr, 262144) }; + } else if (nr > 512) { + elements = { 512, CEIL_DIV(nr, 512), 1 }; + } else { + elements = { nr, 1, 1 }; + } + } break; + case GGML_OP_SOLVE_TRI: + { + uint32_t nr = (uint32_t)(ne02 * ne03); + if (nr > 262144) { + elements = { 512, 512, CEIL_DIV(nr, 262144) }; + } else if (nr > 512) { + elements = { 512, CEIL_DIV(nr, 512), 1 }; + } else { + elements = { nr, 1, 1 }; + } + } + break; + case GGML_OP_RMS_NORM: + if (ctx->do_add_rms_partials) { + // Run one element per thread, 128 threads per workgroup + elements = { (uint32_t)CEIL_DIV(ne00, 128), 1, 1 }; + } else { + elements = { (uint32_t)ne01, (uint32_t)ne02, (uint32_t)ne03 }; + } + break; - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + case GGML_OP_SUM: + // We use GGML_OP_SUM_ROWS with 1 row. + elements = { 1, 1, 1 }; + break; + case GGML_OP_GROUP_NORM: + { + const uint32_t num_groups = dst->op_params[0]; + elements = { num_groups * (uint32_t)src0->ne[3], 1, 1 }; + } break; + case GGML_OP_DIAG_MASK_INF: + elements = { (uint32_t)ggml_nrows(src0), (uint32_t)ne00, 1 }; + break; + case GGML_OP_ROPE: + case GGML_OP_ROPE_BACK: + { + uint32_t nrows = (uint32_t)ggml_nrows(src0); + uint32_t z = 1; + if (nrows > ctx->device->properties.limits.maxComputeWorkGroupCount[0]) { + z = CEIL_DIV(nrows, 32768); + nrows = 32768; + } + elements = { nrows, (uint32_t)ne00, z }; - vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x); - vk_subbuffer g_buf = ggml_vk_tensor_subbuffer(ctx, g); - vk_subbuffer gm_buf = ggml_vk_tensor_subbuffer(ctx, gm); - vk_subbuffer gv_buf = ggml_vk_tensor_subbuffer(ctx, gv); - vk_subbuffer p_buf = ggml_vk_tensor_subbuffer(ctx, p); + } break; + case GGML_OP_GET_ROWS: + elements = { (uint32_t)ne00, (uint32_t)ne10, (uint32_t)(ne11 * ne12) }; + elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + break; + case GGML_OP_GET_ROWS_BACK: + elements = { (uint32_t)dst->ne[0], (uint32_t)dst->ne[1], 1 }; + elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + break; + case GGML_OP_ARGSORT: + GGML_ASSERT(0); + break; + case GGML_OP_IM2COL: + { + const bool is_2D = dst->op_params[6] == 1; - std::array elements = { (uint32_t)ggml_nelements(x), 1, 1 }; + const uint32_t IC = src1->ne[is_2D ? 2 : 1]; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - {x_buf, g_buf, gm_buf, gv_buf, p_buf}, - pc, elements); -} + const uint32_t KH = is_2D ? src0->ne[1] : 1; + const uint32_t KW = src0->ne[0]; -static void ggml_vk_opt_step_adamw(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - const size_t n = ggml_nelements(dst->src[0]); + const uint32_t OH = is_2D ? dst->ne[2] : 1; + const uint32_t OW = dst->ne[1]; - ggml_vk_op_f32_opt_step_adamw( - ctx, subctx, dst, - { (uint32_t)n, 0, 0.0f, 0.0f, 0.0f, 0.0f } - ); -} + const uint32_t batch = src1->ne[is_2D ? 3 : 2]; -static void ggml_vk_opt_step_sgd(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { - const size_t n = ggml_nelements(dst->src[0]); + const uint32_t CHW = IC * KH * KW; + // Cap X workgroups to limit concurrent IC channel reads. + // The shader loops over X to cover the full CHW dimension. + // AMD prefers a lower limit + const uint32_t min_cap = ctx->device->vendor_id == VK_VENDOR_ID_AMD ? 512u : 4096u; + const uint32_t x_elements = std::min(CHW, std::max(min_cap, OW * KH * KW)); + elements = { x_elements, OW, OH * batch }; + elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + } break; + case GGML_OP_IM2COL_3D: + { + const uint32_t IC = ((const uint32_t *)(dst->op_params))[9]; - ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_OPT_STEP_SGD, { (uint32_t)n, 0, 0.0f, 0.0f, 0.0f, 0.0f }); -} + const uint32_t N = ne13 / IC; -static void ggml_vk_concat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - int * op_params = (int *)dst->op_params; + const uint32_t KD = ne02; + const uint32_t KH = ne01; + const uint32_t KW = ne00; - const uint32_t unit_size = ggml_vk_concat_unit_size(dst->type); - const uint32_t units_per_block = ggml_type_size(dst->type) / unit_size; - const uint32_t block_size = ggml_blck_size(dst->type); - const bool quantized = ggml_is_quantized(dst->type); + const uint32_t OD = dst->ne[3] / N; + const uint32_t OH = dst->ne[2]; + const uint32_t OW = dst->ne[1]; - // Address dimension 0 in packed storage units; higher strides may be noncontiguous. - const uint32_t ne00 = src0->ne[0] / block_size * units_per_block; - const uint32_t ne10 = src1->ne[0] / block_size * units_per_block; - const uint32_t ne20 = dst->ne[0] / block_size * units_per_block; - const uint32_t nb00 = quantized ? 1 : src0->nb[0] / unit_size; - const uint32_t nb10 = quantized ? 1 : src1->nb[0] / unit_size; - const uint32_t nb20 = quantized ? 1 : dst->nb[0] / unit_size; + const uint32_t IC_KD_KH_KW = IC*KD*KH*KW; + const uint32_t N_OD_OH = N*OD*OH; - vk_op_concat_push_constants pc {{ - ne20 * (uint32_t)dst->ne[1] * (uint32_t)dst->ne[2] * (uint32_t)dst->ne[3], - ne00, (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], nb00, (uint32_t)src0->nb[1] / unit_size, (uint32_t)src0->nb[2] / unit_size, (uint32_t)src0->nb[3] / unit_size, - ne10, (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], nb10, (uint32_t)src1->nb[1] / unit_size, (uint32_t)src1->nb[2] / unit_size, (uint32_t)src1->nb[3] / unit_size, - ne20, (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], nb20, (uint32_t) dst->nb[1] / unit_size, (uint32_t) dst->nb[2] / unit_size, (uint32_t) dst->nb[3] / unit_size, - 0, - 0.0f, 0.0f, op_params[0], - }}; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONCAT, std::move(pc)); -} + elements = { IC_KD_KH_KW, OW, N_OD_OH }; + elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + } break; + case GGML_OP_TIMESTEP_EMBEDDING: + { + const uint32_t dim = dst->op_params[0]; + uint32_t half_ceil = (dim + 1) / 2; + elements = { half_ceil, (uint32_t)src0->ne[0], 1 }; + } break; + case GGML_OP_CONV_TRANSPOSE_1D: + { + elements = {uint32_t(src0->ne[1]), 1, 1}; // parallelize in {Cout, 1, 1} + } break; + case GGML_OP_COL2IM_1D: + { + elements = { uint32_t(dst->ne[0]), uint32_t(dst->ne[1]), 1 }; + } break; + case GGML_OP_POOL_1D: + { + const uint32_t N = dst->ne[3] * dst->ne[2]; + const uint32_t OC = dst->ne[1]; + const uint32_t OL = dst->ne[0]; + elements = { N * OC * OL, 1, 1}; + } break; + case GGML_OP_POOL_2D: + { + const uint32_t N = dst->ne[3]; + const uint32_t OC = dst->ne[2]; + const uint32_t OH = dst->ne[1]; + const uint32_t OW = dst->ne[0]; + elements = { N * OC * OH * OW, 1, 1}; + } break; + case GGML_OP_CONV_2D: + case GGML_OP_CONV_TRANSPOSE_2D: + if constexpr (std::is_same_v) { + const uint32_t NPQ = pc.N * pc.OH * pc.OW; + const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, pc.Cout, NPQ); + const uint32_t NPQ_blocks = CEIL_DIV(NPQ, vk_conv_block_sizes[shape].NPQ); -static void ggml_vk_upscale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t mode = (uint32_t)ggml_get_op_params_i32(dst, 0); + elements = { pc.Cout, NPQ_blocks, 1 }; + if (elements[1] > 512) { + elements[2] = CEIL_DIV(elements[1], 512); + elements[1] = 512; + } + } else { + GGML_ABORT("invalid push constant type for CONV_2D"); + } + break; + case GGML_OP_CONV_3D: + if constexpr (std::is_same_v) { + const uint32_t NPQ = pc.N * pc.OD * pc.OH * pc.OW; + const vk_conv_shapes shape = ggml_vk_conv_select_shape(ctx, pc.OC, NPQ); + const uint32_t NPQ_blocks = CEIL_DIV(NPQ, vk_conv_block_sizes[shape].NPQ); - GGML_TENSOR_UNARY_OP_LOCALS + elements = { pc.OC, NPQ_blocks, 1 }; + if (elements[1] > 512) { + elements[2] = CEIL_DIV(elements[1], 512); + elements[1] = 512; + } + } else { + GGML_ABORT("invalid push constant type for CONV_3D"); + } + break; + case GGML_OP_ADD: + case GGML_OP_SUB: + case GGML_OP_DIV: + case GGML_OP_MUL: + case GGML_OP_ADD1: + case GGML_OP_OUT_PROD: + case GGML_OP_ARANGE: + case GGML_OP_FILL: + case GGML_OP_SCALE: + case GGML_OP_SQR: + case GGML_OP_SQRT: + case GGML_OP_SIN: + case GGML_OP_COS: + case GGML_OP_LOG: + case GGML_OP_TRI: + case GGML_OP_DIAG: + case GGML_OP_CLAMP: + case GGML_OP_LEAKY_RELU: + case GGML_OP_PAD: + case GGML_OP_PAD_REFLECT_1D: + case GGML_OP_ROLL: + case GGML_OP_REPEAT: + case GGML_OP_REPEAT_BACK: + case GGML_OP_CPY: + case GGML_OP_CONCAT: + case GGML_OP_UPSCALE: + case GGML_OP_UNARY: + case GGML_OP_GLU: + case GGML_OP_CONV_2D_DW: + { + uint32_t ne = ggml_nelements(dst); + if (op == GGML_OP_CPY && ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { + // Convert from number of logical elements to 2- or 4-byte units. + ne /= ggml_blck_size(src0->type); + if ((ggml_type_size(src0->type) % 4) == 0) { + ne *= ggml_type_size(src0->type) / 4; + } else { + ne *= ggml_type_size(src0->type) / 2; + } + } + if (op == GGML_OP_CONCAT && ggml_is_quantized(dst->type)) { + ne = ne / ggml_blck_size(dst->type) * ggml_type_size(dst->type) / ggml_vk_concat_unit_size(dst->type); + } + // copy_to_quant has block size of 32, and each thread does QUANT_K elements. + // Splitting into 512x512xZ wouldn't work well since each workgroup does 1024 elements. + // So divide by block size here before splitting into 512x512 groups. + if (op == GGML_OP_CPY && !ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { + ne = CEIL_DIV(ne, ggml_blck_size(dst->type)); + } + if (ne > 262144) { + elements = { 512, 512, CEIL_DIV(ne, 262144) }; + } else if (ne > 512) { + elements = { 512, CEIL_DIV(ne, 512), 1 }; + } else { + elements = { ne, 1, 1 }; + } - float sf0 = (float)ne0 / ne00; - float sf1 = (float)ne1 / ne01; - float sf2 = (float)ne2 / ne02; - float sf3 = (float)ne3 / ne03; - float pixel_offset = 0.5f; + if (pipeline == ctx->device->pipeline_cpy_transpose_02_32 || + pipeline == ctx->device->pipeline_cpy_transpose_02_16) { + // 32x32 tiles over dims 0 and 2; dim1 and dim3 are the batch + elements[0] = (uint32_t)CEIL_DIV(dst->ne[0], 32); + elements[1] = (uint32_t)CEIL_DIV(dst->ne[2], 32); + elements[2] = (uint32_t)(dst->ne[1]*dst->ne[3]); + elements[0] = std::min(elements[0], ctx->device->properties.limits.maxComputeWorkGroupCount[0]); + elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + } else if (pipeline == ctx->device->pipeline_cpy_transpose_32 || + pipeline == ctx->device->pipeline_cpy_transpose_16) { + // 32x32 tiles + elements[0] = (uint32_t)CEIL_DIV(dst->ne[0], 32); + elements[1] = (uint32_t)CEIL_DIV(dst->ne[1], 32); + elements[2] = (uint32_t)(dst->ne[2]*dst->ne[3]); + elements[0] = std::min(elements[0], ctx->device->properties.limits.maxComputeWorkGroupCount[0]); + elements[1] = std::min(elements[1], ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = std::min(elements[2], ctx->device->properties.limits.maxComputeWorkGroupCount[2]); + } + } break; + case GGML_OP_ADD_ID: + { + elements = { (uint32_t)ne01, (uint32_t)ne02, 1 }; + } break; + case GGML_OP_SET_ROWS: + { + uint32_t ne = ggml_nelements(src0); + if (ggml_is_quantized(dst->type)) { + // quants run 32 threads each doing QUANT_K elements + ne = CEIL_DIV(ne, 32 * ggml_blck_size(dst->type)); + } else { + // scalar types do one element per thread, running 512 threads + ne = CEIL_DIV(ne, 512); + } + if (ne > 262144) { + elements = { 512, 512, CEIL_DIV(ne, 262144) }; + } else if (ne > 512) { + elements = { 512, CEIL_DIV(ne, 512), 1 }; + } else { + elements = { ne, 1, 1 }; + } + } + break; + case GGML_OP_SSM_CONV: + { + const uint32_t nr = src0->ne[1]; + const uint32_t n_t = dst->ne[1]; + const uint32_t n_s = dst->ne[2]; + elements = { nr, n_t, n_s }; + } + break; + default: + elements = { (uint32_t)ggml_nelements(src0), 1, 1 }; + break; + } - if (mode & GGML_SCALE_FLAG_ALIGN_CORNERS) { - sf0 = ne0 > 1 && ne00 > 1 ? (float)(ne0 - 1) / (ne00 - 1) : sf0; - sf1 = ne1 > 1 && ne01 > 1 ? (float)(ne1 - 1) / (ne01 - 1) : sf1; - pixel_offset = 0.0f; + if (op == GGML_OP_ADD || op == GGML_OP_RMS_NORM) { + vk_subbuffer a_buf = src0_buf; + if (ctx->do_add_rms_partials) { + a_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_add_rms_partials, ctx->prealloc_size_add_rms_partials_offset); + } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { src0_buf, src1_buf, dst_buf, a_buf }, pc, elements); + } else if (op == GGML_OP_GLU) { + // Empty src1 is possible in glu, but the shader needs a buffer + vk_subbuffer subbuf1 = use_src1 ? src1_buf : src0_buf; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc, elements); + } else if (op == GGML_OP_SOFT_MAX) { + // Empty src1 and src2 is possible in soft_max, but the shader needs a buffer + vk_subbuffer subbuf1 = use_src1 ? src1_buf : src0_buf; + vk_subbuffer subbuf2 = use_src2 ? src2_buf : src0_buf; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, subbuf2, dst_buf }, pc, elements); + } else if (op == GGML_OP_ROPE || op == GGML_OP_ROPE_BACK) { + // Empty src2 and src3 is possible in rope, but the shader needs a buffer + vk_subbuffer subbuf2 = use_src2 ? src2_buf : src0_buf; + vk_subbuffer subbuf3 = use_src3 ? src3_buf : src0_buf; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, subbuf2, dst_buf, subbuf3 }, pc, elements); + } else if (op == GGML_OP_IM2COL || op == GGML_OP_IM2COL_3D) { + if (ctx->device->shader_int64 && ctx->device->buffer_device_address) { + // buffer device address path doesn't use dst buffer + dst_buf.size = 1; + } + // im2col uses only src1 and dst buffers + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src1_buf, dst_buf }, pc, elements); + } else if (op == GGML_OP_COUNT_EQUAL) { + // count_equal assumes that destination buffer is initialized with zeroes + ggml_vk_buffer_memset_async(subctx, dst_buf.buffer, dst_buf.offset, 0, dst_buf.size); + ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, dst_buf }, pc, elements); + } else if (op == GGML_OP_OPT_STEP_SGD) { + // OPT_STEP_SGD works on src0, it does not need dst + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf }, pc, elements); + } else if (use_src3) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf, src3_buf, dst_buf }, pc, elements); + } else if (use_src2) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, src2_buf, dst_buf }, pc, elements); + } else if (use_src1) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, dst_buf }, pc, elements); + } else { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, dst_buf }, pc, elements); } - - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UPSCALE, { - (uint32_t)ggml_nelements(dst), 0, 0, - (uint32_t)ne00, (uint32_t)ne01, - (uint32_t)nb00 / src0_type_size, (uint32_t)nb01 / src0_type_size, (uint32_t)nb02 / src0_type_size, (uint32_t)nb03 / src0_type_size, - (uint32_t)ne0, (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, - sf0, sf1, sf2, sf3, pixel_offset - }); } -static void ggml_vk_scale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = ggml_get_op_params_f32(dst, 0); - p.param2 = ggml_get_op_params_f32(dst, 1); +void ggml_vk_get_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SCALE, std::move(p)); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GET_ROWS, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); } -static void ggml_vk_sqr(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SQR, vk_op_unary_push_constants_init(src0, dst)); -} +void ggml_vk_get_rows_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); -static void ggml_vk_sqrt(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SQRT, vk_op_unary_push_constants_init(src0, dst)); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GET_ROWS_BACK, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], (uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2], (uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2], (uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); } -static void ggml_vk_add1(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { +void ggml_vk_acc(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { const uint32_t src0_type_size = ggml_type_size(src0->type); const uint32_t src1_type_size = ggml_type_size(src1->type); const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_ADD1, { + int nb1 = dst->op_params[0] / src0_type_size; // 4 bytes of float32 + int nb2 = dst->op_params[1] / src0_type_size; // 4 bytes of float32 + int nb3 = dst->op_params[2] / src0_type_size; // 4 bytes of float32 + int offset = dst->op_params[3] / src0_type_size; // offset in bytes + + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, dst->op, { (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)nb1, (uint32_t)nb2, (uint32_t)nb3, (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t)nb1, (uint32_t)nb2, (uint32_t)nb3, 0, - 0.0f, 0.0f, 0, + 0.0f, 0.0f, offset, }); } -static void ggml_vk_arange(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - VK_LOG_DEBUG("ggml_vk_arange(dst=" << dst << ", ne=" << ggml_nelements(dst) << ")"); - - vk_op_push_constants pc = { - (uint32_t)ggml_nelements(dst), - 1, - ggml_get_op_params_f32(dst, 0), - ggml_get_op_params_f32(dst, 2), - 0.0f, 0.0f, - }; - - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, dst, GGML_OP_ARANGE); - GGML_ASSERT(pipeline != nullptr); +void ggml_vk_multi_add(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { + const ggml_tensor *first_node = cgraph->nodes[node_idx]; + const ggml_tensor *dst = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, false); + // Make a list of all the tensors used by the op. + // Last element of the list is the dest tensor. + const ggml_tensor *tensors[MAX_PARAMETER_COUNT]; + uint32_t num_srcs = ctx->num_additional_fused_ops + 2; + uint32_t num_tensors = num_srcs + 1; + GGML_ASSERT(num_tensors + ctx->do_add_rms_partials <= MAX_PARAMETER_COUNT); - std::array elements = { (uint32_t)ggml_nelements(dst), 1, 1 }; + tensors[0] = first_node->src[0]; + tensors[1] = first_node->src[1]; + for (int32_t i = 0; i < ctx->num_additional_fused_ops; ++i) { + // check whether the previous result is src[0] or src[1] + if (cgraph->nodes[node_idx + i] == cgraph->nodes[node_idx + i + 1]->src[0]) { + tensors[i+2] = cgraph->nodes[node_idx + i + 1]->src[1]; + } else { + tensors[i+2] = cgraph->nodes[node_idx + i + 1]->src[0]; + } + } + tensors[num_srcs] = dst; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { dst_buf }, pc, elements); -} + vk_op_multi_add_push_constants pc; + pc.ne20 = (uint32_t)dst->ne[0]; + pc.ne21 = (uint32_t)dst->ne[1]; + pc.ne22 = (uint32_t)dst->ne[2]; + pc.ne23 = (uint32_t)dst->ne[3]; -static void ggml_vk_fill(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { - VK_LOG_DEBUG("ggml_vk_fill(dst=" << dst << ", ne=" << ggml_nelements(dst) << ")"); + for (uint32_t i = 0; i < num_tensors; ++i) { + const ggml_tensor *t = tensors[i]; + pc.nb[i][0] = (uint32_t)t->nb[0] / sizeof(float); + pc.nb[i][1] = (uint32_t)t->nb[1] / sizeof(float); + pc.nb[i][2] = (uint32_t)t->nb[2] / sizeof(float); + pc.nb[i][3] = (uint32_t)t->nb[3] / sizeof(float); + } + pc.rms_partials = ctx->do_add_rms_partials; - vk_op_push_constants pc = { - (uint32_t)ggml_nelements(dst), - 1, - ggml_get_op_params_f32(dst, 0), - 0.0f, - 0.0f, 0.0f, - }; + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, tensors[0], tensors[1], nullptr, dst, dst->op); - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, dst, GGML_OP_FILL); - GGML_ASSERT(pipeline != nullptr); + if (pipeline == nullptr) { + std::cerr << "ggml_vulkan: Error: Missing multi_add"; + GGML_ABORT("fatal error"); + } ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, false); - - std::array elements = { (uint32_t)ggml_nelements(dst), 1, 1 }; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { dst_buf }, pc, elements); -} + ggml_backend_vk_buffer_context * buf_ctx[MAX_PARAMETER_COUNT]; + vk_buffer buf[MAX_PARAMETER_COUNT]; + size_t offset[MAX_PARAMETER_COUNT]; + bool uma[MAX_PARAMETER_COUNT]; -static void ggml_vk_sin(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SIN, vk_op_unary_push_constants_init(src0, dst)); -} + for (uint32_t i = 0; i < num_tensors; ++i) { + buf_ctx[i] = (ggml_backend_vk_buffer_context *)tensors[i]->buffer->context; + buf[i] = nullptr; + offset[i] = 0; + uma[i] = false; -static void ggml_vk_cos(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_COS, vk_op_unary_push_constants_init(src0, dst)); -} + if (ctx->device->uma) { + ggml_vk_host_get(ctx->device, tensors[i]->data, buf[i], offset[i]); + uma[i] = buf[i] != nullptr; + } + if (!uma[i]) { + buf[i] = buf_ctx[i]->dev_buffer; + offset[i] = vk_tensor_offset(tensors[i]) + tensors[i]->view_offs; + } + GGML_ASSERT(buf[i] != nullptr); + } + // If any remaining descriptors are unused, just point them at src[0] + for (uint32_t i = num_tensors; i < MAX_PARAMETER_COUNT; ++i) { + buf[i] = buf[0]; + offset[i] = 0; + } + if (ctx->do_add_rms_partials) { + buf[num_tensors] = ctx->prealloc_add_rms_partials; + offset[num_tensors] = ctx->prealloc_size_add_rms_partials_offset; + } -static void ggml_vk_log(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_LOG, vk_op_unary_push_constants_init(src0, dst)); -} + std::array elements; -static void ggml_vk_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = ggml_get_op_params_f32(dst, 0); + uint32_t ne = ggml_nelements(dst); + if (ne > 262144) { + elements = { 512, 512, CEIL_DIV(ne, 262144) }; + } else if (ne > 512) { + elements = { 512, CEIL_DIV(ne, 512), 1 }; + } else { + elements = { ne, 1, 1 }; + } - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_TRI, std::move(p)); + static_assert(MAX_PARAMETER_COUNT == 12); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + ggml_vk_subbuffer(ctx, buf[0], offset[0]), + ggml_vk_subbuffer(ctx, buf[1], offset[1]), + ggml_vk_subbuffer(ctx, buf[2], offset[2]), + ggml_vk_subbuffer(ctx, buf[3], offset[3]), + ggml_vk_subbuffer(ctx, buf[4], offset[4]), + ggml_vk_subbuffer(ctx, buf[5], offset[5]), + ggml_vk_subbuffer(ctx, buf[6], offset[6]), + ggml_vk_subbuffer(ctx, buf[7], offset[7]), + ggml_vk_subbuffer(ctx, buf[8], offset[8]), + ggml_vk_subbuffer(ctx, buf[9], offset[9]), + ggml_vk_subbuffer(ctx, buf[10], offset[10]), + ggml_vk_subbuffer(ctx, buf[11], offset[11]), + }, pc, elements); } -static void ggml_vk_diag(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); +void ggml_vk_add(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_DIAG, std::move(p)); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_ADD, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, ctx->do_add_rms_partials, + }); } -static void ggml_vk_clamp(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = ggml_get_op_params_f32(dst, 0); - p.param2 = ggml_get_op_params_f32(dst, 1); +void ggml_vk_out_prod(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CLAMP, std::move(p)); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_OUT_PROD, { + (uint32_t)ggml_nelements(dst), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], + (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], + (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], + (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); } -static void ggml_vk_pad(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_pad_push_constants p = vk_op_pad_push_constants_init(src0, dst); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_PAD, std::move(p)); -} +void ggml_vk_sub(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); -static void ggml_vk_roll(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const int32_t s0 = ggml_get_op_params_i32(dst, 0); - const int32_t s1 = ggml_get_op_params_i32(dst, 1); - const int32_t s2 = ggml_get_op_params_i32(dst, 2); - const int32_t s3 = ggml_get_op_params_i32(dst, 3); - const uint32_t s01_packed = ((s0 + 0x8000) << 16) | (s1 + 0x8000); - const uint32_t s23_packed = ((s2 + 0x8000) << 16) | (s3 + 0x8000); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SUB, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); +} - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - memcpy(&p.param1, &s01_packed, sizeof(float)); - memcpy(&p.param2, &s23_packed, sizeof(float)); +void ggml_vk_mul(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_ROLL, std::move(p)); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_MUL, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); } -static void ggml_vk_repeat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_REPEAT, std::move(p)); +int ggml_vk_unary_mul_op_index(ggml_unary_op op) { + switch (op) { + case GGML_UNARY_OP_GELU: return 0; + case GGML_UNARY_OP_SIGMOID: return 1; + case GGML_UNARY_OP_SILU: return 2; + case GGML_UNARY_OP_SOFTPLUS: return 3; + default: return -1; + } } -static void ggml_vk_repeat_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_REPEAT_BACK, std::move(p)); -} +void ggml_vk_unary_mul(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + const ggml_tensor * unary = cgraph->nodes[node_idx]; + ggml_tensor * mul = cgraph->nodes[node_idx + 1]; -static void ggml_vk_cpy(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - uint32_t ne = (uint32_t)ggml_nelements(src0); - if (ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { - // Convert from number of logical elements to 2- or 4-byte units. - ne /= ggml_blck_size(src0->type); - if ((ggml_type_size(src0->type) % 4) == 0) { - ne *= ggml_type_size(src0->type) / 4; - } else { - ne *= ggml_type_size(src0->type) / 2; - } + // unary on src1 that tiles into src0 + const bool op_on_b = mul->src[1] == unary && + !ggml_are_same_shape(unary->src[0], mul->src[0]) && + ggml_can_repeat(unary, mul->src[0]); + + const ggml_tensor * src0 = op_on_b ? mul->src[0] : unary->src[0]; + const ggml_tensor * src1 = op_on_b ? unary->src[0] : + ((mul->src[0] == unary) ? mul->src[1] : mul->src[0]); + + const bool f16 = src0->type == GGML_TYPE_F16; + const bool norepeat = ggml_are_same_shape(src0, src1); + const int oi = ggml_vk_unary_mul_op_index(ggml_get_unary_op(unary)); + if (oi < 0) { + GGML_ABORT("fatal error"); } + vk_pipeline pipeline = ctx->device->pipeline_unary_mul[oi][f16][norepeat][op_on_b]; - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ne); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CPY, std::move(p)); + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(mul->type); + + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, mul, GGML_OP_UNARY, { + (uint32_t)ggml_nelements(op_on_b ? mul : src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) mul->ne[0], (uint32_t) mul->ne[1], (uint32_t) mul->ne[2],(uint32_t) mul->ne[3], (uint32_t) mul->nb[0] / dst_type_size, (uint32_t) mul->nb[1] / dst_type_size, (uint32_t) mul->nb[2] / dst_type_size, (uint32_t) mul->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }, pipeline); } -static void ggml_vk_set_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { +void ggml_vk_div(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { const uint32_t src0_type_size = ggml_type_size(src0->type); const uint32_t src1_type_size = ggml_type_size(src1->type); const uint32_t dst_type_size = ggml_type_size(dst->type); - // Skip empty skip_rows operations. For most ops the empty check at the start - // of ggml_vk_build_graph is sufficient, but set_rows can have a nonempty dst - // with empty srcs. - if (ggml_is_empty(src0) || ggml_is_empty(src1)) { - return; - } - - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SET_ROWS, { + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_DIV, { (uint32_t)ggml_nelements(src0), (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, @@ -13081,1962 +9770,2015 @@ static void ggml_vk_set_rows(ggml_backend_vk_context * ctx, vk_context& subctx, }); } -static void ggml_vk_silu_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SILU_BACK, { (uint32_t)ggml_nelements(src0), 0, 0.0f, 0.0f, 0.0f, 0.0f }); +void ggml_vk_add_id(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t src2_type_size = ggml_type_size(src2->type); + + ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_ADD_ID, { + (uint32_t)dst->ne[0], + (uint32_t)dst->ne[1], + (uint32_t)src0->nb[1] / src0_type_size, + (uint32_t)src0->nb[2] / src0_type_size, + (uint32_t)src1->nb[1] / src1_type_size, + (uint32_t)src2->nb[1] / src2_type_size, + }); } -static void ggml_vk_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - float * op_params = (float *)dst->op_params; - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = op_params[0]; +static void ggml_vk_op_f32_wkv(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst, const vk_op_rwkv_wkv6_push_constants&& pc, int version) { + GGML_ASSERT(version == 6 || version == 7); + int num_srcs = version == 6 ? 6 : 7; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_NORM, std::move(p)); -} + for (int i = 0; i < num_srcs; i++) { + GGML_ASSERT(!ggml_is_quantized(dst->src[i]->type)); + } -static void ggml_vk_group_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const int * int_op_params = (const int *)dst->op_params; - const float * float_op_params = (const float *)dst->op_params; + GGML_ASSERT(dst->buffer != nullptr); - const uint32_t num_groups = int_op_params[0]; - const float eps = float_op_params[1]; - const uint32_t group_size = src0->ne[0] * src0->ne[1] * ((src0->ne[2] + num_groups - 1) / num_groups); + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); + GGML_ASSERT(pipeline != nullptr); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_GROUP_NORM, { group_size, 0, eps, 0.0f, 0.0f, 0.0f }); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer src_buf[7] = {}; + for (int i = 0; i < num_srcs; i++) { + src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); + } + + std::array elements = { + (uint32_t)(pc.B * pc.H), + 1, + 1 + }; + + if (version == 6) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], dst_buf}, + pc, elements); + } else if (version == 7) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], src_buf[6], dst_buf}, + pc, elements); + } else { + // shouldn't happen + GGML_ASSERT(false); + } } -static uint32_t ggml_vk_rms_num_partials(ggml_backend_vk_context * ctx, const ggml_tensor *node) { - const uint32_t ne = (uint32_t)node->ne[0]; - const uint32_t denom = ctx->device->pipeline_add_rms[0][0][0]->wg_denoms[0]; - const uint32_t num_partials = CEIL_DIV(ne, denom); - return num_partials; +void ggml_vk_rwkv_wkv6(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const size_t seq_length = dst->src[0]->ne[2]; + const size_t n_embed = dst->ne[0]; + const size_t n_heads = dst->src[0]->ne[1]; + const size_t n_seqs = dst->src[5]->ne[1]; + + ggml_vk_op_f32_wkv( + ctx, subctx, dst, + { + (uint32_t)n_seqs, + (uint32_t)seq_length, + (uint32_t)n_embed, + (uint32_t)n_heads, + }, + 6 + ); } -static uint32_t ggml_vk_rms_partials_size(ggml_backend_vk_context * ctx, const ggml_tensor *node) { - const uint32_t num_partials = ggml_vk_rms_num_partials(ctx, node); - const uint32_t num_bytes = ROUNDUP_POW2(num_partials * sizeof(uint32_t), ctx->device->partials_binding_alignment); - return num_bytes; +void ggml_vk_rwkv_wkv7(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const size_t seq_length = dst->src[0]->ne[2]; + const size_t n_embed = dst->ne[0]; + const size_t n_heads = dst->src[0]->ne[1]; + const size_t n_seqs = dst->src[6]->ne[1]; + + ggml_vk_op_f32_wkv( + ctx, subctx, dst, + { + (uint32_t)n_seqs, + (uint32_t)seq_length, + (uint32_t)n_embed, + (uint32_t)n_heads, + }, + 7 + ); } -static vk_op_rope_push_constants ggml_vk_make_rope_constants(const ggml_tensor *dst, const ggml_tensor *src0, const bool has_ff, bool backprop, const uint32_t set_rows_stride) { - const int n_dims = ((const int32_t *) dst->op_params)[1]; - const int mode = ((const int32_t *) dst->op_params)[2]; - // const int n_ctx = ((const int32_t *) dst->op_params)[3]; - const int n_ctx_orig = ((const int32_t *) dst->op_params)[4]; - const float freq_base = ((const float *) dst->op_params)[5]; - const float freq_scale = ((const float *) dst->op_params)[6]; - const float ext_factor = ((const float *) dst->op_params)[7]; - const float attn_factor = ((const float *) dst->op_params)[8]; - const float beta_fast = ((const float *) dst->op_params)[9]; - const float beta_slow = ((const float *) dst->op_params)[10]; - int sections[4] {}; - if (mode & GGML_ROPE_TYPE_MROPE) { - memcpy(sections, (const int32_t *) dst->op_params + 11, sizeof(int)*4); - } +void ggml_vk_gated_linear_attn(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const size_t seq_length = dst->src[0]->ne[2]; + const size_t n_embed = dst->ne[0]; + const size_t n_heads = dst->src[0]->ne[1]; + const size_t n_seqs = dst->src[4]->ne[1]; - const bool is_imrope = mode == GGML_ROPE_TYPE_IMROPE; + float scale; + memcpy(&scale, dst->op_params, sizeof(float)); - float corr_dims[2]; - ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims); + GGML_ASSERT(dst->buffer != nullptr); - const float theta_scale = powf(freq_base, -2.0f/n_dims); + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); + GGML_ASSERT(pipeline != nullptr); - uint32_t nb01 = src0->nb[1] / ggml_type_size(src0->type); - uint32_t nb02 = src0->nb[2] / ggml_type_size(src0->type); - uint32_t nb03 = src0->nb[3] / ggml_type_size(src0->type); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - uint32_t nb11 = dst->nb[1] / ggml_type_size(dst->type); - uint32_t nb12 = dst->nb[2] / ggml_type_size(dst->type); - uint32_t nb13 = dst->nb[3] / ggml_type_size(dst->type); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer src_buf[5] = {}; + for (int i = 0; i < 5; i++) { + src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); + } - vk_op_rope_push_constants rope { - (uint32_t)mode, (uint32_t)ggml_nrows(src0), (uint32_t)n_dims, freq_scale, - freq_base, ext_factor, attn_factor, {corr_dims[0], corr_dims[1]}, theta_scale, has_ff, - { sections[0], sections[1], sections[2], sections[3] }, is_imrope, backprop, set_rows_stride, + const vk_op_gated_linear_attn_push_constants pc = { + (uint32_t)n_seqs, + (uint32_t)seq_length, + (uint32_t)n_embed, + (uint32_t)n_heads, + scale, + }; - (uint32_t)src0->ne[0], - (uint32_t)src0->ne[1], - (uint32_t)src0->ne[2], - nb01, nb02, nb03, - nb11, nb12, nb13, - 0, 0, // a_offset, d_offset filled in by init_pushconst_tensor_offsets + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], dst_buf}, + pc, { (uint32_t)(n_seqs * n_heads), 1, 1 }); +} + +void ggml_vk_lightning_indexer(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const ggml_tensor * q = dst->src[0]; + const ggml_tensor * k = dst->src[1]; + const ggml_tensor * w = dst->src[2]; + const ggml_tensor * m = dst->src[3]; + + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, q, k, w, dst, dst->op); + GGML_ASSERT(pipeline != nullptr); + + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + const uint32_t n_kv = k->ne[2]; + const uint32_t n_heads = q->ne[1]; + const uint32_t n_tokens = q->ne[2]; + const uint32_t n_streams = q->ne[3]; + const uint32_t n_masks = m->ne[3]; + + const uint32_t n_outputs = (uint32_t)(dst->ne[0] * dst->ne[1] * dst->ne[3]); + const uint32_t dispatch_x = std::min(n_outputs, ctx->device->properties.limits.maxComputeWorkGroupCount[0]); + const uint32_t dispatch_y = CEIL_DIV(n_outputs, dispatch_x); + + // q, w and dst are f32 and m is f16, so their strides are passed in elements; + // k may be quantized, so its strides stay in bytes + const uint32_t q_nb1 = q->nb[1] / sizeof(float); + const uint32_t q_nb2 = q->nb[2] / sizeof(float); + const uint32_t q_nb3 = q->nb[3] / sizeof(float); + const uint32_t k_nb2 = k->nb[2]; + const uint32_t k_nb3 = k->nb[3]; + const uint32_t w_nb1 = w->nb[1] / sizeof(float); + const uint32_t w_nb3 = w->nb[3] / sizeof(float); + const uint32_t m_nb1 = m->nb[1] / sizeof(ggml_fp16_t); + const uint32_t m_nb3 = m->nb[3] / sizeof(ggml_fp16_t); + const uint32_t d_nb1 = dst->nb[1] / sizeof(float); + const uint32_t d_nb3 = dst->nb[3] / sizeof(float); + + const vk_op_lightning_indexer_push_constants pc = { + n_kv, n_heads, n_tokens, n_streams, n_masks, dispatch_x, + q_nb1, q_nb2, q_nb3, + k_nb2, k_nb3, + w_nb1, w_nb3, + m_nb1, m_nb3, + d_nb1, d_nb3, }; - return rope; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {ggml_vk_tensor_subbuffer(ctx, q), ggml_vk_tensor_subbuffer(ctx, k), ggml_vk_tensor_subbuffer(ctx, w), ggml_vk_tensor_subbuffer(ctx, m), ggml_vk_tensor_subbuffer(ctx, dst)}, + pc, {dispatch_x, dispatch_y, 1}); } -static void ggml_vk_rms_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx, float * op_params) { - ggml_tensor * dst; - const ggml_tensor * src0; - const ggml_tensor * src1; +void ggml_vk_gated_delta_net(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const ggml_tensor * src_q = dst->src[0]; + const ggml_tensor * src_v = dst->src[2]; + const ggml_tensor * src_beta = dst->src[4]; - if (ctx->num_additional_fused_ops > 0) { - // fused rms_norm + mul - ggml_tensor *mul = cgraph->nodes[node_idx + 1]; - ggml_tensor *other_src = mul->src[0] == cgraph->nodes[node_idx + 0] ? mul->src[1] : mul->src[0]; - dst = mul; - src0 = cgraph->nodes[node_idx]->src[0]; - src1 = other_src; - } else { - dst = cgraph->nodes[node_idx]; - src0 = src1 = dst->src[0]; + GGML_ASSERT(dst->buffer != nullptr); + + const uint32_t S_v = (uint32_t)src_v->ne[0]; + const uint32_t H = (uint32_t)src_v->ne[1]; + const uint32_t n_tokens = (uint32_t)src_v->ne[2]; + const uint32_t n_seqs = (uint32_t)src_v->ne[3]; + + // K (snapshot slot count) is an op param; state holds s0 only [S_v, S_v, H, n_seqs]. + const uint32_t K = (uint32_t)ggml_get_op_params_i32(dst, 0); + + const uint32_t s_off = S_v * H * n_tokens * n_seqs; + + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, dst->src[0], dst->src[1], dst->src[2], dst, dst->op); + GGML_ASSERT(pipeline != nullptr); + + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer src_buf[6] = {}; + for (int i = 0; i < 6; i++) { + src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); } - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); + const uint32_t sq1 = (uint32_t)(src_q->nb[1] / sizeof(float)); + const uint32_t sq2 = (uint32_t)(src_q->nb[2] / sizeof(float)); + const uint32_t sq3 = (uint32_t)(src_q->nb[3] / sizeof(float)); + const uint32_t sv1 = (uint32_t)(src_v->nb[1] / sizeof(float)); + const uint32_t sv2 = (uint32_t)(src_v->nb[2] / sizeof(float)); + const uint32_t sv3 = (uint32_t)(src_v->nb[3] / sizeof(float)); + const uint32_t sb1 = (uint32_t)(src_beta->nb[1] / sizeof(float)); + const uint32_t sb2 = (uint32_t)(src_beta->nb[2] / sizeof(float)); + const uint32_t sb3 = (uint32_t)(src_beta->nb[3] / sizeof(float)); - uint32_t param3 = ctx->do_add_rms_partials ? ggml_vk_rms_num_partials(ctx, dst) : 0; + const uint32_t neq1 = (uint32_t)src_q->ne[1]; + const uint32_t rq3 = (uint32_t)(src_v->ne[3] / src_q->ne[3]); - vk_op_binary_push_constants bin { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - op_params[0], 0.0f, (int32_t)param3, + const float scale = 1.0f / sqrtf((float)S_v); + const vk_op_gated_delta_net_push_constants pc = { + H, n_tokens, n_seqs, s_off, + sq1, sq2, sq3, + sv1, sv2, sv3, + sb1, sb2, sb3, + neq1, rq3, + scale, + K }; - // more than one fused op means rms_norm+mul+rope - if (ctx->num_additional_fused_ops > 1) { - static constexpr uint32_t max_tensors = 7; - const ggml_tensor *tensors[max_tensors] {}; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], dst_buf}, + pc, { H, n_seqs, S_v }); +} - ggml_tensor *rms = cgraph->nodes[node_idx + 0]; - ggml_tensor *mul = cgraph->nodes[node_idx + 1]; - ggml_tensor *rope = cgraph->nodes[node_idx + 2]; +void ggml_vk_ssm_scan(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + const ggml_tensor * src2 = dst->src[2]; + const ggml_tensor * src3 = dst->src[3]; + const ggml_tensor * src4 = dst->src[4]; + const ggml_tensor * src5 = dst->src[5]; - ggml_tensor *other_src = mul->src[0] == rms ? mul->src[1] : mul->src[0]; + GGML_ASSERT(dst->buffer != nullptr); - bool do_set_rows = ctx->num_additional_fused_ops == 4; + const uint32_t head_dim = src0->ne[1]; + const uint32_t n_head = src1->ne[1]; + const uint32_t n_group = src4->ne[1]; + const uint32_t n_tok = src1->ne[2]; + const uint32_t n_seq = src1->ne[3]; - tensors[0] = rms->src[0]; - tensors[1] = other_src; - tensors[2] = mul; - tensors[3] = rope->src[1]; // pos - tensors[4] = rope->src[2]; // ff - tensors[5] = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; // dst - tensors[6] = do_set_rows ? tensors[5]->src[1] : nullptr; - const uint32_t set_rows_stride = do_set_rows ? tensors[5]->nb[1] / ggml_type_size(tensors[5]->type) : 0; + bool is_mamba2 = (src3->nb[1] == sizeof(float)); + GGML_ASSERT(is_mamba2); - vk_op_rms_norm_mul_rope_push_constants pc; - pc.bin = bin; - pc.rope = ggml_vk_make_rope_constants(rope, rope->src[0], tensors[4] != nullptr, false, set_rows_stride); + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, src0, src1, src2, dst, dst->op); + GGML_ASSERT(pipeline != nullptr); - vk_pipeline pipeline = tensors[5]->type == GGML_TYPE_F16 ? ctx->device->pipeline_rms_norm_mul_rope_f32_f16 : ctx->device->pipeline_rms_norm_mul_rope_f32_f32; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + const int64_t s_off = ggml_nelements(src1) * sizeof(float); - ggml_backend_vk_buffer_context * buf_ctx[max_tensors]; - vk_buffer buf[max_tensors]; - size_t offset[max_tensors]; - bool uma[max_tensors]; + const vk_op_ssm_scan_push_constants pc = { + (uint32_t)src0->nb[2], (uint32_t)src0->nb[3], + (uint32_t)src1->nb[2], (uint32_t)src1->nb[3], + (uint32_t)src2->nb[1], (uint32_t)src2->nb[2], + (uint32_t)src3->nb[1], + (uint32_t)src4->nb[2], (uint32_t)src4->nb[3], + (uint32_t)src5->nb[2], (uint32_t)src5->nb[3], + (uint32_t)s_off, + n_head, head_dim, n_group, n_tok, + n_seq, (uint32_t) ggml_get_op_params_i32(dst, 0) + }; - for (uint32_t i = 0; i < max_tensors; ++i) { - if (!tensors[i]) { - // If any remaining descriptors are unused, just point them at src[0] - buf[i] = buf[0]; - offset[i] = 0; - continue; - } - buf_ctx[i] = (ggml_backend_vk_buffer_context *)tensors[i]->buffer->context; - buf[i] = nullptr; - offset[i] = 0; - uma[i] = false; + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer src_buf[7] = {}; + for (int i = 0; i < 7 && dst->src[i] != nullptr; i++) { + src_buf[i] = ggml_vk_tensor_subbuffer(ctx, dst->src[i]); + } - if (ctx->device->uma) { - ggml_vk_host_get(ctx->device, tensors[i]->data, buf[i], offset[i]); - uma[i] = buf[i] != nullptr; - } - if (!uma[i]) { - buf[i] = buf_ctx[i]->dev_buffer; - offset[i] = vk_tensor_offset(tensors[i]) + tensors[i]->view_offs; - } - GGML_ASSERT(buf[i] != nullptr); - } + std::array elements; - // a_offset is unused (the fused path reads from shared memory), but the rope/set_rows dst can be misaligned. - // Round the binding offset down to the storage buffer alignment; the in-element shift goes in pc.rope.d_offset. - pc.rope.d_offset = get_misalign_bytes(ctx, tensors[5]) / ggml_type_size(tensors[5]->type); - offset[5] &= ~(size_t(ctx->device->properties.limits.minStorageBufferOffsetAlignment) - 1); + const uint32_t d_state = src0->ne[0]; + uint32_t num_subgroups = d_state / ctx->device->subgroup_size; + const uint32_t num_workgroups_x = CEIL_DIV(n_head * head_dim, num_subgroups); + const uint32_t num_workgroups_y = n_seq; + elements = { num_workgroups_x, num_workgroups_y, 1 }; - std::array elements; - elements = { (uint32_t)rms->src[0]->ne[1], (uint32_t)rms->src[0]->ne[2], (uint32_t)rms->src[0]->ne[3] }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {src_buf[0], src_buf[1], src_buf[2], src_buf[3], src_buf[4], src_buf[5], src_buf[6], dst_buf}, + pc, elements); +} - static_assert(max_tensors == 7); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, - { - ggml_vk_subbuffer(ctx, buf[0], offset[0]), - ggml_vk_subbuffer(ctx, buf[1], offset[1]), - ggml_vk_subbuffer(ctx, buf[2], offset[2]), - ggml_vk_subbuffer(ctx, buf[3], offset[3]), - ggml_vk_subbuffer(ctx, buf[4], offset[4]), - ggml_vk_subbuffer(ctx, buf[5], offset[5]), - ggml_vk_subbuffer(ctx, buf[6], offset[6]), - }, pc, elements); - } else { - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_RMS_NORM, std::move(bin)); - } +void ggml_vk_ssm_conv(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx) { + ggml_tensor * conv = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = conv->src[0]; + const ggml_tensor * src1 = conv->src[1]; - if (ctx->do_add_rms_partials_offset_calculation) { - ctx->prealloc_size_add_rms_partials_offset += ggml_vk_rms_partials_size(ctx, src0); - ctx->do_add_rms_partials = false; - ctx->do_add_rms_partials_offset_calculation = false; - } -} + // Pick the destination tensor (last node in the fused chain) and the optional bias. + // Fusion modes: 0 = ssm_conv, 1 = ssm_conv+silu, 2 = ssm_conv+add(bias)+silu. + ggml_tensor * dst = conv; + const ggml_tensor * bias = nullptr; -static void ggml_vk_rms_norm_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - float * op_params = (float *)dst->op_params; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_RMS_NORM_BACK, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], op_params[0], 0.0f, 0.0f, 0.0f }); -} + if (ctx->num_additional_fused_ops == 1) { + dst = cgraph->nodes[node_idx + 1]; // silu + } else if (ctx->num_additional_fused_ops == 2) { + ggml_tensor * add = cgraph->nodes[node_idx + 1]; + bias = (add->src[0] == conv) ? add->src[1] : add->src[0]; + dst = cgraph->nodes[node_idx + 2]; // silu + } -static void ggml_vk_l2_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const float * op_params = (const float *)dst->op_params; - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = op_params[0]; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_L2_NORM, std::move(p)); -} + // The shader always declares 4 bindings; bind src0 as a dummy when bias isn't fused. + const ggml_tensor * src2 = bias ? bias : src0; -static void ggml_vk_unary(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UNARY, vk_op_unary_push_constants_init(src0, dst)); + ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_SSM_CONV, { + (uint32_t)src0->nb[1], (uint32_t)src0->nb[2], + (uint32_t)src1->nb[1], + (uint32_t)dst->nb[0], (uint32_t)dst->nb[1], (uint32_t)dst->nb[2], + (uint32_t)src1->ne[0], + (uint32_t)src0->ne[0], + (uint32_t)src0->ne[1], + (uint32_t)dst->ne[1], + (uint32_t)dst->ne[2], + }); } -static void ggml_vk_xielu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - float * op_params = (float *)dst->op_params; - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = op_params[1]; - p.param2 = op_params[2]; - p.param3 = op_params[3]; - p.param4 = op_params[4]; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UNARY, std::move(p)); -} +static void ggml_vk_op_f32_opt_step_adamw(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst, const vk_op_push_constants&& pc) { + const ggml_tensor * x = dst->src[0]; + const ggml_tensor * g = dst->src[1]; + const ggml_tensor * gm = dst->src[2]; + const ggml_tensor * gv = dst->src[3]; + const ggml_tensor * p = dst->src[4]; -static void ggml_vk_glu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const float * op_params_f = (const float *)dst->op_params; + GGML_ASSERT(x->type == GGML_TYPE_F32); + GGML_ASSERT(g->type == GGML_TYPE_F32); + GGML_ASSERT(gm->type == GGML_TYPE_F32); + GGML_ASSERT(gv->type == GGML_TYPE_F32); + GGML_ASSERT(p->type == GGML_TYPE_F32); + GGML_ASSERT(dst->buffer != nullptr); + GGML_ASSERT(ggml_is_contiguous(x)); + GGML_ASSERT(ggml_is_contiguous(g)); + GGML_ASSERT(ggml_is_contiguous(gm)); + GGML_ASSERT(ggml_is_contiguous(gv)); + GGML_ASSERT(ggml_is_contiguous(p)); + GGML_ASSERT(ggml_are_same_shape(x, g)); + GGML_ASSERT(ggml_are_same_shape(x, gm)); + GGML_ASSERT(ggml_are_same_shape(x, gv)); + GGML_ASSERT(ggml_nelements(p) == 7); - const bool swapped = (bool)dst->op_params[1]; - const bool split = src1 != nullptr; - const float alpha = op_params_f[2]; - const float limit = op_params_f[3]; + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, g, gm, gv, dst, GGML_OP_OPT_STEP_ADAMW); + GGML_ASSERT(pipeline != nullptr); - if (!split) { - GGML_ASSERT(src0->ne[0] / 2 == dst->ne[0]); - } else { - GGML_ASSERT(src0->ne[0] == src1->ne[0]); - GGML_ASSERT(src0->ne[0] == dst->ne[0]); - GGML_ASSERT(src0->type == src1->type); - } + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - const uint32_t mode = split ? 2 : (swapped ? 1 : 0); - const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = split ? ggml_type_size(src1->type) : src0_type_size; - const uint32_t dst_type_size = ggml_type_size(dst->type); + vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x); + vk_subbuffer g_buf = ggml_vk_tensor_subbuffer(ctx, g); + vk_subbuffer gm_buf = ggml_vk_tensor_subbuffer(ctx, gm); + vk_subbuffer gv_buf = ggml_vk_tensor_subbuffer(ctx, gv); + vk_subbuffer p_buf = ggml_vk_tensor_subbuffer(ctx, p); - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GLU, - { - (uint32_t)ggml_nelements(dst), - (uint32_t)src0->ne[0], - (uint32_t)dst->ne[0], - mode, - alpha, - limit, - (uint32_t)(src0->nb[0] / src0_type_size), - (uint32_t)(src0->nb[1] / src0_type_size), - (uint32_t)(src0->nb[2] / src0_type_size), - (uint32_t)(src0->nb[3] / src0_type_size), - (uint32_t)((split ? src1->nb[0] : src0->nb[0]) / src1_type_size), - (uint32_t)((split ? src1->nb[1] : src0->nb[1]) / src1_type_size), - (uint32_t)((split ? src1->nb[2] : src0->nb[2]) / src1_type_size), - (uint32_t)((split ? src1->nb[3] : src0->nb[3]) / src1_type_size), - (uint32_t)(dst->nb[0] / dst_type_size), - (uint32_t)(dst->nb[1] / dst_type_size), - (uint32_t)(dst->nb[2] / dst_type_size), - (uint32_t)(dst->nb[3] / dst_type_size), - (uint32_t)dst->ne[1], - (uint32_t)dst->ne[2], - 0, - 0, 0, 0, 0, 0, 0, - }); -} + std::array elements = { (uint32_t)ggml_nelements(x), 1, 1 }; -static void ggml_vk_diag_mask_inf(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - int32_t * op_params = (int32_t *)dst->op_params; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_DIAG_MASK_INF, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], op_params[0] }); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + {x_buf, g_buf, gm_buf, gv_buf, p_buf}, + pc, elements); } -static void ggml_vk_soft_max(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { - float * op_params = (float *)dst->op_params; - - float scale = op_params[0]; - float max_bias = op_params[1]; - - const uint32_t ncols = (uint32_t)src0->ne[0]; - const uint32_t nrows_x = (uint32_t)ggml_nrows(src0); - const uint32_t nrows_y = (uint32_t)src0->ne[1]; +void ggml_vk_opt_step_adamw(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const size_t n = ggml_nelements(dst->src[0]); - const uint32_t ne12 = src1 ? (uint32_t)(src1->ne[2]) : 0u; - const uint32_t ne13 = src1 ? (uint32_t)(src1->ne[3]) : 0u; - const uint32_t nb11 = src1 ? (uint32_t)(src1->nb[1] / src1->nb[0]) : 0u; - const uint32_t nb12 = src1 ? (uint32_t)(src1->nb[2] / src1->nb[0]) : 0u; - const uint32_t nb13 = src1 ? (uint32_t)(src1->nb[3] / src1->nb[0]) : 0u; + ggml_vk_op_f32_opt_step_adamw( + ctx, subctx, dst, + { (uint32_t)n, 0, 0.0f, 0.0f, 0.0f, 0.0f } + ); +} - const uint32_t n_head_kv = src0->ne[2]; - const uint32_t n_head_log2 = 1u << (uint32_t) floorf(log2f((float) n_head_kv)); +void ggml_vk_opt_step_sgd(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { + const size_t n = ggml_nelements(dst->src[0]); - const float m0 = powf(2.0f, -(max_bias ) / n_head_log2); - const float m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2); + ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_OPT_STEP_SGD, { (uint32_t)n, 0, 0.0f, 0.0f, 0.0f, 0.0f }); +} - vk_op_soft_max_push_constants pc { - ncols, - src1 != nullptr ? nrows_y : (uint32_t)0, - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], - ne12, ne13, - nb11, nb12, nb13, - scale, max_bias, - m0, m1, - n_head_log2, - nrows_x, - src2 != nullptr - }; +void ggml_vk_concat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + int * op_params = (int *)dst->op_params; - if (ncols <= 16384) { - ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_SOFT_MAX, std::move(pc)); - } else { + const uint32_t unit_size = ggml_vk_concat_unit_size(dst->type); + const uint32_t units_per_block = ggml_type_size(dst->type) / unit_size; + const uint32_t block_size = ggml_blck_size(dst->type); + const bool quantized = ggml_is_quantized(dst->type); - vk_subbuffer buf_a = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer buf_b = src1 ? ggml_vk_tensor_subbuffer(ctx, src1) : buf_a; - vk_subbuffer buf_c = src2 ? ggml_vk_tensor_subbuffer(ctx, src2) : buf_a; - vk_subbuffer buf_d = ggml_vk_tensor_subbuffer(ctx, dst); + // Address dimension 0 in packed storage units; higher strides may be noncontiguous. + const uint32_t ne00 = src0->ne[0] / block_size * units_per_block; + const uint32_t ne10 = src1->ne[0] / block_size * units_per_block; + const uint32_t ne20 = dst->ne[0] / block_size * units_per_block; + const uint32_t nb00 = quantized ? 1 : src0->nb[0] / unit_size; + const uint32_t nb10 = quantized ? 1 : src1->nb[0] / unit_size; + const uint32_t nb20 = quantized ? 1 : dst->nb[0] / unit_size; - uint32_t elems_per_wg = 128 * 4; - uint32_t num_wgs = CEIL_DIV(ncols, elems_per_wg); - size_t tmp_size = num_wgs * nrows_x * sizeof(float); + vk_op_concat_push_constants pc {{ + ne20 * (uint32_t)dst->ne[1] * (uint32_t)dst->ne[2] * (uint32_t)dst->ne[3], + ne00, (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], nb00, (uint32_t)src0->nb[1] / unit_size, (uint32_t)src0->nb[2] / unit_size, (uint32_t)src0->nb[3] / unit_size, + ne10, (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], nb10, (uint32_t)src1->nb[1] / unit_size, (uint32_t)src1->nb[2] / unit_size, (uint32_t)src1->nb[3] / unit_size, + ne20, (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], nb20, (uint32_t) dst->nb[1] / unit_size, (uint32_t) dst->nb[2] / unit_size, (uint32_t) dst->nb[3] / unit_size, + 0, + 0.0f, 0.0f, op_params[0], + }}; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONCAT, std::move(pc)); +} - if (ctx->prealloc_size_x < tmp_size) { - ctx->prealloc_size_x = tmp_size; - ggml_vk_preallocate_buffers(ctx, subctx); - } - if (ctx->prealloc_size_y < tmp_size) { - ctx->prealloc_size_y = tmp_size; - ggml_vk_preallocate_buffers(ctx, subctx); - } - if (ctx->prealloc_x_need_sync || ctx->prealloc_y_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); - } +void ggml_vk_upscale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t mode = (uint32_t)ggml_get_op_params_i32(dst, 0); - vk_subbuffer buf_x = { ctx->prealloc_x, 0, tmp_size }; - vk_subbuffer buf_y = { ctx->prealloc_y, 0, tmp_size }; + GGML_TENSOR_UNARY_OP_LOCALS - std::array elements = { num_wgs, nrows_x, 1 }; + float sf0 = (float)ne0 / ne00; + float sf1 = (float)ne1 / ne01; + float sf2 = (float)ne2 / ne02; + float sf3 = (float)ne3 / ne03; + float pixel_offset = 0.5f; - vk_pipeline pipeline1 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large1_f32_f16 : ctx->device->pipeline_soft_max_large1_f32; - vk_pipeline pipeline2 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large2_f32_f16 : ctx->device->pipeline_soft_max_large2_f32; - vk_pipeline pipeline3 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large3_f32_f16 : ctx->device->pipeline_soft_max_large3_f32; + if (mode & GGML_SCALE_FLAG_ALIGN_CORNERS) { + sf0 = ne0 > 1 && ne00 > 1 ? (float)(ne0 - 1) / (ne00 - 1) : sf0; + sf1 = ne1 > 1 && ne01 > 1 ? (float)(ne1 - 1) / (ne01 - 1) : sf1; + pixel_offset = 0.0f; + } - ggml_pipeline_request_descriptor_sets(ctx, pipeline1, 1); - ggml_pipeline_request_descriptor_sets(ctx, pipeline2, 1); - ggml_pipeline_request_descriptor_sets(ctx, pipeline3, 1); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UPSCALE, { + (uint32_t)ggml_nelements(dst), 0, 0, + (uint32_t)ne00, (uint32_t)ne01, + (uint32_t)nb00 / src0_type_size, (uint32_t)nb01 / src0_type_size, (uint32_t)nb02 / src0_type_size, (uint32_t)nb03 / src0_type_size, + (uint32_t)ne0, (uint32_t)ne1, (uint32_t)ne2, (uint32_t)ne3, + sf0, sf1, sf2, sf3, pixel_offset + }); +} - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline1, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); - ggml_vk_sync_buffers(ctx, subctx); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline2, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); - ggml_vk_sync_buffers(ctx, subctx); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline3, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); +void ggml_vk_scale(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = ggml_get_op_params_f32(dst, 0); + p.param2 = ggml_get_op_params_f32(dst, 1); - ctx->prealloc_x_need_sync = true; - ctx->prealloc_y_need_sync = true; - } + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SCALE, std::move(p)); } -static void ggml_vk_soft_max_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - float * op_params = (float *)dst->op_params; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SOFT_MAX_BACK, { (uint32_t)src0->ne[0], (uint32_t)ggml_nrows(src0), op_params[0], op_params[1], 0.0f, 0.0f }); +void ggml_vk_sqr(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SQR, vk_op_unary_push_constants_init(src0, dst)); } -static void ggml_vk_topk_moe(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { - topk_moe_mode mode = ctx->fused_topk_moe_mode; - const bool has_bias = mode == TOPK_MOE_SIGMOID_NORM_BIAS || mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS; - ggml_tensor * logits = cgraph->nodes[node_idx + 0]->src[0]; - ggml_tensor * bias = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? cgraph->nodes[node_idx + 2]->src[1] : - mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? cgraph->nodes[node_idx + 3]->src[1] : - logits; - ggml_tensor * weights = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; - ggml_tensor * ids = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? cgraph->nodes[node_idx + 4] : - mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? cgraph->nodes[node_idx + 5] : - mode == TOPK_MOE_LATE_SOFTMAX ? cgraph->nodes[node_idx + 1] : - cgraph->nodes[node_idx + 3]; +void ggml_vk_sqrt(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SQRT, vk_op_unary_push_constants_init(src0, dst)); +} - GGML_ASSERT(logits->type == GGML_TYPE_F32); - GGML_ASSERT(bias->type == GGML_TYPE_F32); - GGML_ASSERT(weights->type == GGML_TYPE_F32); - GGML_ASSERT(ids->type == GGML_TYPE_I32); +void ggml_vk_add1(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - const int n_experts = logits->ne[0]; - const int n_rows = logits->ne[1]; - const int n_expert_used = weights->ne[1]; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_ADD1, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); +} - GGML_ASSERT(ids->nb[1] / ggml_type_size(ids->type) == (size_t) n_experts); +void ggml_vk_arange(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_arange(dst=" << dst << ", ne=" << ggml_nelements(dst) << ")"); - vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, cgraph->nodes[node_idx], GGML_OP_SOFT_MAX); + vk_op_push_constants pc = { + (uint32_t)ggml_nelements(dst), + 1, + ggml_get_op_params_f32(dst, 0), + ggml_get_op_params_f32(dst, 2), + 0.0f, 0.0f, + }; + + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, dst, GGML_OP_ARANGE); + GGML_ASSERT(pipeline != nullptr); ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, false); - vk_subbuffer logits_buf = ggml_vk_tensor_subbuffer(ctx, logits); - vk_subbuffer bias_buf = ggml_vk_tensor_subbuffer(ctx, bias); - vk_subbuffer weights_buf = ggml_vk_tensor_subbuffer(ctx, weights); - vk_subbuffer ids_buf = ggml_vk_tensor_subbuffer(ctx, ids); + std::array elements = { (uint32_t)ggml_nelements(dst), 1, 1 }; - vk_op_topk_moe_push_constants pc {}; - pc.n_rows = n_rows; - pc.n_experts_push = n_experts; - pc.n_expert_used = n_expert_used; - pc.clamp_min = -std::numeric_limits::infinity(); - pc.clamp_max = std::numeric_limits::infinity(); - if (mode == TOPK_MOE_EARLY_SOFTMAX_NORM) { - ggml_tensor * clamp = cgraph->nodes[node_idx + 7]; - GGML_ASSERT(clamp->op == GGML_OP_CLAMP); - pc.clamp_min = ggml_get_op_params_f32(clamp, 0); - pc.clamp_max = ggml_get_op_params_f32(clamp, 1); - } - if (mode == TOPK_MOE_SIGMOID_NORM_BIAS) { - ggml_tensor * clamp = cgraph->nodes[node_idx + 8]; - GGML_ASSERT(clamp->op == GGML_OP_CLAMP); - pc.clamp_min = ggml_get_op_params_f32(clamp, 0); - pc.clamp_max = ggml_get_op_params_f32(clamp, 1); - } - if (mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS) { - ggml_tensor * clamp = cgraph->nodes[node_idx + 9]; - GGML_ASSERT(clamp->op == GGML_OP_CLAMP); - pc.clamp_min = ggml_get_op_params_f32(clamp, 0); - pc.clamp_max = ggml_get_op_params_f32(clamp, 1); - } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { dst_buf }, pc, elements); +} -#define GATING_FUNC_SOFTMAX 0 -#define GATING_FUNC_SIGMOID 1 -#define GATING_FUNC_SOFTMAX_WEIGHT 2 -#define GATING_FUNC_SQRT_SOFTPLUS 3 +void ggml_vk_fill(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + VK_LOG_DEBUG("ggml_vk_fill(dst=" << dst << ", ne=" << ggml_nelements(dst) << ")"); + const uint64_t n = ggml_nelements(dst); + GGML_ASSERT(n > 0); - pc.gating_func = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? GATING_FUNC_SIGMOID : - mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? GATING_FUNC_SQRT_SOFTPLUS : - mode == TOPK_MOE_LATE_SOFTMAX ? GATING_FUNC_SOFTMAX_WEIGHT : - GATING_FUNC_SOFTMAX; - pc.has_bias = has_bias; - pc.with_norm = mode == TOPK_MOE_EARLY_SOFTMAX_NORM || has_bias; - if (ctx->fused_topk_moe_scale) { - GGML_ASSERT(weights->op == GGML_OP_SCALE); - pc.output_scale = ggml_get_op_params_f32(weights, 0); - pc.output_bias = ggml_get_op_params_f32(weights, 1); - } else { - pc.output_scale = 1.0f; - pc.output_bias = 0.0f; - } + vk_op_push_constants pc = { + (uint32_t)n, + 1, + ggml_get_op_params_f32(dst, 0), + 0.0f, + 0.0f, 0.0f, + }; - GGML_ASSERT(n_expert_used <= n_experts); + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, dst, GGML_OP_FILL); + GGML_ASSERT(pipeline != nullptr); - const uint32_t rows_per_block = 4; - std::array elements = { CEIL_DIV(n_rows, rows_per_block), 1, 1 }; + // Split the task distribution to 2D to avoid exceeding maxComputeWorkGroupCount + const uint32_t total_wg = CEIL_DIV(n, pipeline->wg_denoms[0]); + const uint32_t wg_x = std::min(total_wg, ctx->device->properties.limits.maxComputeWorkGroupCount[0]); + const uint32_t wg_y = CEIL_DIV(total_wg, wg_x); + GGML_ASSERT(wg_y <= ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, {logits_buf, bias_buf, weights_buf, ids_buf}, pc, elements); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, false); + + std::array elements = { wg_x * pipeline->wg_denoms[0], wg_y * pipeline->wg_denoms[1], 1 }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { dst_buf }, pc, elements); } -static void ggml_vk_rope(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_cgraph * cgraph, int node_idx, bool backprop) { - ggml_tensor * dst = cgraph->nodes[node_idx]; - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; - const ggml_tensor * src2 = dst->src[2]; - const ggml_tensor * src3 = nullptr; - const int n_dims = ((int32_t *) dst->op_params)[1]; - const int mode = ((int32_t *) dst->op_params)[2]; - // const int n_ctx = ((int32_t *) dst->op_params)[3]; - const int n_ctx_orig = ((int32_t *) dst->op_params)[4]; - const float freq_base = ((float *) dst->op_params)[5]; - const float beta_fast = ((float *) dst->op_params)[9]; - const float beta_slow = ((float *) dst->op_params)[10]; - int sections[4] {}; - if (mode & GGML_ROPE_TYPE_MROPE) { - memcpy(sections, (int32_t *) dst->op_params + 11, sizeof(int)*4); - } +void ggml_vk_sin(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SIN, vk_op_unary_push_constants_init(src0, dst)); +} + +void ggml_vk_cos(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_COS, vk_op_unary_push_constants_init(src0, dst)); +} + +void ggml_vk_log(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_LOG, vk_op_unary_push_constants_init(src0, dst)); +} + +void ggml_vk_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = ggml_get_op_params_f32(dst, 0); - float corr_dims[2]; - ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_TRI, std::move(p)); +} - uint32_t set_rows_stride = 0; - // Fused rope + view + set_rows passes the set_rows destination stride in set_rows_stride - // and overrides the dst and sets src3=row_indices - if (ctx->num_additional_fused_ops > 0) { - set_rows_stride = cgraph->nodes[node_idx + 2]->nb[1] / ggml_type_size(cgraph->nodes[node_idx + 2]->type); - src3 = cgraph->nodes[node_idx + 2]->src[1]; - dst = cgraph->nodes[node_idx + 2]; - } +void ggml_vk_diag(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); - ggml_vk_op_f32(ctx, subctx, src0, src1, src2, src3, dst, GGML_OP_ROPE, - ggml_vk_make_rope_constants(cgraph->nodes[node_idx], src0, src2 != nullptr, backprop, set_rows_stride)); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_DIAG, std::move(p)); } -static void ggml_vk_argsort(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const uint32_t * op_params = (const uint32_t *)dst->op_params; +void ggml_vk_clamp(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = ggml_get_op_params_f32(dst, 0); + p.param2 = ggml_get_op_params_f32(dst, 1); - uint32_t ncols = src0->ne[0]; - uint32_t nrows = ggml_nrows(src0); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CLAMP, std::move(p)); +} - uint32_t ncols_pad_log2 = (uint32_t)ceilf(log2f(float(ncols))); - uint32_t ncolsp2 = 1 << ncols_pad_log2; +void ggml_vk_pad(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_pad_push_constants p = vk_op_pad_push_constants_init(src0, dst); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_PAD, std::move(p)); +} - vk_op_argsort_push_constants pc { ncols, ncolsp2, ncols_pad_log2, nrows, op_params[0], 0, 0, 0, 0, }; +void ggml_vk_pad_reflect_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const uint32_t p0 = (uint32_t)dst->op_params[0]; + const uint32_t p1 = (uint32_t)dst->op_params[1]; - // Pick the largest workgroup size <= ncolsp2 - uint32_t pipeline_idx = std::min(ncols_pad_log2, num_argsort_pipelines - 1); + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); + memcpy(&p.param1, &p0, sizeof(float)); + memcpy(&p.param2, &p1, sizeof(float)); - // Use the "small" argsort shader if the whole sort can be done by a single workgroup. - bool use_small = ncols_pad_log2 <= ctx->device->max_workgroup_size_log2 && - ctx->device->pipeline_argsort_f32[pipeline_idx] != nullptr; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_PAD_REFLECT_1D, std::move(p)); +} - vk_pipeline pipeline = use_small ? ctx->device->pipeline_argsort_f32[pipeline_idx] - : ctx->device->pipeline_argsort_large_f32[pipeline_idx]; +void ggml_vk_roll(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const int32_t s0 = ggml_get_op_params_i32(dst, 0); + const int32_t s1 = ggml_get_op_params_i32(dst, 1); + const int32_t s2 = ggml_get_op_params_i32(dst, 2); + const int32_t s3 = ggml_get_op_params_i32(dst, 3); + const uint32_t s01_packed = ((s0 + 0x8000) << 16) | (s1 + 0x8000); + const uint32_t s23_packed = ((s2 + 0x8000) << 16) | (s3 + 0x8000); - vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer subbuf1 = dst_buf; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + memcpy(&p.param1, &s01_packed, sizeof(float)); + memcpy(&p.param2, &s23_packed, sizeof(float)); - // Reserve space for ivec2 per element, with rows padded to a power of two - if (!use_small) { - const size_t x_sz = size_t{ncolsp2} * nrows * 2 * sizeof(int); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_ROLL, std::move(p)); +} - if (ctx->prealloc_size_x < x_sz) { - ctx->prealloc_size_x = x_sz; - ggml_vk_preallocate_buffers(ctx, subctx); - } - if (ctx->prealloc_x_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); +void ggml_vk_repeat(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_REPEAT, std::move(p)); +} + +void ggml_vk_repeat_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ggml_nelements(dst)); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_REPEAT_BACK, std::move(p)); +} + +void ggml_vk_cpy(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + uint32_t ne = (uint32_t)ggml_nelements(src0); + if (ggml_is_quantized(src0->type) && ggml_is_quantized(dst->type)) { + // Convert from number of logical elements to 2- or 4-byte units. + ne /= ggml_blck_size(src0->type); + if ((ggml_type_size(src0->type) % 4) == 0) { + ne *= ggml_type_size(src0->type) / 4; + } else { + ne *= ggml_type_size(src0->type) / 2; } - subbuf1 = { ctx->prealloc_x, 0, ctx->prealloc_x->size }; } - std::array elements; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst, ne); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CPY, std::move(p)); +} - elements[0] = ncolsp2; - elements[1] = std::min((uint32_t)ggml_nrows(src0), ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - elements[2] = 1; +void ggml_vk_set_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); - // First dispatch initializes tmp_idx and does the first N passes where - // there is only communication between threads in the same workgroup. - { - vk_op_argsort_push_constants pc2 = pc; - pc2.outer_start = 0; - pc2.outer_end = std::min(ncols_pad_log2, ctx->device->max_workgroup_size_log2); - pc2.inner_start = 0; - pc2.inner_end = 100; - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc2, elements); - } - if (!use_small) { - ggml_vk_sync_buffers(ctx, subctx); - // Loop over outer/inner passes, synchronizing between each pass. - for (uint32_t outer = ctx->device->max_workgroup_size_log2; outer < ncols_pad_log2; ++outer) { - for (uint32_t inner = 0; inner < outer + 1; ++inner) { - vk_op_argsort_push_constants pc2 = pc; - pc2.outer_start = outer; - pc2.outer_end = outer + 1; - pc2.inner_start = inner; - pc2.inner_end = inner + 1; - // When the inner idx is large enough, there's only communication - // within a workgroup. So the remaining inner iterations can all - // run in the same dispatch. - if (outer - inner < pipeline_idx) { - pc2.inner_end = 100; - inner = outer; - pipeline = ctx->device->pipeline_argsort_large_f32[pipeline_idx]; - } else { - // Smaller workgroup empirically seems to perform better - pipeline = ctx->device->pipeline_argsort_large_f32[pipeline_idx - 2]; - } - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc2, elements); - ggml_vk_sync_buffers(ctx, subctx); - } - } - ctx->prealloc_x_need_sync = true; + // Skip empty skip_rows operations. For most ops the empty check at the start + // of ggml_vk_build_graph is sufficient, but set_rows can have a nonempty dst + // with empty srcs. + if (ggml_is_empty(src0) || ggml_is_empty(src1)) { + return; } + + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SET_ROWS, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); } -static void ggml_vk_topk(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - uint32_t ncols = src0->ne[0]; - uint32_t nrows = ggml_nrows(src0); - uint32_t k = dst->ne[0]; +void ggml_vk_silu_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SILU_BACK, { (uint32_t)ggml_nelements(src0), 0, 0.0f, 0.0f, 0.0f, 0.0f }); +} - vk_op_topk_push_constants pc { ncols, ncols, ncols, k, nrows, 0, 0 }; +void ggml_vk_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + float * op_params = (float *)dst->op_params; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = op_params[0]; - if (ctx->prealloc_x_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_NORM, std::move(p)); +} + +void ggml_vk_group_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const int * int_op_params = (const int *)dst->op_params; + const float * float_op_params = (const float *)dst->op_params; + + const uint32_t num_groups = int_op_params[0]; + const float eps = float_op_params[1]; + const uint32_t group_size = src0->ne[0] * src0->ne[1] * ((src0->ne[2] + num_groups - 1) / num_groups); + + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_GROUP_NORM, { group_size, 0, eps, 0.0f, 0.0f, 0.0f }); +} + +static uint32_t ggml_vk_rms_num_partials(ggml_backend_vk_context * ctx, const ggml_tensor *node) { + const uint32_t ne = (uint32_t)node->ne[0]; + const uint32_t denom = ctx->device->pipeline_add_rms[0][0][0]->wg_denoms[0]; + const uint32_t num_partials = CEIL_DIV(ne, denom); + return num_partials; +} + +uint32_t ggml_vk_rms_partials_size(ggml_backend_vk_context * ctx, const ggml_tensor *node) { + const uint32_t num_partials = ggml_vk_rms_num_partials(ctx, node); + const uint32_t num_bytes = ROUNDUP_POW2(num_partials * sizeof(uint32_t), ctx->device->partials_binding_alignment); + return num_bytes; +} + +static vk_op_rope_push_constants ggml_vk_make_rope_constants(const ggml_tensor *dst, const ggml_tensor *src0, const bool has_ff, bool backprop, const uint32_t set_rows_stride) { + const int n_dims = ((const int32_t *) dst->op_params)[1]; + const int mode = ((const int32_t *) dst->op_params)[2]; + const int n_offs = ((const int32_t *) dst->op_params)[15]; + // const int n_ctx = ((const int32_t *) dst->op_params)[3]; + const int n_ctx_orig = ((const int32_t *) dst->op_params)[4]; + const float freq_base = ((const float *) dst->op_params)[5]; + const float freq_scale = ((const float *) dst->op_params)[6]; + const float ext_factor = ((const float *) dst->op_params)[7]; + const float attn_factor = ((const float *) dst->op_params)[8]; + const float beta_fast = ((const float *) dst->op_params)[9]; + const float beta_slow = ((const float *) dst->op_params)[10]; + int sections[4] {}; + if (mode & GGML_ROPE_TYPE_MROPE) { + memcpy(sections, (const int32_t *) dst->op_params + 11, sizeof(int)*4); } - std::array elements; - elements[1] = std::min(nrows, ctx->device->properties.limits.maxComputeWorkGroupCount[1]); - elements[2] = 1; + const bool is_imrope = mode == GGML_ROPE_TYPE_IMROPE; - uint32_t num_elements = ncols; + float corr_dims[2]; + ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims); - // Each iteration reduces a workgroup's worth of elements down to the K - // largest elements. Repeat until we have the top K elements. - // Need to do at least one iteration to write out the results. - bool done_one_iter = false; - uint32_t dbl_buf_index = 0; - size_t dbl_buf_size; - while (num_elements > k || !done_one_iter) { + const float theta_scale = powf(freq_base, -2.0f/n_dims); - // Prefer going as small as num_topk_pipelines - 3 for perf reasons. - // But if K is larger, then we need a larger workgroup - uint32_t max_pipeline = num_topk_pipelines - 1; - uint32_t preferred_pipeline = std::max(num_topk_pipelines - 3, (uint32_t)log2f(float(k)) + 2); - max_pipeline = std::min(preferred_pipeline, max_pipeline); - uint32_t min_pipeline = (uint32_t)log2f(float(k)) + 1; - // require full subgroup - min_pipeline = std::max(min_pipeline, ctx->device->subgroup_size_log2); + uint32_t nb01 = src0->nb[1] / ggml_type_size(src0->type); + uint32_t nb02 = src0->nb[2] / ggml_type_size(src0->type); + uint32_t nb03 = src0->nb[3] / ggml_type_size(src0->type); - uint32_t pipeline_idx = (uint32_t)ceilf(log2f(float(num_elements))); - pipeline_idx = std::min(pipeline_idx, max_pipeline); - pipeline_idx = std::max(pipeline_idx, min_pipeline); + uint32_t nb11 = dst->nb[1] / ggml_type_size(dst->type); + uint32_t nb12 = dst->nb[2] / ggml_type_size(dst->type); + uint32_t nb13 = dst->nb[3] / ggml_type_size(dst->type); - if (num_elements > (1u << pipeline_idx)) { - // If we could finish on this loop iteration (i.e. a single workgroup) - // then do so. It's better than the overhead of another pass. - for (uint32_t i = pipeline_idx; i < num_topk_pipelines; ++i) { - if (num_elements <= (1u << i)) { - pipeline_idx = i; - break; - } - } - } + vk_op_rope_push_constants rope { + (uint32_t)mode, (uint32_t)ggml_nrows(src0), (uint32_t)n_dims, (uint32_t)n_offs, freq_scale, + freq_base, ext_factor, attn_factor, {corr_dims[0], corr_dims[1]}, theta_scale, has_ff, + { sections[0], sections[1], sections[2], sections[3] }, is_imrope, backprop, set_rows_stride, - vk_pipeline pipeline = ctx->device->pipeline_topk_f32[pipeline_idx]; - // If the device doesn't support a pipeline this large, use smaller - while (!pipeline) { - pipeline_idx--; - GGML_ASSERT(pipeline_idx >= min_pipeline); - pipeline = ctx->device->pipeline_topk_f32[pipeline_idx]; - } + (uint32_t)src0->ne[0], + (uint32_t)src0->ne[1], + (uint32_t)src0->ne[2], + nb01, nb02, nb03, + nb11, nb12, nb13, + 0, 0, // a_offset, d_offset filled in by init_pushconst_tensor_offsets + }; - vk_op_topk_push_constants pc2 = pc; - pc2.ncols_input = num_elements; + return rope; +} + +static void ggml_vk_rms_norm_finish(ggml_backend_vk_context * ctx, const ggml_tensor * src0) { + if (ctx->do_add_rms_partials_offset_calculation) { + ctx->prealloc_size_add_rms_partials_offset += ggml_vk_rms_partials_size(ctx, src0); + ctx->do_add_rms_partials = false; + ctx->do_add_rms_partials_offset_calculation = false; + } +} - // Number of elements remaining after this pass - uint32_t num_dst_elements = (num_elements / pipeline->wg_denoms[0]) * k + std::min(k, num_elements % pipeline->wg_denoms[0]); +void ggml_vk_rms_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const struct ggml_cgraph * cgraph, int node_idx, float * op_params) { + ggml_tensor * rms = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = rms->src[0]; - pc2.ncols_output = num_dst_elements; + if (ctx->fused_rms_norm_mode == RMS_NORM_VIEW_SET_ROWS) { + GGML_ASSERT(ctx->num_additional_fused_ops == 2); + ggml_tensor * set_rows = cgraph->nodes[node_idx + 2]; + const ggml_tensor * indices = set_rows->src[1]; + vk_op_binary_push_constants pc = ggml_vk_rms_norm_push_constants(src0, src0, set_rows, op_params[0], 0); + init_pushconst_tensor_offsets(ctx, pc, src0, src0, nullptr, nullptr, set_rows); - if (!done_one_iter) { - // Reserve space for ivec2 per element, double buffered - // K per workgroup per row - dbl_buf_size = num_dst_elements * nrows * 2 * sizeof(int); - dbl_buf_size = ROUNDUP_POW2(dbl_buf_size, ctx->device->properties.limits.minStorageBufferOffsetAlignment); - const size_t x_sz = dbl_buf_size * 2; + vk_pipeline pipeline = set_rows->type == GGML_TYPE_F16 ? + ctx->device->pipeline_rms_norm_set_rows_f32_f16 : ctx->device->pipeline_rms_norm_set_rows_f32_f32; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + ggml_vk_tensor_subbuffer(ctx, src0, true), + ggml_vk_tensor_subbuffer(ctx, src0, true), + ggml_vk_tensor_subbuffer(ctx, set_rows, true), + ggml_vk_tensor_subbuffer(ctx, indices), + }, pc, { (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], (uint32_t)src0->ne[3] }); + ggml_vk_rms_norm_finish(ctx, src0); + return; + } - if (ctx->prealloc_size_x < x_sz) { - ctx->prealloc_size_x = x_sz; - ggml_vk_preallocate_buffers(ctx, subctx); - } - } + if (ctx->fused_rms_norm_mode == RMS_NORM_MUL_ADD || ctx->fused_rms_norm_mode == RMS_NORM_MUL_ADD_MUL) { + ggml_tensor * mul = cgraph->nodes[node_idx + 1]; + ggml_tensor * add = cgraph->nodes[node_idx + 2]; + const ggml_tensor * weight = mul->src[0] == rms ? mul->src[1] : mul->src[0]; + const ggml_tensor * residual = add->src[0] == mul ? add->src[1] : add->src[0]; + const bool do_post_multiply = ctx->fused_rms_norm_mode == RMS_NORM_MUL_ADD_MUL; + GGML_ASSERT(ctx->num_additional_fused_ops == (do_post_multiply ? 3 : 2)); + ggml_tensor * dst = do_post_multiply ? cgraph->nodes[node_idx + 3] : add; + const ggml_tensor * post_scale = do_post_multiply ? + (dst->src[0] == add ? dst->src[1] : dst->src[0]) : src0; - vk_subbuffer src_buf; - vk_subbuffer dst_buf; + const uint32_t num_partials = ctx->do_add_rms_partials ? ggml_vk_rms_num_partials(ctx, dst) : 0; + vk_op_binary_push_constants pc = ggml_vk_rms_norm_push_constants(src0, weight, dst, op_params[0], num_partials); + init_pushconst_tensor_offsets(ctx, pc, src0, weight, residual, post_scale, dst); - if (num_elements == ncols) { - pc2.first_pass = 1; - src_buf = ggml_vk_tensor_subbuffer(ctx, src0); + vk_pipeline pipeline; + if (ctx->do_add_rms_partials) { + pipeline = do_post_multiply ? + ctx->device->pipeline_rms_norm_mul_add_mul_partials_f32 : ctx->device->pipeline_rms_norm_mul_add_partials_f32; } else { - src_buf = { ctx->prealloc_x, dbl_buf_index * dbl_buf_size, dbl_buf_size }; + pipeline = do_post_multiply ? + ctx->device->pipeline_rms_norm_mul_add_mul_f32 : ctx->device->pipeline_rms_norm_mul_add_f32; } - if (num_dst_elements == k) { - pc2.last_pass = 1; - dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + if (ctx->do_add_rms_partials) { + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + ggml_vk_tensor_subbuffer(ctx, src0, true), + ggml_vk_tensor_subbuffer(ctx, weight, true), + ggml_vk_tensor_subbuffer(ctx, dst, true), + ggml_vk_subbuffer(ctx, ctx->prealloc_add_rms_partials, ctx->prealloc_size_add_rms_partials_offset), + ggml_vk_tensor_subbuffer(ctx, residual), + ggml_vk_tensor_subbuffer(ctx, post_scale), + }, pc, { (uint32_t)CEIL_DIV(src0->ne[0], 128), 1, 1 }); } else { - dst_buf = { ctx->prealloc_x, (dbl_buf_index ^ 1) * dbl_buf_size, dbl_buf_size }; - } + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + ggml_vk_tensor_subbuffer(ctx, src0, true), + ggml_vk_tensor_subbuffer(ctx, weight, true), + ggml_vk_tensor_subbuffer(ctx, dst, true), + ggml_vk_tensor_subbuffer(ctx, residual), + ggml_vk_tensor_subbuffer(ctx, post_scale), + }, pc, { (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], (uint32_t)src0->ne[3] }); + } + ggml_vk_rms_norm_finish(ctx, src0); + return; + } - elements[0] = num_elements; + ggml_tensor * dst; + const ggml_tensor * src1; - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src_buf, dst_buf }, pc2, elements); - num_elements = num_dst_elements; - dbl_buf_index ^= 1; - if (num_elements > k) { - ggml_vk_sync_buffers(ctx, subctx); - } - done_one_iter = true; + if (ctx->fused_rms_norm_mode != RMS_NORM_COUNT) { + ggml_tensor * mul = cgraph->nodes[node_idx + 1]; + dst = mul; + src1 = mul->src[0] == rms ? mul->src[1] : mul->src[0]; + } else { + dst = rms; + src1 = src0; } - ctx->prealloc_x_need_sync = true; -} -static void ggml_vk_sum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, ggml_nelements(src0)); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SUM, p); -} + const uint32_t num_partials = ctx->do_add_rms_partials ? ggml_vk_rms_num_partials(ctx, dst) : 0; + vk_op_binary_push_constants bin = ggml_vk_rms_norm_push_constants(src0, src1, dst, op_params[0], num_partials); -static void ggml_vk_sum_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SUM_ROWS, p); -} + if (ctx->fused_rms_norm_mode == RMS_NORM_MUL_ROPE || + ctx->fused_rms_norm_mode == RMS_NORM_MUL_ROPE_VIEW_SET_ROWS) { + static constexpr uint32_t max_tensors = 7; + const ggml_tensor *tensors[max_tensors] {}; -static void ggml_vk_mean(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); - p.weight = 1.0f / (float)src0->ne[0]; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_MEAN, p); -} + ggml_tensor *rms = cgraph->nodes[node_idx + 0]; + ggml_tensor *mul = cgraph->nodes[node_idx + 1]; + ggml_tensor *rope = cgraph->nodes[node_idx + 2]; -static void ggml_vk_cumsum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - vk_op_sum_rows_push_constants pc = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); - // Use the single pass shader when the rows are small or there are enough rows to fill the GPU. - // For fewer, larger rows, use the multipass shader to spread each row across SMs. - if (dst->ne[0] <= 4096 || ggml_nrows(dst) >= ctx->device->shader_core_count) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CUMSUM, pc); - return; - } + ggml_tensor *other_src = mul->src[0] == rms ? mul->src[1] : mul->src[0]; - // First pass computes partial sums within a block, and stores the last partial - // to the temp buffer. Second pass sums the block partials from the temp buffer - // and adds that to the result of the first pass. - vk_pipeline pipeline1 = ctx->device->pipeline_cumsum_multipass1_f32; - vk_pipeline pipeline2 = ctx->device->pipeline_cumsum_multipass2_f32; - GGML_ASSERT(pipeline1 != nullptr && pipeline2 != nullptr); + bool do_set_rows = ctx->fused_rms_norm_mode == RMS_NORM_MUL_ROPE_VIEW_SET_ROWS; + GGML_ASSERT(ctx->num_additional_fused_ops == (do_set_rows ? 4 : 2)); - ggml_pipeline_request_descriptor_sets(ctx, pipeline1, 1); - ggml_pipeline_request_descriptor_sets(ctx, pipeline2, 1); + tensors[0] = rms->src[0]; + tensors[1] = other_src; + tensors[2] = mul; + tensors[3] = rope->src[1]; // pos + tensors[4] = rope->src[2]; // ff + tensors[5] = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; // dst + tensors[6] = do_set_rows ? tensors[5]->src[1] : nullptr; + const uint32_t set_rows_stride = do_set_rows ? tensors[5]->nb[1] / ggml_type_size(tensors[5]->type) : 0; - std::array elements; + vk_op_rms_norm_mul_rope_push_constants pc; + pc.bin = bin; + pc.rope = ggml_vk_make_rope_constants(rope, rope->src[0], tensors[4] != nullptr, false, set_rows_stride); - elements[0] = dst->ne[0]; - elements[1] = (uint32_t)ggml_nrows(dst); - elements[2] = 1; + vk_pipeline pipeline = tensors[5]->type == GGML_TYPE_F16 ? ctx->device->pipeline_rms_norm_mul_rope_f32_f16 : ctx->device->pipeline_rms_norm_mul_rope_f32_f32; - size_t temp_size = sizeof(float) * elements[0] * ggml_nrows(dst); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - if (ctx->prealloc_size_split_k < temp_size) { - ctx->prealloc_size_split_k = temp_size; - ggml_vk_preallocate_buffers(ctx, subctx); - } + ggml_backend_vk_buffer_context * buf_ctx[max_tensors]; + vk_buffer buf[max_tensors]; + size_t offset[max_tensors]; + bool uma[max_tensors]; - vk_subbuffer src_buf = ggml_vk_tensor_subbuffer(ctx, src0); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - vk_subbuffer temp_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_split_k, 0); + for (uint32_t i = 0; i < max_tensors; ++i) { + if (!tensors[i]) { + // If any remaining descriptors are unused, just point them at src[0] + buf[i] = buf[0]; + offset[i] = 0; + continue; + } + buf_ctx[i] = (ggml_backend_vk_buffer_context *)tensors[i]->buffer->context; + buf[i] = nullptr; + offset[i] = 0; + uma[i] = false; - if (ctx->prealloc_split_k_need_sync) { - ggml_vk_sync_buffers(ctx, subctx); + if (ctx->device->uma) { + ggml_vk_host_get(ctx->device, tensors[i]->data, buf[i], offset[i]); + uma[i] = buf[i] != nullptr; + } + if (!uma[i]) { + buf[i] = buf_ctx[i]->dev_buffer; + offset[i] = vk_tensor_offset(tensors[i]) + tensors[i]->view_offs; + } + GGML_ASSERT(buf[i] != nullptr); + } + + // a_offset is unused (the fused path reads from shared memory), but the rope/set_rows dst can be misaligned. + // Round the binding offset down to the storage buffer alignment; the in-element shift goes in pc.rope.d_offset. + pc.rope.d_offset = get_misalign_bytes(ctx, tensors[5]) / ggml_type_size(tensors[5]->type); + offset[5] &= ~(size_t(ctx->device->properties.limits.minStorageBufferOffsetAlignment) - 1); + + std::array elements; + elements = { (uint32_t)rms->src[0]->ne[1], (uint32_t)rms->src[0]->ne[2], (uint32_t)rms->src[0]->ne[3] }; + + static_assert(max_tensors == 7); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { + ggml_vk_subbuffer(ctx, buf[0], offset[0]), + ggml_vk_subbuffer(ctx, buf[1], offset[1]), + ggml_vk_subbuffer(ctx, buf[2], offset[2]), + ggml_vk_subbuffer(ctx, buf[3], offset[3]), + ggml_vk_subbuffer(ctx, buf[4], offset[4]), + ggml_vk_subbuffer(ctx, buf[5], offset[5]), + ggml_vk_subbuffer(ctx, buf[6], offset[6]), + }, pc, elements); + } else { + GGML_ASSERT(ctx->fused_rms_norm_mode == RMS_NORM_MUL || ctx->fused_rms_norm_mode == RMS_NORM_COUNT); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_RMS_NORM, std::move(bin)); } - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline1, {src_buf, dst_buf, temp_buf}, pc, elements); - ggml_vk_sync_buffers(ctx, subctx); - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline2, {src_buf, dst_buf, temp_buf}, pc, elements); + ggml_vk_rms_norm_finish(ctx, src0); +} - ctx->prealloc_split_k_need_sync = true; +void ggml_vk_rms_norm_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + float * op_params = (float *)dst->op_params; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_RMS_NORM_BACK, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], op_params[0], 0.0f, 0.0f, 0.0f }); } -static void ggml_vk_argmax(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_ARGMAX, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], 0.0f, 0.0f, 0.0f, 0.0f }); +void ggml_vk_l2_norm(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const float * op_params = (const float *)dst->op_params; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = op_params[0]; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_L2_NORM, std::move(p)); } -static void ggml_vk_count_equal(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_COUNT_EQUAL, { (uint32_t)ggml_nelements(src0), 0, 0.0f, 0.0f, 0.0f, 0.0f }); +void ggml_vk_unary(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UNARY, vk_op_unary_push_constants_init(src0, dst)); +} + +void ggml_vk_xielu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + float * op_params = (float *)dst->op_params; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = op_params[1]; + p.param2 = op_params[2]; + p.param3 = op_params[3]; + p.param4 = op_params[4]; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_UNARY, std::move(p)); } -static void ggml_vk_solve_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { +void ggml_vk_glu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const float * op_params_f = (const float *)dst->op_params; + + const bool swapped = (bool)dst->op_params[1]; + const bool split = src1 != nullptr; + const float alpha = op_params_f[2]; + const float limit = op_params_f[3]; + + if (!split) { + GGML_ASSERT(src0->ne[0] / 2 == dst->ne[0]); + } else { + GGML_ASSERT(src0->ne[0] == src1->ne[0]); + GGML_ASSERT(src0->ne[0] == dst->ne[0]); + GGML_ASSERT(src0->type == src1->type); + } + + const uint32_t mode = split ? 2 : (swapped ? 1 : 0); const uint32_t src0_type_size = ggml_type_size(src0->type); - const uint32_t src1_type_size = ggml_type_size(src1->type); - const uint32_t dst_type_size = ggml_type_size(dst->type); + const uint32_t src1_type_size = split ? ggml_type_size(src1->type) : src0_type_size; + const uint32_t dst_type_size = ggml_type_size(dst->type); - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SOLVE_TRI, { - (uint32_t)ggml_nelements(src0), - (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, - (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, - (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, - 0, - 0.0f, 0.0f, 0, - }); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_GLU, + { + (uint32_t)ggml_nelements(dst), + (uint32_t)src0->ne[0], + (uint32_t)dst->ne[0], + mode, + alpha, + limit, + (uint32_t)(src0->nb[0] / src0_type_size), + (uint32_t)(src0->nb[1] / src0_type_size), + (uint32_t)(src0->nb[2] / src0_type_size), + (uint32_t)(src0->nb[3] / src0_type_size), + (uint32_t)((split ? src1->nb[0] : src0->nb[0]) / src1_type_size), + (uint32_t)((split ? src1->nb[1] : src0->nb[1]) / src1_type_size), + (uint32_t)((split ? src1->nb[2] : src0->nb[2]) / src1_type_size), + (uint32_t)((split ? src1->nb[3] : src0->nb[3]) / src1_type_size), + (uint32_t)(dst->nb[0] / dst_type_size), + (uint32_t)(dst->nb[1] / dst_type_size), + (uint32_t)(dst->nb[2] / dst_type_size), + (uint32_t)(dst->nb[3] / dst_type_size), + (uint32_t)dst->ne[1], + (uint32_t)dst->ne[2], + 0, + 0, 0, 0, 0, 0, 0, + }); } -static void ggml_vk_im2col(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const int32_t s0 = dst->op_params[0]; - const int32_t s1 = dst->op_params[1]; - const int32_t p0 = dst->op_params[2]; - const int32_t p1 = dst->op_params[3]; - const int32_t d0 = dst->op_params[4]; - const int32_t d1 = dst->op_params[5]; +void ggml_vk_diag_mask_inf(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + int32_t * op_params = (int32_t *)dst->op_params; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_DIAG_MASK_INF, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], op_params[0] }); +} + +void ggml_vk_soft_max(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * src2, ggml_tensor * dst) { + float * op_params = (float *)dst->op_params; + + float scale = op_params[0]; + float max_bias = op_params[1]; + + const uint32_t ncols = (uint32_t)src0->ne[0]; + const uint32_t nrows_x = (uint32_t)ggml_nrows(src0); + const uint32_t nrows_y = (uint32_t)src0->ne[1]; + + const uint32_t ne12 = src1 ? (uint32_t)(src1->ne[2]) : 0u; + const uint32_t ne13 = src1 ? (uint32_t)(src1->ne[3]) : 0u; + const uint32_t nb11 = src1 ? (uint32_t)(src1->nb[1] / src1->nb[0]) : 0u; + const uint32_t nb12 = src1 ? (uint32_t)(src1->nb[2] / src1->nb[0]) : 0u; + const uint32_t nb13 = src1 ? (uint32_t)(src1->nb[3] / src1->nb[0]) : 0u; - const bool is_2D = dst->op_params[6] == 1; + const uint32_t n_head_kv = src0->ne[2]; + const uint32_t n_head_log2 = 1u << (uint32_t) floorf(log2f((float) n_head_kv)); - const uint32_t IC = src1->ne[is_2D ? 2 : 1]; - const uint32_t IH = is_2D ? src1->ne[1] : 1; - const uint32_t IW = src1->ne[0]; + const float m0 = powf(2.0f, -(max_bias ) / n_head_log2); + const float m1 = powf(2.0f, -(max_bias / 2.0f) / n_head_log2); - const uint32_t KH = is_2D ? src0->ne[1] : 1; - const uint32_t KW = src0->ne[0]; + vk_op_soft_max_push_constants pc { + ncols, + src1 != nullptr ? nrows_y : (uint32_t)0, + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2], + ne12, ne13, + nb11, nb12, nb13, + scale, max_bias, + m0, m1, + n_head_log2, + nrows_x, + src2 != nullptr + }; - const uint32_t OH = is_2D ? dst->ne[2] : 1; - const uint32_t OW = dst->ne[1]; + if (ncols <= 16384) { + ggml_vk_op_f32(ctx, subctx, src0, src1, src2, nullptr, dst, GGML_OP_SOFT_MAX, std::move(pc)); + } else { - const uint32_t offset_delta = src1->nb[is_2D ? 2 : 1] / 4; // nb is byte offset, src is type float32 - const uint32_t batch_offset = src1->nb[is_2D ? 3 : 2] / 4; // nb is byte offset, src is type float32 + vk_subbuffer buf_a = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer buf_b = src1 ? ggml_vk_tensor_subbuffer(ctx, src1) : buf_a; + vk_subbuffer buf_c = src2 ? ggml_vk_tensor_subbuffer(ctx, src2) : buf_a; + vk_subbuffer buf_d = ggml_vk_tensor_subbuffer(ctx, dst); - const uint32_t batch = src1->ne[is_2D ? 3 : 2]; + uint32_t elems_per_wg = 128 * 4; + uint32_t num_wgs = CEIL_DIV(ncols, elems_per_wg); + size_t tmp_size = num_wgs * nrows_x * sizeof(float); - const ggml_backend_vk_buffer_context * d_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; - const vk_buffer d_buf = d_buf_ctx->dev_buffer; + if (ctx->prealloc_size_x < tmp_size) { + ctx->prealloc_size_x = tmp_size; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if (ctx->prealloc_size_y < tmp_size) { + ctx->prealloc_size_y = tmp_size; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if (ctx->prealloc_x_need_sync || ctx->prealloc_y_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - const vk::DeviceAddress dst_addr = d_buf->bda_addr + vk_tensor_offset(dst) + dst->view_offs; + vk_subbuffer buf_x = { ctx->prealloc_x, 0, tmp_size }; + vk_subbuffer buf_y = { ctx->prealloc_y, 0, tmp_size }; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_IM2COL, { - dst_addr, - batch_offset, offset_delta, - IC, IW, IH, OW, OH, KW, KH, - OH * batch, - IC * KH * KW, - s0, s1, p0, p1, d0, d1, batch * IC - }); -} + std::array elements = { num_wgs, nrows_x, 1 }; -static void ggml_vk_im2col_3d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - GGML_TENSOR_BINARY_OP_LOCALS + vk_pipeline pipeline1 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large1_f32_f16 : ctx->device->pipeline_soft_max_large1_f32; + vk_pipeline pipeline2 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large2_f32_f16 : ctx->device->pipeline_soft_max_large2_f32; + vk_pipeline pipeline3 = src1 && src1->type == GGML_TYPE_F16 ? ctx->device->pipeline_soft_max_large3_f32_f16 : ctx->device->pipeline_soft_max_large3_f32; - const int32_t s0 = ((const int32_t *)(dst->op_params))[0]; - const int32_t s1 = ((const int32_t *)(dst->op_params))[1]; - const int32_t s2 = ((const int32_t *)(dst->op_params))[2]; - const int32_t p0 = ((const int32_t *)(dst->op_params))[3]; - const int32_t p1 = ((const int32_t *)(dst->op_params))[4]; - const int32_t p2 = ((const int32_t *)(dst->op_params))[5]; - const int32_t d0 = ((const int32_t *)(dst->op_params))[6]; - const int32_t d1 = ((const int32_t *)(dst->op_params))[7]; - const int32_t d2 = ((const int32_t *)(dst->op_params))[8]; - const int32_t IC = ((const int32_t *)(dst->op_params))[9]; + ggml_pipeline_request_descriptor_sets(ctx, pipeline1, 1); + ggml_pipeline_request_descriptor_sets(ctx, pipeline2, 1); + ggml_pipeline_request_descriptor_sets(ctx, pipeline3, 1); - const int64_t N = ne13 / IC; - const int64_t ID = ne12; - const int64_t IH = ne11; - const int64_t IW = ne10; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline1, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); + ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline2, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); + ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline3, { buf_a, buf_b, buf_c, buf_d, buf_x, buf_y }, pc, elements); - const int64_t KD = ne02; - const int64_t KH = ne01; - const int64_t KW = ne00; + ctx->prealloc_x_need_sync = true; + ctx->prealloc_y_need_sync = true; + } +} - const int64_t OD = ne3 / N; - const int64_t OH = ne2; - const int64_t OW = ne1; +void ggml_vk_soft_max_back(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + float * op_params = (float *)dst->op_params; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SOFT_MAX_BACK, { (uint32_t)src0->ne[0], (uint32_t)ggml_nrows(src0), op_params[0], op_params[1], 0.0f, 0.0f }); +} - const ggml_backend_vk_buffer_context * d_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; - const vk_buffer d_buf = d_buf_ctx->dev_buffer; +void ggml_vk_topk_moe(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { + topk_moe_mode mode = ctx->fused_topk_moe_mode; + const bool has_bias = mode == TOPK_MOE_SIGMOID_NORM_BIAS || mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS; + ggml_tensor * logits = cgraph->nodes[node_idx + 0]->src[0]; + ggml_tensor * bias = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? cgraph->nodes[node_idx + 2]->src[1] : + mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? cgraph->nodes[node_idx + 3]->src[1] : + logits; + ggml_tensor * weights = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; + ggml_tensor * ids = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? cgraph->nodes[node_idx + 4] : + mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? cgraph->nodes[node_idx + 5] : + mode == TOPK_MOE_LATE_SOFTMAX ? cgraph->nodes[node_idx + 1] : + cgraph->nodes[node_idx + 3]; - const vk::DeviceAddress dst_addr = d_buf->bda_addr + vk_tensor_offset(dst) + dst->view_offs; + GGML_ASSERT(logits->type == GGML_TYPE_F32); + GGML_ASSERT(bias->type == GGML_TYPE_F32); + GGML_ASSERT(weights->type == GGML_TYPE_F32); + GGML_ASSERT(ids->type == GGML_TYPE_I32); - vk_op_im2col_3d_push_constants pc {}; + const int n_experts = logits->ne[0]; + const int n_rows = logits->ne[1]; + const int n_expert_used = weights->ne[1]; - pc.dst_addr = dst_addr; - pc.nb10 = nb10 / ggml_type_size(src1->type); - pc.nb11 = nb11 / ggml_type_size(src1->type); - pc.nb12 = nb12 / ggml_type_size(src1->type); - pc.nb13 = nb13 / ggml_type_size(src1->type); - pc.s0 = s0; - pc.s1 = s1; - pc.s2 = s2; - pc.p0 = p0; - pc.p1 = p1; - pc.p2 = p2; - pc.d0 = d0; - pc.d1 = d1; - pc.d2 = d2; - pc.IW = IW; - pc.IH = IH; - pc.ID = ID; - pc.IC = IC; - pc.KW = KW; - pc.OH = OH; - pc.KD_KH_KW = KD*KH*KW; - pc.KH_KW = KH*KW; - pc.IC_KD_KH_KW = IC*KD*KH*KW; - pc.N_OD_OH = N*OD*OH; - pc.OD_OH = OD*OH; - pc.OD_OH_OW_IC_KD_KH_KW = OD*OH*OW*IC*KD*KH*KW; - pc.OH_OW_IC_KD_KH_KW = OH*OW*IC*KD*KH*KW; - pc.OW_IC_KD_KH_KW = OW*IC*KD*KH*KW; + GGML_ASSERT(ids->nb[1] / ggml_type_size(ids->type) == (size_t) n_experts); - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_IM2COL_3D, std::move(pc)); -} + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, nullptr, nullptr, nullptr, cgraph->nodes[node_idx], GGML_OP_SOFT_MAX); -static void ggml_vk_timestep_embedding(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const uint32_t dim = dst->op_params[0]; - const uint32_t max_period = dst->op_params[1]; - const uint32_t nb1 = dst->nb[1] / ggml_type_size(dst->type); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_TIMESTEP_EMBEDDING, { - nb1, dim, max_period, - }); -} + vk_subbuffer logits_buf = ggml_vk_tensor_subbuffer(ctx, logits); + vk_subbuffer bias_buf = ggml_vk_tensor_subbuffer(ctx, bias); + vk_subbuffer weights_buf = ggml_vk_tensor_subbuffer(ctx, weights); + vk_subbuffer ids_buf = ggml_vk_tensor_subbuffer(ctx, ids); -static void ggml_vk_conv_transpose_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - // src0: (K, Cout, Cin, 1) -- kernel - // src1: (L, Cin, 1, 1) -- input - // dst: (*, Cout, 1, 1) + vk_op_topk_moe_push_constants pc {}; + pc.n_rows = n_rows; + pc.n_experts_push = n_experts; + pc.n_expert_used = n_expert_used; + pc.clamp_min = -std::numeric_limits::infinity(); + pc.clamp_max = std::numeric_limits::infinity(); + if (mode == TOPK_MOE_EARLY_SOFTMAX_NORM) { + ggml_tensor * clamp = cgraph->nodes[node_idx + 7]; + GGML_ASSERT(clamp->op == GGML_OP_CLAMP); + pc.clamp_min = ggml_get_op_params_f32(clamp, 0); + pc.clamp_max = ggml_get_op_params_f32(clamp, 1); + } + if (mode == TOPK_MOE_SIGMOID_NORM_BIAS) { + ggml_tensor * clamp = cgraph->nodes[node_idx + 8]; + GGML_ASSERT(clamp->op == GGML_OP_CLAMP); + pc.clamp_min = ggml_get_op_params_f32(clamp, 0); + pc.clamp_max = ggml_get_op_params_f32(clamp, 1); + } + if (mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS) { + ggml_tensor * clamp = cgraph->nodes[node_idx + 9]; + GGML_ASSERT(clamp->op == GGML_OP_CLAMP); + pc.clamp_min = ggml_get_op_params_f32(clamp, 0); + pc.clamp_max = ggml_get_op_params_f32(clamp, 1); + } - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT( dst->type == GGML_TYPE_F32); +#define GATING_FUNC_SOFTMAX 0 +#define GATING_FUNC_SIGMOID 1 +#define GATING_FUNC_SOFTMAX_WEIGHT 2 +#define GATING_FUNC_SQRT_SOFTPLUS 3 - GGML_TENSOR_BINARY_OP_LOCALS + pc.gating_func = mode == TOPK_MOE_SIGMOID_NORM_BIAS ? GATING_FUNC_SIGMOID : + mode == TOPK_MOE_SQRT_SOFTPLUS_NORM_BIAS ? GATING_FUNC_SQRT_SOFTPLUS : + mode == TOPK_MOE_LATE_SOFTMAX ? GATING_FUNC_SOFTMAX_WEIGHT : + GATING_FUNC_SOFTMAX; + pc.has_bias = has_bias; + pc.with_norm = mode == TOPK_MOE_EARLY_SOFTMAX_NORM || has_bias; + if (ctx->fused_topk_moe_scale) { + GGML_ASSERT(weights->op == GGML_OP_SCALE); + pc.output_scale = ggml_get_op_params_f32(weights, 0); + pc.output_bias = ggml_get_op_params_f32(weights, 1); + } else { + pc.output_scale = 1.0f; + pc.output_bias = 0.0f; + } - GGML_ASSERT(nb00 == sizeof(float)); - GGML_ASSERT(nb10 == sizeof(float)); + GGML_ASSERT(n_expert_used <= n_experts); - const int32_t s0 = dst->op_params[0]; + const uint32_t rows_per_block = 4; + std::array elements = { CEIL_DIV(n_rows, rows_per_block), 1, 1 }; - vk_op_conv_transpose_1d_push_constants p{}; - p.Cout = static_cast(ne01); - p.Cin = static_cast(ne02); - p.K = static_cast(ne00); - p.L = static_cast(ne10); - p.KL = static_cast(ne0); - p.nb01 = static_cast(nb01 / nb00); - p.nb02 = static_cast(nb02 / nb00); - p.nb11 = static_cast(nb11 / nb10); - p.nb1 = static_cast(nb1 / nb0); - p.s0 = static_cast(s0); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, {logits_buf, bias_buf, weights_buf, ids_buf}, pc, elements); +} + +void ggml_vk_rope(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_cgraph * cgraph, int node_idx, bool backprop) { + ggml_tensor * dst = cgraph->nodes[node_idx]; + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; + const ggml_tensor * src2 = dst->src[2]; + const ggml_tensor * src3 = nullptr; + const int n_dims = ((int32_t *) dst->op_params)[1]; + const int mode = ((int32_t *) dst->op_params)[2]; + // const int n_ctx = ((int32_t *) dst->op_params)[3]; + const int n_ctx_orig = ((int32_t *) dst->op_params)[4]; + const float freq_base = ((float *) dst->op_params)[5]; + const float beta_fast = ((float *) dst->op_params)[9]; + const float beta_slow = ((float *) dst->op_params)[10]; + int sections[4] {}; + if (mode & GGML_ROPE_TYPE_MROPE) { + memcpy(sections, (int32_t *) dst->op_params + 11, sizeof(int)*4); + } - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_TRANSPOSE_1D, std::move(p)); -} + float corr_dims[2]; + ggml_rope_yarn_corr_dims(n_dims, n_ctx_orig, freq_base, beta_fast, beta_slow, corr_dims); -static void ggml_vk_col2im_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - // src0: [K_OC, T_in] columns from matmul - // dst: [T_out, OC] + uint32_t set_rows_stride = 0; + // Fused rope + view + set_rows passes the set_rows destination stride in set_rows_stride + // and overrides the dst and sets src3=row_indices + if (ctx->num_additional_fused_ops > 0) { + set_rows_stride = cgraph->nodes[node_idx + 2]->nb[1] / ggml_type_size(cgraph->nodes[node_idx + 2]->type); + src3 = cgraph->nodes[node_idx + 2]->src[1]; + dst = cgraph->nodes[node_idx + 2]; + } - const int32_t stride = dst->op_params[0]; - const int32_t oc = dst->op_params[1]; - const int32_t p0 = dst->op_params[2]; + ggml_vk_op_f32(ctx, subctx, src0, src1, src2, src3, dst, GGML_OP_ROPE, + ggml_vk_make_rope_constants(cgraph->nodes[node_idx], src0, src2 != nullptr, backprop, set_rows_stride)); +} - const uint32_t K_OC = static_cast(src0->ne[0]); - const uint32_t T_in = static_cast(src0->ne[1]); - const uint32_t T_out = static_cast(dst->ne[0]); - const uint32_t OC = static_cast(oc); - const uint32_t K = K_OC / OC; +void ggml_vk_argsort(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const uint32_t * op_params = (const uint32_t *)dst->op_params; - vk_op_col2im_1d_push_constants p{}; - p.T_out = T_out; - p.OC = OC; - p.K_OC = K_OC; - p.T_in = T_in; - p.K = K; - p.stride = stride; - p.p0 = p0; + uint32_t ncols = src0->ne[0]; + uint32_t nrows = ggml_nrows(src0); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_COL2IM_1D, std::move(p)); -} + uint32_t ncols_pad_log2 = (uint32_t)ceilf(log2f(float(ncols))); + uint32_t ncolsp2 = 1 << ncols_pad_log2; -// Dispatch the fused snake activation: y = x + sin^2(a * x) * inv_b. -// Match the naive mul -> sin -> sqr -> mul -> add chain and run the -// dedicated kernel directly. The pattern is validated by -// ggml_vk_can_fuse_snake before this call. -static void ggml_vk_snake_dispatch_fused(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { - const ggml_tensor * mul0 = cgraph->nodes[node_idx + 0]; - const ggml_tensor * sqr = cgraph->nodes[node_idx + 2]; - const ggml_tensor * mul1 = cgraph->nodes[node_idx + 3]; - ggml_tensor * add = cgraph->nodes[node_idx + 4]; + vk_op_argsort_push_constants pc { ncols, ncolsp2, ncols_pad_log2, nrows, op_params[0], 0, 0, 0, 0, }; - // x carries the full activation shape, a is the broadcast operand - const ggml_tensor * x = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; - const ggml_tensor * a = (x == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; + // Pick the largest workgroup size <= ncolsp2 + uint32_t pipeline_idx = std::min(ncols_pad_log2, num_argsort_pipelines - 1); - // mul1 reads sqr and inv_b in either operand order - const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; + // Use the "small" argsort shader if the whole sort can be done by a single workgroup. + bool use_small = ncols_pad_log2 <= ctx->device->max_workgroup_size_log2 && + ctx->device->pipeline_argsort_f32[pipeline_idx] != nullptr; - vk_pipeline pipeline = nullptr; - switch (x->type) { - case GGML_TYPE_F32: pipeline = ctx->device->pipeline_snake_f32; break; - case GGML_TYPE_F16: pipeline = ctx->device->pipeline_snake_f16; break; - case GGML_TYPE_BF16: pipeline = ctx->device->pipeline_snake_bf16; break; - default: GGML_ABORT("unsupported type"); - } - ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + vk_pipeline pipeline = use_small ? ctx->device->pipeline_argsort_f32[pipeline_idx] + : ctx->device->pipeline_argsort_large_f32[pipeline_idx]; - vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x); - vk_subbuffer a_buf = ggml_vk_tensor_subbuffer(ctx, a); - vk_subbuffer inv_b_buf = ggml_vk_tensor_subbuffer(ctx, inv_b); - vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, add); + vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer subbuf1 = dst_buf; - vk_op_snake_push_constants pc{}; - pc.ne0 = static_cast(x->ne[0]); - pc.ne1 = static_cast(x->ne[1]); + // Reserve space for ivec2 per element, with rows padded to a power of two + if (!use_small) { + const size_t x_sz = size_t{ncolsp2} * nrows * 2 * sizeof(int); - std::array elements = { pc.ne0, pc.ne1, 1 }; - ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { x_buf, a_buf, inv_b_buf, dst_buf }, pc, elements); -} + if (ctx->prealloc_size_x < x_sz) { + ctx->prealloc_size_x = x_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } + subbuf1 = { ctx->prealloc_x, 0, ctx->prealloc_x->size }; + } -static void ggml_vk_pool_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - uint32_t op = static_cast(dst->op_params[0]); - const int32_t k0 = dst->op_params[1]; - const int32_t s0 = dst->op_params[2]; - const int32_t p0 = dst->op_params[3]; + std::array elements; - const uint32_t IL = src0->ne[0]; + elements[0] = ncolsp2; + elements[1] = std::min((uint32_t)ggml_nrows(src0), ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = 1; - const uint32_t N = dst->ne[3] * dst->ne[2]; + // First dispatch initializes tmp_idx and does the first N passes where + // there is only communication between threads in the same workgroup. + { + vk_op_argsort_push_constants pc2 = pc; + pc2.outer_start = 0; + pc2.outer_end = std::min(ncols_pad_log2, ctx->device->max_workgroup_size_log2); + pc2.inner_start = 0; + pc2.inner_end = 100; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc2, elements); + } + if (!use_small) { + ggml_vk_sync_buffers(ctx, subctx); + // Loop over outer/inner passes, synchronizing between each pass. + for (uint32_t outer = ctx->device->max_workgroup_size_log2; outer < ncols_pad_log2; ++outer) { + for (uint32_t inner = 0; inner < outer + 1; ++inner) { + vk_op_argsort_push_constants pc2 = pc; + pc2.outer_start = outer; + pc2.outer_end = outer + 1; + pc2.inner_start = inner; + pc2.inner_end = inner + 1; + // When the inner idx is large enough, there's only communication + // within a workgroup. So the remaining inner iterations can all + // run in the same dispatch. + if (outer - inner < pipeline_idx) { + pc2.inner_end = 100; + inner = outer; + pipeline = ctx->device->pipeline_argsort_large_f32[pipeline_idx]; + } else { + // Smaller workgroup empirically seems to perform better + pipeline = ctx->device->pipeline_argsort_large_f32[pipeline_idx - 2]; + } + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, subbuf1, dst_buf }, pc2, elements); + ggml_vk_sync_buffers(ctx, subctx); + } + } + ctx->prealloc_x_need_sync = true; + } +} - const uint32_t OC = dst->ne[1]; - const uint32_t OL = dst->ne[0]; +void ggml_vk_topk(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + uint32_t ncols = src0->ne[0]; + uint32_t nrows = ggml_nrows(src0); + uint32_t k = dst->ne[0]; - const uint32_t parallel_elements = N * OC * OL; + // tournament path is faster where it fits; use radix-select only past its k limit + const uint32_t k_min_pipeline = std::max((uint32_t) log2f(float(k)) + 1, ctx->device->subgroup_size_log2); + if (k_min_pipeline >= num_topk_pipelines || ctx->device->pipeline_topk_f32[k_min_pipeline] == nullptr) { + vk_pipeline pipeline = ctx->device->pipeline_topk_radix_f32; + GGML_ASSERT(pipeline != nullptr); - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_POOL_1D, { - IL, OL, OC, - parallel_elements, - op, - k0, s0, p0, - }); -} + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } -static void ggml_vk_pool_2d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - uint32_t op = static_cast(dst->op_params[0]); - const int32_t k1 = dst->op_params[1]; - const int32_t k0 = dst->op_params[2]; - const int32_t s1 = dst->op_params[3]; - const int32_t s0 = dst->op_params[4]; - const int32_t p1 = dst->op_params[5]; - const int32_t p0 = dst->op_params[6]; + vk_op_topk_radix_push_constants pc { ncols, k, nrows, 0, 0, 0 }; + std::array elements { + pipeline->wg_denoms[0], + std::min(nrows, ctx->device->properties.limits.maxComputeWorkGroupCount[1]), + 1, + }; + // the non-QSA path only uses bindings 0/1; bind valid buffers for the unused QSA slots + vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { src0_buf, dst_buf, src0_buf, src0_buf, src0_buf }, pc, elements); + return; + } - const uint32_t IH = src0->ne[1]; - const uint32_t IW = src0->ne[0]; + vk_op_topk_push_constants pc { ncols, ncols, ncols, k, nrows, 0, 0 }; - const uint32_t N = dst->ne[3]; + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - const uint32_t OC = dst->ne[2]; - const uint32_t OH = dst->ne[1]; - const uint32_t OW = dst->ne[0]; + std::array elements; + elements[1] = std::min(nrows, ctx->device->properties.limits.maxComputeWorkGroupCount[1]); + elements[2] = 1; - const uint32_t parallel_elements = N * OC * OH * OW; + uint32_t num_elements = ncols; - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_POOL_2D, { - IW, IH, OW, OH, OC, - parallel_elements, - op, - k0, k1, s0, s1, p0, p1, - }); -} + // Each iteration reduces a workgroup's worth of elements down to the K + // largest elements. Repeat until we have the top K elements. + // Need to do at least one iteration to write out the results. + bool done_one_iter = false; + uint32_t dbl_buf_index = 0; + size_t dbl_buf_size; + while (num_elements > k || !done_one_iter) { -static void ggml_vk_conv_2d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, - const ggml_tensor * src1, ggml_tensor * dst) { - GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT(dst->type == GGML_TYPE_F32); + // Prefer going as small as num_topk_pipelines - 3 for perf reasons. + // But if K is larger, then we need a larger workgroup + uint32_t max_pipeline = num_topk_pipelines - 1; + uint32_t preferred_pipeline = std::max(num_topk_pipelines - 3, (uint32_t)log2f(float(k)) + 2); + max_pipeline = std::min(preferred_pipeline, max_pipeline); + uint32_t min_pipeline = (uint32_t)log2f(float(k)) + 1; + // require full subgroup + min_pipeline = std::max(min_pipeline, ctx->device->subgroup_size_log2); - GGML_TENSOR_BINARY_OP_LOCALS - GGML_ASSERT(nb00 == sizeof(float) || nb00 == sizeof(ggml_fp16_t)); - GGML_ASSERT(nb10 == sizeof(float)); - GGML_ASSERT(nb0 == sizeof(float)); + uint32_t pipeline_idx = (uint32_t)ceilf(log2f(float(num_elements))); + pipeline_idx = std::min(pipeline_idx, max_pipeline); + pipeline_idx = std::max(pipeline_idx, min_pipeline); - bool transpose = dst->op == GGML_OP_CONV_TRANSPOSE_2D; + if (num_elements > (1u << pipeline_idx)) { + // If we could finish on this loop iteration (i.e. a single workgroup) + // then do so. It's better than the overhead of another pass. + for (uint32_t i = pipeline_idx; i < num_topk_pipelines; ++i) { + if (num_elements <= (1u << i)) { + pipeline_idx = i; + break; + } + } + } - vk_op_conv2d_push_constants p{}; - p.Cout = static_cast(!transpose ? ne03 : ne02); - p.Cin = static_cast(!transpose ? ne02 : ne03); - p.N = static_cast(ne13); - GGML_ASSERT(p.Cout == ne2); - GGML_ASSERT(p.Cin == ne12); + vk_pipeline pipeline = ctx->device->pipeline_topk_f32[pipeline_idx]; + // If the device doesn't support a pipeline this large, use smaller + while (!pipeline) { + pipeline_idx--; + GGML_ASSERT(pipeline_idx >= min_pipeline); + pipeline = ctx->device->pipeline_topk_f32[pipeline_idx]; + } - p.W = static_cast(ne10); - p.H = static_cast(ne11); - p.OW = static_cast(ne0); - p.OH = static_cast(ne1); + vk_op_topk_push_constants pc2 = pc; + pc2.ncols_input = num_elements; - p.nb01 = static_cast(nb01 / nb00); - p.nb02 = static_cast(nb02 / nb00); - p.nb03 = static_cast(nb03 / nb00); + // Number of elements remaining after this pass + uint32_t num_dst_elements = (num_elements / pipeline->wg_denoms[0]) * k + std::min(k, num_elements % pipeline->wg_denoms[0]); - p.nb11 = static_cast(nb11 / nb10); - p.nb12 = static_cast(nb12 / nb10); - p.nb13 = static_cast(nb13 / nb10); + pc2.ncols_output = num_dst_elements; - p.nb1 = static_cast(nb1 / nb0); - p.nb2 = static_cast(nb2 / nb0); - p.nb3 = static_cast(nb3 / nb0); + if (!done_one_iter) { + // Reserve space for ivec2 per element, double buffered + // K per workgroup per row + dbl_buf_size = num_dst_elements * nrows * 2 * sizeof(int); + dbl_buf_size = ROUNDUP_POW2(dbl_buf_size, ctx->device->properties.limits.minStorageBufferOffsetAlignment); + const size_t x_sz = dbl_buf_size * 2; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, dst->op, std::move(p)); -} + if (ctx->prealloc_size_x < x_sz) { + ctx->prealloc_size_x = x_sz; + ggml_vk_preallocate_buffers(ctx, subctx); + } + } -static void ggml_vk_conv_3d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, - const ggml_tensor * src1, ggml_tensor * dst) { - GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); - GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT(dst->type == GGML_TYPE_F32); + vk_subbuffer src_buf; + vk_subbuffer dst_buf; - GGML_TENSOR_BINARY_OP_LOCALS - GGML_ASSERT(nb00 == sizeof(float) || nb00 == sizeof(ggml_fp16_t)); - GGML_ASSERT(nb10 == sizeof(float)); - GGML_ASSERT(nb0 == sizeof(float)); + if (num_elements == ncols) { + pc2.first_pass = 1; + src_buf = ggml_vk_tensor_subbuffer(ctx, src0); + } else { + src_buf = { ctx->prealloc_x, dbl_buf_index * dbl_buf_size, dbl_buf_size }; + } + if (num_dst_elements == k) { + pc2.last_pass = 1; + dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + } else { + dst_buf = { ctx->prealloc_x, (dbl_buf_index ^ 1) * dbl_buf_size, dbl_buf_size }; + } - vk_op_conv3d_push_constants p{}; - p.IC = static_cast(ggml_get_op_params_i32(dst, 9)); - p.N = static_cast(ggml_get_op_params_i32(dst, 10)); - p.OC = static_cast(ggml_get_op_params_i32(dst, 11)); - GGML_ASSERT(src0->ne[3] == (int64_t)p.IC * p.OC); - GGML_ASSERT(src1->ne[3] == (int64_t)p.IC * p.N); - GGML_ASSERT(dst->ne[3] == (int64_t)p.OC * p.N); + elements[0] = num_elements; - p.IW = static_cast(ne10); - p.IH = static_cast(ne11); - p.ID = static_cast(ne12); - p.OW = static_cast(ne0); - p.OH = static_cast(ne1); - p.OD = static_cast(ne2); + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src_buf, dst_buf }, pc2, elements); + num_elements = num_dst_elements; + dbl_buf_index ^= 1; + if (num_elements > k) { + ggml_vk_sync_buffers(ctx, subctx); + } + done_one_iter = true; + } + ctx->prealloc_x_need_sync = true; +} - // the shader clamps src addresses to p.IC * p.N * p.IW * p.IH * p.ID - 1 in uint32, so the - // total input element count must fit in a uint32. - GGML_ASSERT((uint64_t)p.IC * p.N * p.IW * p.IH * p.ID <= 0xFFFFFFFFull); +void ggml_vk_topk_qsa(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_cgraph * cgraph, int node_idx) { + const ggml_tensor * get_rows = cgraph->nodes[node_idx + 0]; + const ggml_tensor * add = cgraph->nodes[node_idx + ctx->num_additional_fused_ops - 1]; + ggml_tensor * top_k = cgraph->nodes[node_idx + ctx->num_additional_fused_ops]; - p.nb01 = static_cast(nb01 / nb00); - p.nb02 = static_cast(nb02 / nb00); - p.nb03 = static_cast(nb03 / nb00); + const ggml_tensor * scores = get_rows->src[0]; // [n_tps, n_blocks, n_stream] + const ggml_tensor * cell_blk = get_rows->src[1]; // [n_kv, n_stream] - p.nb11 = static_cast(nb11 / nb10); - p.nb12 = static_cast(nb12 / nb10); - p.nb13 = static_cast(nb13 / nb10); + // raw f16 mask: follow the reshape/cpy chain back to the materialized input + const ggml_tensor * mask = add->src[1]; + while (mask->op == GGML_OP_RESHAPE || mask->op == GGML_OP_CPY) { + mask = mask->src[0]; + } - p.nb1 = static_cast(nb1 / nb0); - p.nb2 = static_cast(nb2 / nb0); - p.nb3 = static_cast(nb3 / nb0); + const uint32_t n_tps = scores->ne[0]; + const uint32_t n_blocks = scores->ne[1]; + const uint32_t n_stream = scores->ne[2]; + const uint32_t n_kv = cell_blk->ne[0]; + const uint32_t width = top_k->ne[0]; + const uint32_t nrows = n_tps * n_stream; - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_3D, std::move(p)); -} + vk_pipeline pipeline = ctx->device->pipeline_topk_radix_qsa; + GGML_ASSERT(pipeline != nullptr); -static void ggml_vk_conv_2d_dw(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - vk_op_conv2d_dw_push_constants p{}; - p.ne = ggml_nelements(dst); - p.channels = dst->ne[2]; - p.batches = dst->ne[3]; - p.dst_w = dst->ne[0]; - p.dst_h = dst->ne[1]; - p.src_w = src1->ne[0]; - p.src_h = src1->ne[1]; - p.knl_w = src0->ne[0]; - p.knl_h = src0->ne[1]; - p.stride_x = dst->op_params[0]; - p.stride_y = dst->op_params[1]; - p.pad_x = dst->op_params[2]; - p.pad_y = dst->op_params[3]; - p.dilation_x = dst->op_params[4]; - p.dilation_y = dst->op_params[5]; + // scratch holds the gathered+masked input, materialized once and reused across passes + const size_t scratch_size = size_t{ n_kv } * nrows * sizeof(float); + if (ctx->prealloc_size_x < scratch_size) { + ctx->prealloc_size_x = scratch_size; + ggml_vk_preallocate_buffers(ctx, subctx); + } + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); + } - GGML_ASSERT(src0->ne[3] == p.channels); - GGML_ASSERT(src1->ne[3] == p.batches); + vk_op_topk_radix_push_constants pc { n_kv, width, nrows, n_tps, n_blocks, n_stream }; + std::array elements { + pipeline->wg_denoms[0], + std::min(nrows, ctx->device->properties.limits.maxComputeWorkGroupCount[1]), + 1, + }; + vk_subbuffer scratch_buf { ctx->prealloc_x, 0, ctx->prealloc_x->size }; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, + { ggml_vk_tensor_subbuffer(ctx, scores), ggml_vk_tensor_subbuffer(ctx, top_k), + ggml_vk_tensor_subbuffer(ctx, cell_blk), ggml_vk_tensor_subbuffer(ctx, mask), + scratch_buf }, pc, elements); + ctx->prealloc_x_need_sync = true; +} - ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_2D_DW, std::move(p)); +void ggml_vk_sum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, ggml_nelements(src0)); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SUM, p); } -static void ggml_vk_leaky_relu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { - const float * op_params = (const float *)dst->op_params; - vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); - p.param1 = op_params[0]; +void ggml_vk_sum_rows(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_SUM_ROWS, p); +} - ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_LEAKY_RELU, std::move(p)); +void ggml_vk_mean(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_sum_rows_push_constants p = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); + p.weight = 1.0f / (float)src0->ne[0]; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_MEAN, p); } -#ifdef GGML_VULKAN_RUN_TESTS -static void ggml_vk_print_matrix_area(const void * data, ggml_type type, int ne0, int ne1, int i0, int i1, int i2) { - if (type != GGML_TYPE_F32 && type != GGML_TYPE_F16) { +void ggml_vk_cumsum(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + vk_op_sum_rows_push_constants pc = vk_op_sum_rows_push_constants_init(src0, dst, src0->ne[0]); + // Use the single pass shader when the rows are small or there are enough rows to fill the GPU. + // For fewer, larger rows, use the multipass shader to spread each row across SMs. + if (dst->ne[0] <= 4096 || ggml_nrows(dst) >= ctx->device->shader_core_count) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_CUMSUM, pc); return; } - i0 = std::max(i0, 5); - i1 = std::max(i1, 5); - i2 = std::max(i2, 0); - fprintf(stderr, " "); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - fprintf(stderr, "%7d ", idx1); - } - fprintf(stderr, "\n"); - for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { - fprintf(stderr, "%7d: ", idx0); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - if (idx0 >= 0 && idx0 < ne0 && idx1 >= 0 && idx1 < ne1) { - float val; - if (type == GGML_TYPE_F32) { - val = *((const float *) data + i2*ne1*ne0 + idx1*ne0 + idx0); - } else if (type == GGML_TYPE_F16) { - val = ggml_fp16_to_fp32(*((const ggml_fp16_t *) data + i2*ne1*ne0 + idx1*ne0 + idx0)); - } else { - GGML_ABORT("fatal error"); - } - fprintf(stderr, "% 7.2f ", val); - } else { - fprintf(stderr, " "); - } - } - fprintf(stderr, "\n"); - } -} - -template -static void ggml_vk_test_matmul(ggml_backend_vk_context * ctx, size_t m, size_t n, size_t k, size_t batch, size_t num_it, int split_k, int shader_size) { - VK_LOG_DEBUG("ggml_vk_test_matmul(" << m << ", " << n << ", " << k << ", " << batch << ", " << num_it << ", " << split_k << ", " << shader_size << ")"); - const size_t x_ne = m * k * batch; - const size_t y_ne = k * n * batch; - const size_t d_ne = m * n * batch; - - vk_pipeline p; - std::string shname; - if (shader_size == 0) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->a_s; - shname = "F32_ALIGNED_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->a_s; - shname = "F32_F16_ALIGNED_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->a_s; - shname = "F16_F32_ALIGNED_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->a_s; - shname = "F16_ALIGNED_S"; - } else { - GGML_ABORT("fatal error"); - } - } else if (shader_size == 1) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->a_m; - shname = "F32_ALIGNED_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->a_m; - shname = "F32_F16_ALIGNED_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->a_m; - shname = "F16_F32_ALIGNED_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->a_m; - shname = "F16_ALIGNED_M"; - } else { - GGML_ABORT("fatal error"); - } - } else if (shader_size == 2) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->a_l; - shname = "F32_ALIGNED_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->a_l; - shname = "F32_F16_ALIGNED_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->a_l; - shname = "F16_F32_ALIGNED_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->a_l; - shname = "F16_ALIGNED_L"; - } else { - GGML_ABORT("fatal error"); - } - } else { - GGML_ASSERT(0); - } - const size_t kpad = ggml_vk_align_size(k, p->align); - - if (k != kpad) { - if (shader_size == 0) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->s; - shname = "F32_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->s; - shname = "F32_F16_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->s; - shname = "F16_F32_S"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->s; - shname = "F16_S"; - } - } else if (shader_size == 1) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->m; - shname = "F32_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->m; - shname = "F32_F16_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->m; - shname = "F16_F32_M"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->m; - shname = "F16_M"; - } - } else if (shader_size == 2) { - if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32->l; - shname = "F32_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f32_f16->l; - shname = "F32_F16_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16_f32.f32acc->l; - shname = "F16_F32_L"; - } else if (std::is_same() && std::is_same()) { - p = ctx->device->pipeline_matmul_f16.f32acc->l; - shname = "F16_L"; - } - } - } + // First pass computes partial sums within a block, and stores the last partial + // to the temp buffer. Second pass sums the block partials from the temp buffer + // and adds that to the result of the first pass. + vk_pipeline pipeline1 = ctx->device->pipeline_cumsum_multipass1_f32; + vk_pipeline pipeline2 = ctx->device->pipeline_cumsum_multipass2_f32; + GGML_ASSERT(pipeline1 != nullptr && pipeline2 != nullptr); - if (split_k > 1) { - ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, num_it); + ggml_pipeline_request_descriptor_sets(ctx, pipeline1, 1); + ggml_pipeline_request_descriptor_sets(ctx, pipeline2, 1); - if (ctx->prealloc_split_k == nullptr || ctx->prealloc_split_k->size < sizeof(float) * d_ne * split_k) { - // Resize buffer - if (ctx->prealloc_split_k != nullptr) { - ggml_vk_destroy_buffer(ctx->prealloc_split_k); - } - ctx->prealloc_split_k = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne * split_k, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - } - } + std::array elements; - ggml_pipeline_allocate_descriptor_sets(ctx); + elements[0] = dst->ne[0]; + elements[1] = (uint32_t)ggml_nrows(dst); + elements[2] = 1; - vk_buffer d_X = ggml_vk_create_buffer_check(ctx->device, sizeof(X_TYPE) * x_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer d_Y = ggml_vk_create_buffer_check(ctx->device, sizeof(Y_TYPE) * y_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer d_D = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - - X_TYPE* x = (X_TYPE *) malloc(sizeof(X_TYPE) * x_ne); - Y_TYPE* y = (Y_TYPE *) malloc(sizeof(Y_TYPE) * y_ne); - float* d = (float *) malloc(sizeof(float) * d_ne); - - for (size_t i = 0; i < x_ne; i++) { - if (std::is_same()) { - x[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; - // x[i] = 1.0f; - // x[i] = i + 1; - // x[i] = (i % k == i / k) ? 1.0f : 0.0f; - } else if (std::is_same()) { - x[i] = ggml_fp32_to_fp16((rand() / (float)RAND_MAX) * 2.0f - 1.0f); - // x[i] = ggml_fp32_to_fp16(1.0f); - // x[i] = ggml_fp32_to_fp16(i + 1); - // x[i] = ggml_fp32_to_fp16((i % k == i / k) ? 1.0f : 0.0f); - } else { - GGML_ABORT("fatal error"); - } + size_t temp_size = sizeof(float) * elements[0] * ggml_nrows(dst); + + if (ctx->prealloc_size_split_k < temp_size) { + ctx->prealloc_size_split_k = temp_size; + ggml_vk_preallocate_buffers(ctx, subctx); } - for (size_t i = 0; i < y_ne; i++) { - if (std::is_same()) { - y[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; - // y[i] = (i % k == i / k) ? 1.0f : 0.0f; - // y[i] = i + 1; - } else if (std::is_same()) { - y[i] = ggml_fp32_to_fp16((rand() / (float)RAND_MAX) * 2.0f - 1.0f); - // y[i] = ggml_fp32_to_fp16((i % k == i / k) ? 1.0f : 0.0f); - // y[i] = ggml_fp32_to_fp16(i + 1); - } else { - GGML_ABORT("fatal error"); - } + + vk_subbuffer src_buf = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); + vk_subbuffer temp_buf = ggml_vk_subbuffer(ctx, ctx->prealloc_split_k, 0); + + if (ctx->prealloc_split_k_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); } - ggml_vk_buffer_write(d_X, 0, x, sizeof(X_TYPE) * k * m * batch); - ggml_vk_buffer_write(d_Y, 0, y, sizeof(Y_TYPE) * k * n * batch); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline1, {src_buf, dst_buf, temp_buf}, pc, elements); + ggml_vk_sync_buffers(ctx, subctx); + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline2, {src_buf, dst_buf, temp_buf}, pc, elements); + + ctx->prealloc_split_k_need_sync = true; +} - vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); - ggml_vk_ctx_begin(ctx->device, subctx); - for (size_t i = 0; i < num_it; i++) { - ggml_vk_matmul( - ctx, subctx, p, ggml_vk_subbuffer(ctx, d_X), ggml_vk_subbuffer(ctx, d_Y), ggml_vk_subbuffer(ctx, d_D), ggml_vk_subbuffer(ctx, ctx->prealloc_split_k), - m, n, k, - k, k, m, k*m, k*n, m*n, - split_k, batch, batch, batch, 1, 1, n - ); +static std::array ggml_vk_nrows_elements(uint32_t nr) { + if (nr > 262144) { + return { 512, 512, CEIL_DIV(nr, 262144) }; + } + if (nr > 512) { + return { 512, CEIL_DIV(nr, 512), 1 }; } - ggml_vk_ctx_end(subctx); + return { nr, 1, 1 }; +} - auto begin = std::chrono::high_resolution_clock::now(); - ggml_vk_submit(subctx, ctx->fence); - VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_matmul waitForFences", ctx->device); - ctx->device->device.resetFences({ ctx->fence }); - ggml_vk_queue_command_pools_cleanup(ctx->device); +void ggml_vk_cross_entropy_loss(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const ggml_tensor * src1 = dst->src[1]; - auto end = std::chrono::high_resolution_clock::now(); - double time = std::chrono::duration_cast(end-begin).count() / 1000.0; + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(ggml_is_contiguous(src0)); + GGML_ASSERT(ggml_is_contiguous(src1)); + GGML_ASSERT(ggml_is_contiguous(dst)); + GGML_ASSERT(ggml_are_same_shape(src0, src1)); + GGML_ASSERT(ggml_is_scalar(dst)); - // copy dst to host - ggml_vk_buffer_read(d_D, 0, d, sizeof(float) * d_ne); + const uint32_t nclasses = (uint32_t)src0->ne[0]; + const uint32_t nrows = (uint32_t)ggml_nrows(src0); - float * d_chk = (float *) malloc(sizeof(float) * d_ne); + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, src0, src1, nullptr, dst, GGML_OP_CROSS_ENTROPY_LOSS); + GGML_ASSERT(pipeline != nullptr); - ggml_init_params iparams = { - /*.mem_size =*/ 1024*1024*1024, - /*.mem_buffer =*/ NULL, - /*.no_alloc =*/ true, - }; + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); + ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_sum_rows_f32, 1); - ggml_context * ggml_ctx = ggml_init(iparams); + vk_subbuffer src0_buf = ggml_vk_tensor_subbuffer(ctx, src0); + vk_subbuffer src1_buf = ggml_vk_tensor_subbuffer(ctx, src1); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst, true); - ggml_type src0_type; - ggml_type src1_type; + const vk_op_push_constants pc = { nclasses, nrows, 0.0f, 0.0f, 0.0f, 0.0f }; - if (std::is_same()) { - src0_type = GGML_TYPE_F32; - } else if (std::is_same()) { - src0_type = GGML_TYPE_F16; - } else { - GGML_ABORT("fatal error"); + const size_t tmp_size = (size_t)nrows * sizeof(float); + if (ctx->prealloc_size_x < tmp_size) { + ctx->prealloc_size_x = tmp_size; + ggml_vk_preallocate_buffers(ctx, subctx); } - if (std::is_same()) { - src1_type = GGML_TYPE_F32; - } else if (std::is_same()) { - src1_type = GGML_TYPE_F16; - } else { - GGML_ABORT("fatal error"); + if (ctx->prealloc_x_need_sync) { + ggml_vk_sync_buffers(ctx, subctx); } - ggml_tensor * src0_ggml = ggml_new_tensor_3d(ggml_ctx, src0_type, k, m, batch); - ggml_tensor * src1_ggml = ggml_new_tensor_3d(ggml_ctx, src1_type, k, n, batch); - ggml_tensor * tensor_ggml = ggml_mul_mat(ggml_ctx, src0_ggml, src1_ggml); + vk_subbuffer tmp_buf = { ctx->prealloc_x, 0, tmp_size }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { src0_buf, src1_buf, tmp_buf }, pc, ggml_vk_nrows_elements(nrows)); + ggml_vk_sync_buffers(ctx, subctx); - src0_ggml->data = x; - src1_ggml->data = y; - tensor_ggml->data = d_chk; + vk_op_sum_rows_push_constants sp = {}; + sp.n_cols = nrows; + sp.ne01 = 1; + sp.ne02 = 1; + sp.weight = 1.0f; + init_pushconst_fastdiv(sp); + sp.misalign_offsets = get_misalign_bytes(ctx, dst) / ggml_type_size(dst->type); - ggml_cgraph * cgraph = ggml_new_graph(ggml_ctx); - ggml_build_forward_expand(cgraph, tensor_ggml); + ggml_vk_dispatch_pipeline(ctx, subctx, ctx->device->pipeline_sum_rows_f32, { tmp_buf, dst_buf }, sp, { 1, 1, 1 }); + ctx->prealloc_x_need_sync = true; +} - ggml_graph_compute_with_ctx(ggml_ctx, cgraph, 1); +void ggml_vk_cross_entropy_loss_back(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_tensor * dst) { + const ggml_tensor * grad = dst->src[0]; + const ggml_tensor * logits = dst->src[1]; + const ggml_tensor * labels = dst->src[2]; - ggml_free(ggml_ctx); + GGML_ASSERT(grad->type == GGML_TYPE_F32); + GGML_ASSERT(logits->type == GGML_TYPE_F32); + GGML_ASSERT(labels->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); + GGML_ASSERT(ggml_is_scalar(grad)); + GGML_ASSERT(ggml_is_contiguous(grad)); + GGML_ASSERT(ggml_is_contiguous(logits)); + GGML_ASSERT(ggml_is_contiguous(labels)); + GGML_ASSERT(ggml_is_contiguous(dst)); + GGML_ASSERT(ggml_are_same_shape(logits, labels)); + GGML_ASSERT(ggml_are_same_shape(logits, dst)); - double avg_err = 0.0; - int first_err_n = -1; - int first_err_m = -1; - int first_err_b = -1; + const uint32_t nclasses = (uint32_t)logits->ne[0]; + const uint32_t nrows = (uint32_t)ggml_nrows(logits); - for (size_t i = 0; i < m*n*batch; i++) { - double err = std::fabs(d[i] - d_chk[i]); - avg_err += err; + vk_pipeline pipeline = ggml_vk_op_get_pipeline(ctx, grad, logits, labels, dst, GGML_OP_CROSS_ENTROPY_LOSS_BACK); + GGML_ASSERT(pipeline != nullptr); - if ((err > 0.05f || std::isnan(err)) && first_err_n == -1) { - first_err_b = i / (m * n); - first_err_n = (i % (m * n)) / m; - first_err_m = (i % (m * n)) % m; - } - } + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - avg_err /= m * n; + vk_subbuffer grad_buf = ggml_vk_tensor_subbuffer(ctx, grad); + vk_subbuffer logits_buf = ggml_vk_tensor_subbuffer(ctx, logits); + vk_subbuffer labels_buf = ggml_vk_tensor_subbuffer(ctx, labels); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, dst); - double tflops = 2.0*m*n*k*batch*num_it / (time / 1000.0) / (1000.0*1000.0*1000.0*1000.0); + const vk_op_push_constants pc = { nclasses, nrows, 0.0f, 0.0f, 0.0f, 0.0f }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { grad_buf, logits_buf, labels_buf, dst_buf }, pc, ggml_vk_nrows_elements(nrows)); +} - std::cerr << "TEST " << shname << " m=" << m << " n=" << n << " k=" << k << " batch=" << batch << " split_k=" << split_k << " matmul " << time / num_it << "ms " << tflops << " TFLOPS avg_err=" << avg_err << std::endl; +void ggml_vk_argmax(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_ARGMAX, { (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], 0.0f, 0.0f, 0.0f, 0.0f }); +} - if (avg_err > 0.1 || std::isnan(avg_err)) { - std::cerr << "m = " << first_err_m << " n = " << first_err_n << " b = " << first_err_b << std::endl; - std::cerr << "Actual result: " << std::endl << std::endl; - ggml_vk_print_matrix_area(d, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); - std::cerr << "Expected result: " << std::endl << std::endl; - ggml_vk_print_matrix_area(d_chk, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); +void ggml_vk_count_equal(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_COUNT_EQUAL, { (uint32_t)ggml_nelements(src0), 0, 0.0f, 0.0f, 0.0f, 0.0f }); +} - if (split_k > 1) { - float * split_k_buf = (float *) malloc(sizeof(float) * d_ne * split_k); - ggml_vk_buffer_read(ctx->prealloc_split_k, 0, split_k_buf, sizeof(float) * d_ne * split_k); +void ggml_vk_solve_tri(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const uint32_t src0_type_size = ggml_type_size(src0->type); + const uint32_t src1_type_size = ggml_type_size(src1->type); + const uint32_t dst_type_size = ggml_type_size(dst->type); + + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_SOLVE_TRI, { + (uint32_t)ggml_nelements(src0), + (uint32_t)src0->ne[0], (uint32_t)src0->ne[1], (uint32_t)src0->ne[2],(uint32_t)src0->ne[3], (uint32_t)src0->nb[0] / src0_type_size, (uint32_t)src0->nb[1] / src0_type_size, (uint32_t)src0->nb[2] / src0_type_size, (uint32_t)src0->nb[3] / src0_type_size, + (uint32_t)src1->ne[0], (uint32_t)src1->ne[1], (uint32_t)src1->ne[2],(uint32_t)src1->ne[3], (uint32_t)src1->nb[0] / src1_type_size, (uint32_t)src1->nb[1] / src1_type_size, (uint32_t)src1->nb[2] / src1_type_size, (uint32_t)src1->nb[3] / src1_type_size, + (uint32_t) dst->ne[0], (uint32_t) dst->ne[1], (uint32_t) dst->ne[2],(uint32_t) dst->ne[3], (uint32_t) dst->nb[0] / dst_type_size, (uint32_t) dst->nb[1] / dst_type_size, (uint32_t) dst->nb[2] / dst_type_size, (uint32_t) dst->nb[3] / dst_type_size, + 0, + 0.0f, 0.0f, 0, + }); +} + +void ggml_vk_im2col(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + const int32_t s0 = dst->op_params[0]; + const int32_t s1 = dst->op_params[1]; + const int32_t p0 = dst->op_params[2]; + const int32_t p1 = dst->op_params[3]; + const int32_t d0 = dst->op_params[4]; + const int32_t d1 = dst->op_params[5]; - std::cerr << "d_buf0: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + const bool is_2D = dst->op_params[6] == 1; - std::cerr << "d_buf1: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + const uint32_t IC = src1->ne[is_2D ? 2 : 1]; + const uint32_t IH = is_2D ? src1->ne[1] : 1; + const uint32_t IW = src1->ne[0]; - std::cerr << "d_buf2: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + 2 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + const uint32_t KH = is_2D ? src0->ne[1] : 1; + const uint32_t KW = src0->ne[0]; - std::cerr << "d_buf3: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + 3 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + const uint32_t OH = is_2D ? dst->ne[2] : 1; + const uint32_t OW = dst->ne[1]; - free(split_k_buf); - } - } + const uint32_t offset_delta = src1->nb[is_2D ? 2 : 1] / 4; // nb is byte offset, src is type float32 + const uint32_t batch_offset = src1->nb[is_2D ? 3 : 2] / 4; // nb is byte offset, src is type float32 - free(d_chk); + const uint32_t batch = src1->ne[is_2D ? 3 : 2]; - ggml_vk_command_pool_cleanup(ctx->device, ctx->compute_cmd_pool); + const ggml_backend_vk_buffer_context * d_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; + const vk_buffer d_buf = d_buf_ctx->dev_buffer; - ggml_vk_destroy_buffer(d_X); - ggml_vk_destroy_buffer(d_Y); - ggml_vk_destroy_buffer(d_D); + const vk::DeviceAddress dst_addr = d_buf->bda_addr + vk_tensor_offset(dst) + dst->view_offs; - free(x); - free(y); - free(d); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_IM2COL, { + dst_addr, + batch_offset, offset_delta, + IC, IW, IH, OW, OH, KW, KH, + OH * batch, + IC * KH * KW, + s0, s1, p0, p1, d0, d1, batch * IC + }); } -static void ggml_vk_print_tensor_area(const ggml_tensor * tensor, int i0, int i1, int i2, int i3) { - if (tensor->type != GGML_TYPE_F32 && tensor->type != GGML_TYPE_F16) { - return; - } - i0 = std::max(i0, 5); - i1 = std::max(i1, 5); - i2 = std::max(i2, 0); - i3 = std::max(i3, 0); - fprintf(stderr, " "); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - fprintf(stderr, "%7d ", idx1); - } - fprintf(stderr, "\n"); - for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { - fprintf(stderr, "%7d: ", idx0); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - if (idx0 >= 0 && idx0 < tensor->ne[0] && idx1 >= 0 && idx1 < tensor->ne[1] && i2 >= 0 && i2 < tensor->ne[2] && i3 >= 0 && i3 < tensor->ne[3]) { - float val; - if (tensor->type == GGML_TYPE_F32) { - val = *(float *) ((char *) tensor->data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); - } else if (tensor->type == GGML_TYPE_F16) { - val = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) tensor->data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0])); - } else { - GGML_ABORT("fatal error"); - } - fprintf(stderr, "% 7.2f ", val); - } else { - fprintf(stderr, " "); - } - } - fprintf(stderr, "\n"); - } -} +void ggml_vk_im2col_3d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + GGML_TENSOR_BINARY_OP_LOCALS -static void ggml_vk_quantize_data(const float * from, void * to, size_t ne, ggml_type quant) { - ggml_quantize_chunk(quant, from, to, 0, 1, ne, nullptr); -} + const int32_t s0 = ((const int32_t *)(dst->op_params))[0]; + const int32_t s1 = ((const int32_t *)(dst->op_params))[1]; + const int32_t s2 = ((const int32_t *)(dst->op_params))[2]; + const int32_t p0 = ((const int32_t *)(dst->op_params))[3]; + const int32_t p1 = ((const int32_t *)(dst->op_params))[4]; + const int32_t p2 = ((const int32_t *)(dst->op_params))[5]; + const int32_t d0 = ((const int32_t *)(dst->op_params))[6]; + const int32_t d1 = ((const int32_t *)(dst->op_params))[7]; + const int32_t d2 = ((const int32_t *)(dst->op_params))[8]; + const int32_t IC = ((const int32_t *)(dst->op_params))[9]; -static void ggml_vk_dequantize_data(const void * from, float * to, size_t ne, ggml_type quant) { - if (quant == GGML_TYPE_F32) { - memcpy(to, from, sizeof(float) * ne); - return; - } + const int64_t N = ne13 / IC; + const int64_t ID = ne12; + const int64_t IH = ne11; + const int64_t IW = ne10; - const auto * tt = ggml_get_type_traits(quant); + const int64_t KD = ne02; + const int64_t KH = ne01; + const int64_t KW = ne00; - ggml_to_float_t dequant_fn = tt->to_float; + const int64_t OD = ne3 / N; + const int64_t OH = ne2; + const int64_t OW = ne1; - dequant_fn(from, to, ne); -} + const ggml_backend_vk_buffer_context * d_buf_ctx = (ggml_backend_vk_buffer_context *)dst->buffer->context; + const vk_buffer d_buf = d_buf_ctx->dev_buffer; -static void ggml_vk_test_dequant(ggml_backend_vk_context * ctx, size_t ne, ggml_type quant) { - VK_LOG_DEBUG("ggml_vk_test_dequant(" << ne << ")"); - const size_t x_sz = sizeof(float) * ne; - const size_t x_sz_f16 = sizeof(ggml_fp16_t) * ne; - const size_t qx_sz = ne * ggml_type_size(quant)/ggml_blck_size(quant); - float * x = (float *) malloc(x_sz); - void * qx = malloc(qx_sz); - vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer x_buf = ggml_vk_create_buffer_check(ctx->device, x_sz_f16, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - float * x_ref = (float *) malloc(x_sz); - ggml_fp16_t * x_chk = (ggml_fp16_t *) malloc(x_sz_f16); + const vk::DeviceAddress dst_addr = d_buf->bda_addr + vk_tensor_offset(dst) + dst->view_offs; - for (size_t i = 0; i < ne; i++) { - x[i] = rand() / (float)RAND_MAX; - } + vk_op_im2col_3d_push_constants pc {}; - vk_pipeline p = ggml_vk_get_to_fp16(ctx, quant); + pc.dst_addr = dst_addr; + pc.nb10 = nb10 / ggml_type_size(src1->type); + pc.nb11 = nb11 / ggml_type_size(src1->type); + pc.nb12 = nb12 / ggml_type_size(src1->type); + pc.nb13 = nb13 / ggml_type_size(src1->type); + pc.s0 = s0; + pc.s1 = s1; + pc.s2 = s2; + pc.p0 = p0; + pc.p1 = p1; + pc.p2 = p2; + pc.d0 = d0; + pc.d1 = d1; + pc.d2 = d2; + pc.IW = IW; + pc.IH = IH; + pc.ID = ID; + pc.IC = IC; + pc.KW = KW; + pc.OH = OH; + pc.KD_KH_KW = KD*KH*KW; + pc.KH_KW = KH*KW; + pc.IC_KD_KH_KW = IC*KD*KH*KW; + pc.N_OD_OH = N*OD*OH; + pc.OD_OH = OD*OH; + pc.OD_OH_OW_IC_KD_KH_KW = OD*OH*OW*IC*KD*KH*KW; + pc.OH_OW_IC_KD_KH_KW = OH*OW*IC*KD*KH*KW; + pc.OW_IC_KD_KH_KW = OW*IC*KD*KH*KW; - ggml_vk_quantize_data(x, qx, ne, quant); - ggml_vk_dequantize_data(qx, x_ref, ne, quant); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_IM2COL_3D, std::move(pc)); +} - ggml_pipeline_request_descriptor_sets(ctx, p, 1); +void ggml_vk_timestep_embedding(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const uint32_t dim = dst->op_params[0]; + const uint32_t max_period = dst->op_params[1]; + const uint32_t nb1 = dst->nb[1] / ggml_type_size(dst->type); - ggml_pipeline_allocate_descriptor_sets(ctx); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_TIMESTEP_EMBEDDING, { + nb1, dim, max_period, + }); +} - ggml_vk_buffer_write(qx_buf, 0, qx, qx_sz); +void ggml_vk_conv_transpose_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + // src0: (K, Cout, Cin, 1) -- kernel + // src1: (L, Cin, 1, 1) -- input + // dst: (*, Cout, 1, 1) - vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); - ggml_vk_ctx_begin(ctx->device, subctx); - const std::vector pc = { 1, (uint32_t)ne, (uint32_t)ne, (uint32_t)ne, (uint32_t)ne }; - ggml_vk_dispatch_pipeline(ctx, subctx, p, { vk_subbuffer{ qx_buf, 0, qx_sz }, vk_subbuffer{ x_buf, 0, x_sz_f16 } }, pc, { (uint32_t)ne, 1, 1}); - ggml_vk_ctx_end(subctx); + GGML_ASSERT(src0->type == GGML_TYPE_F32); + GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT( dst->type == GGML_TYPE_F32); - auto begin = std::chrono::high_resolution_clock::now(); + GGML_TENSOR_BINARY_OP_LOCALS - ggml_vk_submit(subctx, ctx->fence); - VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_dequant waitForFences", ctx->device); - ctx->device->device.resetFences({ ctx->fence }); - ggml_vk_queue_command_pools_cleanup(ctx->device); + GGML_ASSERT(nb00 == sizeof(float)); + GGML_ASSERT(nb10 == sizeof(float)); - auto end = std::chrono::high_resolution_clock::now(); + const int32_t s0 = dst->op_params[0]; - double ms_dequant = std::chrono::duration_cast(end-begin).count() / 1000.0; - ggml_vk_buffer_read(x_buf, 0, x_chk, x_sz_f16); + vk_op_conv_transpose_1d_push_constants p{}; + p.Cout = static_cast(ne01); + p.Cin = static_cast(ne02); + p.K = static_cast(ne00); + p.L = static_cast(ne10); + p.KL = static_cast(ne0); + p.nb01 = static_cast(nb01 / nb00); + p.nb02 = static_cast(nb02 / nb00); + p.nb11 = static_cast(nb11 / nb10); + p.nb1 = static_cast(nb1 / nb0); + p.s0 = static_cast(s0); - int first_err = -1; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_TRANSPOSE_1D, std::move(p)); +} - double avg_err = 0.0; - for (size_t i = 0; i < ne; i++) { - double error = std::fabs(x_ref[i] - ggml_fp16_to_fp32(x_chk[i])); - avg_err += error; +void ggml_vk_col2im_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + // src0: [K_OC, T_in] columns from matmul + // dst: [T_out, OC] - if (first_err < 0 && error > 0.05) { - first_err = i; - } - } + const int32_t stride = dst->op_params[0]; + const int32_t oc = dst->op_params[1]; + const int32_t p0 = dst->op_params[2]; - avg_err /= ne; + const uint32_t K_OC = static_cast(src0->ne[0]); + const uint32_t T_in = static_cast(src0->ne[1]); + const uint32_t T_out = static_cast(dst->ne[0]); + const uint32_t OC = static_cast(oc); + const uint32_t K = K_OC / OC; - std::cerr << "TEST DEQUANT " << ggml_type_name(quant) << " time=" << ms_dequant << "ms avg_err=" << avg_err << std::endl; + vk_op_col2im_1d_push_constants p{}; + p.T_out = T_out; + p.OC = OC; + p.K_OC = K_OC; + p.T_in = T_in; + p.K = K; + p.stride = stride; + p.p0 = p0; - if (avg_err > 0.1) { - std::cerr << "first_error = " << first_err << std::endl; - std::cerr << "Actual result: " << std::endl << std::endl; - for (int i = std::max(0, first_err - 5); i < std::min((int)ne, first_err + 5); i++) { - std::cerr << ggml_fp16_to_fp32(x_chk[i]) << ", "; - } - std::cerr << std::endl << "Expected result: " << std::endl << std::endl; - for (int i = std::max(0, first_err - 5); i < std::min((int)ne, first_err + 5); i++) { - std::cerr << x_ref[i] << ", "; - } - std::cerr << std::endl; - } + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_COL2IM_1D, std::move(p)); +} - ggml_vk_destroy_buffer(x_buf); - ggml_vk_destroy_buffer(qx_buf); - - free(x); - free(qx); - free(x_ref); - free(x_chk); -} - -// This does not work without ggml q8_1 quantization support -// -// typedef uint16_t ggml_half; -// typedef uint32_t ggml_half2; -// -// #define QK8_1 32 -// typedef struct { -// union { -// struct { -// ggml_half d; // delta -// ggml_half s; // d * sum(qs[i]) -// } GGML_COMMON_AGGR_S; -// ggml_half2 ds; -// } GGML_COMMON_AGGR_U; -// int8_t qs[QK8_1]; // quants -// } block_q8_1; -// -// static void ggml_vk_test_quantize(ggml_backend_vk_context * ctx, size_t ne, ggml_type quant) { -// VK_LOG_DEBUG("ggml_vk_test_quantize(" << ne << ")"); -// GGML_ASSERT(quant == GGML_TYPE_Q8_1); -// -// const size_t x_sz = sizeof(float) * ne; -// const size_t qx_sz = ne * ggml_type_size(quant)/ggml_blck_size(quant); -// float * x = (float *) malloc(x_sz); -// block_q8_1 * qx = (block_q8_1 *)malloc(qx_sz); -// block_q8_1 * qx_res = (block_q8_1 *)malloc(qx_sz); -// vk_buffer x_buf = ggml_vk_create_buffer_check(ctx->device, x_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); -// vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); -// -// for (size_t i = 0; i < ne; i++) { -// x[i] = rand() / (float)RAND_MAX; -// } -// -// vk_pipeline p = ggml_vk_get_quantize_pipeline(ctx, quant); -// -// ggml_pipeline_request_descriptor_sets(ctx, p, 1); -// -// ggml_pipeline_allocate_descriptor_sets(ctx); -// -// ggml_vk_buffer_write(x_buf, 0, x, x_sz); -// -// vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); -// ggml_vk_ctx_begin(ctx->device, subctx); -// ggml_vk_quantize_q8_1(ctx, subctx, ggml_vk_subbuffer(ctx, x_buf), ggml_vk_subbuffer(ctx, qx_buf), ne); -// ggml_vk_ctx_end(subctx); -// -// auto begin = std::chrono::high_resolution_clock::now(); -// -// ggml_vk_submit(subctx, ctx->fence); -// VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_quantize waitForFences"); -// ctx->device->device.resetFences({ ctx->fence }); -// ggml_vk_queue_command_pools_cleanup(ctx->device); -// -// auto end = std::chrono::high_resolution_clock::now(); -// -// double ms_quant = std::chrono::duration_cast(end-begin).count() / 1000.0; -// ggml_vk_buffer_read(qx_buf, 0, qx, qx_sz); -// -// ggml_vk_quantize_data(x, qx_res, ne, quant); -// -// int first_err = -1; -// -// for (size_t i = 0; i < ne / 32; i++) { -// double error = std::fabs(ggml_fp16_to_fp32(qx_res[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) - ggml_fp16_to_fp32(qx[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d)); -// -// if (first_err < 0 && error > 0.1) { -// first_err = i; -// } -// -// error = std::fabs(ggml_fp16_to_fp32(qx_res[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) - ggml_fp16_to_fp32(qx[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s)); -// -// if (first_err < 0 && error > 0.1) { -// first_err = i; -// } -// -// for (size_t j = 0; j < 32; j++) { -// uint64_t error = std::abs(qx_res[i].qs[j] - qx[i].qs[j]); -// -// if (first_err < 0 && error > 1) { -// first_err = i; -// } -// } -// } -// -// std::cerr << "TEST QUANTIZE " << ggml_type_name(quant) << " time=" << ms_quant << "ms " << (first_err == -1 ? "CORRECT" : "INCORRECT") << std::endl; -// -// if (first_err != -1) { -// std::cerr << "first_error = " << first_err << std::endl; -// std::cerr << "Actual result: " << std::endl << std::endl; -// std::cout << "d=" << ggml_fp16_to_fp32(qx[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) << " s=" << ggml_fp16_to_fp32(qx[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) << " "; -// for (size_t j = 0; j < 32; j++) { -// std::cout << " qs" << j << "=" << (uint32_t)qx[first_err].qs[j] << " "; -// } -// std::cerr << std::endl << std::endl << "Expected result: " << std::endl << std::endl; -// std::cout << "d=" << ggml_fp16_to_fp32(qx_res[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d) << " s=" << ggml_fp16_to_fp32(qx_res[first_err].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.s) << " "; -// for (size_t j = 0; j < 32; j++) { -// std::cout << " qs" << j << "=" << (uint32_t)qx_res[first_err].qs[j] << " "; -// } -// std::cerr << std::endl; -// } -// -// ggml_vk_destroy_buffer(x_buf); -// ggml_vk_destroy_buffer(qx_buf); -// -// free(x); -// free(qx); -// free(qx_res); -// } - -static void ggml_vk_test_dequant_matmul(ggml_backend_vk_context * ctx, size_t m, size_t n, size_t k, size_t batch, size_t num_it, size_t split_k, size_t shader_size, ggml_type quant, bool mmq = false) { - VK_LOG_DEBUG("ggml_vk_test_dequant_matmul(" << m << ", " << n << ", " << k << ", " << batch << ", " << num_it << ", " << split_k << ", " << ggml_type_name(quant) << ")"); - const size_t x_ne = m * k * batch; - const size_t y_ne = k * n * batch; - const size_t d_ne = m * n * batch; - - vk_matmul_pipeline2 * pipelines; +void ggml_vk_snake_dispatch_fused(ggml_backend_vk_context * ctx, vk_context& subctx, ggml_cgraph * cgraph, int node_idx) { + const ggml_tensor * mul0 = cgraph->nodes[node_idx + 0]; + const ggml_tensor * sqr = cgraph->nodes[node_idx + 2]; + const ggml_tensor * mul1 = cgraph->nodes[node_idx + 3]; + ggml_tensor * add = cgraph->nodes[node_idx + 4]; - if (mmq) { - pipelines = ctx->device->pipeline_dequant_mul_mat_mat_q8_1; - } else { - pipelines = ctx->device->pipeline_dequant_mul_mat_mat; - } - - const bool fp16acc = ctx->device->fp16; - - vk_pipeline p; - std::string shname; - if (shader_size == 0) { - p = fp16acc ? pipelines[quant].f16acc->a_s : pipelines[quant].f32acc->a_s; - shname = std::string(ggml_type_name(quant)) + "_ALIGNED_S"; - } else if (shader_size == 1) { - p = fp16acc ? pipelines[quant].f16acc->a_m : pipelines[quant].f32acc->a_m; - shname = std::string(ggml_type_name(quant)) + "_ALIGNED_M"; - } else if (shader_size == 2) { - p = fp16acc ? pipelines[quant].f16acc->a_l : pipelines[quant].f32acc->a_l; - shname = std::string(ggml_type_name(quant)) + "_ALIGNED_L"; - } else { - GGML_ASSERT(0); - } + // x carries the full activation shape, a is the broadcast operand + const ggml_tensor * x = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; + const ggml_tensor * a = (x == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; - const size_t kpad = mmq ? 0 : ggml_vk_align_size(k, p->align); + // mul1 reads sqr and inv_b in either operand order + const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; - if (mmq || k != kpad) { - if (shader_size == 0) { - p = fp16acc ? pipelines[quant].f16acc->s : pipelines[quant].f32acc->s; - shname = std::string(ggml_type_name(quant)) + "_S"; - } else if (shader_size == 1) { - p = fp16acc ? pipelines[quant].f16acc->m : pipelines[quant].f32acc->m; - shname = std::string(ggml_type_name(quant)) + "_M"; - } else if (shader_size == 2) { - p = fp16acc ? pipelines[quant].f16acc->l : pipelines[quant].f32acc->l; - shname = std::string(ggml_type_name(quant)) + "_L"; - } else { - GGML_ASSERT(0); - } + vk_pipeline pipeline = nullptr; + switch (x->type) { + case GGML_TYPE_F32: pipeline = ctx->device->pipeline_snake_f32; break; + case GGML_TYPE_F16: pipeline = ctx->device->pipeline_snake_f16; break; + case GGML_TYPE_BF16: pipeline = ctx->device->pipeline_snake_bf16; break; + default: GGML_ABORT("unsupported type"); } + ggml_pipeline_request_descriptor_sets(ctx, pipeline, 1); - if (p == nullptr) { - std::cerr << "error: no pipeline for ggml_vk_test_dequant_matmul " << ggml_type_name(quant) << std::endl; - return; - } + vk_subbuffer x_buf = ggml_vk_tensor_subbuffer(ctx, x); + vk_subbuffer a_buf = ggml_vk_tensor_subbuffer(ctx, a); + vk_subbuffer inv_b_buf = ggml_vk_tensor_subbuffer(ctx, inv_b); + vk_subbuffer dst_buf = ggml_vk_tensor_subbuffer(ctx, add); - const size_t x_sz = sizeof(float) * x_ne; - const size_t y_sz = sizeof(float) * y_ne; - const size_t qx_sz = x_ne * ggml_type_size(quant)/ggml_blck_size(quant); - const size_t qy_sz = mmq ? y_ne * ggml_type_size(GGML_TYPE_Q8_1)/ggml_blck_size(GGML_TYPE_Q8_1) : y_sz; - const size_t d_sz = sizeof(float) * d_ne; - float * x = (float *) malloc(x_sz); - float * y = (float *) malloc(y_sz); - void * qx = malloc(qx_sz); - vk_buffer qx_buf = ggml_vk_create_buffer_check(ctx->device, qx_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer y_buf = ggml_vk_create_buffer_check(ctx->device, y_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer qy_buf = ggml_vk_create_buffer_check(ctx->device, qy_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - vk_buffer d_buf = ggml_vk_create_buffer_check(ctx->device, d_sz, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - float * d = (float *) malloc(d_sz); - float * d_chk = (float *) malloc(d_sz); + vk_op_snake_push_constants pc{}; + pc.ne0 = static_cast(x->ne[0]); + pc.ne1 = static_cast(x->ne[1]); - for (size_t i = 0; i < x_ne; i++) { - x[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; - // x[i] = (i % k == i / k) ? 1.0f : 0.0f; - // x[i] = i % k; - } + std::array elements = { pc.ne0, pc.ne1, 1 }; + ggml_vk_dispatch_pipeline(ctx, subctx, pipeline, { x_buf, a_buf, inv_b_buf, dst_buf }, pc, elements); +} - ggml_vk_quantize_data(x, qx, x_ne, quant); +void ggml_vk_pool_1d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + uint32_t op = static_cast(dst->op_params[0]); + const int32_t k0 = dst->op_params[1]; + const int32_t s0 = dst->op_params[2]; + const int32_t p0 = dst->op_params[3]; - for (size_t i = 0; i < y_ne; i++) { - y[i] = (rand() / (float)RAND_MAX) * 2.0f - 1.0f; - // y[i] = (i % k == i / k) ? 1.0f : 0.0f; - // y[i] = i % k; - } + const uint32_t IL = src0->ne[0]; - if (split_k > 1) { - ggml_pipeline_request_descriptor_sets(ctx, ctx->device->pipeline_matmul_split_k_reduce, num_it); + const uint32_t N = dst->ne[3] * dst->ne[2]; - if (ctx->prealloc_split_k == nullptr || ctx->prealloc_split_k->size < sizeof(float) * d_ne * split_k) { - // Resize buffer - if (ctx->prealloc_split_k != nullptr) { - ggml_vk_destroy_buffer(ctx->prealloc_split_k); - } - ctx->prealloc_split_k = ggml_vk_create_buffer_check(ctx->device, sizeof(float) * d_ne * split_k, {vk::MemoryPropertyFlagBits::eDeviceLocal}); - } - } - if (mmq) { - vk_pipeline pipeline_quantize_q8_1 = ggml_vk_get_quantize_pipeline(ctx, GGML_TYPE_Q8_1); - ggml_pipeline_request_descriptor_sets(ctx, pipeline_quantize_q8_1, num_it); - } + const uint32_t OC = dst->ne[1]; + const uint32_t OL = dst->ne[0]; - ggml_pipeline_allocate_descriptor_sets(ctx); + const uint32_t parallel_elements = N * OC * OL; - ggml_vk_buffer_write(qx_buf, 0, qx, qx_sz); - ggml_vk_buffer_write(y_buf, 0, y, y_sz); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_POOL_1D, { + IL, OL, OC, + parallel_elements, + op, + k0, s0, p0, + }); +} - vk_context subctx = ggml_vk_create_context(ctx, ctx->compute_cmd_pool); - ggml_vk_ctx_begin(ctx->device, subctx); - if (mmq) { - for (size_t i = 0; i < num_it; i++) { - ggml_vk_quantize_q8_1(ctx, subctx, { y_buf, 0, y_sz }, { qy_buf, 0, qy_sz }, y_ne); - ggml_vk_matmul( - ctx, subctx, p, { qx_buf, 0, qx_sz }, { qy_buf, 0, qy_sz }, { d_buf, 0, d_sz }, { ctx->prealloc_split_k, 0, ctx->prealloc_size_split_k }, - m, n, k, - k, k, m, k*m, k*n, m*n, - split_k, batch, batch, batch, 1, 1, n - ); - } - } else { - for (size_t i = 0; i < num_it; i++) { - ggml_vk_matmul( - ctx, subctx, p, { qx_buf, 0, qx_sz }, { y_buf, 0, y_sz }, { d_buf, 0, d_sz }, { ctx->prealloc_split_k, 0, ctx->prealloc_size_split_k }, - m, n, k, - k, k, m, k*m, k*n, m*n, - split_k, batch, batch, batch, 1, 1, n - ); - } - } - ggml_vk_ctx_end(subctx); +void ggml_vk_pool_2d(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + uint32_t op = static_cast(dst->op_params[0]); + const int32_t k1 = dst->op_params[1]; + const int32_t k0 = dst->op_params[2]; + const int32_t s1 = dst->op_params[3]; + const int32_t s0 = dst->op_params[4]; + const int32_t p1 = dst->op_params[5]; + const int32_t p0 = dst->op_params[6]; - auto begin = std::chrono::high_resolution_clock::now(); + const uint32_t IH = src0->ne[1]; + const uint32_t IW = src0->ne[0]; - ggml_vk_submit(subctx, ctx->fence); - VK_CHECK(ctx->device->device.waitForFences({ ctx->fence }, true, UINT64_MAX), "ggml_vk_test_dequant waitForFences", ctx->device); - ctx->device->device.resetFences({ ctx->fence }); - ggml_vk_queue_command_pools_cleanup(ctx->device); + const uint32_t N = dst->ne[3]; - auto end = std::chrono::high_resolution_clock::now(); + const uint32_t OC = dst->ne[2]; + const uint32_t OH = dst->ne[1]; + const uint32_t OW = dst->ne[0]; - double time_ms = std::chrono::duration_cast(end-begin).count() / 1000.0; - ggml_vk_buffer_read(d_buf, 0, d, d_sz); + const uint32_t parallel_elements = N * OC * OH * OW; - ggml_init_params iparams = { - /*.mem_size =*/ 1024*1024*1024, - /*.mem_buffer =*/ NULL, - /*.no_alloc =*/ true, - }; + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_POOL_2D, { + IW, IH, OW, OH, OC, + parallel_elements, + op, + k0, k1, s0, s1, p0, p1, + }); +} + +void ggml_vk_conv_2d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst) { + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); - ggml_context * ggml_ctx = ggml_init(iparams); + GGML_TENSOR_BINARY_OP_LOCALS + GGML_ASSERT(nb00 == sizeof(float) || nb00 == sizeof(ggml_fp16_t)); + GGML_ASSERT(nb10 == sizeof(float)); + GGML_ASSERT(nb0 == sizeof(float)); - ggml_tensor * src0_ggml = ggml_new_tensor_3d(ggml_ctx, quant, k, m, batch); - ggml_tensor * src1_ggml = ggml_new_tensor_3d(ggml_ctx, GGML_TYPE_F32, k, n, batch); - ggml_tensor * tensor_ggml = ggml_mul_mat(ggml_ctx, src0_ggml, src1_ggml); + bool transpose = dst->op == GGML_OP_CONV_TRANSPOSE_2D; - src0_ggml->data = qx; - src1_ggml->data = y; - tensor_ggml->data = d_chk; + vk_op_conv2d_push_constants p{}; + p.Cout = static_cast(!transpose ? ne03 : ne02); + p.Cin = static_cast(!transpose ? ne02 : ne03); + p.N = static_cast(ne13); + GGML_ASSERT(p.Cout == ne2); + GGML_ASSERT(p.Cin == ne12); - ggml_cgraph * cgraph = ggml_new_graph(ggml_ctx); - ggml_build_forward_expand(cgraph, tensor_ggml); + p.W = static_cast(ne10); + p.H = static_cast(ne11); + p.OW = static_cast(ne0); + p.OH = static_cast(ne1); - ggml_graph_compute_with_ctx(ggml_ctx, cgraph, 1); + p.nb01 = static_cast(nb01 / nb00); + p.nb02 = static_cast(nb02 / nb00); + p.nb03 = static_cast(nb03 / nb00); - ggml_free(ggml_ctx); + p.nb11 = static_cast(nb11 / nb10); + p.nb12 = static_cast(nb12 / nb10); + p.nb13 = static_cast(nb13 / nb10); - double avg_err = 0.0; - int first_err_n = -1; - int first_err_m = -1; - int first_err_b = -1; + p.nb1 = static_cast(nb1 / nb0); + p.nb2 = static_cast(nb2 / nb0); + p.nb3 = static_cast(nb3 / nb0); - for (size_t i = 0; i < m*n*batch; i++) { - double err = std::fabs(d[i] - d_chk[i]); - avg_err += err; + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, dst->op, std::move(p)); +} - if ((err > 0.05f || std::isnan(err)) && first_err_n == -1) { - first_err_b = i / (m * n); - first_err_n = (i % (m * n)) / m; - first_err_m = (i % (m * n)) % m; - } - } +void ggml_vk_conv_3d(ggml_backend_vk_context * ctx, vk_context & subctx, const ggml_tensor * src0, + const ggml_tensor * src1, ggml_tensor * dst) { + GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); + GGML_ASSERT(src1->type == GGML_TYPE_F32); + GGML_ASSERT(dst->type == GGML_TYPE_F32); - avg_err /= m * n; + GGML_TENSOR_BINARY_OP_LOCALS + GGML_ASSERT(nb00 == sizeof(float) || nb00 == sizeof(ggml_fp16_t)); + GGML_ASSERT(nb10 == sizeof(float)); + GGML_ASSERT(nb0 == sizeof(float)); - double tflops = 2.0*m*n*k*batch*num_it / (time_ms / 1000.0) / (1000.0*1000.0*1000.0*1000.0); + vk_op_conv3d_push_constants p{}; + p.IC = static_cast(ggml_get_op_params_i32(dst, 9)); + p.N = static_cast(ggml_get_op_params_i32(dst, 10)); + p.OC = static_cast(ggml_get_op_params_i32(dst, 11)); + GGML_ASSERT(src0->ne[3] == (int64_t)p.IC * p.OC); + GGML_ASSERT(src1->ne[3] == (int64_t)p.IC * p.N); + GGML_ASSERT(dst->ne[3] == (int64_t)p.OC * p.N); - std::cerr << "TEST dequant matmul " << shname; - if (mmq) { - std::cerr << " mmq"; - } - std::cerr << " m=" << m << " n=" << n << " k=" << k << " batch=" << batch << " split_k=" << split_k << " matmul " << time_ms / num_it << "ms " << tflops << " TFLOPS avg_err=" << avg_err << std::endl; + p.IW = static_cast(ne10); + p.IH = static_cast(ne11); + p.ID = static_cast(ne12); + p.OW = static_cast(ne0); + p.OH = static_cast(ne1); + p.OD = static_cast(ne2); - if (avg_err > 0.01 || std::isnan(avg_err)) { - std::cerr << "m = " << first_err_m << " n = " << first_err_n << " b = " << first_err_b << std::endl; - std::cerr << "Actual result: " << std::endl << std::endl; - ggml_vk_print_matrix_area(d, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); - std::cerr << std::endl; - std::cerr << "Expected result: " << std::endl << std::endl; - ggml_vk_print_matrix_area(d_chk, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + // the shader clamps src addresses to p.IC * p.N * p.IW * p.IH * p.ID - 1 in uint32, so the + // total input element count must fit in a uint32. + GGML_ASSERT((uint64_t)p.IC * p.N * p.IW * p.IH * p.ID <= 0xFFFFFFFFull); - std::cerr << "src0: " << std::endl << std::endl; - ggml_vk_print_matrix_area(x, GGML_TYPE_F32, k, m, first_err_m, first_err_n, first_err_b); - std::cerr << std::endl; - std::cerr << "src1: " << std::endl << std::endl; - ggml_vk_print_matrix_area(y, GGML_TYPE_F32, k, n, first_err_m, first_err_n, first_err_b); + p.nb01 = static_cast(nb01 / nb00); + p.nb02 = static_cast(nb02 / nb00); + p.nb03 = static_cast(nb03 / nb00); - if (split_k > 1) { - float * split_k_buf = (float *) malloc(sizeof(float) * d_ne * split_k); - ggml_vk_buffer_read(ctx->prealloc_split_k, 0, split_k_buf, sizeof(float) * d_ne * split_k); + p.nb11 = static_cast(nb11 / nb10); + p.nb12 = static_cast(nb12 / nb10); + p.nb13 = static_cast(nb13 / nb10); - std::cerr << "d_buf0: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + p.nb1 = static_cast(nb1 / nb0); + p.nb2 = static_cast(nb2 / nb0); + p.nb3 = static_cast(nb3 / nb0); - std::cerr << "d_buf1: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_3D, std::move(p)); +} - std::cerr << "d_buf2: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + 2 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); +void ggml_vk_conv_2d_dw(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + vk_op_conv2d_dw_push_constants p{}; + p.ne = ggml_nelements(dst); + p.channels = dst->ne[2]; + p.batches = dst->ne[3]; + p.dst_w = dst->ne[0]; + p.dst_h = dst->ne[1]; + p.src_w = src1->ne[0]; + p.src_h = src1->ne[1]; + p.knl_w = src0->ne[0]; + p.knl_h = src0->ne[1]; + p.stride_x = dst->op_params[0]; + p.stride_y = dst->op_params[1]; + p.pad_x = dst->op_params[2]; + p.pad_y = dst->op_params[3]; + p.dilation_x = dst->op_params[4]; + p.dilation_y = dst->op_params[5]; - std::cerr << "d_buf3: " << std::endl << std::endl; - ggml_vk_print_matrix_area(split_k_buf + 3 * d_ne, GGML_TYPE_F32, m, n, first_err_m, first_err_n, first_err_b); + GGML_ASSERT(src0->ne[3] == p.channels); + GGML_ASSERT(src1->ne[3] == p.batches); - free(split_k_buf); - } - } + ggml_vk_op_f32(ctx, subctx, src0, src1, nullptr, nullptr, dst, GGML_OP_CONV_2D_DW, std::move(p)); +} - ggml_vk_destroy_buffer(qx_buf); - ggml_vk_destroy_buffer(y_buf); - ggml_vk_destroy_buffer(qy_buf); - ggml_vk_destroy_buffer(d_buf); +void ggml_vk_leaky_relu(ggml_backend_vk_context * ctx, vk_context& subctx, const ggml_tensor * src0, ggml_tensor * dst) { + const float * op_params = (const float *)dst->op_params; + vk_op_unary_push_constants p = vk_op_unary_push_constants_init(src0, dst); + p.param1 = op_params[0]; - free(x); - free(qx); - free(y); - free(d); - free(d_chk); + ggml_vk_op_f32(ctx, subctx, src0, nullptr, nullptr, nullptr, dst, GGML_OP_LEAKY_RELU, std::move(p)); } -#endif -static void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_context subctx) { +void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_context subctx) { #if defined(GGML_VULKAN_RUN_TESTS) const std::vector vals { 512, 512, 128, @@ -15154,7 +11896,7 @@ static void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_contex ctx->prealloc_y = ggml_vk_create_buffer_device(ctx->device, ctx->prealloc_size_y); ctx->prealloc_y_last_pipeline_used = nullptr; ctx->prealloc_y_last_tensor_used = nullptr; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; } if (ctx->prealloc_split_k == nullptr || (ctx->prealloc_size_split_k > 0 && ctx->prealloc_split_k->size < ctx->prealloc_size_split_k)) { VK_LOG_MEMORY("ggml_vk_preallocate_buffers(split_k_size: " << ctx->prealloc_size_split_k << ")"); @@ -15174,11 +11916,7 @@ static void ggml_vk_preallocate_buffers(ggml_backend_vk_context * ctx, vk_contex } } -static void ggml_vk_compute_forward(ggml_backend_vk_context* ctx, ggml_cgraph * cgraph, ggml_tensor* tensor, int tensor_idx, bool almost_ready); - -// Returns true if node has enqueued work into the queue, false otherwise -// If submit is true the current all operations queued so far are being submitted to Vulkan to overlap cmdlist creation and GPU execution. -static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int node_idx, ggml_tensor *node_begin, int node_idx_begin, bool last_node, bool almost_ready, bool submit){ +bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int node_idx, ggml_tensor *node_begin, int node_idx_begin, bool last_node, bool almost_ready, bool submit){ ggml_tensor * node = cgraph->nodes[node_idx]; if (ggml_is_empty(node) || ggml_op_is_empty(node->op) || !node->buffer) { return false; @@ -15315,6 +12053,9 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr } } + // closed explicitly below, and by the destructor on the early returns + ggml_vk_debug_label dbg(compute_ctx, cgraph, node_idx, ctx->num_additional_fused_ops); + switch (node->op) { case GGML_OP_REPEAT: ggml_vk_repeat(ctx, compute_ctx, src0, node); @@ -15330,7 +12071,11 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr break; case GGML_OP_GET_ROWS: - ggml_vk_get_rows(ctx, compute_ctx, src0, src1, node); + if (ctx->fused_topk_qsa) { + ggml_vk_topk_qsa(ctx, compute_ctx, cgraph, node_idx); + } else { + ggml_vk_get_rows(ctx, compute_ctx, src0, src1, node); + } break; case GGML_OP_GET_ROWS_BACK: @@ -15426,6 +12171,10 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr case GGML_OP_PAD: ggml_vk_pad(ctx, compute_ctx, src0, node); + break; + case GGML_OP_PAD_REFLECT_1D: + ggml_vk_pad_reflect_1d(ctx, compute_ctx, src0, node); + break; case GGML_OP_ROLL: ggml_vk_roll(ctx, compute_ctx, src0, node); @@ -15469,6 +12218,10 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr ggml_vk_topk_moe(ctx, compute_ctx, cgraph, node_idx); break; } + if (ctx->num_additional_fused_ops) { + ggml_vk_unary_mul(ctx, compute_ctx, cgraph, node_idx); + break; + } switch (ggml_get_unary_op(node)) { case GGML_UNARY_OP_ELU: @@ -15509,6 +12262,7 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: ggml_vk_glu(ctx, compute_ctx, src0, src1, node); break; default: @@ -15562,6 +12316,18 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr case GGML_OP_CUMSUM: ggml_vk_cumsum(ctx, compute_ctx, src0, node); + break; + case GGML_OP_DSV4_HC_COMB: + ggml_vk_dsv4_hc_comb(ctx, compute_ctx, src0, src1, src2, node); + + break; + case GGML_OP_DSV4_HC_PRE: + ggml_vk_dsv4_hc_pre(ctx, compute_ctx, src0, src1, node); + + break; + case GGML_OP_DSV4_HC_POST: + ggml_vk_dsv4_hc_post(ctx, compute_ctx, src0, src1, src2, src3, node); + break; case GGML_OP_MEAN: ggml_vk_mean(ctx, compute_ctx, src0, node); @@ -15570,6 +12336,14 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr case GGML_OP_ARGMAX: ggml_vk_argmax(ctx, compute_ctx, src0, node); + break; + case GGML_OP_CROSS_ENTROPY_LOSS: + ggml_vk_cross_entropy_loss(ctx, compute_ctx, node); + + break; + case GGML_OP_CROSS_ENTROPY_LOSS_BACK: + ggml_vk_cross_entropy_loss_back(ctx, compute_ctx, node); + break; case GGML_OP_COUNT_EQUAL: ggml_vk_count_equal(ctx, compute_ctx, src0, src1, node); @@ -15653,6 +12427,11 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr break; + case GGML_OP_LIGHTNING_INDEXER: + ggml_vk_lightning_indexer(ctx, compute_ctx, node); + + break; + case GGML_OP_GATED_DELTA_NET: ggml_vk_gated_delta_net(ctx, compute_ctx, node); @@ -15681,6 +12460,9 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr return false; } + // the submit path below can end the command buffer, so close the region first + dbg.close(); + ctx->tensor_ctxs[node_idx] = compute_ctx; #if defined(GGML_VULKAN_CHECK_RESULTS) @@ -15707,7 +12489,7 @@ static bool ggml_vk_build_graph(ggml_backend_vk_context * ctx, ggml_cgraph * cgr return true; } -static void ggml_vk_compute_forward(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, ggml_tensor * tensor, int tensor_idx, bool almost_ready = false) { +void ggml_vk_compute_forward(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, ggml_tensor * tensor, int tensor_idx, bool almost_ready) { GGML_UNUSED(cgraph); GGML_UNUSED(tensor); @@ -15757,12 +12539,11 @@ static void ggml_vk_compute_forward(ggml_backend_vk_context * ctx, ggml_cgraph * } } -// Clean up after graph processing is done -static void ggml_vk_graph_cleanup(ggml_backend_vk_context * ctx) { +void ggml_vk_graph_cleanup(ggml_backend_vk_context * ctx) { VK_LOG_DEBUG("ggml_vk_graph_cleanup()"); ctx->prealloc_y_last_pipeline_used = {}; ctx->prealloc_y_last_tensor_used = nullptr; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; ctx->unsynced_nodes_written.clear(); ctx->unsynced_nodes_read.clear(); @@ -15796,8 +12577,7 @@ static void ggml_vk_graph_cleanup(ggml_backend_vk_context * ctx) { ctx->descriptor_set_idx = 0; } -// Clean up on backend free -static void ggml_vk_cleanup(ggml_backend_vk_context * ctx) { +void ggml_vk_cleanup(ggml_backend_vk_context * ctx) { VK_LOG_DEBUG("ggml_vk_cleanup(" << ctx->name << ")"); // discard any unsubmitted command buffers ctx->compute_ctx.reset(); @@ -15814,7 +12594,7 @@ static void ggml_vk_cleanup(ggml_backend_vk_context * ctx) { ctx->prealloc_y_last_pipeline_used = nullptr; ctx->prealloc_y_last_tensor_used = nullptr; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; ctx->prealloc_size_x = 0; ctx->prealloc_size_y = 0; @@ -15845,13 +12625,13 @@ static void ggml_vk_cleanup(ggml_backend_vk_context * ctx) { } } -static int ggml_vk_get_device_count() { +int ggml_vk_get_device_count() { ggml_vk_instance_init(); return vk_instance.device_indices.size(); } -static void ggml_vk_get_device_description(int device, char * description, size_t description_size) { +void ggml_vk_get_device_description(int device, char * description, size_t description_size) { ggml_vk_instance_init(); std::vector devices = vk_instance.instance.enumeratePhysicalDevices(); @@ -15862,30 +12642,24 @@ static void ggml_vk_get_device_description(int device, char * description, size_ snprintf(description, description_size, "%s", props.deviceName.data()); } -// backend interface - -#define UNUSED GGML_UNUSED - -// device backend - -static bool ggml_backend_buffer_is_vk(ggml_backend_buffer_t buffer) { +bool ggml_backend_buffer_is_vk(ggml_backend_buffer_t buffer) { return buffer->buft->iface.get_name == ggml_backend_vk_buffer_type_name; } -static void ggml_backend_vk_buffer_free_buffer(ggml_backend_buffer_t buffer) { +void ggml_backend_vk_buffer_free_buffer(ggml_backend_buffer_t buffer) { VK_LOG_MEMORY("ggml_backend_vk_buffer_free_buffer()"); ggml_backend_vk_buffer_context * ctx = (ggml_backend_vk_buffer_context *)buffer->context; ggml_vk_destroy_buffer(ctx->dev_buffer); delete ctx; } -static void * ggml_backend_vk_buffer_get_base(ggml_backend_buffer_t buffer) { +void * ggml_backend_vk_buffer_get_base(ggml_backend_buffer_t buffer) { return vk_ptr_base; UNUSED(buffer); } -static enum ggml_status ggml_backend_vk_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor) { +enum ggml_status ggml_backend_vk_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor) { VK_LOG_DEBUG("ggml_backend_vk_buffer_init_tensor(" << buffer << " (" << buffer->context << "), " << tensor << ")"); if (tensor->view_src != nullptr) { GGML_ASSERT(tensor->view_src->buffer->buft == buffer->buft); @@ -15893,7 +12667,7 @@ static enum ggml_status ggml_backend_vk_buffer_init_tensor(ggml_backend_buffer_t return GGML_STATUS_SUCCESS; } -static void ggml_backend_vk_buffer_memset_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) { +void ggml_backend_vk_buffer_memset_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) { VK_LOG_DEBUG("ggml_backend_vk_buffer_memset_tensor(" << buffer << ", " << tensor << ", " << value << ", " << offset << ", " << size << ")"); ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)buffer->context; vk_buffer buf = buf_ctx->dev_buffer; @@ -15906,7 +12680,7 @@ static void ggml_backend_vk_buffer_memset_tensor(ggml_backend_buffer_t buffer, g ggml_vk_buffer_memset(buf, vk_tensor_offset(tensor) + tensor->view_offs + offset, val32, size); } -static void ggml_backend_vk_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { +void ggml_backend_vk_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size) { VK_LOG_DEBUG("ggml_backend_vk_buffer_set_tensor(" << buffer << ", " << tensor << ", " << data << ", " << offset << ", " << size << ")"); ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)buffer->context; vk_buffer buf = buf_ctx->dev_buffer; @@ -15918,7 +12692,7 @@ static void ggml_backend_vk_buffer_set_tensor(ggml_backend_buffer_t buffer, ggml ggml_vk_buffer_write(buf, vk_tensor_offset(tensor) + tensor->view_offs + offset, data, size); } -static void ggml_backend_vk_buffer_set_tensor_2d(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, +void ggml_backend_vk_buffer_set_tensor_2d(ggml_backend_buffer_t buffer, ggml_tensor * tensor, const void * data, size_t offset, size_t size, size_t n_copies, size_t stride_tensor, size_t stride_data) { VK_LOG_DEBUG("ggml_backend_vk_buffer_set_tensor_2d(" << buffer << ", " << tensor << ", " << data << ", " << offset << ", " << size << ", " << n_copies << ", " << stride_tensor << ", " << stride_data << ")"); @@ -15932,7 +12706,7 @@ static void ggml_backend_vk_buffer_set_tensor_2d(ggml_backend_buffer_t buffer, g ggml_vk_buffer_write_2d(buf, vk_tensor_offset(tensor) + tensor->view_offs + offset, data, stride_data, stride_tensor, size, n_copies); } -static void ggml_backend_vk_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { +void ggml_backend_vk_buffer_get_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size) { VK_LOG_DEBUG("ggml_backend_vk_buffer_get_tensor(" << buffer << ", " << tensor << ", " << data << ", " << offset << ", " << size << ")"); ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)buffer->context; @@ -15945,7 +12719,7 @@ static void ggml_backend_vk_buffer_get_tensor(ggml_backend_buffer_t buffer, cons ggml_vk_buffer_read(buf, vk_tensor_offset(tensor) + tensor->view_offs + offset, data, size); } -static void ggml_backend_vk_buffer_get_tensor_2d(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, +void ggml_backend_vk_buffer_get_tensor_2d(ggml_backend_buffer_t buffer, const ggml_tensor * tensor, void * data, size_t offset, size_t size, size_t n_copies, size_t stride_tensor, size_t stride_data) { VK_LOG_DEBUG("ggml_backend_vk_buffer_get_tensor_2d(" << buffer << ", " << tensor << ", " << data << ", " << offset << ", " << size << ", " << n_copies << ", " << stride_tensor << ", " << stride_data << ")"); @@ -15960,7 +12734,7 @@ static void ggml_backend_vk_buffer_get_tensor_2d(ggml_backend_buffer_t buffer, c ggml_vk_buffer_read_2d(buf, vk_tensor_offset(tensor) + tensor->view_offs + offset, data, stride_tensor, stride_data, size, n_copies); } -static bool ggml_backend_vk_buffer_cpy_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * src, ggml_tensor * dst) { +bool ggml_backend_vk_buffer_cpy_tensor(ggml_backend_buffer_t buffer, const ggml_tensor * src, ggml_tensor * dst) { if (ggml_nbytes(src) == 0) { return true; } @@ -15981,34 +12755,19 @@ static bool ggml_backend_vk_buffer_cpy_tensor(ggml_backend_buffer_t buffer, cons UNUSED(buffer); } -static void ggml_backend_vk_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) { +void ggml_backend_vk_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) { ggml_backend_vk_buffer_context * ctx = (ggml_backend_vk_buffer_context *)buffer->context; ggml_vk_buffer_memset(ctx->dev_buffer, 0, value, buffer->size); } -static ggml_backend_buffer_i ggml_backend_vk_buffer_interface = { - /* .free_buffer = */ ggml_backend_vk_buffer_free_buffer, - /* .get_base = */ ggml_backend_vk_buffer_get_base, - /* .init_tensor = */ ggml_backend_vk_buffer_init_tensor, - /* .memset_tensor = */ ggml_backend_vk_buffer_memset_tensor, - /* .set_tensor = */ ggml_backend_vk_buffer_set_tensor, - /* .get_tensor = */ ggml_backend_vk_buffer_get_tensor, - /* .set_tensor_2d = */ ggml_backend_vk_buffer_set_tensor_2d, - /* .get_tensor_2d = */ ggml_backend_vk_buffer_get_tensor_2d, - /* .cpy_tensor = */ ggml_backend_vk_buffer_cpy_tensor, - /* .clear = */ ggml_backend_vk_buffer_clear, - /* .reset = */ NULL, -}; - -// vk buffer type -static const char * ggml_backend_vk_buffer_type_name(ggml_backend_buffer_type_t buft) { +const char * ggml_backend_vk_buffer_type_name(ggml_backend_buffer_type_t buft) { ggml_backend_vk_buffer_type_context * ctx = (ggml_backend_vk_buffer_type_context *)buft->context; return ctx->name.c_str(); } -static ggml_backend_buffer_t ggml_backend_vk_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) { +ggml_backend_buffer_t ggml_backend_vk_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) { VK_LOG_MEMORY("ggml_backend_vk_buffer_type_alloc_buffer(" << size << ")"); ggml_backend_vk_buffer_type_context * ctx = (ggml_backend_vk_buffer_type_context *) buft->context; @@ -16024,17 +12783,17 @@ static ggml_backend_buffer_t ggml_backend_vk_buffer_type_alloc_buffer(ggml_backe return ggml_backend_buffer_init(buft, ggml_backend_vk_buffer_interface, bufctx, size); } -static size_t ggml_backend_vk_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) { +size_t ggml_backend_vk_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) { ggml_backend_vk_buffer_type_context * ctx = (ggml_backend_vk_buffer_type_context *) buft->context; return ctx->device->properties.limits.minStorageBufferOffsetAlignment; } -static size_t ggml_backend_vk_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) { +size_t ggml_backend_vk_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) { ggml_backend_vk_buffer_type_context * ctx = (ggml_backend_vk_buffer_type_context *) buft->context; return ctx->device->suballocation_block_size; } -static size_t ggml_backend_vk_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor) { +size_t ggml_backend_vk_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor) { return ggml_nbytes(tensor); UNUSED(buft); @@ -16050,8 +12809,6 @@ ggml_backend_buffer_type_t ggml_backend_vk_buffer_type(size_t dev_num) { return &dev->buffer_type; } -// host buffer type - static const char * ggml_backend_vk_host_buffer_type_name(ggml_backend_buffer_type_t buft) { return GGML_VK_NAME "_Host"; @@ -16097,8 +12854,6 @@ static size_t ggml_backend_vk_host_buffer_type_get_max_size(ggml_backend_buffer_ UNUSED(buft); } -// Should be changed to return device-specific host buffer type -// but that probably requires changes in llama.cpp ggml_backend_buffer_type_t ggml_backend_vk_host_buffer_type() { static struct ggml_backend_buffer_type ggml_backend_vk_buffer_type_host = { /* .iface = */ { @@ -16120,16 +12875,13 @@ ggml_backend_buffer_type_t ggml_backend_vk_host_buffer_type() { return &ggml_backend_vk_buffer_type_host; } - -// backend - static const char * ggml_backend_vk_name(ggml_backend_t backend) { ggml_backend_vk_context * ctx = (ggml_backend_vk_context *)backend->context; return ctx->name.c_str(); } -static void ggml_backend_vk_free(ggml_backend_t backend) { +void ggml_backend_vk_free(ggml_backend_t backend) { ggml_backend_vk_context * ctx = (ggml_backend_vk_context *)backend->context; VK_LOG_DEBUG("ggml_backend_vk_free(" << ctx->name << ")"); @@ -16304,6 +13056,22 @@ static bool ggml_backend_vk_cpy_tensor_async(ggml_backend_t backend_src, ggml_ba return false; } + // If the backend is idle, use a CPU copy to avoid GPU synchronization overhead. + static constexpr size_t max_cpu_copy_size = 128 * 1024; + const bool src_backend_synchronous = backend_src->iface.synchronize == nullptr; + const bool transfer_idle = !ctx->device->async_use_transfer_queue || + ctx->transfer_semaphore_last_submitted == ctx->transfer_semaphore.value; + const bool backend_idle = ctx->compute_ctx.expired() && ctx->transfer_ctx.expired() && + !ctx->submit_pending && !ctx->almost_ready_fence_pending && transfer_idle; + const bool dst_host_coherent = + (dst_buf->memory_property_flags & (vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent)) == + (vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent); + + if ((backend_src == backend_dst || src_backend_synchronous) && backend_idle && dst_host_coherent && ggml_nbytes(src) <= max_cpu_copy_size) { + ggml_vk_buffer_write(dst_buf, vk_tensor_offset(dst) + dst->view_offs, src->data, ggml_nbytes(src)); + return true; + } + vk_context cpy_ctx; if (ctx->device->async_use_transfer_queue) { cpy_ctx = ggml_vk_get_transfer_ctx(ctx); @@ -16316,11 +13084,10 @@ static bool ggml_backend_vk_cpy_tensor_async(ggml_backend_t backend_src, ggml_ba src->data, ggml_nbytes(src)); } - GGML_UNUSED(backend_src); return false; } -static void ggml_vk_synchronize(ggml_backend_vk_context * ctx) { +void ggml_vk_synchronize(ggml_backend_vk_context * ctx) { VK_LOG_DEBUG("ggml_vk_synchronize()"); bool do_transfer = !ctx->compute_ctx.expired(); @@ -16400,16 +13167,58 @@ static void ggml_backend_vk_synchronize(ggml_backend_t backend) { ggml_vk_graph_cleanup(ctx); } -static bool ggml_vk_is_empty(ggml_tensor * node) { +bool ggml_vk_is_empty(ggml_tensor * node) { return ggml_is_empty(node) || node->op == GGML_OP_NONE || node->op == GGML_OP_RESHAPE || node->op == GGML_OP_TRANSPOSE || node->op == GGML_OP_VIEW || node->op == GGML_OP_PERMUTE; } -static bool ggml_vk_can_fuse(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, std::initializer_list ops) { +static bool ggml_vk_can_fuse_unary_mul(const struct ggml_cgraph * cgraph, int unary_idx, int mul_idx) { + const ggml_tensor * unary = cgraph->nodes[unary_idx]; + const ggml_tensor * mul = cgraph->nodes[mul_idx]; + + if (ggml_vk_unary_mul_op_index(ggml_get_unary_op(unary)) < 0) { + return false; + } + if (unary->type != GGML_TYPE_F32 && unary->type != GGML_TYPE_F16) { + return false; + } + if (unary->type != mul->type) { + return false; + } + if (mul->src[0] != unary && mul->src[1] != unary) { + return false; + } + const ggml_tensor * other = (mul->src[0] == unary) ? mul->src[1] : mul->src[0]; + if (other == nullptr || other->type != unary->type) { + return false; + } + if (!ggml_is_contiguous_1(other) || !ggml_is_contiguous_1(unary->src[0])) { + return false; + } + // fastmod needs src to tile into dst + if (mul->src[0] == unary) { + return ggml_can_repeat(other, unary); + } + return ggml_can_repeat(unary, mul->src[0]); +} + +static bool ggml_vk_can_fuse_unary_mul_pair(const struct ggml_cgraph * cgraph, int node_idx) { + const enum ggml_op ops[] = { GGML_OP_UNARY, GGML_OP_MUL }; + const int outputs[] = { node_idx + 1 }; + return ggml_can_fuse_subgraph(cgraph, node_idx, 2, ops, outputs, 1) && + ggml_vk_can_fuse_unary_mul(cgraph, node_idx, node_idx + 1); +} + +bool ggml_vk_can_fuse(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, std::initializer_list ops) { + if (ops.size() == 2 && ops.begin()[0] == GGML_OP_UNARY && ops.begin()[1] == GGML_OP_MUL) { + return ggml_vk_can_fuse_unary_mul_pair(cgraph, node_idx); + } + if (!ggml_can_fuse(cgraph, node_idx, ops)) { return false; } - if (ops.size() == 2 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { + if ((ops.size() == 2 || ops.size() == 3 || ops.size() == 4) && + ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { // additional constraints specific to this fusion const ggml_tensor *rms_norm = cgraph->nodes[node_idx]; const ggml_tensor *mul = cgraph->nodes[node_idx + 1]; @@ -16431,7 +13240,45 @@ static bool ggml_vk_can_fuse(const ggml_backend_vk_context * ctx, const struct g if (!ggml_is_contiguous_rows(mul->src[0]) || !ggml_is_contiguous_rows(mul->src[1])) { return false; } + + if (ops.size() >= 3 && ops.begin()[2] == GGML_OP_ADD) { + const ggml_tensor *add = cgraph->nodes[node_idx + 2]; + const ggml_tensor *residual = add->src[0] == mul ? add->src[1] : add->src[0]; + if (add->src[0] != mul && add->src[1] != mul) { + return false; + } + if (residual->type != GGML_TYPE_F32 || add->type != GGML_TYPE_F32 || + !ggml_are_same_shape(add, residual) || !ggml_is_contiguous(residual) || + !ggml_is_contiguous(add) || get_misalign_bytes(ctx, residual) != 0) { + return false; + } + + const ggml_tensor *dst = add; + if (ops.size() == 4) { + if (ops.begin()[3] != GGML_OP_MUL) { + return false; + } + + const ggml_tensor *post_mul = cgraph->nodes[node_idx + 3]; + const ggml_tensor *scale = post_mul->src[0] == add ? post_mul->src[1] : post_mul->src[0]; + if (post_mul->src[0] != add && post_mul->src[1] != add) { + return false; + } + // The shader reads data_e[0], so the final multiply must use a scalar. + if (scale->type != GGML_TYPE_F32 || post_mul->type != GGML_TYPE_F32 || + ggml_nelements(scale) != 1 || !ggml_is_contiguous(post_mul) || + get_misalign_bytes(ctx, scale) != 0) { + return false; + } + dst = post_mul; + } + + if (get_misalign_bytes(ctx, dst) != 0) { + return false; + } + } } + auto const &mm_add_ok = [&](const ggml_tensor *mul, const ggml_tensor *add) { const ggml_tensor *bias = add->src[0] == mul ? add->src[1] : add->src[0]; @@ -16557,8 +13404,7 @@ static bool ggml_vk_can_fuse(const ggml_backend_vk_context * ctx, const struct g return true; } -// Match SSM_CONV + UNARY(SILU) or SSM_CONV + ADD + UNARY(SILU). num_extra is 1 or 2. -static bool ggml_vk_can_fuse_ssm_conv(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, +bool ggml_vk_can_fuse_ssm_conv(const ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, int num_extra) { const ggml_tensor * conv = cgraph->nodes[node_idx]; if (conv->op != GGML_OP_SSM_CONV) { @@ -16613,7 +13459,7 @@ static bool ggml_vk_can_fuse_ssm_conv(const ggml_backend_vk_context * ctx, const return true; } -static bool ggml_vk_can_fuse_topk_moe(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, +bool ggml_vk_can_fuse_topk_moe(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx, topk_moe_mode mode) { const ggml_tensor * softmax; @@ -16724,14 +13570,97 @@ static bool ggml_vk_can_fuse_topk_moe(ggml_backend_vk_context * ctx, const struc return true; } -static bool ggml_vk_can_fuse_rope_set_rows(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, +static bool ggml_vk_match_ops(const struct ggml_cgraph * cgraph, int node_idx, + const std::initializer_list & ops) { + if (node_idx + (int) ops.size() > cgraph->n_nodes) { + return false; + } + for (size_t j = 0; j < ops.size(); ++j) { + const ggml_tensor * node = cgraph->nodes[node_idx + j]; + if (node->op != ops.begin()[j] || + (node->flags & GGML_TENSOR_FLAG_COMPUTE) == 0 || + (node->flags & GGML_TENSOR_FLAG_OUTPUT) != 0) { + return false; + } + } + return true; +} + +bool ggml_vk_can_fuse_topk_qsa(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { + if (ctx->device->disable_fusion || !ctx->device->pipeline_topk_radix_qsa) { + return false; + } + + const int n_ops = topk_qsa_pattern.size(); + if (!ggml_vk_match_ops(cgraph, node_idx, topk_qsa_pattern) || + !ggml_check_edges(cgraph, node_idx, topk_qsa_edges)) { + return false; + } + + // elided nodes must be single-use (cpy counts its own src[1] self-reference) + for (int j = 0; j < n_ops - 1; ++j) { + const ggml_tensor * node = cgraph->nodes[node_idx + j]; + const int32_t want = node->op == GGML_OP_CPY ? 2 : 1; + if (ggml_node_get_use_count(cgraph, node_idx + j) != want) { + return false; + } + } + + const ggml_tensor * get_rows = cgraph->nodes[node_idx + 0]; + const ggml_tensor * add = cgraph->nodes[node_idx + n_ops - 2]; + const ggml_tensor * top_k = cgraph->nodes[node_idx + n_ops - 1]; + + const ggml_tensor * scores = get_rows->src[0]; // [n_tps, n_blocks, n_stream] + const ggml_tensor * cell_blk = get_rows->src[1]; // [n_kv, n_stream] + const ggml_tensor * expanded = add->src[0]; // [n_kv, n_tps, n_stream] + + // raw mask: follow the reshape/cpy chain back to the materialized f16 input + const ggml_tensor * mask = add->src[1]; + while (mask && (mask->op == GGML_OP_RESHAPE || mask->op == GGML_OP_CPY)) { + mask = mask->src[0]; + } + if (!mask || mask->type != GGML_TYPE_F16) { + return false; + } + + if (scores->type != GGML_TYPE_F32 || cell_blk->type != GGML_TYPE_I32 || top_k->type != GGML_TYPE_I32) { + return false; + } + if (!ggml_is_contiguous(scores) || !ggml_is_contiguous(cell_blk) || !ggml_is_contiguous(mask) || + !ggml_is_contiguous(expanded) || !ggml_is_contiguous(top_k)) { + return false; + } + + const int64_t n_tps = scores->ne[0]; + const int64_t n_blocks = scores->ne[1]; + const int64_t n_stream = scores->ne[2]; + const int64_t n_kv = cell_blk->ne[0]; + const int64_t width = top_k->ne[0]; + + // pin the indexer layout the shader's addressing assumes + if (scores->ne[3] != 1 || cell_blk->ne[1] != n_stream || ggml_nrows(cell_blk) != n_stream || + ggml_nelements(mask) != n_kv * n_tps * n_stream || + expanded->ne[0] != n_kv || expanded->ne[1] != n_tps || expanded->ne[2] != n_stream || + top_k->ne[1] != n_tps || top_k->ne[2] != n_stream || top_k->ne[3] != 1 || + n_blocks <= 0 || n_kv <= 0 || width <= 0 || width > n_kv) { + return false; + } + + // only worth it in the radix regime; small k uses the faster tournament unfused + const uint32_t k_min_pipeline = std::max((uint32_t) log2f(float(width)) + 1, ctx->device->subgroup_size_log2); + if (k_min_pipeline < num_topk_pipelines && ctx->device->pipeline_topk_f32[k_min_pipeline]) { + return false; + } + return true; +} + +bool ggml_vk_can_fuse_rope_set_rows(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { - GGML_UNUSED(ctx); const ggml_tensor *rope = cgraph->nodes[node_idx + 0]; const ggml_tensor *view = cgraph->nodes[node_idx + 1]; const ggml_tensor *set_rows = cgraph->nodes[node_idx + 2]; - // ne3 not tested + // The set_rows epilogue uses one index per ne2 slice and does not encode ne3. if (rope->src[0]->ne[3] != 1) { return false; } @@ -16740,29 +13669,57 @@ static bool ggml_vk_can_fuse_rope_set_rows(ggml_backend_vk_context * ctx, const return false; } - if (set_rows->src[1]->type != GGML_TYPE_I64) { + // The shader reads each aligned I64 index as a uvec2 and uses its low 32 bits. + if (set_rows->src[1]->type != GGML_TYPE_I64 || !ggml_is_contiguous(set_rows->src[1]) || + set_rows->nb[0] != ggml_type_size(set_rows->type) || get_misalign_bytes(ctx, set_rows->src[1]) != 0) { return false; } - // The view should flatten two dims of rope into one dim + // SET_ROWS consumes one flattened [ne0*ne1] row for each ne2 slice. if (!ggml_is_contiguous(view) || - view->ne[0] != rope->ne[0] * rope->ne[1]) { + view->ne[0] != rope->ne[0] * rope->ne[1] || view->ne[1] != rope->ne[2] || + view->ne[2] != 1 || view->ne[3] != 1 || + ggml_nelements(set_rows->src[1]) != rope->ne[2]) { return false; } - // Only norm/neox/mrope shaders have the fusion code + // Only norm/neox/mrope/imrope shaders have the fusion code const int mode = ((const int32_t *) rope->op_params)[2]; - if (mode != GGML_ROPE_TYPE_NORMAL && mode != GGML_ROPE_TYPE_NEOX && mode != GGML_ROPE_TYPE_MROPE) { + if (mode != GGML_ROPE_TYPE_NORMAL && mode != GGML_ROPE_TYPE_NEOX && + mode != GGML_ROPE_TYPE_MROPE && mode != GGML_ROPE_TYPE_IMROPE) { + return false; + } + + return true; +} + +bool ggml_vk_can_fuse_rms_norm_set_rows(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, + int node_idx) { + const ggml_tensor * rms = cgraph->nodes[node_idx]; + const ggml_tensor * view = cgraph->nodes[node_idx + 1]; + const ggml_tensor * set_rows = cgraph->nodes[node_idx + 2]; + + // The RMS kernel reads F32 and writes directly to the F32 or F16 SET_ROWS destination. + if (rms->src[0]->type != GGML_TYPE_F32 || rms->type != GGML_TYPE_F32 || + (set_rows->type != GGML_TYPE_F32 && set_rows->type != GGML_TYPE_F16) || + set_rows->src[1]->type != GGML_TYPE_I64 || !ggml_is_contiguous(set_rows->src[1]) || + set_rows->nb[0] != ggml_type_size(set_rows->type) || get_misalign_bytes(ctx, set_rows->src[1]) != 0) { + return false; + } + // As with the ROPE epilogue, each ne2 slice supplies one flattened row and ne3 is not encoded. + if (rms->ne[3] != 1 || !ggml_is_contiguous(rms->src[0]) || !ggml_is_contiguous(view)) { + return false; + } + if (view->ne[0] != rms->ne[0] * rms->ne[1] || view->ne[1] != rms->ne[2] || + view->ne[2] != 1 || view->ne[3] != 1 || + ggml_nelements(set_rows->src[1]) != rms->ne[2]) { return false; } return true; } -// Pattern check for the 5-op Snake fusion: mul -> sin -> sqr -> mul -> add. -// Verifies the chain shape, the closure x_in_add == x_in_mul0, and that -// the broadcast operands a and inv_b share a [1, C] layout. -static bool ggml_vk_can_fuse_snake(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { +bool ggml_vk_can_fuse_snake(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { GGML_UNUSED(ctx); if (!ggml_can_fuse(cgraph, node_idx, snake_pattern)) { return false; @@ -16818,11 +13775,7 @@ static bool ggml_vk_can_fuse_snake(ggml_backend_vk_context * ctx, const struct g return true; } -// Check whether the tensors overlap in memory. -// Fusions can potentially overwrite src tensors in ways that are not prevented -// by ggml-alloc. If the fusion src is being applied in a way that's elementwise -// with the destination, then it's OK for them to overlap if they are exactly equal. -static bool ggml_vk_tensors_overlap(const ggml_tensor * a, const ggml_tensor * b, bool elementwise) { +bool ggml_vk_tensors_overlap(const ggml_tensor * a, const ggml_tensor * b, bool elementwise) { ggml_backend_vk_buffer_context * a_buf_ctx = (ggml_backend_vk_buffer_context *)a->buffer->context; vk_buffer a_buf = a_buf_ctx->dev_buffer; ggml_backend_vk_buffer_context * b_buf_ctx = (ggml_backend_vk_buffer_context *)b->buffer->context; @@ -16845,9 +13798,8 @@ static bool ggml_vk_tensors_overlap(const ggml_tensor * a, const ggml_tensor * b return false; } -static bool ggml_vk_can_fuse_rms_norm_mul_rope(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, +bool ggml_vk_can_fuse_rms_norm_mul_rope(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { - GGML_UNUSED(ctx); const ggml_tensor *rms = cgraph->nodes[node_idx + 0]; const ggml_tensor *mul = cgraph->nodes[node_idx + 1]; const ggml_tensor *rope = cgraph->nodes[node_idx + 2]; @@ -16879,7 +13831,7 @@ static bool ggml_vk_can_fuse_rms_norm_mul_rope(ggml_backend_vk_context * ctx, co return true; } -static uint32_t ggml_vk_fuse_multi_add(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { +uint32_t ggml_vk_fuse_multi_add(ggml_backend_vk_context * ctx, const struct ggml_cgraph * cgraph, int node_idx) { const ggml_tensor *first_node = cgraph->nodes[node_idx]; if (first_node->op != GGML_OP_ADD) { @@ -16952,14 +13904,13 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg ctx->device->diag_prev_end = -1; if (vk_instance.debug_utils_support) { - vk::DebugUtilsLabelEXT dul = {}; - dul.pLabelName = "ggml_backend_vk_graph_compute"; - dul.color = std::array{1.0f, 1.0f, 1.0f, 1.0f}; - - std::lock_guard guard(*ctx->device->compute_queue->handle); - vk_instance.pfn_vkQueueBeginDebugUtilsLabelEXT(ctx->device->compute_queue->handle->queue, reinterpret_cast(&dul)); + ctx->device->debug_cmdbuf_idx = 0; } + // queue scope, so it encloses every submit this evaluation makes. + // closed when the function returns + ggml_vk_debug_label queue_dbg(ctx->device->compute_queue->handle.get(), "ggml_backend_vk_graph_compute"); + ctx->prealloc_size_add_rms_partials_offset = 0; ctx->do_add_rms_partials = false; ctx->do_add_rms_partials_offset_calculation = false; @@ -17012,7 +13963,7 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg ctx->prealloc_y_last_pipeline_used = nullptr; ctx->prealloc_y_last_tensor_used = nullptr; - ctx->prealloc_y_last_decode_vector_staging = false; + ctx->prealloc_y_last_k_padded = false; if (ctx->prealloc_size_add_rms_partials) { ggml_vk_preallocate_buffers(ctx, nullptr); @@ -17103,6 +14054,8 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg ctx->fused_topk_moe_mode = TOPK_MOE_COUNT; ctx->fused_topk_moe_scale = false; + ctx->fused_topk_qsa = false; + ctx->fused_rms_norm_mode = RMS_NORM_COUNT; const char *fusion_string {}; if (!ctx->device->disable_fusion) { uint32_t num_adds = ggml_vk_fuse_multi_add(ctx, cgraph, i); @@ -17137,32 +14090,62 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg fusion_string = "MUL_MAT_ID_MUL"; op_srcs_fused_elementwise[0] = false; op_srcs_fused_elementwise[1] = true; - } else if (ggml_can_fuse_subgraph(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ROPE, GGML_OP_VIEW, GGML_OP_SET_ROWS }, { i + 4 }) && + } else if (ggml_can_fuse_subgraph(cgraph, i, rms_norm_mul_rope_view_set_rows_pattern, { i + 4 }) && ggml_check_edges(cgraph, i, rms_norm_mul_rope_view_set_rows_edges) && ggml_vk_can_fuse_rms_norm_mul_rope(ctx, cgraph, i) && ggml_vk_can_fuse_rope_set_rows(ctx, cgraph, i + 2)) { ctx->num_additional_fused_ops = 4; + ctx->fused_rms_norm_mode = RMS_NORM_MUL_ROPE_VIEW_SET_ROWS; fusion_string = "RMS_NORM_MUL_ROPE_VIEW_SET_ROWS"; op_srcs_fused_elementwise[0] = false; op_srcs_fused_elementwise[1] = false; op_srcs_fused_elementwise[2] = false; op_srcs_fused_elementwise[3] = false; op_srcs_fused_elementwise[4] = false; - } else if (ggml_vk_can_fuse(ctx, cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ROPE })&& + } else if (ggml_vk_can_fuse(ctx, cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ROPE }) && ggml_vk_can_fuse_rms_norm_mul_rope(ctx, cgraph, i)) { ctx->num_additional_fused_ops = 2; + ctx->fused_rms_norm_mode = RMS_NORM_MUL_ROPE; fusion_string = "RMS_NORM_MUL_ROPE"; // rope is approximately elementwise - whole rows are done by a single workgroup and it's row-wise op_srcs_fused_elementwise[0] = false; op_srcs_fused_elementwise[1] = true; op_srcs_fused_elementwise[2] = true; + } else if (ggml_vk_can_fuse(ctx, cgraph, i, rms_norm_mul_add_mul_pattern)) { + ctx->num_additional_fused_ops = 3; + ctx->fused_rms_norm_mode = RMS_NORM_MUL_ADD_MUL; + fusion_string = "RMS_NORM_MUL_ADD_MUL"; + std::fill_n(op_srcs_fused_elementwise, 4, true); + } else if (ggml_vk_can_fuse(ctx, cgraph, i, rms_norm_mul_add_pattern)) { + ctx->num_additional_fused_ops = 2; + ctx->fused_rms_norm_mode = RMS_NORM_MUL_ADD; + fusion_string = "RMS_NORM_MUL_ADD"; + std::fill_n(op_srcs_fused_elementwise, 3, true); + } else if (ggml_can_fuse_subgraph(cgraph, i, rms_norm_view_set_rows_pattern, { i + 2 }) && + ggml_check_edges(cgraph, i, rms_norm_view_set_rows_edges) && + ggml_vk_can_fuse_rms_norm_set_rows(ctx, cgraph, i)) { + ctx->num_additional_fused_ops = 2; + ctx->fused_rms_norm_mode = RMS_NORM_VIEW_SET_ROWS; + fusion_string = "RMS_NORM_VIEW_SET_ROWS"; + std::fill_n(op_srcs_fused_elementwise, 3, false); } else if (ggml_vk_can_fuse(ctx, cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) { ctx->num_additional_fused_ops = 1; + ctx->fused_rms_norm_mode = RMS_NORM_MUL; fusion_string = "RMS_NORM_MUL"; // rms_norm is not elementwise, but whole rows must be consumed and the scale factor computed before // they are overwritten, and one workgroup per row. So close enough. op_srcs_fused_elementwise[0] = true; op_srcs_fused_elementwise[1] = true; + } else if (ggml_vk_can_fuse(ctx, cgraph, i, { GGML_OP_UNARY, GGML_OP_MUL })) { + ctx->num_additional_fused_ops = 1; + switch (ggml_get_unary_op(cgraph->nodes[i])) { + case GGML_UNARY_OP_GELU: fusion_string = "GELU_MUL"; break; + case GGML_UNARY_OP_SIGMOID: fusion_string = "SIGMOID_MUL"; break; + case GGML_UNARY_OP_SILU: fusion_string = "SILU_MUL"; break; + default: fusion_string = "SOFTPLUS_MUL"; break; + } + op_srcs_fused_elementwise[0] = true; + op_srcs_fused_elementwise[1] = true; } else if (ggml_vk_can_fuse_ssm_conv(ctx, cgraph, i, 2)) { ctx->num_additional_fused_ops = 2; fusion_string = "SSM_CONV_BIAS_SILU"; @@ -17176,7 +14159,7 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg fusion_string = "SSM_CONV_SILU"; op_srcs_fused_elementwise[0] = false; op_srcs_fused_elementwise[1] = true; - } else if (ggml_can_fuse_subgraph(cgraph, i, { GGML_OP_ROPE, GGML_OP_VIEW, GGML_OP_SET_ROWS }, { i + 2 }) && + } else if (ggml_can_fuse_subgraph(cgraph, i, rope_view_set_rows_pattern, { i + 2 }) && ggml_check_edges(cgraph, i, rope_view_set_rows_edges) && ggml_vk_can_fuse_rope_set_rows(ctx, cgraph, i)) { ctx->num_additional_fused_ops = 2; @@ -17192,6 +14175,11 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg // with a data dependency on that register. The overlap check still // rejects partial overlaps (different base or size). std::fill_n(op_srcs_fused_elementwise, 5, true); + } else if (ggml_vk_can_fuse_topk_qsa(ctx, cgraph, i)) { + ctx->num_additional_fused_ops = topk_qsa_pattern.size() - 1; + ctx->fused_topk_qsa = true; + fusion_string = "TOPK_QSA"; + std::fill_n(op_srcs_fused_elementwise, ctx->num_additional_fused_ops + 1, false); } else if (ggml_can_fuse_subgraph(cgraph, i, topk_moe_early_softmax_norm, { i + 3, i + 9 }) && ggml_check_edges(cgraph, i, topk_moe_early_softmax_norm_edges) && ggml_vk_can_fuse_topk_moe(ctx, cgraph, i, TOPK_MOE_EARLY_SOFTMAX_NORM)) { @@ -17266,39 +14254,31 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg bool need_disable = false; - // topk_moe often overwrites the source, but for a given row all the src values are - // loaded before anything is stored. If there's only one row, this is safe, so treat - // this as a special case. - bool is_topk_moe_single_row = ctx->fused_topk_moe_mode != TOPK_MOE_COUNT && - ggml_nrows(cgraph->nodes[i]->src[0]) == 1; - - if (!is_topk_moe_single_row) { - for (int j = 0; j < 2; ++j) { - ggml_tensor *dst = output_nodes[j]; - if (!dst) { - continue; - } - // Loop over all srcs of all nodes in the fusion. If the src overlaps - // the destination and the src is not an intermediate node that's being - // elided, then disable fusion. - for (int k = 0; k <= ctx->num_additional_fused_ops; ++k) { - for (uint32_t s = 0; s < GGML_MAX_SRC; ++s) { - ggml_tensor *src = cgraph->nodes[i + k]->src[s]; - if (!src || src->op == GGML_OP_NONE) { - continue; - } - if (ggml_vk_tensors_overlap(src, dst, op_srcs_fused_elementwise[k])) { - bool found = false; - for (int n = 0; n < k; ++n) { - if (cgraph->nodes[i + n] == src) { - found = true; - break; - } - } - if (!found) { - need_disable = true; + for (int j = 0; j < 2; ++j) { + ggml_tensor *dst = output_nodes[j]; + if (!dst) { + continue; + } + // Loop over all srcs of all nodes in the fusion. If the src overlaps + // the destination and the src is not an intermediate node that's being + // elided, then disable fusion. + for (int k = 0; k <= ctx->num_additional_fused_ops; ++k) { + for (uint32_t s = 0; s < GGML_MAX_SRC; ++s) { + ggml_tensor *src = cgraph->nodes[i + k]->src[s]; + if (!src || src->op == GGML_OP_NONE) { + continue; + } + if (ggml_vk_tensors_overlap(src, dst, op_srcs_fused_elementwise[k])) { + bool found = false; + for (int n = 0; n < k; ++n) { + if (cgraph->nodes[i + n] == src) { + found = true; + break; } } + if (!found) { + need_disable = true; + } } } } @@ -17308,6 +14288,9 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg ctx->fused_ops_write_mask = 1; ctx->fused_topk_moe_mode = TOPK_MOE_COUNT; ctx->fused_topk_moe_scale = false; + ctx->fused_topk_qsa = false; + ctx->fused_rms_norm_mode = RMS_NORM_COUNT; + fusion_string = nullptr; } } @@ -17407,8 +14390,7 @@ static ggml_status ggml_backend_vk_graph_compute(ggml_backend_t backend, ggml_cg UNUSED(backend); } -// Sort the graph for improved parallelism. -static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * graph) +void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * graph, struct ggml_backend_graph_optimize_params * params) { VK_LOG_DEBUG("ggml_vk_graph_optimize(" << graph->n_nodes << " nodes)"); ggml_backend_vk_context * ctx = (ggml_backend_vk_context *)backend->context; @@ -17417,20 +14399,32 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * return; } - auto const &is_empty = [](ggml_tensor * node) -> bool { + auto const &is_empty = [](const ggml_tensor * node) -> bool { return node->op == GGML_OP_NONE || node->op == GGML_OP_RESHAPE || node->op == GGML_OP_TRANSPOSE || node->op == GGML_OP_VIEW || node->op == GGML_OP_PERMUTE; }; - auto const &is_src_of = [](const ggml_tensor *dst, const ggml_tensor *src) -> bool { + auto const &is_src_of = [&is_empty](const ggml_tensor *dst, const ggml_tensor *src) -> bool { + auto const &base = [](const ggml_tensor * tensor) { + return tensor->view_src ? tensor->view_src : tensor; + }; for (uint32_t s = 0; s < GGML_MAX_SRC; ++s) { if (dst->src[s] == src) { return true; } + if (is_empty(dst) || is_empty(src)) { + continue; + } + // A source view of dst may read storage written through a different view by src. + if (dst->src[s] && base(dst->src[s]) == base(src)) { + return true; + } + // Moving dst forward may overwrite storage still read through a view by src. + if (src->src[s] && base(dst) == base(src->src[s])) { + return true; + } } // implicit dependency if they view the same tensor - const ggml_tensor *dst2 = dst->view_src ? dst->view_src : dst; - const ggml_tensor *src2 = src->view_src ? src->view_src : src; - if (dst2 == src2) { + if (base(dst) == base(src)) { return true; } return false; @@ -17441,6 +14435,16 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * std::set used_node_set; int first_unused = 0; + + // scheduled or zero-compute nodes in [lo, hi) + auto const &empty_or_scheduled_between = [&](int lo, int hi) -> bool { + for (int v = lo; v < hi; ++v) { + if (!used[v] && !is_empty(graph->nodes[v])) { + return false; + } + } + return true; + }; while (first_unused < graph->n_nodes) { std::vector current_set; @@ -17473,24 +14477,74 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * return false; }; - if (keep_pattern(topk_moe_early_softmax_norm)) { + auto const &add_pattern_alloc_deps = [&](const std::initializer_list &pattern, int last_node) { + // Keep external inputs alive through the fused output. + std::set seen; + for (size_t j = 0; j < pattern.size(); ++j) { + ggml_tensor * node = graph->nodes[first_unused + j]; + for (uint32_t s = 0; s < GGML_MAX_SRC; ++s) { + ggml_tensor * src = node->src[s]; + if (src && seen.insert(src).second) { + params->add_alloc_dep(params->user_data, src, graph->nodes[last_node]); + } + } + seen.insert(node); + } + }; + + auto const &keep_topk_moe_pattern = [&](const std::initializer_list &pattern) -> bool { + if (!match_pattern(pattern, first_unused)) { + return false; + } + + int last_node = first_unused + (int) pattern.size() - 1; + // Some TOPK_MOE variants fuse a trailing scale. + if (last_node + 1 < graph->n_nodes && graph->nodes[last_node + 1]->op == GGML_OP_SCALE) { + last_node++; + } + + add_pattern_alloc_deps(pattern, last_node); + + return keep_pattern(pattern); + }; + + if (keep_topk_moe_pattern(topk_moe_early_softmax_norm)) { continue; } - if (keep_pattern(topk_moe_sigmoid_norm_bias)) { + if (keep_topk_moe_pattern(topk_moe_sigmoid_norm_bias)) { continue; } - if (keep_pattern(topk_moe_sqrt_softplus_norm_bias)) { + if (keep_topk_moe_pattern(topk_moe_sqrt_softplus_norm_bias)) { continue; } - if (keep_pattern(topk_moe_early_softmax)) { + if (keep_topk_moe_pattern(topk_moe_early_softmax)) { continue; } - if (keep_pattern(topk_moe_late_softmax)) { + if (keep_topk_moe_pattern(topk_moe_late_softmax)) { continue; } if (keep_pattern(snake_pattern)) { continue; } + if (keep_pattern(topk_qsa_pattern)) { + continue; + } + + if (keep_pattern(rms_norm_mul_add_mul_pattern)) { + continue; + } + if (keep_pattern(rms_norm_mul_add_pattern)) { + continue; + } + if (keep_pattern(rms_norm_mul_rope_view_set_rows_pattern)) { + continue; + } + if (keep_pattern(rms_norm_view_set_rows_pattern)) { + continue; + } + if (keep_pattern(rope_view_set_rows_pattern)) { + continue; + } // First, grab the next unused node. current_set.push_back(first_unused); @@ -17509,20 +14563,36 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * if (is_empty(graph->nodes[j])) { continue; } - // Don't pull forward nodes from fusion patterns + // Protect every interior QSA node (not just the start): the mask branch is + // independent, so it gets pulled out and breaks keep_pattern otherwise. + auto const &in_qsa_pattern = [&](int n) -> bool { + for (int o = 0; o < (int) topk_qsa_pattern.size(); ++o) { + if (n - o >= 0 && match_pattern(topk_qsa_pattern, n - o)) { + return true; + } + } + return false; + }; if (match_pattern(topk_moe_early_softmax_norm, j) || match_pattern(topk_moe_sigmoid_norm_bias, j) || match_pattern(topk_moe_sqrt_softplus_norm_bias, j) || match_pattern(topk_moe_early_softmax, j) || match_pattern(topk_moe_late_softmax, j) || - match_pattern(snake_pattern, j)) { + match_pattern(snake_pattern, j) || + in_qsa_pattern(j) || + match_pattern(rms_norm_mul_add_mul_pattern, j) || + match_pattern(rms_norm_mul_add_pattern, j) || + match_pattern(rms_norm_mul_rope_view_set_rows_pattern, j) || + match_pattern(rms_norm_view_set_rows_pattern, j) || + match_pattern(rope_view_set_rows_pattern, j)) { continue; } bool ok = true; for (int c = first_unused; c < j; ++c) { if (!used[c] && is_src_of(graph->nodes[j], graph->nodes[c]) && - !(j == c+1 && c == current_set.back() && graph->nodes[c]->op == GGML_OP_RMS_NORM && graph->nodes[j]->op == GGML_OP_MUL) && + !(c == current_set.back() && graph->nodes[c]->op == GGML_OP_RMS_NORM && graph->nodes[j]->op == GGML_OP_MUL && empty_or_scheduled_between(c+1, j)) && + !(c == current_set.back() && graph->nodes[c]->op == GGML_OP_UNARY && graph->nodes[j]->op == GGML_OP_MUL && empty_or_scheduled_between(c+1, j)) && !(j == c+1 && c == current_set.back() && graph->nodes[c]->op == GGML_OP_MUL_MAT && graph->nodes[j]->op == GGML_OP_ADD) && !(j == c+1 && c == current_set.back() && graph->nodes[c]->op == GGML_OP_MUL_MAT_ID && graph->nodes[j]->op == GGML_OP_ADD_ID) && !(j == c+1 && c == current_set.back() && graph->nodes[c]->op == GGML_OP_MUL_MAT_ID && graph->nodes[j]->op == GGML_OP_MUL) && @@ -17555,30 +14625,41 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * } } } - // Look for ROPE + VIEW + SET_ROWS and make them consecutive - if (graph->nodes[rope_idx]->op == GGML_OP_ROPE) { + // Look for ROPE/RMS_NORM + VIEW + SET_ROWS and make them consecutive + if (graph->nodes[rope_idx]->op == GGML_OP_ROPE || graph->nodes[rope_idx]->op == GGML_OP_RMS_NORM) { int view_idx = -1; int set_rows_idx = -1; - for (int k = rope_idx+1; k < std::min(rope_idx + 10, graph->n_nodes); ++k) { - if (view_idx == -1 && - graph->nodes[k]->op == GGML_OP_VIEW && - graph->nodes[k]->src[0] == graph->nodes[rope_idx]) { + for (int k = rope_idx + 1; k < std::min(rope_idx + 15, graph->n_nodes); ++k) { + if (used[k]) { + continue; + } + if (view_idx == -1 && graph->nodes[k]->op == GGML_OP_VIEW && graph->nodes[k]->src[0] == graph->nodes[rope_idx]) { view_idx = k; continue; } - if (view_idx != -1 && - set_rows_idx == -1 && - graph->nodes[k]->op == GGML_OP_SET_ROWS && - graph->nodes[k]->src[0] == graph->nodes[view_idx]) { + if (view_idx != -1 && graph->nodes[k]->op == GGML_OP_SET_ROWS && graph->nodes[k]->src[0] == graph->nodes[view_idx]) { set_rows_idx = k; break; } } if (set_rows_idx != -1) { - current_set.push_back(view_idx); - current_set.push_back(set_rows_idx); - used[view_idx] = true; - used[set_rows_idx] = true; + const int node_idxs[] = { rope_idx, view_idx, set_rows_idx }; + const ggml_op ops[] = { graph->nodes[rope_idx]->op, GGML_OP_VIEW, GGML_OP_SET_ROWS }; + bool can_pull = ggml_can_fuse_subgraph_ext(graph, node_idxs, 3, ops, &set_rows_idx, 1); + + for (int c = rope_idx + 1; can_pull && c < set_rows_idx; ++c) { + if (!used[c] && c != view_idx && !is_empty(graph->nodes[c]) && + is_src_of(graph->nodes[set_rows_idx], graph->nodes[c])) { + can_pull = false; + } + } + + if (can_pull) { + current_set.push_back(view_idx); + current_set.push_back(set_rows_idx); + used[view_idx] = true; + used[set_rows_idx] = true; + } } } // Look for MUL_MAT_ID + ADD_ID + MUL @@ -17624,6 +14705,27 @@ static void ggml_vk_graph_optimize(ggml_backend_t backend, struct ggml_cgraph * } } } + // UNARY + MUL: pull the consuming MUL forward + if (j > 0 && + graph->nodes[j]->op == GGML_OP_UNARY) { + for (int k = j + 1; k < std::min(j + 15, graph->n_nodes); ++k) { + ggml_tensor * mul = graph->nodes[k]; + if (mul->op != GGML_OP_MUL || (mul->src[0] != graph->nodes[j] && mul->src[1] != graph->nodes[j])) { + continue; + } + ggml_tensor * other = (mul->src[0] == graph->nodes[j]) ? mul->src[1] : mul->src[0]; + // the other src must either be weights or already processed + if (!(other->op == GGML_OP_NONE || used_node_set.find(other) != used_node_set.end())) { + continue; + } + if (!ggml_vk_can_fuse_unary_mul(graph, j, k)) { + continue; + } + current_set.push_back(k); + used[k] = true; + break; + } + } } } // Second pass grabs view nodes. @@ -17725,7 +14827,6 @@ static void ggml_backend_vk_event_wait(ggml_backend_t backend, ggml_backend_even } } -// TODO: enable async and synchronize static ggml_backend_i ggml_backend_vk_interface = { /* .get_name = */ ggml_backend_vk_name, /* .free = */ ggml_backend_vk_free, @@ -17866,17 +14967,6 @@ static std::string ggml_backend_vk_get_device_pci_id(int device_idx) { return std::string(pci_bus_id); } -////////////////////////// - -struct ggml_backend_vk_device_context { - size_t device; - std::string name; - std::string description; - bool is_integrated_gpu; - std::string pci_bus_id; - int op_offload_min_batch_size; -}; - static const char * ggml_backend_vk_device_get_name(ggml_backend_dev_t dev) { ggml_backend_vk_device_context * ctx = (ggml_backend_vk_device_context *)dev->context; return ctx->name.c_str(); @@ -18000,6 +15090,7 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_GLU_OP_SWIGLU_OAI: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: return (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16) && (op->type == GGML_TYPE_F32 || op->type == GGML_TYPE_F16) && (op->src[0]->type == op->type) && @@ -18016,6 +15107,9 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm // If there's not enough shared memory for row_ids and the result tile, fallback to CPU return false; } + if (ggml_get_op_params_i32(op, 3) == GGML_PREC_F32) { + return false; + } } switch (src0_type) { case GGML_TYPE_F32: @@ -18044,6 +15138,7 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_TYPE_IQ4_NL: case GGML_TYPE_MXFP4: case GGML_TYPE_NVFP4: + case GGML_TYPE_TQ1_0: case GGML_TYPE_TQ2_0: break; default: @@ -18150,6 +15245,7 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_TYPE_IQ4_NL: case GGML_TYPE_MXFP4: case GGML_TYPE_NVFP4: + case GGML_TYPE_TQ1_0: case GGML_TYPE_TQ2_0: case GGML_TYPE_I32: return true; @@ -18317,15 +15413,14 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm if (!ggml_is_contiguous(op) || !ggml_is_contiguous(op->src[0])) { return false; } - // We could potentially support larger, using argsort to sort the - // whole thing. Not clear if this is needed. - uint32_t min_pipeline = (uint32_t)log2f(float(op->ne[0])) + 1; - if (min_pipeline >= num_topk_pipelines || - !device->pipeline_topk_f32[min_pipeline]) { - return false; + // large k falls back to radix-select + const uint32_t min_pipeline = + std::max((uint32_t) log2f(float(op->ne[0])) + 1, device->subgroup_size_log2); + if (min_pipeline < num_topk_pipelines && device->pipeline_topk_f32[min_pipeline]) { + return true; } + return device->pipeline_topk_radix_f32 != nullptr; } - return true; case GGML_OP_UPSCALE: if (op->op_params[0] & GGML_SCALE_FLAG_ANTIALIAS) { if ((op->op_params[0] & 0xFF) != GGML_SCALE_MODE_BILINEAR) { @@ -18352,6 +15447,7 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_OP_SCALE: return ggml_is_contiguous(op->src[0]) && op->src[0]->type == GGML_TYPE_F32; case GGML_OP_PAD: + case GGML_OP_PAD_REFLECT_1D: case GGML_OP_ROLL: return op->src[0]->type == GGML_TYPE_F32; case GGML_OP_DIAG_MASK_INF: @@ -18373,6 +15469,31 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm } return false; } + case GGML_OP_DSV4_HC_COMB: + case GGML_OP_DSV4_HC_PRE: + case GGML_OP_DSV4_HC_POST: + { + if (op->type != GGML_TYPE_F32) { + return false; + } + for (uint32_t i = 0; i < GGML_MAX_SRC; ++i) { + if (op->src[i] && op->src[i]->type != GGML_TYPE_F32) { + return false; + } + } + // hc is hardcoded to 4 in the shaders. ggml only constrains it + // to 4 for COMB, so PRE/POST have to be checked here. + if (op->op == GGML_OP_DSV4_HC_PRE && op->src[0]->ne[1] != 4) { + return false; + } + if (op->op == GGML_OP_DSV4_HC_POST && op->src[1]->ne[1] != 4) { + return false; + } + if (op->op == GGML_OP_DSV4_HC_COMB) { + return device->pipeline_dsv4_hc_comb_f32 != nullptr; + } + return true; + } case GGML_OP_SOLVE_TRI: { if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32) { @@ -18393,6 +15514,18 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm } case GGML_OP_ARGMAX: return ggml_is_contiguous(op->src[0]) && op->src[0]->type == GGML_TYPE_F32; + case GGML_OP_CROSS_ENTROPY_LOSS: + return ggml_is_contiguous(op->src[0]) && op->src[0]->type == GGML_TYPE_F32 + && ggml_is_contiguous(op->src[1]) && op->src[1]->type == GGML_TYPE_F32 + && ggml_are_same_shape(op->src[0], op->src[1]) + && ggml_is_contiguous(op) && ggml_is_scalar(op) && op->type == GGML_TYPE_F32; + case GGML_OP_CROSS_ENTROPY_LOSS_BACK: + return ggml_is_contiguous(op->src[0]) && op->src[0]->type == GGML_TYPE_F32 && ggml_is_scalar(op->src[0]) + && ggml_is_contiguous(op->src[1]) && op->src[1]->type == GGML_TYPE_F32 + && ggml_is_contiguous(op->src[2]) && op->src[2]->type == GGML_TYPE_F32 + && ggml_are_same_shape(op->src[1], op->src[2]) + && ggml_are_same_shape(op->src[1], op) + && ggml_is_contiguous(op) && op->type == GGML_TYPE_F32; case GGML_OP_COUNT_EQUAL: return ggml_is_contiguous(op->src[0]) && op->src[0]->type == GGML_TYPE_I32 && ggml_is_contiguous(op->src[1]) && op->src[1]->type == GGML_TYPE_I32; @@ -18418,6 +15551,40 @@ static bool ggml_backend_vk_device_supports_op(ggml_backend_dev_t dev, const ggm case GGML_OP_GATED_LINEAR_ATTN: // the shader block size is hardcoded to head_size 64 return op->src[0]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32 && op->src[0]->ne[0] == 64; + case GGML_OP_LIGHTNING_INDEXER: + { + const ggml_tensor * q = op->src[0]; + const ggml_tensor * k = op->src[1]; + const ggml_tensor * w = op->src[2]; + const ggml_tensor * m = op->src[3]; + + // the q/w/m types and the shape relationships between q, k, w, m and dst + // are already asserted in ggml_lightning_indexer() + if (!ggml_vk_lightning_indexer_k_type_supported(k->type) || !device->fp16) { + return false; + } + + // the shader block size is hardcoded to head size 128 + if (q->ne[0] != 128) { + return false; + } + + // the shader indexes the buffers by element stride, and is dispatched + // without allow_misalign + for (const ggml_tensor * t : {q, k, w, m, op}) { + if (t->nb[0] != ggml_type_size(t->type) || + (vk_tensor_offset(t) + t->view_offs) % device->properties.limits.minStorageBufferOffsetAlignment != 0) { + return false; + } + // the strides get scaled down from bytes, so the division must be exact + for (int i = 1; i < GGML_MAX_DIMS; ++i) { + if (t->nb[i] % ggml_type_size(t->type) != 0) { + return false; + } + } + } + return true; + } case GGML_OP_GATED_DELTA_NET: { const uint32_t S_v = op->src[2]->ne[0]; @@ -18523,21 +15690,6 @@ static bool ggml_backend_vk_device_supports_buft(ggml_backend_dev_t dev, ggml_ba return buft_ctx->device->idx == ctx->device; } -static int64_t ggml_vk_get_op_batch_size(const ggml_tensor * op) { - switch (op->op) { - case GGML_OP_GET_ROWS: - return 0; - case GGML_OP_MUL_MAT: - return op->ne[1]; - case GGML_OP_MUL_MAT_ID: - case GGML_OP_ROPE: - case GGML_OP_ROPE_BACK: - return op->ne[2]; - default: - return ggml_nrows(op); - } -} - static bool ggml_backend_vk_device_offload_op(ggml_backend_dev_t dev, const ggml_tensor * op) { ggml_backend_vk_device_context * dev_ctx = (ggml_backend_vk_device_context *)dev->context; @@ -18619,31 +15771,6 @@ static void ggml_backend_vk_device_event_synchronize(ggml_backend_dev_t dev, ggm } } -static vk_buffer ggml_vk_buffer_from_host_ptr(vk_device & device, void * ptr, size_t size) { - if (!device->external_memory_host) { - return {}; - } - - uintptr_t uptr = reinterpret_cast(ptr); - if (uptr & (device->min_imported_host_pointer_alignment - 1)) { - return {}; - } - if (size & (device->min_imported_host_pointer_alignment - 1)) { - return {}; - } - - const vk::MemoryPropertyFlags property_flags = vk::MemoryPropertyFlagBits::eHostVisible | vk::MemoryPropertyFlagBits::eHostCoherent | vk::MemoryPropertyFlagBits::eHostCached; - - vk_buffer buf {}; - try { - buf = ggml_vk_create_buffer(device, size, { property_flags }, ptr); - } catch (vk::SystemError& e) { - GGML_LOG_WARN("ggml_vulkan: Failed ggml_vk_create_buffer (%s)\n", e.what()); - } - - return buf; -} - static ggml_backend_buffer_t ggml_backend_vk_device_buffer_from_host_ptr(ggml_backend_dev_t dev, void * ptr, size_t size, size_t max_tensor_size) { VK_LOG_DEBUG("ggml_backend_vk_device_buffer_from_host_ptr(backend=" << dev << ", ptr=" << ptr << ", size=" << size << ")"); GGML_UNUSED(max_tensor_size); @@ -18754,8 +15881,7 @@ ggml_backend_reg_t ggml_backend_vk_reg() { } } -// Extension availability -static bool ggml_vk_instance_layer_settings_available() { +bool ggml_vk_instance_layer_settings_available() { #ifdef GGML_VULKAN_VALIDATE // Check if validation layer provides the extension const std::string layer_name = "VK_LAYER_KHRONOS_validation"; @@ -18773,7 +15899,8 @@ static bool ggml_vk_instance_layer_settings_available() { #endif return false; } -static bool ggml_vk_instance_portability_enumeration_ext_available(const std::vector& instance_extensions) { + +bool ggml_vk_instance_portability_enumeration_ext_available(const std::vector& instance_extensions) { #ifdef __APPLE__ // Check for portability enumeration extension for MoltenVK support for (const auto& properties : instance_extensions) { @@ -18788,8 +15915,7 @@ static bool ggml_vk_instance_portability_enumeration_ext_available(const std::ve UNUSED(instance_extensions); } -// Extension availability -static bool ggml_vk_instance_debug_utils_ext_available( +bool ggml_vk_instance_debug_utils_ext_available( const std::vector & instance_extensions) { // Check for portability enumeration extension for MoltenVK support for (const auto & properties : instance_extensions) { @@ -18804,7 +15930,7 @@ static bool ggml_vk_instance_debug_utils_ext_available( UNUSED(instance_extensions); } -static bool ggml_vk_device_is_supported(const vk::PhysicalDevice & vkdev) { +bool ggml_vk_device_is_supported(const vk::PhysicalDevice & vkdev) { VkPhysicalDeviceFeatures2 device_features2; device_features2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2; @@ -18818,7 +15944,7 @@ static bool ggml_vk_device_is_supported(const vk::PhysicalDevice & vkdev) { return vk11_features.storageBuffer16BitAccess; } -static bool ggml_vk_khr_cooperative_matrix_support(const vk::PhysicalDeviceProperties& props, const vk::PhysicalDeviceDriverProperties& driver_props, vk_device_architecture arch) { +bool ggml_vk_khr_cooperative_matrix_support(const vk::PhysicalDeviceProperties& props, const vk::PhysicalDeviceDriverProperties& driver_props, vk_device_architecture arch) { switch (props.vendorID) { case VK_VENDOR_ID_INTEL: // Only allowing Xe2/Xe3 GPU and integrated Xe GPUs at the moment since older hardware (ex. Arc A770) has performance regressions. @@ -18835,7 +15961,7 @@ static bool ggml_vk_khr_cooperative_matrix_support(const vk::PhysicalDevicePrope } } -static uint32_t ggml_vk_intel_shader_core_count(const vk::PhysicalDevice& vkdev) { +uint32_t ggml_vk_intel_shader_core_count(const vk::PhysicalDevice& vkdev) { VkPhysicalDeviceProperties2 props = vkdev.getProperties2(); if (props.properties.vendorID != VK_VENDOR_ID_INTEL) { @@ -18878,8 +16004,7 @@ static uint32_t ggml_vk_intel_shader_core_count(const vk::PhysicalDevice& vkdev) } } -// checks whether lower <= driver_version < upper, with each bound given as xxx.yyyy -static bool ggml_vk_intel_windows_driver_in_range(uint32_t driver_version, uint32_t lower_major, uint32_t lower_minor, uint32_t upper_major, uint32_t upper_minor) { +bool ggml_vk_intel_windows_driver_in_range(uint32_t driver_version, uint32_t lower_major, uint32_t lower_minor, uint32_t upper_major, uint32_t upper_minor) { #if defined(_WIN32) // Intel Windows encodes xxx.yyyy as [31:14].[13:0]. const uint32_t major = driver_version >> 14; @@ -18899,765 +16024,251 @@ static bool ggml_vk_intel_windows_driver_in_range(uint32_t driver_version, uint3 #endif } +GGML_BACKEND_DL_IMPL(ggml_backend_vk_reg) -// checks - -#ifdef GGML_VULKAN_CHECK_RESULTS -static void ggml_vk_print_graph_origin(const ggml_tensor * tensor, std::vector& done, int level = 0) { - if (std::find(done.begin(), done.end(), tensor) != done.end() || level > 10) { - return; - } - for (int j = 0; j < level; j++) { - std::cerr << " "; - } - std::cerr << ggml_op_name(tensor->op) << " gpu=" << (tensor->extra != nullptr) << std::endl; - done.push_back(tensor); +// out-of-lined header method definitions - for (int i = 0; i < GGML_MAX_SRC; i++) { - if (tensor->src[i] != nullptr) { - ggml_vk_print_graph_origin(tensor->src[i], done, level + 1); +void vk_queue_handle_synchronized::submit(vk::ArrayProxy submits, vk::Fence fence) { + // Workaround for NVIDIA driver bug + std::unique_lock device_guard; + if (device_submit_mutex) { + device_guard = std::unique_lock(*device_submit_mutex); + } + std::lock_guard guard(mutex); + try { + queue.submit(submits, fence); + } catch (vk::DeviceLostError &) { + if (auto dev = device.lock()) { + ggml_vk_print_device_lost_info(dev); } + throw; } } -static void ggml_vk_print_tensor_area(const ggml_tensor * tensor, const void * data, int i0, int i1, int i2, int i3) { - if (tensor->type != GGML_TYPE_F32 && tensor->type != GGML_TYPE_F16 && tensor->type != GGML_TYPE_I32) { - return; +void vk_queue_handle_unsynchronized::submit(vk::ArrayProxy submits, vk::Fence fence) { + // Workaround for NVIDIA driver bug + std::unique_lock device_guard; + if (device_submit_mutex) { + device_guard = std::unique_lock(*device_submit_mutex); } - i0 = std::max(i0, 5); - i1 = std::max(i1, 5); - i2 = std::max(i2, 0); - i3 = std::max(i3, 0); - fprintf(stderr, " "); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - fprintf(stderr, "%7d ", idx1); - } - fprintf(stderr, "\n"); - for (int idx0 = i0 - 5; idx0 < i0 + 5; idx0++) { - fprintf(stderr, "%7d: ", idx0); - for (int idx1 = i1 - 5; idx1 < i1 + 5; idx1++) { - if (idx0 >= 0 && idx0 < tensor->ne[0] && idx1 >= 0 && idx1 < tensor->ne[1] && i2 >= 0 && i2 < tensor->ne[2] && i3 >= 0 && i3 < tensor->ne[3]) { - float val; - if (tensor->type == GGML_TYPE_F32) { - val = *(const float *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); - } else if (tensor->type == GGML_TYPE_F16) { - val = ggml_fp16_to_fp32(*(const ggml_fp16_t *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0])); - } else if (tensor->type == GGML_TYPE_I32) { - val = *(const int32_t *) ((const char *) data + i3*tensor->nb[3] + i2*tensor->nb[2] + idx1*tensor->nb[1] + idx0*tensor->nb[0]); - } else { - GGML_ABORT("fatal error"); - } - fprintf(stderr, "% 7.2f ", val); - } else { - fprintf(stderr, " "); - } + try { + queue.submit(submits, fence); + } catch (vk::DeviceLostError &) { + if (auto dev = device.lock()) { + ggml_vk_print_device_lost_info(dev); } - fprintf(stderr, "\n"); + throw; } } -static void ggml_vk_print_tensor(const ggml_tensor * tensor, const char * name) { - void * tensor_data = tensor->data; +vk_device_struct::~vk_device_struct() { + VK_LOG_DEBUG("destroy device " << name); - const bool is_gpu = tensor->buffer != nullptr && ggml_backend_buffer_is_vk(tensor->buffer); + device.destroyFence(fence); - if (is_gpu) { - const size_t tensor_size = ggml_nbytes(tensor); - tensor_data = malloc(tensor_size); + ggml_vk_destroy_buffer(sync_staging); - ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; + if (compute_queue) compute_queue->cmd_pool.destroy(device); + if (transfer_queue) transfer_queue->cmd_pool.destroy(device); - vk_buffer buffer_gpu = buf_ctx->dev_buffer; - ggml_vk_buffer_read(buffer_gpu, vk_tensor_offset(tensor) + tensor->view_offs, tensor_data, tensor_size); - } + // Explicitly clear to ensure queues drop their shared_ptrs to handles + // before the Vulkan logical device instance is destroyed + compute_queue.reset(); + transfer_queue.reset(); - std::cerr << "TENSOR CHECK " << name << " (" << tensor->name << "): " << ggml_op_name(tensor->op) << std::endl; - std::cerr << "tensor=" << tensor << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << std::endl; - if (tensor->src[0] != nullptr) { - std::cerr << "tensor->src[0]=" << tensor->src[0] << " name=" << tensor->src[0]->name << " op=" << ggml_op_name(tensor->src[0]->op) << " type=" << ggml_type_name(tensor->src[0]->type) << " ne0=" << tensor->src[0]->ne[0] << " nb0=" << tensor->src[0]->nb[0] << " ne1=" << tensor->src[0]->ne[1] << " nb1=" << tensor->src[0]->nb[1] << " ne2=" << tensor->src[0]->ne[2] << " nb2=" << tensor->src[0]->nb[2] << " ne3=" << tensor->src[0]->ne[3] << " nb3=" << tensor->src[0]->nb[3] << std::endl; - } - if (tensor->src[1] != nullptr) { - std::cerr << "tensor->src[1]=" << tensor->src[1] << " name=" << tensor->src[1]->name << " op=" << ggml_op_name(tensor->src[1]->op) << " type=" << ggml_type_name(tensor->src[1]->type) << " ne0=" << tensor->src[1]->ne[0] << " nb0=" << tensor->src[1]->nb[0] << " ne1=" << tensor->src[1]->ne[1] << " nb1=" << tensor->src[1]->nb[1] << " ne2=" << tensor->src[1]->ne[2] << " nb2=" << tensor->src[1]->nb[2] << " ne3=" << tensor->src[1]->ne[3] << " nb3=" << tensor->src[1]->nb[3] << std::endl; - } - std::cerr << std::endl << "Result:" << std::endl; - ggml_vk_print_tensor_area(tensor, tensor_data, 5, 5, 0, 0); - std::cerr << std::endl; - std::vector done; - ggml_vk_print_graph_origin(tensor, done); + for (auto& pipeline : all_pipelines) { + if (pipeline.expired()) { + continue; + } - if (is_gpu) { - free(tensor_data); + vk_pipeline pl = pipeline.lock(); + ggml_vk_destroy_pipeline(device, pl); } + all_pipelines.clear(); + + device.destroyDescriptorSetLayout(dsl); + + device.destroy(); } -void * comp_result; -size_t comp_size; -size_t comp_nb[GGML_MAX_DIMS]; -size_t check_counter = 0; -static void ggml_vk_check_results_0(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx) { - ggml_tensor * tensor = cgraph->nodes[tensor_idx + ctx->num_additional_fused_ops]; - if (tensor->op == GGML_OP_TRANSPOSE || tensor->op == GGML_OP_SET_ROWS) { +void vk_perf_logger::print_timings(bool force) { + if (timings.empty()) { return; } - - check_counter++; - if (!(vk_output_tensor > 0 && vk_output_tensor == check_counter) && check_counter <= vk_skip_checks) { + print_count++; + if ((print_count % vk_perf_logger_frequency) != 0 && !force) { return; } - - VK_LOG_DEBUG("ggml_vk_check_results_0(" << tensor->name << ")"); - - struct ggml_init_params iparams = { - /*.mem_size =*/ 2ul*1024ul*1024ul*1024ul, - /*.mem_buffer =*/ NULL, - /*.no_alloc =*/ false, - }; - - struct ggml_context * ggml_ctx = ggml_init(iparams); - - std::array src_clone = {nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr, nullptr}; - const char * srci_name[GGML_MAX_SRC] = {"src0", "src1", "src2", "src3", "src4", "src5", "src6", "src7", "src8", "src9"}; - - std::map cloned_tensors; - std::vector cloned_mallocs; - - struct ggml_tensor * tensor_clone = nullptr; - - for (int f = 0; f < ctx->num_additional_fused_ops + 1; ++f) { - tensor = cgraph->nodes[tensor_idx + f]; - for (int i = 0; i < GGML_MAX_SRC; i++) { - ggml_tensor * srci = tensor->src[i]; - if (srci == nullptr) { - continue; - } - // If a src tensor has been cloned, use that one - auto it = cloned_tensors.find(srci); - if (it != cloned_tensors.end()) { - src_clone[i] = it->second; - continue; - } - ggml_tensor * srci_clone = ggml_dup_tensor(ggml_ctx, srci); - size_t srci_size = ggml_nbytes(srci); - - src_clone[i] = srci_clone; - void *src_buffer = malloc(srci_size); - cloned_mallocs.push_back(src_buffer); - - srci_clone->data = src_buffer; - if (ggml_backend_buffer_is_host(srci->buffer)) { - memcpy(srci_clone->data, srci->data, srci_size); - memcpy(srci_clone->nb, srci->nb, sizeof(size_t) * GGML_MAX_DIMS); - } else if (ggml_backend_buffer_is_vk(srci->buffer)) { - ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)srci->buffer->context; - vk_buffer& buffer_gpu = buf_ctx->dev_buffer; - uint64_t offset = vk_tensor_offset(srci) + srci->view_offs; - if (!ggml_is_contiguous(srci) && ggml_vk_dim01_contiguous(srci)) { - for (int i3 = 0; i3 < srci->ne[3]; i3++) { - for (int i2 = 0; i2 < srci->ne[2]; i2++) { - const int idx = i3*srci->ne[2] + i2; - ggml_vk_buffer_read(buffer_gpu, offset + idx * srci->nb[2], ((char *)srci_clone->data + idx * srci_clone->nb[2]), srci->ne[1] * srci->nb[1]); - } - } - - srci_clone->nb[0] = srci->nb[0]; - srci_clone->nb[1] = srci->nb[1]; - for (int i = 2; i < GGML_MAX_DIMS; i++) { - srci_clone->nb[i] = srci_clone->nb[i - 1]*srci_clone->ne[i - 1]; - } - } else { - if (offset + srci_size >= buffer_gpu->size) { - srci_size = buffer_gpu->size - offset; - } - ggml_vk_buffer_read(buffer_gpu, offset, srci_clone->data, srci_size); - memcpy(srci_clone->nb, srci->nb, sizeof(size_t) * GGML_MAX_DIMS); - } - } else { - GGML_ABORT("fatal error"); - } - - if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { - ggml_vk_print_tensor(srci, srci_name[i]); - } + print_count = 0; + uint64_t total_all_op_times = 0; + std::cerr << "----------------\nVulkan Timings:" << std::endl; + for (const auto & t : timings) { + uint64_t total_op_times = 0; + for (const auto & time : t.second) { + total_op_times += time; } + std::cerr << t.first << ": " << t.second.size() << " x " << (total_op_times / t.second.size() / 1000.0) + << " us = " << (total_op_times / 1000.0) << " us"; - if (tensor->op == GGML_OP_FLASH_ATTN_EXT) { - const float * params = (const float *)tensor->op_params; - tensor_clone = ggml_flash_attn_ext(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3], params[0], params[1], params[2]); - if (src_clone[4]) { - ggml_flash_attn_ext_add_sinks(tensor_clone, src_clone[4]); - } - } else if (tensor->op == GGML_OP_MUL_MAT) { - tensor_clone = ggml_mul_mat(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_MUL_MAT_ID) { - tensor_clone = ggml_mul_mat_id(ggml_ctx, src_clone[0], src_clone[1], src_clone[2]); - } else if (tensor->op == GGML_OP_SUB) { - tensor_clone = ggml_sub(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_MUL) { - tensor_clone = ggml_mul(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_DIV) { - tensor_clone = ggml_div(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_CONCAT) { - tensor_clone = ggml_concat(ggml_ctx, src_clone[0], src_clone[1], *(int *)tensor->op_params); - } else if (tensor->op == GGML_OP_UPSCALE) { - tensor_clone = ggml_interpolate(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], (ggml_scale_mode) tensor->op_params[0]); - } else if (tensor->op == GGML_OP_SCALE) { - const float * params = (const float *)tensor->op_params; - tensor_clone = ggml_scale_bias(ggml_ctx, src_clone[0], params[0], params[1]); - } else if (tensor->op == GGML_OP_ADD1) { - tensor_clone = ggml_add1(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_ARANGE) { - const float start = ggml_get_op_params_f32(tensor, 0); - const float stop = ggml_get_op_params_f32(tensor, 1); - const float step = ggml_get_op_params_f32(tensor, 2); - tensor_clone = ggml_arange(ggml_ctx, start, stop, step); - } else if (tensor->op == GGML_OP_FILL) { - const float value = ggml_get_op_params_f32(tensor, 0); - tensor_clone = ggml_fill(ggml_ctx, src_clone[0], value); - } else if (tensor->op == GGML_OP_SQR) { - tensor_clone = ggml_sqr(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_SQRT) { - tensor_clone = ggml_sqrt(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_SIN) { - tensor_clone = ggml_sin(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_COS) { - tensor_clone = ggml_cos(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_LOG) { - tensor_clone = ggml_log(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_TRI) { - tensor_clone = ggml_tri(ggml_ctx, src_clone[0], (ggml_tri_type)ggml_get_op_params_i32(tensor, 0)); - } else if (tensor->op == GGML_OP_DIAG) { - tensor_clone = ggml_diag(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_CLAMP) { - const float * params = (const float *)tensor->op_params; - tensor_clone = ggml_clamp(ggml_ctx, src_clone[0], params[0], params[1]); - } else if (tensor->op == GGML_OP_PAD) { - tensor_clone = ggml_pad_ext(ggml_ctx, src_clone[0], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3], - tensor->op_params[4], tensor->op_params[5], tensor->op_params[6], tensor->op_params[7]); - } else if (tensor->op == GGML_OP_REPEAT) { - tensor_clone = ggml_repeat(ggml_ctx, src_clone[0], tensor); - } else if (tensor->op == GGML_OP_REPEAT_BACK) { - tensor_clone = ggml_repeat_back(ggml_ctx, src_clone[0], tensor); - } else if (tensor->op == GGML_OP_ADD) { - tensor_clone = ggml_add(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_ACC) { - tensor_clone = ggml_acc(ggml_ctx, src_clone[0], src_clone[1], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3]); - } else if (tensor->op == GGML_OP_SET) { - tensor_clone = ggml_set(ggml_ctx, src_clone[0], src_clone[1], tensor->op_params[0], tensor->op_params[1], tensor->op_params[2], tensor->op_params[3]); - } else if (tensor->op == GGML_OP_NORM) { - tensor_clone = ggml_norm(ggml_ctx, src_clone[0], *(float *)tensor->op_params); - } else if (tensor->op == GGML_OP_GROUP_NORM) { - const float * float_params = (const float *)tensor->op_params; - tensor_clone = ggml_group_norm(ggml_ctx, src_clone[0], tensor->op_params[0], float_params[1]); - } else if (tensor->op == GGML_OP_RMS_NORM) { - tensor_clone = ggml_rms_norm(ggml_ctx, src_clone[0], *(float *)tensor->op_params); - } else if (tensor->op == GGML_OP_RMS_NORM_BACK) { - const float eps = ((float *) tensor->op_params)[0]; - tensor_clone = ggml_rms_norm_back(ggml_ctx, src_clone[0], src_clone[1], eps); - } else if (tensor->op == GGML_OP_SILU_BACK) { - tensor_clone = ggml_silu_back(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_L2_NORM) { - const float eps = ((float *) tensor->op_params)[0]; - tensor_clone = ggml_l2_norm(ggml_ctx, src_clone[0], eps); - } else if (tensor->op == GGML_OP_SOFT_MAX) { - if (tensor->src[1] != nullptr) { - const float * params = (const float *)tensor->op_params; - tensor_clone = ggml_soft_max_ext(ggml_ctx, src_clone[0], src_clone[1], params[0], params[1]); - } else { - tensor_clone = ggml_soft_max(ggml_ctx, src_clone[0]); - } - } else if (tensor->op == GGML_OP_SOFT_MAX_BACK) { - tensor_clone = ggml_soft_max_ext_back(ggml_ctx, src_clone[0], src_clone[1], ((float *)tensor->op_params)[0], ((float *)tensor->op_params)[1]); - } else if (tensor->op == GGML_OP_DIAG_MASK_INF) { - tensor_clone = ggml_diag_mask_inf(ggml_ctx, src_clone[0], tensor->op_params[0]); - } else if (tensor->op == GGML_OP_ROPE || tensor->op == GGML_OP_ROPE_BACK) { - const int n_dims = ((int32_t *) tensor->op_params)[1]; - const int mode = ((int32_t *) tensor->op_params)[2]; - //const int n_ctx_ggml = ((int32_t *) tensor->op_params)[3]; - const int n_ctx_orig_ggml = ((int32_t *) tensor->op_params)[4]; - const float freq_base = ((float *) tensor->op_params)[5]; - const float freq_scale = ((float *) tensor->op_params)[6]; - const float ext_factor = ((float *) tensor->op_params)[7]; - const float attn_factor = ((float *) tensor->op_params)[8]; - const float beta_fast = ((float *) tensor->op_params)[9]; - const float beta_slow = ((float *) tensor->op_params)[10]; - if (mode & GGML_ROPE_TYPE_MROPE) { - int32_t *sections = ((int32_t *) tensor->op_params) + 11; - if (tensor->op == GGML_OP_ROPE) { - tensor_clone = ggml_rope_multi(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, sections, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); - } else { - tensor_clone = ggml_rope_multi_back(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, sections, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); - } - } else { - if (tensor->op == GGML_OP_ROPE) { - tensor_clone = ggml_rope_ext(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); - } else { - tensor_clone = ggml_rope_ext_back(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], n_dims, mode, n_ctx_orig_ggml, freq_base, freq_scale, ext_factor, attn_factor, beta_fast, beta_slow); - } - } - } else if (tensor->op == GGML_OP_UNARY) { - switch (ggml_get_unary_op(tensor)) { - case GGML_UNARY_OP_EXP: - tensor_clone = ggml_exp(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_EXPM1: - tensor_clone = ggml_expm1(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_ELU: - tensor_clone = ggml_elu(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_SILU: - tensor_clone = ggml_silu(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_GELU: - tensor_clone = ggml_gelu(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_GELU_ERF: - tensor_clone = ggml_gelu_erf(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_GELU_QUICK: - tensor_clone = ggml_gelu_quick(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_RELU: - tensor_clone = ggml_relu(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_XIELU: - tensor_clone = ggml_xielu(ggml_ctx, src_clone[0], 0, 0, 0, 0); - ggml_set_op_params_f32(tensor_clone, 1, ggml_get_op_params_f32(tensor, 1)); - ggml_set_op_params_f32(tensor_clone, 2, ggml_get_op_params_f32(tensor, 2)); - ggml_set_op_params_f32(tensor_clone, 3, ggml_get_op_params_f32(tensor, 3)); - ggml_set_op_params_f32(tensor_clone, 4, ggml_get_op_params_f32(tensor, 4)); - break; - case GGML_UNARY_OP_NEG: - tensor_clone = ggml_neg(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_TANH: - tensor_clone = ggml_tanh(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_SIGMOID: - tensor_clone = ggml_sigmoid(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_HARDSIGMOID: - tensor_clone = ggml_hardsigmoid(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_HARDSWISH: - tensor_clone = ggml_hardswish(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_ABS: - tensor_clone = ggml_abs(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_SOFTPLUS: - tensor_clone = ggml_softplus(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_STEP: - tensor_clone = ggml_step(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_ROUND: - tensor_clone = ggml_round(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_CEIL: - tensor_clone = ggml_ceil(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_FLOOR: - tensor_clone = ggml_floor(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_TRUNC: - tensor_clone = ggml_trunc(ggml_ctx, src_clone[0]); - break; - case GGML_UNARY_OP_SGN: - tensor_clone = ggml_sgn(ggml_ctx, src_clone[0]); - break; - default: - std::cerr << "Missing vk_check_results OP: " << ggml_op_name(tensor->op) << std::endl; - GGML_ABORT("fatal error"); - } - } else if (tensor->op == GGML_OP_GLU) { - if (src_clone[1] == nullptr) { - tensor_clone = ggml_glu(ggml_ctx, src_clone[0], (ggml_glu_op) tensor->op_params[0], tensor->op_params[1]); - } else { - tensor_clone = ggml_glu_split(ggml_ctx, src_clone[0], src_clone[1], (ggml_glu_op) tensor->op_params[0]); + // If we have as many flops entries as timing entries for the op, then compute and log the flops/S. + auto it = flops.find(t.first); + if (it != flops.end() && (it->second).size() == t.second.size()) { + uint64_t total_op_flops = 0; + for (const auto & elem : it->second) { + total_op_flops += elem; } - ggml_set_op_params_i32(tensor_clone, 2, ggml_get_op_params_i32(tensor, 2)); - ggml_set_op_params_i32(tensor_clone, 3, ggml_get_op_params_i32(tensor, 3)); - } else if (tensor->op == GGML_OP_CPY || tensor->op == GGML_OP_DUP) { - if (tensor->src[1] == nullptr) { - tensor_clone = ggml_dup(ggml_ctx, src_clone[0]); - tensor_clone->type = tensor->type; - } else { - tensor_clone = ggml_cpy(ggml_ctx, src_clone[0], src_clone[1]); - } - } else if (tensor->op == GGML_OP_CONT) { - tensor_clone = ggml_cont_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3]); - } else if (tensor->op == GGML_OP_RESHAPE) { - tensor_clone = ggml_reshape_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3]); - } else if (tensor->op == GGML_OP_VIEW) { - tensor_clone = ggml_view_4d(ggml_ctx, src_clone[0], tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], tensor->nb[1], tensor->nb[2], tensor->nb[3], ((int32_t *) tensor->op_params)[0]); - } else if (tensor->op == GGML_OP_PERMUTE) { - int32_t * params = (int32_t *)tensor->op_params; - tensor_clone = ggml_permute(ggml_ctx, src_clone[0], params[0], params[1], params[2], params[3]); - } else if (tensor->op == GGML_OP_TRANSPOSE) { - tensor_clone = ggml_transpose(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_GET_ROWS) { - tensor_clone = ggml_get_rows(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_ARGSORT) { - tensor_clone = ggml_argsort(ggml_ctx, src_clone[0], (ggml_sort_order) *(int *)tensor->op_params); - } else if (tensor->op == GGML_OP_TOP_K) { - tensor_clone = ggml_top_k(ggml_ctx, src_clone[0], tensor->ne[0]); - } else if (tensor->op == GGML_OP_SUM) { - tensor_clone = ggml_sum(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_SUM_ROWS) { - tensor_clone = ggml_sum_rows(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_CUMSUM) { - tensor_clone = ggml_cumsum(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_MEAN) { - tensor_clone = ggml_mean(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_ARGMAX) { - tensor_clone = ggml_argmax(ggml_ctx, src_clone[0]); - } else if (tensor->op == GGML_OP_COUNT_EQUAL) { - tensor_clone = ggml_count_equal(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_SOLVE_TRI) { - tensor_clone = ggml_solve_tri(ggml_ctx, src_clone[0], src_clone[1], true, true, false); - } else if (tensor->op == GGML_OP_IM2COL) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t p0 = tensor->op_params[2]; - const int32_t p1 = tensor->op_params[3]; - const int32_t d0 = tensor->op_params[4]; - const int32_t d1 = tensor->op_params[5]; - - const bool is_2D = tensor->op_params[6] == 1; - tensor_clone = ggml_im2col(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1, is_2D, tensor->type); - } else if (tensor->op == GGML_OP_IM2COL_3D) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t s2 = tensor->op_params[2]; - const int32_t p0 = tensor->op_params[3]; - const int32_t p1 = tensor->op_params[4]; - const int32_t p2 = tensor->op_params[5]; - const int32_t d0 = tensor->op_params[6]; - const int32_t d1 = tensor->op_params[7]; - const int32_t d2 = tensor->op_params[8]; - const int32_t IC = tensor->op_params[9]; - - tensor_clone = ggml_im2col_3d(ggml_ctx, src_clone[0], src_clone[1], IC, s0, s1, s2, p0, p1, p2, d0, d1, d2, tensor->type); - } else if (tensor->op == GGML_OP_TIMESTEP_EMBEDDING) { - const int32_t dim = tensor->op_params[0]; - const int32_t max_period = tensor->op_params[1]; - tensor_clone = ggml_timestep_embedding(ggml_ctx, src_clone[0], dim, max_period); - } else if (tensor->op == GGML_OP_CONV_TRANSPOSE_1D){ - const int32_t s0 = tensor->op_params[0]; - const int32_t p0 = tensor->op_params[1]; - const int32_t d0 = tensor->op_params[2]; - tensor_clone = ggml_conv_transpose_1d(ggml_ctx, src_clone[0], src_clone[1], s0, p0, d0); - } else if (tensor->op == GGML_OP_COL2IM_1D) { - const int32_t stride = tensor->op_params[0]; - const int32_t oc = tensor->op_params[1]; - const int32_t p0 = tensor->op_params[2]; - tensor_clone = ggml_col2im_1d(ggml_ctx, src_clone[0], stride, oc, p0); - } else if (tensor->op == GGML_OP_POOL_1D) { - enum ggml_op_pool op = static_cast(tensor->op_params[0]); - const int32_t k0 = tensor->op_params[1]; - const int32_t s0 = tensor->op_params[2]; - const int32_t p0 = tensor->op_params[3]; - - tensor_clone = ggml_pool_1d(ggml_ctx, src_clone[0], op, k0, s0, p0); - } else if (tensor->op == GGML_OP_POOL_2D) { - enum ggml_op_pool op = static_cast(tensor->op_params[0]); - const int32_t k0 = tensor->op_params[1]; - const int32_t k1 = tensor->op_params[2]; - const int32_t s0 = tensor->op_params[3]; - const int32_t s1 = tensor->op_params[4]; - const int32_t p0 = tensor->op_params[5]; - const int32_t p1 = tensor->op_params[6]; - - tensor_clone = ggml_pool_2d(ggml_ctx, src_clone[0], op, k0, k1, s0, s1, p0, p1); - } else if (tensor->op == GGML_OP_CONV_2D) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t p0 = tensor->op_params[2]; - const int32_t p1 = tensor->op_params[3]; - const int32_t d0 = tensor->op_params[4]; - const int32_t d1 = tensor->op_params[5]; - tensor_clone = ggml_conv_2d(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1); - } else if (tensor->op == GGML_OP_CONV_3D) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t s2 = tensor->op_params[2]; - const int32_t p0 = tensor->op_params[3]; - const int32_t p1 = tensor->op_params[4]; - const int32_t p2 = tensor->op_params[5]; - const int32_t d0 = tensor->op_params[6]; - const int32_t d1 = tensor->op_params[7]; - const int32_t d2 = tensor->op_params[8]; - const int32_t IC = tensor->op_params[9]; - const int32_t N = tensor->op_params[10]; - const int32_t OC = tensor->op_params[11]; - tensor_clone = ggml_conv_3d_direct(ggml_ctx, src_clone[0], src_clone[1], s0, s1, s2, p0, p1, p2, d0, d1, d2, IC, N, OC); - } else if (tensor->op == GGML_OP_CONV_2D_DW) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t p0 = tensor->op_params[2]; - const int32_t p1 = tensor->op_params[3]; - const int32_t d0 = tensor->op_params[4]; - const int32_t d1 = tensor->op_params[5]; - tensor_clone = ggml_conv_2d_dw_direct(ggml_ctx, src_clone[0], src_clone[1], s0, s1, p0, p1, d0, d1); - } else if (tensor->op == GGML_OP_CONV_TRANSPOSE_2D) { - const int32_t s = tensor->op_params[0]; - tensor_clone = ggml_conv_transpose_2d_p0(ggml_ctx, src_clone[0], src_clone[1], s); - } else if (tensor->op == GGML_OP_LEAKY_RELU) { - const float * op_params = (const float *)tensor->op_params; - tensor_clone = ggml_leaky_relu(ggml_ctx, src_clone[0], op_params[0], false); - } else if (tensor->op == GGML_OP_RWKV_WKV6) { - tensor_clone = ggml_rwkv_wkv6(ggml_ctx, src_clone[0], src_clone[1], - src_clone[2], src_clone[3], src_clone[4], src_clone[5]); - } else if (tensor->op == GGML_OP_RWKV_WKV7) { - tensor_clone = ggml_rwkv_wkv7(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], src_clone[3], - src_clone[4], src_clone[5], src_clone[6]); - } else if (tensor->op == GGML_OP_GATED_LINEAR_ATTN) { - const float * op_params = (const float *)tensor->op_params; - tensor_clone = ggml_gated_linear_attn(ggml_ctx, src_clone[0], src_clone[1], - src_clone[2], src_clone[3], src_clone[4], op_params[0]); - } else if (tensor->op == GGML_OP_GATED_DELTA_NET) { - tensor_clone = ggml_gated_delta_net(ggml_ctx, src_clone[0], src_clone[1], - src_clone[2], src_clone[3], src_clone[4], src_clone[5], - ggml_get_op_params_i32(tensor, 0)); - } else if (tensor->op == GGML_OP_OPT_STEP_ADAMW) { - src_clone[0]->flags = tensor->src[0]->flags; - tensor_clone = ggml_opt_step_adamw(ggml_ctx, src_clone[0], src_clone[1], - src_clone[2], src_clone[3], src_clone[4]); - } else if (tensor->op == GGML_OP_OPT_STEP_SGD) { - src_clone[0]->flags = tensor->src[0]->flags; - tensor_clone = ggml_opt_step_sgd(ggml_ctx, src_clone[0], src_clone[1], - src_clone[2]); - } else if (tensor->op == GGML_OP_ADD_ID) { - tensor_clone = ggml_add_id(ggml_ctx, src_clone[0], src_clone[1], src_clone[2]); - } else if (tensor->op == GGML_OP_SSM_SCAN) { - const int32_t K = ggml_get_op_params_i32(tensor, 0); - tensor_clone = ggml_ssm_scan(ggml_ctx, src_clone[0], src_clone[1], src_clone[2], - src_clone[3], src_clone[4], src_clone[5], src_clone[6], K); - } else if (tensor->op == GGML_OP_SSM_CONV) { - tensor_clone = ggml_ssm_conv(ggml_ctx, src_clone[0], src_clone[1]); - } else if (tensor->op == GGML_OP_ROLL) { - const int32_t s0 = tensor->op_params[0]; - const int32_t s1 = tensor->op_params[1]; - const int32_t s2 = tensor->op_params[2]; - const int32_t s3 = tensor->op_params[3]; - tensor_clone = ggml_roll(ggml_ctx, src_clone[0], s0, s1, s2, s3); - } - else { - std::cerr << "Missing vk_check_results OP: " << ggml_op_name(tensor->op) << std::endl; - GGML_ABORT("fatal error"); + std::cerr << " (" + << (double(total_op_flops) / (1000.0 * 1000.0 * 1000.0)) / + (double(total_op_times) / (1000.0 * 1000.0 * 1000.0)) + << " GFLOPS/s)"; } - cloned_tensors[tensor] = tensor_clone; - } - ggml_cgraph * cgraph_cpu = ggml_new_graph(ggml_ctx); - ggml_build_forward_expand(cgraph_cpu, tensor_clone); + total_all_op_times += total_op_times; - ggml_graph_compute_with_ctx(ggml_ctx, cgraph_cpu, 8); - - if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { - ggml_vk_print_tensor(tensor_clone, "tensor_clone"); + std::cerr << std::endl; } - comp_size = ggml_nbytes(tensor_clone); - - comp_result = malloc(comp_size); - memcpy(comp_result, tensor_clone->data, comp_size); - memcpy(comp_nb, tensor_clone->nb, sizeof(size_t) * GGML_MAX_DIMS); - - for (auto m : cloned_mallocs) { - free(m); + if (timings.size() > 0) { + std::cerr << "Total time: " << total_all_op_times / 1000.0 << " us." << std::endl; } - ggml_free(ggml_ctx); - - VK_LOG_DEBUG("END ggml_vk_check_results_0(" << tensor->name << ")"); + timings.clear(); + flops.clear(); } -static void ggml_vk_check_results_1(ggml_backend_vk_context * ctx, ggml_cgraph * cgraph, int tensor_idx) { - ggml_tensor * tensor = cgraph->nodes[tensor_idx + ctx->num_additional_fused_ops]; - if (tensor->op == GGML_OP_TRANSPOSE || tensor->op == GGML_OP_SET_ROWS) { +std::string vk_perf_logger::get_node_fusion_name(const ggml_tensor * node, const char *fusion_name, uint64_t *n_flops) { + *n_flops = ggml_vk_get_node_flops(node); + std::string fusion_str; + if (fusion_name) { + fusion_str = fusion_name + std::string(" "); + } + if (node->op == GGML_OP_UNARY) { + return fusion_str + ggml_unary_op_name(ggml_get_unary_op(node)); + } + if (node->op == GGML_OP_MUL_MAT || node->op == GGML_OP_MUL_MAT_ID) { + const uint64_t m = node->ne[0]; + const uint64_t n = node->ne[1]; + const uint64_t k = node->src[1]->ne[0]; + const uint64_t batch = node->ne[2] * node->ne[3]; + std::string name = ggml_op_name(node->op); + if ((node->op == GGML_OP_MUL_MAT && n <= mul_mat_vec_max_cols) || + (node->op == GGML_OP_MUL_MAT_ID && node->src[2]->ne[1] == 1)) { + name += "_VEC"; + } + name += " "; + name += ggml_type_name(node->src[0]->type); + name += " m=" + std::to_string(m) + " n=" + std::to_string(n) + " k=" + std::to_string(k); + if (node->op == GGML_OP_MUL_MAT_ID) { + name += " n_expert=" + std::to_string(node->src[0]->ne[2]); + } + if (batch > 1) { + name += " batch=" + std::to_string(batch); + } + return fusion_str + name; + } + if (node->op == GGML_OP_CONV_2D || node->op == GGML_OP_CONV_TRANSPOSE_2D) { + std::string name = ggml_op_name(node->op); + const ggml_tensor * knl = node->src[0]; + uint64_t Cout = node->ne[2]; + uint64_t size_K = node->src[1]->ne[2] * knl->ne[0] * knl->ne[1]; + uint64_t size_N = node->ne[3] * node->ne[0] * node->ne[1]; + name += " M=Cout=" + std::to_string(Cout) + ", K=Cin*KW*KH=" + std::to_string(size_K) + + ", N=N*OW*OH=" + std::to_string(size_N); + return fusion_str + name; + } + if (node->op == GGML_OP_RMS_NORM) { + std::string name = ggml_op_name(node->op); + name += "(" + std::to_string(node->ne[0]) + "," + std::to_string(node->ne[1]) + "," + std::to_string(node->ne[2]) + "," + std::to_string(node->ne[3]) + ")"; + return fusion_str + name; + } + if (node->op == GGML_OP_FLASH_ATTN_EXT) { + const ggml_tensor * dst = node; + const ggml_tensor * q = node->src[0]; + const ggml_tensor * k = node->src[1]; + const ggml_tensor * v = node->src[2]; + const ggml_tensor * m = node->src[3]; + std::stringstream name; + name << fusion_str; + name << ggml_op_name(node->op) << + " dst(" << dst->ne[0] << "," << dst->ne[1] << "," << dst->ne[2] << "," << dst->ne[3] << "), " << + " q(" << q->ne[0] << "," << q->ne[1] << "," << q->ne[2] << "," << q->ne[3] << "), " << + " k(" << k->ne[0] << "," << k->ne[1] << "," << k->ne[2] << "," << k->ne[3] << "), " << + " v(" << v->ne[0] << "," << v->ne[1] << "," << v->ne[2] << "," << v->ne[3] << "), " << + " m(" << (m?m->ne[0]:0) << "," << (m?m->ne[1]:0) << "," << (m?m->ne[2]:0) << "," << (m?m->ne[3]:0) << ")"; + return name.str(); + } + if (node->op == GGML_OP_TOP_K) { + std::stringstream name; + name << fusion_str; + name << ggml_op_name(node->op) << + " K=" << node->ne[0] << + " (" << node->src[0]->ne[0] << "," << node->src[0]->ne[1] << "," << node->src[0]->ne[2] << "," << node->src[0]->ne[3] << ")"; + return name.str(); + } + return fusion_str + ggml_op_name(node->op); +} + +ggml_backend_vk_buffer_context::~ggml_backend_vk_buffer_context() { + ggml_vk_destroy_buffer(dev_buffer); +} + +ggml_vk_debug_label::ggml_vk_debug_label(vk_context & ctx, const std::string & pipeline_name, uint32_t wg0, uint32_t wg1, uint32_t wg2) { + if (!vk_instance.debug_utils_support || ctx->s == nullptr) { return; } + begin(ctx, pipeline_name + " (" + std::to_string(wg0) + "," + std::to_string(wg1) + "," + std::to_string(wg2) + ")"); +} - if (!(vk_output_tensor > 0 && vk_output_tensor == check_counter) && check_counter <= vk_skip_checks) { +ggml_vk_debug_label::ggml_vk_debug_label(vk_context & ctx, const ggml_cgraph * cgraph, int node_idx, int n_fused) { + if (!vk_instance.debug_utils_support || ctx->s == nullptr) { return; } + std::string name = ggml_op_name(cgraph->nodes[node_idx]->op); + for (int i = 1; i <= n_fused; i++) { + name += "+"; + name += ggml_op_name(cgraph->nodes[node_idx + i]->op); + } + name += " "; + name += cgraph->nodes[node_idx]->name; + begin(ctx, name); +} - VK_LOG_DEBUG("ggml_vk_check_results_1(" << tensor->name << ")"); - - ggml_tensor * src0 = tensor->src[0]; - ggml_tensor * src1 = tensor->src[1]; - ggml_tensor * src2 = tensor->src[2]; - ggml_tensor * src3 = tensor->src[3]; - - void * tensor_data = tensor->data; - - if (ggml_backend_buffer_is_vk(tensor->buffer)) { - size_t tensor_size = ggml_nbytes(tensor); - tensor_data = malloc(tensor_size); - - ggml_backend_vk_buffer_context * buf_ctx = (ggml_backend_vk_buffer_context *)tensor->buffer->context; - - vk_buffer& buffer_gpu = buf_ctx->dev_buffer; - uint64_t offset = vk_tensor_offset(tensor) + tensor->view_offs; - if (offset + tensor_size >= buffer_gpu->size) { - tensor_size = buffer_gpu->size - offset; - } - - ggml_vk_buffer_read(buffer_gpu, offset, tensor_data, tensor_size); - } - - float first_error_result = -1.0f; - float first_error_correct = -1.0f; - std::array first_error = { -1, -1, -1, -1 }; - double avg_err = 0.0; - size_t counter = 0; - - for (int i3 = 0; i3 < tensor->ne[3]; i3++) { - for (int i2 = 0; i2 < tensor->ne[2]; i2++) { - for (int i1 = 0; i1 < tensor->ne[1]; i1++) { - for (int i0 = 0; i0 < tensor->ne[0]; i0++) { - const bool buffer_size_fit = i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0] < comp_size; - float correct = 0.0f; - float result = 0.0f; - - if (buffer_size_fit) { - if (tensor->type == GGML_TYPE_F32) { - correct = *(float *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); - result = *(float *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); - } else if (tensor->type == GGML_TYPE_F16) { - correct = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0])); - result = ggml_fp16_to_fp32(*(ggml_fp16_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0])); - } else if (tensor->type == GGML_TYPE_BF16) { - correct = ggml_bf16_to_fp32(*(ggml_bf16_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0])); - result = ggml_bf16_to_fp32(*(ggml_bf16_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0])); - } else if (tensor->type == GGML_TYPE_I32) { - correct = *(int32_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); - result = *(int32_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); - } else if (tensor->type == GGML_TYPE_I64) { - correct = *(int64_t *) ((char *) comp_result + i3*comp_nb[3] + i2*comp_nb[2] + i1*comp_nb[1] + i0*comp_nb[0]); - result = *(int64_t *) ((char *) tensor_data + i3*tensor->nb[3] + i2*tensor->nb[2] + i1*tensor->nb[1] + i0*tensor->nb[0]); - } else { - std::cerr << "Results check not implemented for type " << ggml_type_name(tensor->type) << std::endl; - } - } else { - std::cerr << "Missing debug code for type " << ggml_type_name(tensor->type) << std::endl; - GGML_ABORT("fatal error"); - } - - if ((std::isnan(correct) != std::isnan(result)) || (std::isinf(correct) != std::isinf(result)) || !buffer_size_fit) { - std::cerr << "ERROR: Invalid value in " << ggml_op_name(tensor->op) << " i3=" << i3 << " i2=" << i2 << " i1=" << i1 << " i0=" << i0 << " result=" << result << " correct=" << correct << " avg_err=" << (avg_err / counter) << std::endl; - std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; - if (src0 != nullptr) { - std::cerr << "src0=" << src0 << " src0->name=" << src0->name << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; - } - if (src1 != nullptr) { - std::cerr << "src1=" << src1 << " src1->name=" << src1->name << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; - } - if (src2 != nullptr) { - std::cerr << "src2=" << src2 << " src2->name=" << src2->name << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; - } - if (src3 != nullptr) { - std::cerr << "src3=" << src3 << " src3->name=" << src3->name << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; - } - std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; - std::cerr << std::endl << "Result:" << std::endl; - ggml_vk_print_tensor_area(tensor, tensor_data, i0, i1, i2, i3); - std::cerr << std::endl << "Correct:" << std::endl; - ggml_vk_print_tensor_area(tensor, comp_result, i0, i1, i2, i3); - std::cerr << std::endl; - std::vector done; - ggml_vk_print_graph_origin(tensor, done); - GGML_ABORT("fatal error"); - } - const double denom = std::fabs(correct) > 1.0f ? (std::fabs(correct) > 1e-8 ? std::fabs(correct) : 1e-8) : 1.0f; - if (first_error[0] == -1 && std::fabs(correct - result) / denom > 0.5) { - first_error[0] = i0; - first_error[1] = i1; - first_error[2] = i2; - first_error[3] = i3; - first_error_result = result; - first_error_correct = correct; - } - - // Special case, value is infinite, avoid NaN result in avg_err - // NaN also appears in results, if both are nan error is 0 - if (!std::isinf(correct) && !std::isinf(result) && !std::isnan(correct) && !std::isnan(result)) { - avg_err += std::fabs(correct - result) / denom; - } - counter++; - } - } - } +ggml_vk_debug_label::ggml_vk_debug_label(vk_queue_handle * handle, const char * name) { + if (!vk_instance.debug_utils_support || handle == nullptr) { + return; } + vk::DebugUtilsLabelEXT label = {}; + label.pLabelName = name; + label.color = std::array{1.0f, 1.0f, 1.0f, 1.0f}; - avg_err /= counter; + qhandle = handle; + std::lock_guard guard(*qhandle); + vk_instance.pfn_vkQueueBeginDebugUtilsLabelEXT(qhandle->queue, reinterpret_cast(&label)); +} - if (vk_output_tensor > 0 && vk_output_tensor == check_counter) { - std::cerr << "TENSOR CHECK: avg_err=" << avg_err << " in " << ggml_op_name(tensor->op) << " (check " << check_counter << ")" << std::endl; - std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; - if (src0 != nullptr) { - std::cerr << "src0=" << src0 << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; - } - if (src1 != nullptr) { - std::cerr << "src1=" << src1 << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; - } - if (src2 != nullptr) { - std::cerr << "src2=" << src2 << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; - } - if (src3 != nullptr) { - std::cerr << "src3=" << src3 << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; +void ggml_vk_debug_label::close() { + if (subctx != nullptr) { + // close on the current command buffer, which may differ from the one begin used + if (subctx->s != nullptr) { + vk_instance.pfn_vkCmdEndDebugUtilsLabelEXT(subctx->s->buffer->buf); } - std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; - std::cerr << std::endl << "Result:" << std::endl; - ggml_vk_print_tensor_area(tensor, tensor_data, 5, 5, 0, 0); - std::cerr << std::endl << "Correct:" << std::endl; - ggml_vk_print_tensor_area(tensor, comp_result, 5, 5, 0, 0); - std::cerr << std::endl; - std::vector done; - ggml_vk_print_graph_origin(tensor, done); + subctx->debug_labels.pop_back(); + subctx = nullptr; } - - if (avg_err > 0.01 || std::isnan(avg_err)) { - std::cerr << "ERROR: avg_err=" << avg_err << " in " << ggml_op_name(tensor->op) << " (check " << check_counter << ")" << std::endl; - std::cerr << "tensor=" << tensor << " tensor->name=" << tensor->name << " tensor->type: " << ggml_type_name(tensor->type) << " ne0=" << tensor->ne[0] << " nb0=" << tensor->nb[0] << " ne1=" << tensor->ne[1] << " nb1=" << tensor->nb[1] << " ne2=" << tensor->ne[2] << " nb2=" << tensor->nb[2] << " ne3=" << tensor->ne[3] << " nb3=" << tensor->nb[3] << " offset=" << tensor->view_offs << std::endl; - if (src0 != nullptr) { - std::cerr << "src0=" << src0 << " op=" << ggml_op_name(src0->op) << " type=" << ggml_type_name(src0->type) << " ne0=" << src0->ne[0] << " nb0=" << src0->nb[0] << " ne1=" << src0->ne[1] << " nb1=" << src0->nb[1] << " ne2=" << src0->ne[2] << " nb2=" << src0->nb[2] << " ne3=" << src0->ne[3] << " nb3=" << src0->nb[3] << " offset=" << src0->view_offs << std::endl; - } - if (src1 != nullptr) { - std::cerr << "src1=" << src1 << " op=" << ggml_op_name(src1->op) << " type=" << ggml_type_name(src1->type) << " ne0=" << src1->ne[0] << " nb0=" << src1->nb[0] << " ne1=" << src1->ne[1] << " nb1=" << src1->nb[1] << " ne2=" << src1->ne[2] << " nb2=" << src1->nb[2] << " ne3=" << src1->ne[3] << " nb3=" << src1->nb[3] << " offset=" << src1->view_offs << std::endl; - } - if (src2 != nullptr) { - std::cerr << "src2=" << src2 << " op=" << ggml_op_name(src2->op) << " type=" << ggml_type_name(src2->type) << " ne0=" << src2->ne[0] << " nb0=" << src2->nb[0] << " ne1=" << src2->ne[1] << " nb1=" << src2->nb[1] << " ne2=" << src2->ne[2] << " nb2=" << src2->nb[2] << " ne3=" << src2->ne[3] << " nb3=" << src2->nb[3] << " offset=" << src2->view_offs << std::endl; - } - if (src3 != nullptr) { - std::cerr << "src3=" << src3 << " op=" << ggml_op_name(src3->op) << " type=" << ggml_type_name(src3->type) << " ne0=" << src3->ne[0] << " nb0=" << src3->nb[0] << " ne1=" << src3->ne[1] << " nb1=" << src3->nb[1] << " ne2=" << src3->ne[2] << " nb2=" << src3->nb[2] << " ne3=" << src3->ne[3] << " nb3=" << src3->nb[3] << " offset=" << src3->view_offs << std::endl; - } - std::cerr << "First error: result=" << first_error_result << " correct=" << first_error_correct << " i3=" << first_error[3] << " i2=" << first_error[2] << " i1=" << first_error[1] << " i0=" << first_error[0] << std::endl; - std::cerr << std::endl << "Result:" << std::endl; - ggml_vk_print_tensor_area(tensor, tensor_data, first_error[0], first_error[1], first_error[2], first_error[3]); - std::cerr << std::endl << "Correct:" << std::endl; - ggml_vk_print_tensor_area(tensor, comp_result, first_error[0], first_error[1], first_error[2], first_error[3]); - std::cerr << std::endl; - std::vector done; - ggml_vk_print_graph_origin(tensor, done); - GGML_ABORT("fatal error"); - } else { - std::cerr << check_counter << " " << tensor->name << " op=" << ggml_op_name(tensor->op) << " avg_err=" << avg_err << std::endl; + if (qhandle != nullptr) { + std::lock_guard guard(*qhandle); + vk_instance.pfn_vkQueueEndDebugUtilsLabelEXT(qhandle->queue); + qhandle = nullptr; } +} - free(comp_result); - comp_result = nullptr; - comp_size = 0; - - if (ggml_backend_buffer_is_vk(tensor->buffer)) { - free(tensor_data); +void ggml_vk_debug_label::begin(vk_context & ctx, const std::string & name) { + if (!vk_instance.debug_utils_support || ctx->s == nullptr) { + return; } - - VK_LOG_DEBUG("END ggml_vk_check_results_1(" << tensor->name << ")"); + subctx = ctx.get(); + subctx->debug_labels.push_back(name); + ggml_vk_cmd_label_begin(subctx->s->buffer->buf, subctx->debug_labels.back().c_str()); } -#endif -GGML_BACKEND_DL_IMPL(ggml_backend_vk_reg) diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/argsort.comp b/ggml/src/ggml-vulkan/vulkan-shaders/argsort.comp index 0fc2b9b7..4ba63f7a 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/argsort.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/argsort.comp @@ -33,7 +33,11 @@ void argsort(bool needs_bounds_check, const uint row) { const uint row_offset = row * p.ncols; // initialize indices - dst_row[col] = ivec2(col, floatBitsToInt(data_a[row_offset + col])); + ivec2 value = ivec2(col, 0); + if (!needs_bounds_check || col < p.ncols) { + value.y = floatBitsToInt(data_a[row_offset + col]); + } + dst_row[col] = value; barrier(); uint num_outer_loop_iters = NCOLS_PADDED_LOG2; @@ -42,18 +46,20 @@ void argsort(bool needs_bounds_check, const uint row) { [[unroll]] for (uint j = k / 2, inner_idx = 0; inner_idx < num_inner_loop_iters; j /= 2, inner_idx++) { const int ixj = int(col ^ j); - int idx_0 = (col & k) == 0 ? col : ixj; - int idx_1 = (col & k) == 0 ? ixj : col; + if (ixj > col) { + int idx_0 = (col & k) == 0 ? col : ixj; + int idx_1 = (col & k) == 0 ? ixj : col; - ivec2 sh_idx_0 = dst_row[idx_0]; - ivec2 sh_idx_1 = dst_row[idx_1]; - bool idx_0_oob = needs_bounds_check ? sh_idx_0.x >= p.ncols : false; - bool idx_1_oob = needs_bounds_check ? sh_idx_1.x >= p.ncols : false; + ivec2 sh_idx_0 = dst_row[idx_0]; + ivec2 sh_idx_1 = dst_row[idx_1]; + bool idx_0_oob = needs_bounds_check ? sh_idx_0.x >= p.ncols : false; + bool idx_1_oob = needs_bounds_check ? sh_idx_1.x >= p.ncols : false; - if ((idx_0_oob || - (!idx_1_oob && intBitsToFloat(sh_idx_0.y) > intBitsToFloat(sh_idx_1.y))) && (ixj > col)) { - dst_row[idx_0] = sh_idx_1; - dst_row[idx_1] = sh_idx_0; + if (idx_0_oob || + (!idx_1_oob && intBitsToFloat(sh_idx_0.y) > intBitsToFloat(sh_idx_1.y))) { + dst_row[idx_0] = sh_idx_1; + dst_row[idx_1] = sh_idx_0; + } } barrier(); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/argsort_large.comp b/ggml/src/ggml-vulkan/vulkan-shaders/argsort_large.comp index 920bac6b..f6a29be2 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/argsort_large.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/argsort_large.comp @@ -27,6 +27,8 @@ layout (push_constant) uniform parameter { uint inner_end; } p; +shared int s; + void argsort(bool needs_bounds_check, const uint row) { // bitonic sort int col = int(gl_GlobalInvocationID.x); @@ -35,6 +37,12 @@ void argsort(bool needs_bounds_check, const uint row) { const uint row_offset = row * p.ncols; uint idx_offset = row * p.ncols_padded; + // workaround for NV driver/compiler bug - dummy use of shared memory + if (gl_LocalInvocationIndex == 0) { + s = 0; + col += s; + } + bool need_barrier = false; // initialize indices @@ -42,7 +50,10 @@ void argsort(bool needs_bounds_check, const uint row) { [[unroll]] for (int u = 0; u < WG_UNROLL_FACTOR; ++u) { uint c = u*BLOCK_SIZE + col; if (c < p.ncols_padded) { - ivec2 v = ivec2(c, floatBitsToInt(data_a[row_offset + c])); + ivec2 v = ivec2(c, 0); + if (!needs_bounds_check || c < p.ncols) { + v.y = floatBitsToInt(data_a[row_offset + c]); + } tmp_idx[idx_offset + c] = v; } } diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp b/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp index 99400098..c64004cd 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp @@ -19,6 +19,7 @@ #endif #include "types.glsl" +#include "utils.glsl" // shape notation: [dim(N), ..., dim(0)] -- stride(dim(j)) >= stride(dim(i)) if i > j layout(binding = 0) readonly buffer A { @@ -193,14 +194,6 @@ uint32_t Br = tid / BS_NPQ; uint32_t Bc = tid % BS_NPQ; const uint32_t BrpWg = WG_SIZE / BS_NPQ; -// see init_fastdiv_values in ggml-vulkan.cpp -uint fastdiv(uint n, uint mp, uint L) { - uint msbs, lsbs; - // msbs = mulhi(n, mp) - umulExtended(n, mp, msbs, lsbs); - return (msbs + n) >> L; -} - #ifdef COOPMAT2 #define ACC_TYPE float16_t diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/conv3d_mm.comp b/ggml/src/ggml-vulkan/vulkan-shaders/conv3d_mm.comp index f66f299f..d5ce4290 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/conv3d_mm.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/conv3d_mm.comp @@ -15,6 +15,7 @@ #endif #include "types.glsl" +#include "utils.glsl" // shape notation: [dim(N), ..., dim(0)] -- stride(dim(j)) >= stride(dim(i)) if i > j layout(binding = 0) readonly buffer A { @@ -178,14 +179,6 @@ uint32_t Br = tid / BS_NPQ; uint32_t Bc = tid % BS_NPQ; const uint32_t BrpWg = WG_SIZE / BS_NPQ; -// see init_fastdiv_values in ggml-vulkan.cpp -uint fastdiv(uint n, uint mp, uint L) { - uint msbs, lsbs; - // msbs = mulhi(n, mp) - umulExtended(n, mp, msbs, lsbs); - return (msbs + n) >> L; -} - void split_crs(uint32_t crs_idx, out uint32_t ic, out uint32_t kd, out uint32_t kh, out uint32_t kw) { const uint32_t KHKW = KH * KW; const uint32_t KDKHKW = KD * KHKW; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/copy_transpose_02.comp b/ggml/src/ggml-vulkan/vulkan-shaders/copy_transpose_02.comp new file mode 100644 index 00000000..5a3d66da --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/copy_transpose_02.comp @@ -0,0 +1,61 @@ +#version 450 + +#include "types.glsl" +#include "generic_unary_head.glsl" + +// workgroup does 32x32 tile, but uses 32x8 threads +#define TILE_DIM 32 +layout(local_size_x = 32, local_size_y = 8, local_size_z = 1) in; + +// +1 padding avoids shared-memory bank conflicts on the transposed read +shared uint sh[TILE_DIM][TILE_DIM + 1]; + +void iter(uvec3 wg_id) { + const uint tile_i0 = wg_id.x; // tiles dst ne10 (== src ne00) + const uint tile_i2 = wg_id.y; // tiles dst ne12 (== src ne02) + + const uint tid_col = gl_LocalInvocationID.x; + const uint tid_row = gl_LocalInvocationID.y; + + const uint i1 = wg_id.z % p.ne11; + const uint i3 = wg_id.z / p.ne11; + const uint i01 = i1; + const uint i03 = i3; + + [[unroll]] for (uint y = 0; y < 4; ++y) { + const uint i00 = tile_i0 * TILE_DIM + tid_row + 8 * y; + const uint i02 = tile_i2 * TILE_DIM + tid_col; + if (i00 < p.ne00 && i01 < p.ne01 && i02 < p.ne02 && i03 < p.ne03) { + const uint src_idx = i00 * p.nb00 + i01 * p.nb01 + i02 * p.nb02 + i03 * p.nb03; + sh[tid_row + 8 * y][tid_col] = uint(data_a[get_aoffset() + src_idx]); + } + } + + barrier(); + + [[unroll]] for (uint y = 0; y < 4; ++y) { + const uint i0 = tile_i0 * TILE_DIM + tid_col; + const uint i2 = tile_i2 * TILE_DIM + tid_row + 8 * y; + if (i0 < p.ne10 && i1 < p.ne11 && i2 < p.ne12 && i3 < p.ne13) { + const uint dst_idx = i0 * p.nb10 + i1 * p.nb11 + i2 * p.nb12 + i3 * p.nb13; + data_d[get_doffset() + dst_idx] = D_TYPE(sh[tid_col][tid_row + 8 * y]); + } + } +} + +#define CEIL_DIV(a, b) (((a) + (b) - 1) / (b)) + +void main() { + bool need_barrier = false; + for (uint z = gl_WorkGroupID.z; z < p.ne11 * p.ne13; z += gl_NumWorkGroups.z) { + for (uint y = gl_WorkGroupID.y; y < CEIL_DIV(p.ne12, TILE_DIM); y += gl_NumWorkGroups.y) { + for (uint x = gl_WorkGroupID.x; x < CEIL_DIV(p.ne10, TILE_DIM); x += gl_NumWorkGroups.x) { + if (need_barrier) { + barrier(); + } + need_barrier = true; + iter(uvec3(x, y, z)); + } + } + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/count_experts.comp b/ggml/src/ggml-vulkan/vulkan-shaders/count_experts.comp index ffc86086..06a50181 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/count_experts.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/count_experts.comp @@ -2,7 +2,13 @@ #extension GL_EXT_control_flow_attributes : enable +#ifdef USE_SUBGROUPS +#extension GL_KHR_shader_subgroup_basic : enable +#extension GL_KHR_shader_subgroup_arithmetic : enable +#endif + #include "types.glsl" +#include "utils.glsl" layout (push_constant) uniform parameter { @@ -11,6 +17,10 @@ layout (push_constant) uniform parameter uint32_t nb00; uint32_t nb01; uint32_t a_offset; + uint32_t n_experts; + uint32_t hoist_row_ids; + uint32_t ne00mp; + uint32_t ne00L; } p; #define BLOCK_SIZE 256 @@ -20,17 +30,96 @@ layout(local_size_x = BLOCK_SIZE, local_size_y = 1, local_size_z = 1) in; layout (binding = 0) readonly buffer A {uint data_a[];}; layout (binding = 1) writeonly buffer D {uint data_d[];}; -shared uint vals[BLOCK_SIZE]; +// Upper bound on n_experts for the hoisted row-id path. Must match the limit in +// ggml_vk_mul_mat_id_q_f16 (hoist_row_ids). The non-hoisted reduction below only +// needs BLOCK_SIZE entries. +#define MAX_EXPERTS 1024 + +shared uint vals[MAX_EXPERTS]; +shared uint offsets[MAX_EXPERTS]; +shared uint cursors[MAX_EXPERTS]; +// data_d layout when p.hoist_row_ids is set: +// [0, n_experts) per-expert row count +// [n_experts, 2*n_experts) per-expert start offset into the row id region +// [2*n_experts] total row count +// [2*n_experts + 1, ) row ids grouped by expert, packed as (i01 << 16) | (i00 & 0xffff) +// Otherwise only data_d[expert_id] is written, holding that expert's row count. void main() { const uint expert_id = gl_WorkGroupID.x; const uint num_elements = p.ne00 * p.ne01; const uint tid = gl_LocalInvocationID.x; + if (p.hoist_row_ids != 0) { + for (uint e = tid; e < p.n_experts; e += BLOCK_SIZE) { + vals[e] = 0; + } + barrier(); + + for (uint idx = tid; idx < num_elements; idx += BLOCK_SIZE) { + const uint i01 = fastdiv(idx, p.ne00mp, p.ne00L); + const uint i00 = idx - i01 * p.ne00; + const uint expert = data_a[p.a_offset + i01 * p.nb01 + i00 * p.nb00]; + if (expert < p.n_experts) { + atomicAdd(vals[expert], 1); + } + } + barrier(); + +#ifdef USE_SUBGROUPS + if (gl_SubgroupID == 0) { + // pad the trip count so the subgroup ops stay in uniform control flow + const uint n_experts_padded = (p.n_experts + gl_SubgroupSize - 1) & ~(gl_SubgroupSize - 1); + uint base = 0; + for (uint expert = gl_SubgroupInvocationID; expert < n_experts_padded; expert += gl_SubgroupSize) { + const bool in_range = expert < p.n_experts; + const uint count = in_range ? vals[expert] : 0; + const uint offset = base + subgroupExclusiveAdd(count); + if (in_range) { + data_d[expert] = count; + data_d[p.n_experts + expert] = offset; + offsets[expert] = offset; + cursors[expert] = 0; + } + base += subgroupAdd(count); + } + if (subgroupElect()) { + data_d[2 * p.n_experts] = base; + } + } +#else + if (tid == 0) { + uint offset = 0; + for (uint expert = 0; expert < p.n_experts; ++expert) { + const uint count = vals[expert]; + data_d[expert] = count; + data_d[p.n_experts + expert] = offset; + offsets[expert] = offset; + cursors[expert] = 0; + offset += count; + } + data_d[2 * p.n_experts] = offset; + } +#endif + barrier(); + + for (uint idx = tid; idx < num_elements; idx += BLOCK_SIZE) { + const uint i01 = fastdiv(idx, p.ne00mp, p.ne00L); + const uint i00 = idx - i01 * p.ne00; + const uint expert = data_a[p.a_offset + i01 * p.nb01 + i00 * p.nb00]; + if (expert < p.n_experts) { + const uint row = atomicAdd(cursors[expert], 1); + const uint packed_row_id = (i01 << 16) | (i00 & 0xffffu); + data_d[2 * p.n_experts + 1 + offsets[expert] + row] = packed_row_id; + } + } + return; + } + uint count = 0; for (uint idx = tid; idx < num_elements; idx += BLOCK_SIZE) { - const uint i01 = idx / p.ne00; - const uint i00 = idx % p.ne00; + const uint i01 = fastdiv(idx, p.ne00mp, p.ne00L); + const uint i00 = idx - i01 * p.ne00; const uint a = data_a[p.a_offset + i01 * p.nb01 + i00 * p.nb00]; count += uint(a == expert_id); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss.comp b/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss.comp new file mode 100644 index 00000000..0c135c6f --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss.comp @@ -0,0 +1,78 @@ +#version 450 + +#include "generic_head.glsl" +#include "types.glsl" + +#extension GL_EXT_control_flow_attributes : enable + +layout(constant_id = 0) const uint BLOCK_SIZE = 32; +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer A {A_TYPE data_a[];}; +layout (binding = 1) readonly buffer B {B_TYPE data_b[];}; +layout (binding = 2) writeonly buffer D {D_TYPE data_d[];}; + +shared FLOAT_TYPE tmp[BLOCK_SIZE]; + +FLOAT_TYPE wg_reduce_max(FLOAT_TYPE v) { + const uint tid = gl_LocalInvocationID.x; + tmp[tid] = v; + barrier(); + [[unroll]] for (uint s = BLOCK_SIZE / 2; s > 0; s >>= 1) { + if (tid < s) { + tmp[tid] = max(tmp[tid], tmp[tid + s]); + } + barrier(); + } + v = tmp[0]; + barrier(); + return v; +} + +FLOAT_TYPE wg_reduce_sum(FLOAT_TYPE v) { + const uint tid = gl_LocalInvocationID.x; + tmp[tid] = v; + barrier(); + [[unroll]] for (uint s = BLOCK_SIZE / 2; s > 0; s >>= 1) { + if (tid < s) { + tmp[tid] += tmp[tid + s]; + } + barrier(); + } + v = tmp[0]; + barrier(); + return v; +} + +void main() { + const uint row = gl_WorkGroupID.z * 262144 + gl_WorkGroupID.y * 512 + gl_WorkGroupID.x; + const uint tid = gl_LocalInvocationID.x; + + if (row >= p.KY) { + return; + } + + const uint off = row * p.KX; + + FLOAT_TYPE max_logit = FLOAT_TYPE(uintBitsToFloat(0xFF800000)); + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + max_logit = max(max_logit, FLOAT_TYPE(data_a[off + i])); + } + max_logit = wg_reduce_max(max_logit); + + FLOAT_TYPE sum_exp = FLOAT_TYPE(0.0f); + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + sum_exp += exp(FLOAT_TYPE(data_a[off + i]) - max_logit); + } + const FLOAT_TYPE log_sum = log(wg_reduce_sum(sum_exp)); + + FLOAT_TYPE loss = FLOAT_TYPE(0.0f); + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + loss += (FLOAT_TYPE(data_a[off + i]) - max_logit - log_sum) * FLOAT_TYPE(data_b[off + i]); + } + loss = -wg_reduce_sum(loss) / FLOAT_TYPE(p.KY); + + if (tid == 0) { + data_d[row] = D_TYPE(loss); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss_back.comp b/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss_back.comp new file mode 100644 index 00000000..3cdebe86 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/cross_entropy_loss_back.comp @@ -0,0 +1,75 @@ +#version 450 + +#include "generic_head.glsl" +#include "types.glsl" + +#extension GL_EXT_control_flow_attributes : enable + +layout(constant_id = 0) const uint BLOCK_SIZE = 32; +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer G {A_TYPE data_g[];}; +layout (binding = 1) readonly buffer X {B_TYPE data_x[];}; +layout (binding = 2) readonly buffer Y {B_TYPE data_y[];}; +layout (binding = 3) writeonly buffer D {D_TYPE data_d[];}; + +shared FLOAT_TYPE tmp[BLOCK_SIZE]; + +FLOAT_TYPE wg_reduce_max(FLOAT_TYPE v) { + const uint tid = gl_LocalInvocationID.x; + tmp[tid] = v; + barrier(); + [[unroll]] for (uint s = BLOCK_SIZE / 2; s > 0; s >>= 1) { + if (tid < s) { + tmp[tid] = max(tmp[tid], tmp[tid + s]); + } + barrier(); + } + v = tmp[0]; + barrier(); + return v; +} + +FLOAT_TYPE wg_reduce_sum(FLOAT_TYPE v) { + const uint tid = gl_LocalInvocationID.x; + tmp[tid] = v; + barrier(); + [[unroll]] for (uint s = BLOCK_SIZE / 2; s > 0; s >>= 1) { + if (tid < s) { + tmp[tid] += tmp[tid + s]; + } + barrier(); + } + v = tmp[0]; + barrier(); + return v; +} + +void main() { + const uint row = gl_WorkGroupID.z * 262144 + gl_WorkGroupID.y * 512 + gl_WorkGroupID.x; + const uint tid = gl_LocalInvocationID.x; + + if (row >= p.KY) { + return; + } + + const uint off = row * p.KX; + const FLOAT_TYPE d_by_nrows = FLOAT_TYPE(data_g[0]) / FLOAT_TYPE(p.KY); + + FLOAT_TYPE max_logit = FLOAT_TYPE(uintBitsToFloat(0xFF800000)); + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + max_logit = max(max_logit, FLOAT_TYPE(data_x[off + i])); + } + max_logit = wg_reduce_max(max_logit); + + FLOAT_TYPE sum_exp = FLOAT_TYPE(0.0f); + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + sum_exp += exp(FLOAT_TYPE(data_x[off + i]) - max_logit); + } + const FLOAT_TYPE inv_sum = FLOAT_TYPE(1.0f) / wg_reduce_sum(sum_exp); + + for (uint i = tid; i < p.KX; i += BLOCK_SIZE) { + const FLOAT_TYPE sm = exp(FLOAT_TYPE(data_x[off + i]) - max_logit) * inv_sum; + data_d[off + i] = D_TYPE((sm - FLOAT_TYPE(data_y[off + i])) * d_by_nrows); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl index 627932bd..9df66cb4 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl @@ -608,6 +608,21 @@ vec2 get_dm(uint ib, uint a_offset) { } #endif +#if defined(DATA_A_TQ1_0) +float tq1_0_val(uint ib, uint e, uint a_offset) { + const uint bidx = tq1_0_byte_of(e); + const uint qbyte = uint(bidx < 48u ? data_a[a_offset + ib].qs[bidx] + : data_a[a_offset + ib].qh[bidx - 48u]); + return float(tq1_0_trit(qbyte, tq1_0_digit_of(e))) - 1.0; +} +vec2 dequantize(uint ib, uint iqs, uint a_offset) { + return vec2(tq1_0_val(ib, iqs, a_offset), tq1_0_val(ib, iqs + 1u, a_offset)); +} +vec2 get_dm(uint ib, uint a_offset) { + return vec2(float(data_a[a_offset + ib].d), 0); +} +#endif + #if defined(DATA_A_TQ2_0) vec2 dequantize(uint ib, uint iqs, uint a_offset) { // elem e -> byte qs[(e/128)*32 + e%32], bits 2*((e%128)/32); w = q - 1 (d applied via get_dm) diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl index 46cc69cb..cc6e242a 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl @@ -247,6 +247,33 @@ f16vec4 dequantFuncQ8_0_v(const in decodeBufQ8_0 bl, const in uint blockCoords[2 return f16vec4(vec4(qi) * vec4(float(d))); } +layout(buffer_reference, std430, buffer_reference_align = 2) buffer decodeBufTQ1_0 { + block_tq1_0 block; +}; + +float16_t dequantFuncTQ1_0(const in decodeBufTQ1_0 bl, const in uint blockCoords[2], const in uint coordInBlock[2]) +{ + const uint e = coordInBlock[1]; + const uint bidx = tq1_0_byte_of(e); + const uint qbyte = uint(bidx < 48u ? bl.block.qs[bidx] : bl.block.qh[bidx - 48u]); + const uint xi = tq1_0_trit(qbyte, tq1_0_digit_of(e)); + return bl.block.d * (float16_t(int(xi)) - float16_t(1.0)); +} + +f16vec4 dequantFuncTQ1_0_v(const in decodeBufTQ1_0 bl, const in uint blockCoords[2], const in uint coordInBlock[2]) +{ + const uint e = coordInBlock[1]; + f16vec4 v; + [[unroll]] for (uint k = 0u; k < 4u; ++k) { + const uint ee = e + k; + const uint bidx = tq1_0_byte_of(ee); + const uint qbyte = uint(bidx < 48u ? bl.block.qs[bidx] : bl.block.qh[bidx - 48u]); + const uint xi = tq1_0_trit(qbyte, tq1_0_digit_of(ee)); + v[k] = bl.block.d * (float16_t(int(xi)) - float16_t(1.0)); + } + return v; +} + layout(buffer_reference, std430, buffer_reference_align = 2) buffer decodeBufTQ2_0 { block_tq2_0 block; }; @@ -1041,7 +1068,7 @@ float16_t dequantFuncIQ2_S(const in decodeBufIQ2_S bl, const in uint blockCoords const uint scale = (bl.block.scales[ib32] >> ((idx & 0x10) >> 2)) & 0xf; const uint qs = bl.block.qs[ib8]; const uint qh = bl.block.qh[ib32]; - const uint sign = bl.block.qs[QUANT_K / 8 + ib8] >> (idx & 0x6); + const uint sign = bl.block.qs[QUANT_K_IQ2_S / 8 + ib8] >> (idx & 0x6); const float d = float(bl.block.d); const float db = d * 0.25 * (0.5 + scale); @@ -1063,7 +1090,7 @@ f16vec4 dequantFuncIQ2_S_v(const in decodeBufIQ2_S bl, const in uint blockCoords const uint scale = (bl.block.scales[ib32] >> ((idx & 0x10) >> 2)) & 0xf; const uint qs = bl.block.qs[ib8]; const uint qh = bl.block.qh[ib32]; - const uint sb = uint(bl.block.qs[QUANT_K / 8 + ib8]) >> (idx & 0x6u); + const uint sb = uint(bl.block.qs[QUANT_K_IQ2_S / 8 + ib8]) >> (idx & 0x6u); const float d = float(bl.block.d); const float db = d * 0.25 * (0.5 + scale); @@ -1094,7 +1121,7 @@ float16_t dequantFuncIQ3_XXS(const in decodeBufIQ3_XXS bl, const in uint blockCo uint idx = coordInBlock[1]; const uint iqs = (idx & 0xFC) >> 2; // 0..63 - const uint is = QUANT_K / 4 + ((idx & 0xE0) >> 3);// 8 values + const uint is = QUANT_K_IQ3_XXS / 4 + ((idx & 0xE0) >> 3);// 8 values const float d = float(bl.block.d); const uint qs = bl.block.qs[iqs]; @@ -1117,7 +1144,7 @@ f16vec4 dequantFuncIQ3_XXS_v(const in decodeBufIQ3_XXS bl, const in uint blockCo const uint idx = coordInBlock[1]; const uint iqs = idx >> 2; - const uint is = QUANT_K / 4 + ((idx & 0xE0) >> 3); + const uint is = QUANT_K_IQ3_XXS / 4 + ((idx & 0xE0) >> 3); const float d = float(bl.block.d); const uint qs = bl.block.qs[iqs]; @@ -1406,6 +1433,8 @@ f16vec4 dequantFuncNVFP4_v(const in decodeBufNVFP4 bl, const in uint blockCoords #elif defined(DATA_A_Q8_0) #define dequantFuncA dequantFuncQ8_0 #define dequantFuncA_v dequantFuncQ8_0_v +#elif defined(DATA_A_TQ1_0) +#define dequantFuncA dequantFuncTQ1_0 #elif defined(DATA_A_TQ2_0) #define dequantFuncA dequantFuncTQ2_0 #define dequantFuncA_v dequantFuncTQ2_0_v diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q8_0.comp b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q8_0.comp index 10844ddf..3b3fbbe8 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q8_0.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q8_0.comp @@ -18,7 +18,18 @@ void main() { return; } +#ifdef DEQUANT_TRANSPOSE + // read [HS, NH, KV, NS], write [HS, KV, NH, NS] + const uint HS = p.M, NH = p.K, KVn = p.stride_a; + const uint e0 = ib * 32; + const uint b_idx = (e0 % HS) + + ((e0 / (HS * NH)) % KVn) * HS + + ((e0 / HS) % NH) * (HS * KVn) + + (e0 / (HS * NH * KVn)) * (HS * KVn * NH) + + 16 * il; +#else const uint b_idx = 1024*i + 32*ir + 16*il; +#endif const float d = float(data_a[ib].d); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dequant_tq1_0.comp b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_tq1_0.comp new file mode 100644 index 00000000..1632e746 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dequant_tq1_0.comp @@ -0,0 +1,28 @@ +#version 450 + +#include "dequant_head.glsl" + +layout (local_size_x = 256, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer A {block_tq1_0 data_a[];}; +layout (binding = 1) writeonly buffer D {D_TYPE data_b[];}; + +void main() { + const uint i = gl_GlobalInvocationID.x * 4; + + if (i >= p.nel) { + return; + } + + const uint ib = i / QUANT_K_TQ1_0; + const float d = float(data_a[ib].d); + + [[unroll]] for (uint j = 0; j < 4 && (i + j) < p.nel; ++j) { + const uint e = (i + j) % QUANT_K_TQ1_0; + const uint bidx = tq1_0_byte_of(e); + const uint qbyte = uint(bidx < 48u ? data_a[ib].qs[bidx] + : data_a[ib].qh[bidx - 48u]); + const uint xi = tq1_0_trit(qbyte, tq1_0_digit_of(e)); + data_b[i + j] = D_TYPE(d * (float(xi) - 1.0f)); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_comb.comp b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_comb.comp new file mode 100644 index 00000000..f4ac0378 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_comb.comp @@ -0,0 +1,90 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : require +#extension GL_KHR_shader_subgroup_basic : require +#extension GL_KHR_shader_subgroup_shuffle : require + +// 16 lanes per token, indexed idst + hc*isrc: idst in bits 0..1, isrc in bits 2..3, +// so subgroupShuffleXor by 1|2 reduces a row and by 4|8 a column. + +layout(constant_id = 0) const uint SUBGROUP_SIZE = 32; + +layout(local_size_x_id = 0, local_size_y = 4, local_size_z = 1) in; + +layout(push_constant) uniform parameter +{ + uint n_tokens; + + uint nbm0; uint nbm1; // mixes + uint nbs0; // scale + uint nbb0; // base + uint nbd0; uint nbd1; uint nbd2; // dst + + uint m_offset; + uint s_offset; + uint b_offset; + uint d_offset; + + float eps; + uint n_iter; +}; + +layout(binding = 0, std430) readonly buffer M { float data_m[]; }; +layout(binding = 1, std430) readonly buffer S { float data_s[]; }; +layout(binding = 2, std430) readonly buffer B { float data_b[]; }; +layout(binding = 3, std430) writeonly buffer D { float data_d[]; }; + +const uint hc = 4; +const uint comb_offset = 2 * hc; + +const uint TOKENS_PER_SUBGROUP = SUBGROUP_SIZE / 16; + +void main() { + const uint lane = gl_SubgroupInvocationID; + const uint blk = lane >> 4; // which 16-lane block, i.e. which token + const uint idx = lane & 15; // idst + hc*isrc + + const uint sg = gl_WorkGroupID.x * gl_WorkGroupSize.y + gl_SubgroupID; + const uint it = sg * TOKENS_PER_SUBGROUP + blk; + + // no early return, the shuffles need every lane; out-of-range blocks compute a discarded value + const bool in_range = it < n_tokens; + + const float scale_comb = data_s[s_offset + 2 * nbs0]; + + float v = 0.0f; + if (in_range) { + v = data_m[m_offset + (comb_offset + idx) * nbm0 + it * nbm1] * scale_comb + + data_b[b_offset + (comb_offset + idx) * nbb0]; + } + + // Softmax across destinations: the four lanes sharing an isrc. + float vmax = max(v, subgroupShuffleXor(v, 1)); + vmax = max(vmax, subgroupShuffleXor(vmax, 2)); + v = exp(v - vmax); + + float sum = v + subgroupShuffleXor(v, 1); + sum += subgroupShuffleXor(sum, 2); + v = v / sum + eps; + + // Normalize columns: equal destination indices are four lanes apart. + sum = v + subgroupShuffleXor(v, 4); + sum += subgroupShuffleXor(sum, 8); + v /= sum + eps; + + for (uint i = 1; i < n_iter; ++i) { + sum = v + subgroupShuffleXor(v, 1); + sum += subgroupShuffleXor(sum, 2); + v /= sum + eps; + + sum = v + subgroupShuffleXor(v, 4); + sum += subgroupShuffleXor(sum, 8); + v /= sum + eps; + } + + if (in_range) { + const uint idst = idx & 3; + const uint isrc = idx >> 2; + data_d[d_offset + idst * nbd0 + isrc * nbd1 + it * nbd2] = v; + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_post.comp b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_post.comp new file mode 100644 index 00000000..e521fd9d --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_post.comp @@ -0,0 +1,92 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : require + +// Fan one stream back out to hc streams and add the combination-weighted +// residuals: +// +// dst[i0, idst, it] = x[i0, it]*post[idst, it] +// + sum_isrc residual[i0, isrc, it]*comb[idst, isrc, it] +// +// HAS_COMB == 0: identity mixing, each stream keeps its own residual: +// +// dst[i0, idst, it] = x[i0, it]*post[idst, it] + residual[i0, idst, it] + +layout(constant_id = 0) const uint BLOCK_SIZE = 256; +layout(constant_id = 1) const uint HAS_COMB = 1; + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout(push_constant) uniform parameter +{ + uint n_embd; + uint n_tokens; + + uint nbx0; uint nbx1; // x + uint nbr0; uint nbr1; uint nbr2; // residual + uint nbp0; uint nbp1; // post + uint nbc0; uint nbc1; uint nbc2; // comb + uint nbd0; uint nbd1; uint nbd2; // dst + + uint x_offset; + uint r_offset; + uint p_offset; + uint c_offset; + uint d_offset; +}; + +layout(binding = 0, std430) readonly buffer X { float data_x[]; }; +layout(binding = 1, std430) readonly buffer R { float data_r[]; }; +layout(binding = 2, std430) readonly buffer P { float data_p[]; }; +layout(binding = 3, std430) readonly buffer C { float data_c[]; }; +layout(binding = 4, std430) writeonly buffer D { float data_d[]; }; + +const uint hc = 4; + +shared float post_s[hc]; +shared float comb_s[hc * hc]; + +void main() { + const uint tid = gl_LocalInvocationID.x; + const uint it = gl_WorkGroupID.y; + + if (tid < hc) { + post_s[tid] = data_p[p_offset + tid * nbp0 + it * nbp1]; + } + if (HAS_COMB == 1 && tid < hc * hc) { + const uint idst = tid & 3; + const uint isrc = tid >> 2; + comb_s[tid] = data_c[c_offset + idst * nbc0 + isrc * nbc1 + it * nbc2]; + } + barrier(); + + // After the barrier, so every invocation reaches it. + const uint i0 = gl_WorkGroupID.x * BLOCK_SIZE + tid; + if (i0 >= n_embd) { + return; + } + + const float xv = data_x[x_offset + i0 * nbx0 + it * nbx1]; + + const uint rb = r_offset + i0 * nbr0 + it * nbr2; + + float r[hc]; + [[unroll]] + for (uint isrc = 0; isrc < hc; ++isrc) { + r[isrc] = data_r[rb + isrc * nbr1]; + } + + [[unroll]] + for (uint idst = 0; idst < hc; ++idst) { + float result = xv * post_s[idst]; + if (HAS_COMB == 1) { + [[unroll]] + for (uint isrc = 0; isrc < hc; ++isrc) { + result = fma(r[isrc], comb_s[idst + hc * isrc], result); + } + } else { + result += r[idst]; + } + data_d[d_offset + i0 * nbd0 + idst * nbd1 + it * nbd2] = result; + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_pre.comp b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_pre.comp new file mode 100644 index 00000000..fa301547 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/dsv4_hc_pre.comp @@ -0,0 +1,75 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : require + +// Collapse the hc residual streams of a token into one, weighted per stream: +// +// dst[i0, it] = scale * sum_ih x[i0, ih, it] * weights[ih, it] +// +// GATED: weights is a per-element gate [n_embd, hc, n_tokens], applied as sigmoid: +// +// dst[i0, it] = scale * sum_ih x[i0, ih, it] * sigmoid(gate[i0, ih, it]) + +layout(constant_id = 0) const uint BLOCK_SIZE = 256; +layout(constant_id = 1) const uint GATED = 0; + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout(push_constant) uniform parameter +{ + uint n_embd; + uint n_tokens; + + uint nbx0; uint nbx1; uint nbx2; // x + uint nbw0; uint nbw1; uint nbw2; // weights / gate + uint nbd0; uint nbd1; // dst + + uint x_offset; + uint w_offset; + uint d_offset; + + float scale; +}; + +layout(binding = 0, std430) readonly buffer X { float data_x[]; }; +layout(binding = 1, std430) readonly buffer W { float data_w[]; }; +layout(binding = 2, std430) writeonly buffer D { float data_d[]; }; + +const uint hc = 4; + +shared float w[hc]; + +void main() { + const uint tid = gl_LocalInvocationID.x; + const uint it = gl_WorkGroupID.y; + + if (GATED == 0) { + if (tid < hc) { + w[tid] = data_w[w_offset + tid * nbw0 + it * nbw1]; + } + barrier(); + } + + // After the barrier, so every invocation reaches it. + const uint i0 = gl_WorkGroupID.x * BLOCK_SIZE + tid; + if (i0 >= n_embd) { + return; + } + + const uint xb = x_offset + i0 * nbx0 + it * nbx2; + const uint wb = w_offset + i0 * nbw0 + it * nbw2; + + float result = 0.0f; + [[unroll]] + for (uint ih = 0; ih < hc; ++ih) { + float wv; + if (GATED == 1) { + wv = 1.0f / (1.0f + exp(-data_w[wb + ih * nbw1])); + } else { + wv = w[ih]; + } + result = fma(data_x[xb + ih * nbx1], wv, result); + } + + data_d[d_offset + i0 * nbd0 + it * nbd1] = scale * result; +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/fa_types.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/fa_types.glsl new file mode 100644 index 00000000..1e732a9a --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/fa_types.glsl @@ -0,0 +1,45 @@ +#if !defined(GGML_FA_TYPES_COMP) +#define GGML_FA_TYPES_COMP + +#include "ggml_type_ids.glsl" + +// Number of matrix elements per buffer block, derived from the K/V type spec +// constant. F32 is treated as a vec4 "block" of 4 floats. F16 uses block size 1 +// and bypasses the dequant path entirely. Quants follow their ggml block sizes. +uint fa_block_elems(uint ty) { + switch (ty) { + case GGML_TYPE_F32: return 4u; + case GGML_TYPE_F16: return 1u; + case GGML_TYPE_Q4_0: return uint(QUANT_K_Q4_0); + case GGML_TYPE_Q4_1: return uint(QUANT_K_Q4_1); + case GGML_TYPE_Q5_0: return uint(QUANT_K_Q5_0); + case GGML_TYPE_Q5_1: return uint(QUANT_K_Q5_1); + case GGML_TYPE_Q8_0: return uint(QUANT_K_Q8_0); + case GGML_TYPE_IQ4_NL: return uint(QUANT_K_IQ4_NL); + case GGML_TYPE_BF16: return 1u; + default: return 1u; + } +} + +// QUANT_R_MMQ for FA-eligible K types. Q4_*/Q5_* store two nibbles per byte +// (R==2); Q8_0 stores one byte per element (R==1). Used to derive the number +// of int32s per 32-element block on the MMQ K path: ints_per_block == 8 / R. +uint fa_quant_r_mmq(uint ty) { + switch (ty) { + case GGML_TYPE_Q4_0: return uint(QUANT_R_Q4_0); + case GGML_TYPE_Q4_1: return uint(QUANT_R_Q4_1); + case GGML_TYPE_Q5_0: return uint(QUANT_R_Q5_0); + case GGML_TYPE_Q5_1: return uint(QUANT_R_Q5_1); + case GGML_TYPE_Q8_0: return uint(QUANT_R_Q8_0); + default: return 1u; + } +} + +bool fa_type_needs_shmem(uint ty) { + switch (ty) { + case GGML_TYPE_IQ4_NL: return true; + default: return false; + } +} + +#endif // !defined(GGML_FA_TYPES_COMP) diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/fill.comp b/ggml/src/ggml-vulkan/vulkan-shaders/fill.comp index a56be76c..b5cc3332 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/fill.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/fill.comp @@ -8,7 +8,9 @@ layout(local_size_x = 512, local_size_y = 1, local_size_z = 1) in; layout (binding = 0) writeonly buffer D {D_TYPE data_d[];}; void main() { - const uint i = gl_GlobalInvocationID.x; + // 2D grid flattening: each x workgroup covers gl_WorkGroupSize.x elements, + // each y workgroup covers gl_NumWorkGroups.x * gl_WorkGroupSize.x elements. + const uint i = (gl_GlobalInvocationID.y * gl_NumWorkGroups.x * gl_WorkGroupSize.x) + gl_GlobalInvocationID.x; if (i >= p.KX) { return; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp index 6c264c78..107d44aa 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp @@ -121,26 +121,26 @@ void main() { const uint buf_ib = r * qf_stride + d / 8; const uint buf_iqs = d % 8; - FLOAT_TYPEV4 vals = is_in_bounds ? FLOAT_TYPEV4(data_qv4[q_offset / 4 + (i * Br + r) * q_stride / 4 + d] * p.scale) : FLOAT_TYPEV4(0.0f); - const FLOAT_TYPEV4 abs_vals = abs(vals); + vec4 vals = is_in_bounds ? data_qv4[q_offset / 4 + (i * Br + r) * q_stride / 4 + d] * p.scale : vec4(0.0f); + const vec4 abs_vals = abs(vals); - const FLOAT_TYPE thread_max = max(max(abs_vals.x, abs_vals.y), max(abs_vals.z, abs_vals.w)); - const FLOAT_TYPE amax = subgroupClusteredMax(thread_max, 8); - const FLOAT_TYPE qd = amax / FLOAT_TYPE(127.0); - const FLOAT_TYPE qd_inv = qd != FLOAT_TYPE(0.0) ? FLOAT_TYPE(1.0) / qd : FLOAT_TYPE(0.0); + const float thread_max = max(max(abs_vals.x, abs_vals.y), max(abs_vals.z, abs_vals.w)); + const float amax = subgroupClusteredMax(thread_max, 8); + const float qd = amax / 127.0f; + const float qd_inv = qd != 0.0f ? 1.0f / qd : 0.0f; vals = round(vals * qd_inv); Qf[buf_ib].qs[buf_iqs] = pack32(i8vec4(vals)); // Q8_0 K only needs (qd, _); the asymmetric Q4_*/Q5_* family also stores // the row-sum scaled by qd, used in k_dot_correction. - if (FaTypeK == FA_TYPE_Q8_0) { + if (FaTypeK == GGML_TYPE_Q8_0) { if (buf_iqs == 0) { - Qf[buf_ib].ds = FLOAT_TYPEV2(qd, 0.0); + Qf[buf_ib].ds = FLOAT_TYPEV2(qd, 0.0f); } } else { - const FLOAT_TYPE thread_sum = vals.x + vals.y + vals.z + vals.w; - const FLOAT_TYPE sum = subgroupClusteredAdd(thread_sum, 8); + const float thread_sum = vals.x + vals.y + vals.z + vals.w; + const float sum = subgroupClusteredAdd(thread_sum, 8); if (buf_iqs == 0) { Qf[buf_ib].ds = FLOAT_TYPEV2(qd, sum * qd); @@ -218,12 +218,14 @@ void main() { uint32_t c = (idx + tid) % Bc; uint32_t r = (idx + tid) / Bc; if (idx + tid < Bc * Br) { - if ((!KV_bounds_check || j * Bc + c < KV) && (!nem1_bounds_check || i * Br + r < p.nem1)) { - FLOAT_TYPE m = FLOAT_TYPE(data_m[m_offset + (i * Br + r) * m_stride + (j * Bc + c)]); + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + c, kcol); + if (kv_active && (!nem1_bounds_check || i * Br + r < p.nem1)) { + FLOAT_TYPE m = FLOAT_TYPE(data_m[m_offset + (i * Br + r) * m_stride + kcol]); masksh[c * masksh_stride + r] = m; max_mask = max(max_mask, float(m)); } else { - masksh[c * masksh_stride + r] = FLOAT_TYPE(0); + masksh[c * masksh_stride + r] = USE_SPARSE ? FLOAT_TYPE(NEG_FLT_MAX_OVER_2) : FLOAT_TYPE(0); } } } @@ -258,14 +260,15 @@ void main() { uint32_t c = (idx + tid) / (HSK / 4); if (idx + gl_WorkGroupSize.x <= Bc * HSK / 4 || c < Bc) { FLOAT_TYPEV4 K_Tf = FLOAT_TYPEV4(0); - if (!KV_bounds_check || j * Bc + c < KV) { + uint32_t kcol; + if (fa_kv_index(j * Bc + c, kcol)) { if (USE_DECODE_K) { - uint coord = (j * Bc + c) * k_stride * BLOCK_SIZE_K + 4 * d; + uint coord = kcol * k_stride * BLOCK_SIZE_K + 4 * d; uint ib = coord / BLOCK_SIZE_K; uint iqs = (coord % BLOCK_SIZE_K); K_Tf = dequantize4(ib, iqs, k_offset, BINDING_IDX_K); } else { - K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + (j * Bc + c) * k_stride / 4 + d]); + K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + kcol * k_stride / 4 + d]); } } @@ -305,7 +308,9 @@ void main() { } [[unroll]] for (uint32_t c = 0; c < cols_per_thread; ++c) { - if (KV_bounds_check && j * Bc + c * cols_per_iter + col_tid >= KV) { + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + c * cols_per_iter + col_tid, kcol); + if (!kv_active) { continue; } @@ -313,12 +318,12 @@ void main() { if (SHMEM_STAGING != 0) { K_Tf = kvsh[(c * cols_per_iter + col_tid) * kvsh_stride + (d * D_split + d_tid)]; } else if (USE_DECODE_K) { - uint coord = (j * Bc + c * cols_per_iter + col_tid) * k_stride * BLOCK_SIZE_K + 4 * (d * D_split + d_tid); + uint coord = kcol * k_stride * BLOCK_SIZE_K + 4 * (d * D_split + d_tid); uint ib = coord / BLOCK_SIZE_K; uint iqs = (coord % BLOCK_SIZE_K); K_Tf = dequantize4(ib, iqs, k_offset, BINDING_IDX_K); } else { - K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + (j * Bc + c * cols_per_iter + col_tid) * k_stride / 4 + d * D_split + d_tid]); + K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + kcol * k_stride / 4 + d * D_split + d_tid]); } [[unroll]] for (uint32_t r = 0; r < rows_per_thread; ++r) { Sf[r][c] = dot_product(Q_cache[r], K_Tf, Sf[r][c]); @@ -327,7 +332,9 @@ void main() { } } else { [[unroll]] for (uint32_t c = 0; c < cols_per_thread; ++c) { - if (KV_bounds_check && j * Bc + c * cols_per_iter + col_tid >= KV) { + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + c * cols_per_iter + col_tid, kcol); + if (!kv_active) { continue; } @@ -336,12 +343,12 @@ void main() { if (SHMEM_STAGING != 0) { K_Tf = kvsh[(c * cols_per_iter + col_tid) * kvsh_stride + (d * D_split + d_tid)]; } else if (USE_DECODE_K) { - uint coord = (j * Bc + c * cols_per_iter + col_tid) * k_stride * BLOCK_SIZE_K + 4 * (d * D_split + d_tid); + uint coord = kcol * k_stride * BLOCK_SIZE_K + 4 * (d * D_split + d_tid); uint ib = coord / BLOCK_SIZE_K; uint iqs = (coord % BLOCK_SIZE_K); K_Tf = dequantize4(ib, iqs, k_offset, BINDING_IDX_K); } else { - K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + (j * Bc + c * cols_per_iter + col_tid) * k_stride / 4 + d * D_split + d_tid]); + K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + kcol * k_stride / 4 + d * D_split + d_tid]); } [[unroll]] for (uint32_t r = 0; r < rows_per_thread; ++r) { Sf[r][c] = dot_product(Qf[tile_row(r) * qf_stride + d * D_split + d_tid], K_Tf, Sf[r][c]); @@ -367,7 +374,7 @@ void main() { // Q4_*/Q5_* take the block-8 fast path when one step covers a full // block; Q8_0 always goes through the per-int get_k_qs* helpers // (its qs is byte-packed, not nibble-packed). - const bool block8_fast = (d_per_step == 8) && (FaTypeK != FA_TYPE_Q8_0); + const bool block8_fast = (d_per_step == 8) && (FaTypeK != GGML_TYPE_Q8_0); if (SHMEM_STAGING != 0) { const uint k_block_idx = (d_tid * (HSK_per_thread / 4) + d_block) / 8; @@ -375,7 +382,7 @@ void main() { k_dm = ACC_TYPEV2(kblocksh[buf_ib].dm); if (block8_fast) { - const bool has_qh = (FaTypeK == FA_TYPE_Q5_0) || (FaTypeK == FA_TYPE_Q5_1); + const bool has_qh = (FaTypeK == GGML_TYPE_Q5_0) || (FaTypeK == GGML_TYPE_Q5_1); [[unroll]] for (uint32_t d = 0; d < 4; d++) { uint vui = kblocksh[buf_ib].qs[d]; k_quants[d ] = int32_t( vui & 0x0F0F0F0F); @@ -489,14 +496,15 @@ void main() { uint32_t c = (idx + tid) / (HSV / 4); if (idx + gl_WorkGroupSize.x <= Bc * HSV / 4 || c < Bc) { FLOAT_TYPEV4 V_Tf = FLOAT_TYPEV4(0); - if (!KV_bounds_check || j * Bc + c < KV) { + uint32_t vcol; + if (fa_kv_index(j * Bc + c, vcol)) { if (USE_DECODE_V) { - uint coord = (j * Bc + c) * v_stride * BLOCK_SIZE_V + 4 * d; + uint coord = vcol * v_stride * BLOCK_SIZE_V + 4 * d; uint ib = coord / BLOCK_SIZE_V; uint iqs = (coord % BLOCK_SIZE_V); V_Tf = dequantize4(ib, iqs, v_offset, BINDING_IDX_V); } else { - V_Tf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + (j * Bc + c) * v_stride / 4 + d]); + V_Tf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + vcol * v_stride / 4 + d]); } } @@ -507,7 +515,9 @@ void main() { } [[unroll]] for (uint32_t c = 0; c < cols_per_thread; ++c) { - if (KV_bounds_check && j * Bc + c * cols_per_iter + col_tid >= KV) { + uint32_t vcol; + bool kv_active = fa_kv_index(j * Bc + c * cols_per_iter + col_tid, vcol); + if (!kv_active) { continue; } @@ -522,12 +532,12 @@ void main() { if (SHMEM_STAGING != 0) { Vf = kvsh[(c * cols_per_iter + col_tid) * kvsh_stride + (d * D_split + d_tid)]; } else if (USE_DECODE_V) { - uint coord = (j * Bc + c * cols_per_iter + col_tid) * v_stride * BLOCK_SIZE_V + 4 * (d * D_split + d_tid); + uint coord = vcol * v_stride * BLOCK_SIZE_V + 4 * (d * D_split + d_tid); uint ib = coord / BLOCK_SIZE_V; uint iqs = (coord % BLOCK_SIZE_V); Vf = dequantize4(ib, iqs, v_offset, BINDING_IDX_V); } else { - Vf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + (j * Bc + c * cols_per_iter + col_tid) * v_stride / 4 + d * D_split + d_tid]); + Vf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + vcol * v_stride / 4 + d * D_split + d_tid]); } [[unroll]] for (uint32_t r = 0; r < rows_per_thread; ++r) { Of[r][d] += FLOAT_TYPEV4(Pf[r] * Vf); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl index 3c64f91d..2e0e23bc 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl @@ -24,6 +24,8 @@ const bool USE_MASK_OPT = (Flags & 1) != 0; const bool MASK_ENABLE = (Flags & 2) != 0; const bool LOGIT_SOFTCAP = (Flags & 4) != 0; const bool OLD_AMD_WINDOWS = (Flags & 8) != 0; +// Sparse: gather binding-7 indices instead of scanning [0,KV); p.split_kv = n_kv_max. +const bool USE_SPARSE = (Flags & 16) != 0; // Round up head sizes to a multiple of 16, for coopmat1/coopmat2 paths const uint32_t HSK_pad = (HSK + 15) & ~15; @@ -82,23 +84,15 @@ layout (binding = 5) writeonly buffer OV4 {D_TYPEV4 data_ov4[];}; layout (binding = 6) readonly buffer MO {uint32_t data_mask_opt[];}; +layout (binding = 7) readonly buffer SP {int32_t data_sparse[];}; + #define MASK_OPT_ALL_NEG_INF 1 #define MASK_OPT_ALL_ZERO 2 #define BINDING_IDX_K 0 #define BINDING_IDX_V 1 -// FaTypeK / FaTypeV spec constant values. These mirror enum ggml_type so the -// host can pass the type directly. Keep in sync with ggml.h. -#define FA_TYPE_F32 0u -#define FA_TYPE_F16 1u -#define FA_TYPE_Q4_0 2u -#define FA_TYPE_Q4_1 3u -#define FA_TYPE_Q5_0 6u -#define FA_TYPE_Q5_1 7u -#define FA_TYPE_Q8_0 8u -#define FA_TYPE_IQ4_NL 20u -#define FA_TYPE_BF16 30u +#include "fa_types.glsl" #if defined(BFLOAT16) #define O_TYPE float @@ -108,45 +102,6 @@ layout (binding = 6) readonly buffer MO {uint32_t data_mask_opt[];}; #define O_TYPEV4 FLOAT_TYPEV4 #endif -// Number of matrix elements per buffer block, derived from the K/V type spec -// constant. F32 is treated as a vec4 "block" of 4 floats. F16 uses block size 1 -// and bypasses the dequant path entirely. Quants follow their ggml block sizes. -uint fa_block_elems(uint ty) { - switch (ty) { - case FA_TYPE_F32: return 4u; - case FA_TYPE_F16: return 1u; - case FA_TYPE_Q4_0: return uint(QUANT_K_Q4_0); - case FA_TYPE_Q4_1: return uint(QUANT_K_Q4_1); - case FA_TYPE_Q5_0: return uint(QUANT_K_Q5_0); - case FA_TYPE_Q5_1: return uint(QUANT_K_Q5_1); - case FA_TYPE_Q8_0: return uint(QUANT_K_Q8_0); - case FA_TYPE_IQ4_NL: return uint(QUANT_K_IQ4_NL); - case FA_TYPE_BF16: return 1u; - default: return 1u; - } -} - -// QUANT_R_MMQ for FA-eligible K types. Q4_*/Q5_* store two nibbles per byte -// (R==2); Q8_0 stores one byte per element (R==1). Used to derive the number -// of int32s per 32-element block on the MMQ K path: ints_per_block == 8 / R. -uint fa_quant_r_mmq(uint ty) { - switch (ty) { - case FA_TYPE_Q4_0: return uint(QUANT_R_Q4_0); - case FA_TYPE_Q4_1: return uint(QUANT_R_Q4_1); - case FA_TYPE_Q5_0: return uint(QUANT_R_Q5_0); - case FA_TYPE_Q5_1: return uint(QUANT_R_Q5_1); - case FA_TYPE_Q8_0: return uint(QUANT_R_Q8_0); - default: return 1u; - } -} - -bool fa_type_needs_shmem(uint ty) { - switch (ty) { - case FA_TYPE_IQ4_NL: return true; - default: return false; - } -} - // These can't be `const` globals because GLSL forbids function calls in global // const initializers, even when the spec constants would let the driver fold // them. Macros expand at the use site and fold after specialization. @@ -154,8 +109,8 @@ bool fa_type_needs_shmem(uint ty) { #define BLOCK_SIZE_V fa_block_elems(FaTypeV) // F16 reads f16 elements directly from the binding; everything else routes // through dequantize4 / the MMQ helpers to unpack from the packed block layout. -#define USE_DECODE_K (FaTypeK != FA_TYPE_F16) -#define USE_DECODE_V (FaTypeV != FA_TYPE_F16) +#define USE_DECODE_K (FaTypeK != GGML_TYPE_F16) +#define USE_DECODE_V (FaTypeV != GGML_TYPE_F16) #define CEIL_DIV(a, b) (((a) + (b) - 1) / (b)) @@ -193,7 +148,7 @@ ACC_TYPE perElemOpGetSink(const in uint32_t r, const in uint32_t c, const in ACC uint32_t i, N, KV, split_k_index, Tr, start_j, end_j, gqa_iq1, iq2, iq3, rk2, rk3, rv2, rv3, ik2, ik3, iv2, iv3, - q_stride, k_stride, v_stride, m_stride; + q_stride, k_stride, v_stride, m_stride, sparse_base; void init_indices() { @@ -257,6 +212,33 @@ void init_indices() // that prevents the compiler from folding the "&" through the select // and breaking the alignment detection. m_stride = (p.gqa_ratio > 1) ? (p.gqa_ratio >> 16) : KV; + + // Sparse: the tile shares one mask row (gqa heads, or Br==1). split_k + // partitions the n_kv_max blocks. + if (USE_SPARSE) { + uint32_t qrow = (p.gqa_ratio > 1) ? gqa_iq1 : (i * Br); + sparse_base = (((iq3 % p.nem3) * p.nem2 + (iq2 % p.nem2)) * p.nem1 + qrow) * p.split_kv; + + uint32_t total_blocks = CEIL_DIV(p.split_kv, Bc); + uint32_t per_blocks = CEIL_DIV(total_blocks, p.k_num); + start_j = min(split_k_index * per_blocks, total_blocks); + end_j = min((split_k_index + 1) * per_blocks, total_blocks); + } +} + +// Resolve a linear KV slot to a real column; false for inactive (sparse padding/-1, or dense OOB). +bool fa_kv_index(uint lin, out uint kv_col) { + if (USE_SPARSE) { + if (lin >= p.split_kv) { + kv_col = 0; + return false; + } + int idx = data_sparse[sparse_base + lin]; + kv_col = idx >= 0 ? uint(idx) : 0; + return idx >= 0; + } + kv_col = lin; + return !KV_bounds_check || lin < KV; } // Bias applied to softmax to stay in fp16 range. diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp index 057ed739..aa9dd624 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp @@ -176,9 +176,16 @@ void main() { uint32_t c = (idx + tid) / (Br / 4); uint32_t r = (idx + tid) % (Br / 4); if (idx + tid < Bc * Br / 4 || idx + gl_WorkGroupSize.x <= Bc * Br / 4) { - if ((!KV_bounds_check || j * Bc + c < KV)) { + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + c, kcol); + if (kv_active) { f16vec4 m; - if (!nem1_bounds_check || i * Br + r * 4 + 3 < p.nem1) { + if (USE_SPARSE) { + // sparse is gqa-gated (m_stride == 0): all four rows share the value + FLOAT_TYPE mv = FLOAT_TYPE(data_m[m_offset + kcol]); + m = f16vec4(mv); + max_mask = max(max_mask, float(mv)); + } else if (!nem1_bounds_check || i * Br + r * 4 + 3 < p.nem1) { m = f16vec4(data_m[m_offset + (i * Br + r * 4 ) * m_stride + (j * Bc + c)], data_m[m_offset + (i * Br + r * 4 + 1) * m_stride + (j * Bc + c)], data_m[m_offset + (i * Br + r * 4 + 2) * m_stride + (j * Bc + c)], @@ -206,6 +213,8 @@ void main() { m = f16vec4(0.0); } mask_cache[idx / WorkGroupSize] = m; + } else if (USE_SPARSE) { + mask_cache[idx / WorkGroupSize] = f16vec4(NEG_FLT_MAX_OVER_2); } } } @@ -231,17 +240,19 @@ void main() { uint32_t c = (idx + tid) / (HSK_pad / 4); if (idx + gl_WorkGroupSize.x <= Bc * HSK_pad / 4 || c < Bc) { FLOAT_TYPEV4 K_Tf = FLOAT_TYPEV4(0); - if ((!KV_bounds_check || j * Bc + c < KV) && (HSK == HSK_pad || d < HSK / 4)) { + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + c, kcol); + if (kv_active && (HSK == HSK_pad || d < HSK / 4)) { #if !defined(BFLOAT16) if (USE_DECODE_K) { - uint coord = (j * Bc + c) * k_stride * BLOCK_SIZE_K + 4 * d; + uint coord = kcol * k_stride * BLOCK_SIZE_K + 4 * d; uint ib = coord / BLOCK_SIZE_K; uint iqs = (coord % BLOCK_SIZE_K); K_Tf = dequantize4(ib, iqs, k_offset, BINDING_IDX_K); } else #endif { - K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + (j * Bc + c) * k_stride / 4 + d]); + K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + kcol * k_stride / 4 + d]); } } @@ -266,7 +277,7 @@ void main() { if (SHMEM_STAGING == 0) { // For quants we always need to dequant into kvsh; for f16/bf16 we can load // directly from global memory when alignment / bounds allow it. - const bool stage_k = USE_DECODE_K || KV_bounds_check || d * 16 + 16 > HSK; + const bool stage_k = USE_DECODE_K || KV_bounds_check || USE_SPARSE || d * 16 + 16 > HSK; if (stage_k) { barrier(); [[unroll]] for (uint32_t idx = 0; idx < Bc * MatBr / 4; idx += gl_WorkGroupSize.x) { @@ -274,17 +285,19 @@ void main() { uint32_t row = (idx + tid) / (MatBr / 4); if (idx + tid < Bc * MatBr / 4) { FLOAT_TYPEV4 K_Tf = FLOAT_TYPEV4(0); - if ((!KV_bounds_check || j * Bc + row < KV) && (HSK == HSK_pad || d * 16 + col_vec * 4 < HSK)) { + uint32_t kcol; + bool kv_active = fa_kv_index(j * Bc + row, kcol); + if (kv_active && (HSK == HSK_pad || d * 16 + col_vec * 4 < HSK)) { #if !defined(BFLOAT16) if (USE_DECODE_K) { - uint coord = (j * Bc + row) * k_stride * BLOCK_SIZE_K + d * 16 + col_vec * 4; + uint coord = kcol * k_stride * BLOCK_SIZE_K + d * 16 + col_vec * 4; uint ib = coord / BLOCK_SIZE_K; uint iqs = (coord % BLOCK_SIZE_K); K_Tf = dequantize4(ib, iqs, k_offset, BINDING_IDX_K); } else #endif { - K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + (j * Bc + row) * k_stride / 4 + d * 16 / 4 + col_vec]); + K_Tf = FLOAT_TYPEV4(data_kv4[k_offset / 4 + kcol * k_stride / 4 + d * 16 / 4 + col_vec]); } } @@ -401,17 +414,19 @@ void main() { uint32_t c = (idx + tid) / (HSV_pad / 4); if (idx + gl_WorkGroupSize.x <= Bc * HSV_pad / 4 || c < Bc) { FLOAT_TYPEV4 V_Tf = FLOAT_TYPEV4(0); - if ((!KV_bounds_check || j * Bc + c < KV) && (HSV == HSV_pad || d < HSV / 4)) { + uint32_t v_row; + bool kv_active = fa_kv_index(j * Bc + c, v_row); + if (kv_active && (HSV == HSV_pad || d < HSV / 4)) { #if !defined(BFLOAT16) if (USE_DECODE_V) { - uint coord = (j * Bc + c) * v_stride * BLOCK_SIZE_V + 4 * d; + uint coord = v_row * v_stride * BLOCK_SIZE_V + 4 * d; uint ib = coord / BLOCK_SIZE_V; uint iqs = (coord % BLOCK_SIZE_V); V_Tf = dequantize4(ib, iqs, v_offset, BINDING_IDX_V); } else #endif { - V_Tf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + (j * Bc + c) * v_stride / 4 + d]); + V_Tf = FLOAT_TYPEV4(data_vv4[v_offset / 4 + v_row * v_stride / 4 + d]); } } @@ -441,21 +456,22 @@ void main() { if (SHMEM_STAGING == 0) { // For quants we always preload via kvsh. For f16/bf16 we only preload when // alignment / bounds force it (otherwise we coopMatLoad direct from data_vv4). - const bool stage_v = USE_DECODE_V || KV_bounds_check; + const bool stage_v = USE_DECODE_V || KV_bounds_check || USE_SPARSE; if (stage_v) { [[unroll]] for (uint32_t i = 0; i < v_loads_per_thread; ++i) { const uint idx = i * gl_WorkGroupSize.x + tid; const uint row = idx / v_cols; const uint col = idx % v_cols; - const uint v_row = j * Bc + row; + uint32_t v_row; + bool kv_active = fa_kv_index(j * Bc + row, v_row); const uint v_col = hsv_tile * MatBc * row_split + col * 4; const uint coord = v_row * v_stride * BLOCK_SIZE_V + v_col; const uint ib = coord / BLOCK_SIZE_V; const uint iqs = coord % BLOCK_SIZE_V; - if (!KV_bounds_check || (v_row < KV && v_col < HSV)) { + if (USE_SPARSE ? (kv_active && v_col < HSV) : (!KV_bounds_check || (v_row < KV && v_col < HSV))) { #if !defined(BFLOAT16) if (USE_DECODE_V) { kvsh[row * vsh_stride + col] = dequantize4(ib, iqs, v_offset, BINDING_IDX_V); @@ -479,7 +495,7 @@ void main() { coopMatLoad(KMat, Psh, bc_chunk * MatBc * psh_stride, psh_stride, gl_CooperativeMatrixLayoutColumnMajor); if (SHMEM_STAGING == 0) { - if (!USE_DECODE_V && !KV_bounds_check) { + if (!USE_DECODE_V && !KV_bounds_check && !USE_SPARSE) { // F16/BF16 values can be loaded directly from global memory const uint v_tile_row = j * Bc + bc_chunk * MatBc; const uint v_tile_offset = v_offset / 4 + v_tile_row * v_stride / 4 + hsv_offset / 4; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp index 31741115..c6ed63dd 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp @@ -29,6 +29,12 @@ #include "dequant_funcs_cm2.glsl" #endif +#ifdef GL_NV_cooperative_matrix_decode_vector +#define FA_GATHER_BS 4u +#else +#define FA_GATHER_BS 1u +#endif + // buffer_reference stride = sizeof(struct) = FaBlockBytesK/V. layout(buffer_reference, std430, buffer_reference_align = 1) buffer decodeBufFA_K { uint8_t raw[FaBlockBytesK]; @@ -40,26 +46,28 @@ layout(buffer_reference, std430, buffer_reference_align = 1) buffer decodeBufFA_ #if !defined(BFLOAT16) float16_t faDecodeK(const decodeBufFA_K bl_in, const uint blockCoords[2], const uint coordInBlock[2]) { switch (FaTypeK) { - case FA_TYPE_F32: return dequantFuncF32 (decodeBufF32 (bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_0: return dequantFuncQ4_0(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_1: return dequantFuncQ4_1(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_0: return dequantFuncQ5_0(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_1: return dequantFuncQ5_1(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q8_0: return dequantFuncQ8_0(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_IQ4_NL: return dequantFuncIQ4_NL(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_F32: return dequantFuncF32 (decodeBufF32 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_IQ4_NL: return dequantFuncIQ4_NL(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q1_0: return dequantFuncQ1_0(decodeBufQ1_0(bl_in), blockCoords, coordInBlock); default: return float16_t(0); } } float16_t faDecodeV(const decodeBufFA_V bl_in, const uint blockCoords[2], const uint coordInBlock[2]) { switch (FaTypeV) { - case FA_TYPE_F32: return dequantFuncF32 (decodeBufF32 (bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_0: return dequantFuncQ4_0(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_1: return dequantFuncQ4_1(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_0: return dequantFuncQ5_0(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_1: return dequantFuncQ5_1(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q8_0: return dequantFuncQ8_0(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_IQ4_NL: return dequantFuncIQ4_NL(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_F32: return dequantFuncF32 (decodeBufF32 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_IQ4_NL: return dequantFuncIQ4_NL(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q1_0: return dequantFuncQ1_0(decodeBufQ1_0(bl_in), blockCoords, coordInBlock); default: return float16_t(0); } } @@ -67,26 +75,26 @@ float16_t faDecodeV(const decodeBufFA_V bl_in, const uint blockCoords[2], const // V=4 vector decode for K/V; dispatches to per-format _v decoders. f16vec4 faDecodeKVector(const decodeBufFA_K bl_in, const uint blockCoords[2], const uint coordInBlock[2]) { switch (FaTypeK) { - case FA_TYPE_F32: return f16vec4(decodeBufF32(bl_in).block); - case FA_TYPE_Q4_0: return dequantFuncQ4_0_v(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_1: return dequantFuncQ4_1_v(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_0: return dequantFuncQ5_0_v(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_1: return dequantFuncQ5_1_v(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q8_0: return dequantFuncQ8_0_v(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_IQ4_NL: return dequantFuncIQ4_NL_v(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_F32: return f16vec4(decodeBufF32(bl_in).block); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0_v(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1_v(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0_v(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1_v(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0_v(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_IQ4_NL: return dequantFuncIQ4_NL_v(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); default: return f16vec4(0); } } f16vec4 faDecodeVVector(const decodeBufFA_V bl_in, const uint blockCoords[2], const uint coordInBlock[2]) { switch (FaTypeV) { - case FA_TYPE_F32: return f16vec4(decodeBufF32(bl_in).block); - case FA_TYPE_Q4_0: return dequantFuncQ4_0_v(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q4_1: return dequantFuncQ4_1_v(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_0: return dequantFuncQ5_0_v(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q5_1: return dequantFuncQ5_1_v(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); - case FA_TYPE_Q8_0: return dequantFuncQ8_0_v(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); - case FA_TYPE_IQ4_NL: return dequantFuncIQ4_NL_v(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_F32: return f16vec4(decodeBufF32(bl_in).block); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0_v(decodeBufQ4_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1_v(decodeBufQ4_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0_v(decodeBufQ5_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1_v(decodeBufQ5_1(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0_v(decodeBufQ8_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_IQ4_NL: return dequantFuncIQ4_NL_v(decodeBufIQ4_NL(bl_in), blockCoords, coordInBlock); default: return f16vec4(0); } } @@ -105,6 +113,67 @@ layout (binding = 1) readonly buffer K {uint8_t data_k[];}; layout (binding = 2) readonly buffer V {uint8_t data_v[];}; layout (binding = 3) readonly buffer M {uint8_t data_m[];}; +// f16 aliases for the sparse gather callbacks. +layout (binding = 1) readonly buffer KF16 {float16_t data_kf16[];}; +layout (binding = 2) readonly buffer VF16 {float16_t data_vf16[];}; +layout (binding = 3) readonly buffer MF16 {float16_t data_mf16[];}; +#ifdef GL_NV_cooperative_matrix_decode_vector +layout (binding = 1) readonly buffer KF16V4 {f16vec4 data_kf16v4[];}; +layout (binding = 2) readonly buffer VF16V4 {f16vec4 data_vf16v4[];}; +#endif + +// K/V/mask f16-element offsets for the current head/batch, set in main(). +uint32_t g_k_off_elem, g_v_off_elem, g_m_off_elem; + +#if !defined(BFLOAT16) +// blockCoords are in block units: KV slot = blockCoords[0], +// head dim = blockCoords[1]*FA_GATHER_BS + coordInBlock[1]. +float16_t faGatherK(const decodeBufFA_K unused, const uint32_t blockCoords[2], const uint32_t coordInBlock[2]) { + if (blockCoords[0] >= p.split_kv) { return float16_t(0); } + const int r = data_sparse[sparse_base + blockCoords[0]]; + return r < 0 ? float16_t(0) : data_kf16[g_k_off_elem + uint(r) * k_stride + blockCoords[1] * FA_GATHER_BS + coordInBlock[1]]; +} + +float16_t faGatherV(const decodeBufFA_V unused, const uint32_t blockCoords[2], const uint32_t coordInBlock[2]) { + if (blockCoords[0] >= p.split_kv) { return float16_t(0); } + const int r = data_sparse[sparse_base + blockCoords[0]]; + return r < 0 ? float16_t(0) : data_vf16[g_v_off_elem + uint(r) * v_stride + blockCoords[1] * FA_GATHER_BS + coordInBlock[1]]; +} + +#ifdef GL_NV_cooperative_matrix_decode_vector +f16vec4 faGatherKVector(const decodeBufFA_K unused, const uint32_t blockCoords[2], const uint32_t coordInBlock[2]) { + if (blockCoords[0] >= p.split_kv) { return f16vec4(0); } + const int r = data_sparse[sparse_base + blockCoords[0]]; + if (r < 0) { return f16vec4(0); } + const uint32_t o = g_k_off_elem + uint(r) * k_stride + blockCoords[1] * FA_GATHER_BS + coordInBlock[1]; + return data_kf16v4[o / 4]; +} + +f16vec4 faGatherVVector(const decodeBufFA_V unused, const uint32_t blockCoords[2], const uint32_t coordInBlock[2]) { + if (blockCoords[0] >= p.split_kv) { return f16vec4(0); } + const int r = data_sparse[sparse_base + blockCoords[0]]; + if (r < 0) { return f16vec4(0); } + const uint32_t o = g_v_off_elem + uint(r) * v_stride + blockCoords[1] * FA_GATHER_BS + coordInBlock[1]; + return data_vf16v4[o / 4]; +} + +#define FAGATHERK , faGatherK, faGatherKVector +#define FAGATHERV , faGatherV, faGatherVVector +#else +#define FAGATHERK , faGatherK +#define FAGATHERV , faGatherV +#endif +#endif + +// Add gathered mask to S (slope==1 since sparse requires max_bias==0). col = slot in block jblk. +ACC_TYPE faAddSparseMask(const uint32_t row, const uint32_t col, const ACC_TYPE elem, const uint32_t jblk) { + const float NEG = uintBitsToFloat(0xFEFFFFFF); + const uint32_t kvslot = jblk * Bc + col; + if (kvslot >= p.split_kv) { return ACC_TYPE(NEG); } + const int r = data_sparse[sparse_base + kvslot]; + return r < 0 ? ACC_TYPE(NEG) : elem + ACC_TYPE(data_mf16[g_m_off_elem + row * m_stride + uint(r)]); +} + ACC_TYPE maxReduce(const in ACC_TYPE x, const in ACC_TYPE y) { return max(x, y); } @@ -183,14 +252,16 @@ void main() { tensorViewNV<2, false, 1, 0> tensorViewTranspose = createTensorViewNV(2, false, 1, 0); - const uint bs_k = fa_block_elems(FaTypeK); - const uint bs_v = fa_block_elems(FaTypeV); + const uint bs_k = USE_SPARSE ? FA_GATHER_BS : fa_block_elems(FaTypeK); + const uint bs_v = USE_SPARSE ? FA_GATHER_BS : fa_block_elems(FaTypeV); tensorLayoutK = setTensorLayoutBlockSizeNV(tensorLayoutK, 1, bs_k); tensorLayoutV = setTensorLayoutBlockSizeNV(tensorLayoutV, 1, bs_v); + // Sparse iterates n_kv_max (in split_kv); the decode callbacks remap each slot. + const uint32_t KV_iter = USE_SPARSE ? p.split_kv : KV; tensorLayoutQ = setTensorLayoutDimensionNV(tensorLayoutQ, N, HSK); - tensorLayoutK = setTensorLayoutDimensionNV(tensorLayoutK, KV, HSK); - tensorLayoutV = setTensorLayoutDimensionNV(tensorLayoutV, KV, HSV); + tensorLayoutK = setTensorLayoutDimensionNV(tensorLayoutK, KV_iter, HSK); + tensorLayoutV = setTensorLayoutDimensionNV(tensorLayoutV, KV_iter, HSV); // hint to the compiler that strides are aligned for the aligned variant of the shader if (Clamp != gl_CooperativeMatrixClampModeConstantNV) @@ -248,6 +319,10 @@ void main() { mo_offset += ((iq3 % p.nem3) * p.nem2 + (iq2 % p.nem2)) * CEIL_DIV(p.nem1, Br) * mo_stride; } + g_k_off_elem = (ik2*p.nb12 + ik3*p.nb13) / 2; + g_v_off_elem = (iv2*p.nb22 + iv3*p.nb23) / 2; + g_m_off_elem = m_offset / 2; + uint32_t mask_opt = 0; uint32_t mask_opt_idx = ~0; @@ -255,7 +330,7 @@ void main() { for (uint32_t j = start_j; j < end_j; ++j) { coopmat mv = coopmat(0); - if (MASK_ENABLE) { + if (MASK_ENABLE && !USE_SPARSE) { if (USE_MASK_OPT && mask_opt_idx != j / 16) { mask_opt_idx = j / 16; @@ -313,7 +388,9 @@ void main() { coopMatLoadTensorNV(K_T, data_k, k_offset, sliceTensorLayoutNV(tensorLayoutK, j * Bc, Bc, 0, HSK_pad), tensorViewTranspose); #else const bool k_use_decode = (bs_k > 1u); - if (k_use_decode) { + if (USE_SPARSE) { + coopMatLoadTensorNV(K_T, data_k, k_offset, sliceTensorLayoutNV(tensorLayoutK, j * Bc, Bc, 0, HSK_pad), tensorViewTranspose FAGATHERK); + } else if (k_use_decode) { coopMatLoadTensorNV(K_T, data_k, k_offset, sliceTensorLayoutNV(tensorLayoutK, j * Bc, Bc, 0, HSK_pad), tensorViewTranspose FADECODEK); } else { coopMatLoadTensorNV(K_T, data_k, k_offset, sliceTensorLayoutNV(tensorLayoutK, j * Bc, Bc, 0, HSK_pad), tensorViewTranspose); @@ -328,7 +405,9 @@ void main() { } } - if (MASK_ENABLE) { + if (MASK_ENABLE && USE_SPARSE) { + coopMatPerElementNV(S, S, faAddSparseMask, j); + } else if (MASK_ENABLE) { S += slopeMat*coopmat(mv); } @@ -383,7 +462,9 @@ void main() { coopMatLoadTensorNV(V, data_v, v_offset, sliceTensorLayoutNV(tensorLayoutV, j * Bc, Bc, 0, HSV_pad)); #else const bool v_use_decode = (bs_v > 1u); - if (v_use_decode) { + if (USE_SPARSE) { + coopMatLoadTensorNV(V, data_v, v_offset, sliceTensorLayoutNV(tensorLayoutV, j * Bc, Bc, 0, HSV_pad) FAGATHERV); + } else if (v_use_decode) { coopMatLoadTensorNV(V, data_v, v_offset, sliceTensorLayoutNV(tensorLayoutV, j * Bc, Bc, 0, HSV_pad) FADECODEV); } else { coopMatLoadTensorNV(V, data_v, v_offset, sliceTensorLayoutNV(tensorLayoutV, j * Bc, Bc, 0, HSV_pad)); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_1.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_1.comp new file mode 100644 index 00000000..b5f95aaa --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_1.comp @@ -0,0 +1,263 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : enable +#extension GL_EXT_shader_16bit_storage : require +#extension GL_EXT_shader_explicit_arithmetic_types_float16 : require +#extension GL_EXT_shader_explicit_arithmetic_types_int16 : require +#extension GL_KHR_memory_scope_semantics : enable +#extension GL_KHR_shader_subgroup_basic : enable +#extension GL_KHR_shader_subgroup_ballot : enable +#extension GL_KHR_shader_subgroup_arithmetic : enable +#extension GL_KHR_cooperative_matrix : enable +#extension GL_EXT_shared_memory_block : enable + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer Q {float16_t qState[];}; +layout (binding = 1) readonly buffer K_VEC4 {f16vec4 kStateVec4[];}; +layout (binding = 2) buffer MASK_F16 {float16_t mState_f16[];}; +layout (binding = 3) buffer P_FP16 {float16_t matP_f16[];}; +layout (binding = 4) buffer OUT_MAX {float out_max_f32[];}; + +layout (push_constant) uniform parameter +{ + uint kvSeqLen; + uint activationLength; + uint qHead; + uint kvHead; + uint qkRatio; + uint qkSubGroups; + uint flag; + uint kvStride1; + uint kvStride2; + uint batchStrideQ; + uint batchStrideK; + uint batchStrideV; + uint batchStrideM; + uint batchStrideO; + float softMaxScale; +} p; + +layout (constant_id = 0) const uint GROUPSIZE = 128; +layout (constant_id = 1) const uint GQA_RATIO = 8; +layout (constant_id = 2) const uint HEAD_DIM = 128; +layout (constant_id = 3) const uint WARPSIZE = 16; +layout (constant_id = 4) const uint MATP_REDUCE = 32; +layout (constant_id = 5) const uint N_TOK = 1; +layout (constant_id = 6) const uint COOP_MAT_P_PER_LOOP = 4; + +#define MAX_HEADS 8 + +#define TN WARPSIZE +#define TM 8 +#define TK 16 +#define SUBGROUP_COUNT (GROUPSIZE / WARPSIZE) +#define MATP_PER_LOOP (COOP_MAT_P_PER_LOOP * TM) +#define P_LOOP_COUNT (MATP_REDUCE / MATP_PER_LOOP) + +#define COOP_MAT_Q_PER_TOKEN ((GQA_RATIO + TN - 1) / TN) +#define COOP_MAT_P_M COOP_MAT_Q_PER_TOKEN +#define COOP_MAT_P_N (MATP_REDUCE / TM) +#define SLM_PV_SIZE (MATP_REDUCE * COOP_MAT_P_M * TN) +#define SLM_MASK_SIZE (N_TOK * MATP_REDUCE) +#define SLM_POOL_SIZE_K (MATP_PER_LOOP * HEAD_DIM) +#define K_LOAD_PER_LOOP (GROUPSIZE * 4) +#define HEAD_DIM_VEC4 (HEAD_DIM / 4) +#define SLM_CHUNK_SIZE (TK / 4) +#define K_LOAD_LOOPS ((SLM_POOL_SIZE_K + K_LOAD_PER_LOOP - 1) / K_LOAD_PER_LOOP) +#define O_COUNT ((GQA_RATIO + SUBGROUP_COUNT - 1) / SUBGROUP_COUNT) + +shared slm_pool_block { + float slm_pool_pv[SLM_PV_SIZE + SLM_MASK_SIZE]; +} slm_pool_f32; + +shared slm_pool_alias_block { + float16_t slm_pool_k[SLM_POOL_SIZE_K]; +} slm_pool_f16; + +void main() { + const uint lane = gl_SubgroupInvocationID; + const uint kHeadIdx = gl_WorkGroupID.x % p.kvHead; + const uint outGroupIdx = gl_WorkGroupID.x / p.kvHead; + const uint v = gl_WorkGroupID.y; + const uint d = gl_WorkGroupID.z; + const uint localLinearId = gl_SubgroupID; + const uint wgLane = localLinearId * WARPSIZE + lane; + const uint qDim = p.qHead * HEAD_DIM; + const uint kvDim = p.kvStride1; + const uint maskDim = p.kvSeqLen; + const uint maxDim = (p.kvSeqLen + MATP_REDUCE - 1) / MATP_REDUCE; + const uint pDim = maxDim * MATP_REDUCE; + const uint tokFlatIdx = localLinearId + outGroupIdx * N_TOK; + uint offsetBaseQ = min(tokFlatIdx, p.activationLength - 1) * qDim; + offsetBaseQ = offsetBaseQ + d * p.batchStrideQ + kHeadIdx * HEAD_DIM * GQA_RATIO; + const uint offsetBaseK = (d * p.batchStrideK + (v * MATP_REDUCE) * kvDim + kHeadIdx * p.kvStride2) / 4; + uint offsetOut = d * p.qHead * p.activationLength * pDim + v * MATP_REDUCE + kHeadIdx * GQA_RATIO * pDim + (localLinearId * O_COUNT + outGroupIdx * N_TOK * p.qHead) * pDim + lane; + uint offsetMax = d * p.qHead * p.activationLength * maxDim + v + kHeadIdx * GQA_RATIO * maxDim + (localLinearId * O_COUNT + outGroupIdx * N_TOK * p.qHead) * maxDim; + const uint offsetSlmLoadPv = (localLinearId * O_COUNT * MATP_REDUCE + lane); + const uint offsetBaseM = v * MATP_REDUCE + lane; + const float fp32Min = uintBitsToFloat(0xFEFFFFFF); + + const uint loopCount = HEAD_DIM / TK; + float maskFp32[MATP_REDUCE / WARPSIZE]; + + if (tokFlatIdx < p.activationLength) { + [[unroll]] for (uint mk = 0; mk < MATP_REDUCE / WARPSIZE; mk++) { + const uint maskOffset = mk * WARPSIZE + offsetBaseM; + if (maskOffset < maskDim) { + maskFp32[mk] = float(mState_f16[d * p.batchStrideM + tokFlatIdx * maskDim + maskOffset]); + } else { + maskFp32[mk] = fp32Min; + } + } + } + + coopmat matP[COOP_MAT_P_M][COOP_MAT_P_N]; + + [[unroll]] for (uint mp = 0; mp < COOP_MAT_P_M; mp++) { + [[unroll]] for (uint np = 0; np < COOP_MAT_P_N; np++) { + matP[mp][np] = coopmat(0.0f); + } + } + + [[unroll]] for (uint kLoad = 0; kLoad < K_LOAD_LOOPS; kLoad++) { + const uint flatOffset = kLoad * GROUPSIZE + wgLane; + const uint kRowIdx = flatOffset / HEAD_DIM_VEC4; + const uint kColIdx = flatOffset % HEAD_DIM_VEC4; + const uint slmChunkCol = kColIdx % SLM_CHUNK_SIZE; + const uint slmChunkRow = kColIdx / SLM_CHUNK_SIZE; + const uint offsetK = offsetBaseK + kRowIdx * kvDim / 4 + kColIdx; + const uint offsetSlmK = kRowIdx * TK + slmChunkRow * TK * MATP_PER_LOOP + slmChunkCol * 4; + slm_pool_f16.slm_pool_k[offsetSlmK + 0] = kStateVec4[offsetK].x; + slm_pool_f16.slm_pool_k[offsetSlmK + 1] = kStateVec4[offsetK].y; + slm_pool_f16.slm_pool_k[offsetSlmK + 2] = kStateVec4[offsetK].z; + slm_pool_f16.slm_pool_k[offsetSlmK + 3] = kStateVec4[offsetK].w; + } + + [[unroll]] for (uint pLoop = 0; pLoop < P_LOOP_COUNT; pLoop++) { + f16vec4 kTemp[K_LOAD_LOOPS]; + + if (pLoop + 1 < P_LOOP_COUNT) { + [[unroll]] for (uint kLoad = 0; kLoad < K_LOAD_LOOPS; kLoad++) { + const uint flatOffset = kLoad * GROUPSIZE + wgLane; + const uint kRowIdx = flatOffset / HEAD_DIM_VEC4 + (pLoop + 1) * MATP_PER_LOOP; + const uint kColIdx = flatOffset % HEAD_DIM_VEC4; + const uint offsetK = offsetBaseK + kRowIdx * kvDim / 4 + kColIdx; + kTemp[kLoad] = kStateVec4[offsetK]; + } + } + + barrier(); + if (localLinearId < N_TOK) { + [[unroll]] for (uint loop = 0; loop < loopCount; loop++) { + coopmat matQ[COOP_MAT_P_M]; + coopmat matK[COOP_MAT_P_PER_LOOP]; + + [[unroll]] for (uint mq = 0; mq < COOP_MAT_P_M; mq++) { + coopMatLoad( + matQ[mq], + qState, + offsetBaseQ + mq * TN * HEAD_DIM + loop * TK, + HEAD_DIM, + gl_CooperativeMatrixLayoutColumnMajor); + } + + [[unroll]] for (uint np = 0; np < COOP_MAT_P_PER_LOOP; np++) { + coopMatLoad( + matK[np], + slm_pool_f16.slm_pool_k, + loop * TK * MATP_PER_LOOP + np * TM * TK, + TK, + gl_CooperativeMatrixLayoutRowMajor); + } + + [[unroll]] for (uint mp = 0; mp < COOP_MAT_P_M; mp++) { + [[unroll]] for (uint np = 0; np < COOP_MAT_P_PER_LOOP; np++) { + matP[mp][pLoop * COOP_MAT_P_PER_LOOP + np] = coopMatMulAdd(matK[np], matQ[mp], matP[mp][pLoop * COOP_MAT_P_PER_LOOP + np]); + } + } + } + } + + barrier(); + + if (pLoop + 1 < P_LOOP_COUNT) { + [[unroll]] for (uint kLoad = 0; kLoad < K_LOAD_LOOPS; kLoad++) { + const uint flatOffset = kLoad * GROUPSIZE + wgLane; + const uint kRowIdx = flatOffset / HEAD_DIM_VEC4; + const uint kColIdx = flatOffset % HEAD_DIM_VEC4; + const uint slmChunkCol = kColIdx % SLM_CHUNK_SIZE; + const uint slmChunkRow = kColIdx / SLM_CHUNK_SIZE; + const uint offsetSlmK = kRowIdx * TK + slmChunkRow * TK * MATP_PER_LOOP + slmChunkCol * 4; + slm_pool_f16.slm_pool_k[offsetSlmK + 0] = kTemp[kLoad].x; + slm_pool_f16.slm_pool_k[offsetSlmK + 1] = kTemp[kLoad].y; + slm_pool_f16.slm_pool_k[offsetSlmK + 2] = kTemp[kLoad].z; + slm_pool_f16.slm_pool_k[offsetSlmK + 3] = kTemp[kLoad].w; + } + } + } + + barrier(); + + if (tokFlatIdx < p.activationLength) { + [[unroll]] for (uint mk = 0; mk < MATP_REDUCE / WARPSIZE; mk++) { + slm_pool_f32.slm_pool_pv[SLM_PV_SIZE + localLinearId * MATP_REDUCE + mk * WARPSIZE + lane] = maskFp32[mk]; + } + } + + [[unroll]] for (uint oLoop = 0; oLoop < N_TOK; oLoop++) { + if (oLoop + outGroupIdx * N_TOK < p.activationLength) { + if (localLinearId == oLoop) { + [[unroll]] for (uint mp = 0; mp < COOP_MAT_P_M; mp++) { + [[unroll]] for (uint np = 0; np < COOP_MAT_P_N; np++) { + coopMatStore(matP[mp][np], slm_pool_f32.slm_pool_pv, mp * MATP_REDUCE * TN + np * TM, MATP_REDUCE, gl_CooperativeMatrixLayoutColumnMajor); + } + } + } + + barrier(); + + [[unroll]] for (uint maskIdx = 0; maskIdx < MATP_REDUCE / WARPSIZE; maskIdx++) { + maskFp32[maskIdx] = slm_pool_f32.slm_pool_pv[SLM_PV_SIZE + oLoop * MATP_REDUCE + maskIdx * WARPSIZE + lane]; + } + + float fp32O[O_COUNT][MATP_REDUCE / WARPSIZE]; + float maxOut[O_COUNT]; + + [[unroll]] for (uint oc = 0; oc < O_COUNT; oc++) { + [[unroll]] for (uint os = 0; os < MATP_REDUCE / WARPSIZE; os++) { + fp32O[oc][os] = slm_pool_f32.slm_pool_pv[offsetSlmLoadPv + os * WARPSIZE + oc * MATP_REDUCE] * p.softMaxScale; + } + + [[unroll]] for (uint os = 0; os < MATP_REDUCE / WARPSIZE; os++) { + fp32O[oc][os] = fp32O[oc][os] + maskFp32[os]; + } + + float maxTemp = fp32Min; + [[unroll]] for (uint os = 0; os < MATP_REDUCE / WARPSIZE; os++) { + maxTemp = max(maxTemp, fp32O[oc][os]); + } + maxOut[oc] = subgroupMax(maxTemp); + [[unroll]] for (uint os = 0; os < MATP_REDUCE / WARPSIZE; os++) { + fp32O[oc][os] = exp(fp32O[oc][os] - maxOut[oc]); + } + } + + [[unroll]] for (uint oc = 0; oc < O_COUNT; oc++) { + if (localLinearId * O_COUNT + oc < GQA_RATIO) { + [[unroll]] for (uint os = 0; os < MATP_REDUCE / WARPSIZE; os++) { + matP_f16[offsetOut + oc * pDim + os * WARPSIZE] = float16_t(fp32O[oc][os]); + } + + if (lane == 0) { + out_max_f32[offsetMax + oc * maxDim] = maxOut[oc]; + } + } + } + + offsetOut = offsetOut + p.qHead * pDim; + offsetMax = offsetMax + p.qHead * maxDim; + barrier(); + } + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_2.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_2.comp new file mode 100644 index 00000000..60a1c2ce --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_decode_phase_2.comp @@ -0,0 +1,408 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : enable +#extension GL_EXT_shader_16bit_storage : require +#extension GL_EXT_shader_explicit_arithmetic_types_float16 : require +#extension GL_EXT_shader_explicit_arithmetic_types_int16 : require +#extension GL_KHR_memory_scope_semantics : enable +#extension GL_KHR_shader_subgroup_basic : enable +#extension GL_KHR_shader_subgroup_ballot : enable +#extension GL_KHR_shader_subgroup_arithmetic : enable +#extension GL_KHR_cooperative_matrix : enable +#extension GL_EXT_shared_memory_block : enable + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer P {f16vec4 pStateVec4[];}; +layout (binding = 1) readonly buffer V {float16_t vState[];}; +layout (binding = 1) readonly buffer V_VEC4 {f16vec4 vStateVec4[];}; +layout (binding = 2) buffer MAX_FP32 {float max_f32[];}; +layout (binding = 3) buffer SINK_FP32 {float sink_f32[];}; +layout (binding = 4) buffer OUT_FP32 {float out_f32[];}; +layout (binding = 4) buffer OUT_VEC4 {vec4 out_f32_vec4[];}; +layout (binding = 4) buffer OUT_F16 {float16_t out_f16[];}; + +layout (push_constant) uniform parameter +{ + uint kvSeqLen; + uint activationLength; + uint qHead; + uint kvHead; + uint qkRatio; + uint qkSubGroups; + uint flag; + uint kvStride1; + uint kvStride2; + uint batchStrideQ; + uint batchStrideK; + uint batchStrideV; + uint batchStrideM; + uint batchStrideO; + float softMaxScale; +} p; + +layout (constant_id = 0) const uint GROUPSIZE = 256; +layout (constant_id = 1) const uint GQA_RATIO = 8; +layout (constant_id = 2) const uint HEAD_DIM = 128; +layout (constant_id = 3) const uint N_TOKS_PER_GROUP = 1; +layout (constant_id = 4) const uint WARPSIZE = 16; +layout (constant_id = 5) const uint MATP_PER_LOOP = 64; +layout (constant_id = 6) const uint MATP_REDUCE = 32; +layout (constant_id = 7) const uint WARP_V_DIM = 16; + +#define TN WARPSIZE +#define TM 8 +#define TK 16 +#define MAT_O_N (WARP_V_DIM / TM) +#define MAT_P_M (GQA_RATIO * N_TOKS_PER_GROUP) +#define ALIGNED_P_M ((MAT_P_M + WARPSIZE - 1) / WARPSIZE) +#define V_HEAD_GROUPS (HEAD_DIM / WARP_V_DIM) + +#define SUBGROUP_COUNT (GROUPSIZE / WARPSIZE) +#define SPLIT_P_GROUPS (MATP_PER_LOOP / TK) + +#define SLM_POOL_SIZE_O (SUBGROUP_COUNT * ALIGNED_P_M * TN * MAT_O_N * TM) + +#define P_LOAD_PER_LOOP (GROUPSIZE * 4) +#define P_LOAD_LOOPS ((MAT_P_M * MATP_PER_LOOP + P_LOAD_PER_LOOP - 1) / P_LOAD_PER_LOOP) +#define SLM_POOL_SIZE_P (P_LOAD_LOOPS * P_LOAD_PER_LOOP) +#define SIZE_LOCAL_MAX (MAT_P_M * MATP_PER_LOOP / MATP_REDUCE) +#define MAX_LOAD_LOOPS ((SIZE_LOCAL_MAX + GROUPSIZE - 1) / GROUPSIZE) +#define SLM_POOL_SIZE_LOCAL_MAX (MAX_LOAD_LOOPS * GROUPSIZE) +#define MAX_REDUCE_COUNT ((MAT_P_M + SUBGROUP_COUNT - 1) / SUBGROUP_COUNT) +#define GLOBAL_MAX_SIZE (MAX_REDUCE_COUNT * SUBGROUP_COUNT) + +#define SLM_POOL_SIZE_SOFTMAX_SUM (SUBGROUP_COUNT * P_LOAD_LOOPS) + +#define SLM_OFFSET_P (GLOBAL_MAX_SIZE * 2 + SLM_POOL_SIZE_SOFTMAX_SUM * 2 + SLM_POOL_SIZE_LOCAL_MAX * 2 * 2) + +#define SLM_OFFSET_GLOBAL_MAX 0 +#define SLM_OFFSET_SOFTMAX_SUM (SLM_OFFSET_GLOBAL_MAX + GLOBAL_MAX_SIZE) +#define SLM_OFFSET_O (SLM_OFFSET_SOFTMAX_SUM + SLM_POOL_SIZE_SOFTMAX_SUM) +#define SLM_OFFSET_LOCAL_MAX (GLOBAL_MAX_SIZE + SLM_POOL_SIZE_SOFTMAX_SUM) + +#define P_REDUCE_VEC4 (MATP_PER_LOOP / 4) +#define MAX_PER_LOOP (MATP_PER_LOOP / MATP_REDUCE) +#define SLM_MAX_STRIDE (MATP_REDUCE / 4) +#define SUB_GROUPS_PER_LINE (MATP_PER_LOOP / WARPSIZE / 4) + +shared slm_pool_block { + float slm_pool_o[GLOBAL_MAX_SIZE + SLM_POOL_SIZE_SOFTMAX_SUM + SLM_POOL_SIZE_O]; +} slm_pool_f32; + +shared slm_pool_alias_block { + float16_t slm_pool_pv[GLOBAL_MAX_SIZE * 2 + SLM_POOL_SIZE_SOFTMAX_SUM * 2 + SLM_POOL_SIZE_LOCAL_MAX * 2 * 2 + SLM_POOL_SIZE_P * 2]; +} slm_pool_alias_f16; + +void main() { + const uint lane = gl_SubgroupInvocationID; + const uint v = gl_WorkGroupID.y; + const uint d = gl_WorkGroupID.z; + const uint vWarpIdx = gl_WorkGroupID.x % V_HEAD_GROUPS; + const uint outTokIdx = gl_WorkGroupID.x / V_HEAD_GROUPS; + const uint localLinearId = gl_SubgroupID; + const uint wgLane = localLinearId * WARPSIZE + lane; + const uint splitIdx = localLinearId; + const uint maxDim = (p.kvSeqLen + MATP_REDUCE - 1) / MATP_REDUCE; + const uint pDim = maxDim * MATP_REDUCE; + const uint kvDim = p.kvStride1; + const uint oDim = p.qHead * HEAD_DIM; + const uint offsetBaseP = (d * p.activationLength * p.qHead + v * GQA_RATIO + outTokIdx * N_TOKS_PER_GROUP * p.qHead) * pDim / 4; + const uint offsetBaseMax = (d * p.activationLength * p.qHead + v * GQA_RATIO + outTokIdx * N_TOKS_PER_GROUP * p.qHead) * maxDim; + const uint offsetBaseV = (d * p.batchStrideV + v * p.kvStride2 + vWarpIdx * WARP_V_DIM + splitIdx * TK * kvDim); + const uint offsetSlmP = (SLM_OFFSET_P + wgLane * 4); + const float fp32Min = uintBitsToFloat(0xFEFFFFFF); + const float fp32Max = uintBitsToFloat(0x7EFFFFFF); + uint offsetV = offsetBaseV; + + coopmat sums[ALIGNED_P_M][MAT_O_N]; + f16vec4 pStateTemp[P_LOAD_LOOPS]; + + float fp32CompensationP[P_LOAD_LOOPS]; + + uint loadRowBase[P_LOAD_LOOPS]; + uint loadColBase[P_LOAD_LOOPS]; + float fp32SoftMaxSum[P_LOAD_LOOPS]; + float fp32GlobalMaxP[P_LOAD_LOOPS]; + uint maxRowBase[MAX_LOAD_LOOPS]; + uint maxColBase[MAX_LOAD_LOOPS]; + uint outOffsets[ALIGNED_P_M]; + bool outputMask[ALIGNED_P_M]; + float fp32SinkCoeff[ALIGNED_P_M]; + + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + const uint flatOffset = pm * WARPSIZE + lane; + const uint inGroupTokIdx = flatOffset / GQA_RATIO; + const uint inGroupHeadIdx = flatOffset % GQA_RATIO; + outputMask[pm] = (N_TOKS_PER_GROUP * outTokIdx + inGroupTokIdx < p.activationLength) && (inGroupHeadIdx < GQA_RATIO) && (inGroupTokIdx < N_TOKS_PER_GROUP); + outOffsets[pm] = (inGroupTokIdx * oDim + inGroupHeadIdx * HEAD_DIM) / 4; + if ((0x1 & p.flag) != 0) { + fp32SinkCoeff[pm] = sink_f32[inGroupHeadIdx + v * GQA_RATIO]; + } + } + + [[unroll]] for (uint maxCount = 0; maxCount < MAX_REDUCE_COUNT; maxCount++) { + const uint flatIdx = maxCount * SUBGROUP_COUNT + localLinearId; + const uint rowIdx = flatIdx % GQA_RATIO; + const uint tokIdx = flatIdx / GQA_RATIO; + + if (tokIdx < N_TOKS_PER_GROUP) { + float fp32MaxReduce = fp32Min; + const uint maxOffset = offsetBaseMax + (tokIdx * p.qHead + rowIdx) * maxDim; + [[unroll]] for (uint maxReduce = 0; maxReduce < (maxDim + WARPSIZE - 1) / WARPSIZE; maxReduce++) { + if (maxReduce * WARPSIZE + lane < maxDim) { + fp32MaxReduce = max(fp32MaxReduce, max_f32[maxOffset + maxReduce * WARPSIZE + lane]); + } + } + fp32MaxReduce = subgroupMax(fp32MaxReduce); + if (lane == 0) { + slm_pool_f32.slm_pool_o[SLM_OFFSET_GLOBAL_MAX + maxCount * SUBGROUP_COUNT + localLinearId] = fp32MaxReduce; + } + } else { + if (lane == 0) { + slm_pool_f32.slm_pool_o[SLM_OFFSET_GLOBAL_MAX + maxCount * SUBGROUP_COUNT + localLinearId] = fp32Max; + } + } + } + + barrier(); + + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + const uint flatOffset = (pLoad * GROUPSIZE + wgLane) / P_REDUCE_VEC4; + const uint rowIdxFlat = flatOffset % GQA_RATIO; + const uint tokenIdxFlat = min(flatOffset / GQA_RATIO, N_TOKS_PER_GROUP - 1); + loadColBase[pLoad] = (pLoad * GROUPSIZE + wgLane) % P_REDUCE_VEC4; + loadRowBase[pLoad] = (tokenIdxFlat * p.qHead + rowIdxFlat); + fp32SoftMaxSum[pLoad] = 0.0f; + fp32GlobalMaxP[pLoad] = slm_pool_f32.slm_pool_o[SLM_OFFSET_GLOBAL_MAX + flatOffset]; + } + + [[unroll]] for (uint maxLoad = 0; maxLoad < MAX_LOAD_LOOPS; maxLoad++) { + const uint flatOffset = (maxLoad * GROUPSIZE + wgLane) / MAX_PER_LOOP; + const uint rowIdxFlat = flatOffset % GQA_RATIO; + const uint tokenIdxFlat = min(flatOffset / GQA_RATIO, N_TOKS_PER_GROUP - 1); + maxColBase[maxLoad] = (maxLoad * GROUPSIZE + wgLane) % MAX_PER_LOOP; + maxRowBase[maxLoad] = (tokenIdxFlat * p.qHead + rowIdxFlat); + } + + [[unroll]] for (uint maxLoad = 0; maxLoad < MAX_LOAD_LOOPS; maxLoad++) { + const uint flatMaxOffset = maxRowBase[maxLoad] * maxDim + maxColBase[maxLoad]; + slm_pool_f32.slm_pool_o[SLM_OFFSET_LOCAL_MAX + maxLoad * GROUPSIZE + wgLane] = max_f32[offsetBaseMax + flatMaxOffset]; + maxColBase[maxLoad] = maxColBase[maxLoad] + MATP_PER_LOOP / MATP_REDUCE; + } + + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + const uint flatOffset = loadRowBase[pLoad] * pDim / 4 + loadColBase[pLoad]; + pStateTemp[pLoad] = pStateVec4[offsetBaseP + flatOffset]; + } + + [[unroll]] for (uint n = 0; n < ALIGNED_P_M; n++) { + [[unroll]] for (uint i = 0; i < MAT_O_N; i++) { + sums[n][i] = coopmat(0.0f); + } + } + + barrier(); + + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + const uint maxOffset = (pLoad * GROUPSIZE + wgLane) / SLM_MAX_STRIDE; + if (loadColBase[pLoad] < pDim / 4) { + fp32CompensationP[pLoad] = slm_pool_f32.slm_pool_o[SLM_OFFSET_LOCAL_MAX + maxOffset]; + float pTemp[4] = float[4](pStateTemp[pLoad].x, pStateTemp[pLoad].y, pStateTemp[pLoad].z, pStateTemp[pLoad].w); + float compTemp = exp(fp32CompensationP[pLoad] - fp32GlobalMaxP[pLoad]); + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + pTemp[kk] = pTemp[kk] * compTemp; + fp32SoftMaxSum[pLoad] = fp32SoftMaxSum[pLoad] + pTemp[kk]; + slm_pool_alias_f16.slm_pool_pv[offsetSlmP + pLoad * GROUPSIZE * 4 + kk] = float16_t(pTemp[kk]); + } + } else { + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + slm_pool_alias_f16.slm_pool_pv[offsetSlmP + pLoad * GROUPSIZE * 4 + kk] = float16_t(0.0f); + } + } + + loadColBase[pLoad] = loadColBase[pLoad] + P_REDUCE_VEC4; + } + + const uint loopCount = (p.kvSeqLen + MATP_PER_LOOP - 1) / MATP_PER_LOOP; + + for (uint loop = 0; loop < loopCount; loop++) { + const uint slmPingPongLoad = (loop & 0x1); + const uint slmPingPongStore = ((loop + 1) & 0x1); + + if (loop + 1 < loopCount) { + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + const uint flatOffset = loadRowBase[pLoad] * pDim / 4 + loadColBase[pLoad]; + pStateTemp[pLoad] = pStateVec4[offsetBaseP + flatOffset]; + } + + [[unroll]] for (uint maxLoad = 0; maxLoad < MAX_LOAD_LOOPS; maxLoad++) { + const uint flatMaxOffset = maxRowBase[maxLoad] * maxDim + maxColBase[maxLoad]; + slm_pool_f32.slm_pool_o[SLM_OFFSET_LOCAL_MAX + slmPingPongStore * SLM_POOL_SIZE_LOCAL_MAX + maxLoad * GROUPSIZE + wgLane] = max_f32[offsetBaseMax + flatMaxOffset]; + maxColBase[maxLoad] = maxColBase[maxLoad] + MATP_PER_LOOP / MATP_REDUCE; + } + } + + barrier(); + + { + const uint coopMatOffsetP = SLM_OFFSET_P + slmPingPongLoad * SLM_POOL_SIZE_P + splitIdx * TK; + coopmat matV[MAT_O_N]; + [[unroll]] for (uint cc = 0; cc < MAT_O_N; cc++) { + coopMatLoad( + matV[cc], + vState, + offsetV + TM * cc, + kvDim, + gl_CooperativeMatrixLayoutColumnMajor); + } + [[unroll]] for (uint mo = 0; mo < ALIGNED_P_M; mo++) { + coopmat matP; + coopMatLoad( + matP, + slm_pool_alias_f16.slm_pool_pv, + coopMatOffsetP + mo * TN * MATP_PER_LOOP, + MATP_PER_LOOP, + gl_CooperativeMatrixLayoutColumnMajor); + + [[unroll]] for (uint no = 0; no < MAT_O_N; no++) { + sums[mo][no] = coopMatMulAdd(matV[no], matP, sums[mo][no]); + } + } + } + + offsetV += MATP_PER_LOOP * kvDim; + if (loop * MATP_PER_LOOP + splitIdx * TK >= p.kvSeqLen) { + offsetV = 0; + } + if (loop + 1 < loopCount) { + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + const uint maxOffset = (pLoad * GROUPSIZE + wgLane) / SLM_MAX_STRIDE; + if (loadColBase[pLoad] < pDim / 4) { + fp32CompensationP[pLoad] = slm_pool_f32.slm_pool_o[SLM_OFFSET_LOCAL_MAX + slmPingPongStore * SLM_POOL_SIZE_LOCAL_MAX + maxOffset]; + float pTemp[4] = float[4](pStateTemp[pLoad].x, pStateTemp[pLoad].y, pStateTemp[pLoad].z, pStateTemp[pLoad].w); + float compTemp = exp(fp32CompensationP[pLoad] - fp32GlobalMaxP[pLoad]); + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + pTemp[kk] = pTemp[kk] * compTemp; + fp32SoftMaxSum[pLoad] = fp32SoftMaxSum[pLoad] + pTemp[kk]; + slm_pool_alias_f16.slm_pool_pv[offsetSlmP + slmPingPongStore * SLM_POOL_SIZE_P + pLoad * GROUPSIZE * 4 + kk] = float16_t(pTemp[kk]); + } + } else { + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + slm_pool_alias_f16.slm_pool_pv[offsetSlmP + slmPingPongStore * SLM_POOL_SIZE_P + pLoad * GROUPSIZE * 4 + kk] = float16_t(0.0f); + } + } + loadColBase[pLoad] = loadColBase[pLoad] + P_REDUCE_VEC4; + } + } + } + + barrier(); + + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + fp32SoftMaxSum[pLoad] = subgroupAdd(fp32SoftMaxSum[pLoad]); + } + + [[unroll]] for (uint mo = 0; mo < ALIGNED_P_M; mo++) { + [[unroll]] for (uint no = 0; no < MAT_O_N; no++) { + coopMatStore( + sums[mo][no], + slm_pool_f32.slm_pool_o, + SLM_OFFSET_O + mo * TN * WARP_V_DIM + TM * no + localLinearId * ALIGNED_P_M * TN * WARP_V_DIM, + WARP_V_DIM, + gl_CooperativeMatrixLayoutColumnMajor); + } + } + + [[unroll]] for (uint pLoad = 0; pLoad < P_LOAD_LOOPS; pLoad++) { + slm_pool_f32.slm_pool_o[SLM_OFFSET_SOFTMAX_SUM + pLoad * SUBGROUP_COUNT + localLinearId] = fp32SoftMaxSum[pLoad]; + } + + barrier(); + + if (localLinearId == 1) { + const uint sumBase = SLM_OFFSET_SOFTMAX_SUM + lane * SUB_GROUPS_PER_LINE; + float sumTemp[ALIGNED_P_M][SUB_GROUPS_PER_LINE]; + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + [[unroll]] for (uint reduce = 0; reduce < SUB_GROUPS_PER_LINE; reduce++) { + sumTemp[pm][reduce] = slm_pool_f32.slm_pool_o[sumBase + pm * WARPSIZE * SUB_GROUPS_PER_LINE + reduce]; + } + } + + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + [[unroll]] for (uint reduce = 1; reduce < SUB_GROUPS_PER_LINE; reduce++) { + sumTemp[pm][0] = sumTemp[pm][0] + sumTemp[pm][reduce]; + } + } + + if ((0x1 & p.flag) != 0) { + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + float fp32GlobalMax = slm_pool_f32.slm_pool_o[SLM_OFFSET_GLOBAL_MAX + pm * WARPSIZE + lane]; + float sinkCompensation = fp32GlobalMax - fp32SinkCoeff[pm]; + sinkCompensation = exp(sinkCompensation); + float softmaxSumTemp = sumTemp[pm][0] * sinkCompensation; + sumTemp[pm][0] = sumTemp[pm][0] + 1.0f / sinkCompensation; + sumTemp[pm][0] = 1.0f / sumTemp[pm][0]; + sinkCompensation = sinkCompensation / (1.0f + softmaxSumTemp); + sumTemp[pm][0] = fp32GlobalMax < fp32SinkCoeff[pm] ? sinkCompensation : sumTemp[pm][0]; + slm_pool_f32.slm_pool_o[SLM_OFFSET_SOFTMAX_SUM + pm * WARPSIZE + lane] = sumTemp[pm][0]; + } + } else { + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + slm_pool_f32.slm_pool_o[SLM_OFFSET_SOFTMAX_SUM + pm * WARPSIZE + lane] = 1.0f / sumTemp[pm][0]; + } + } + } + + [[unroll]] for (uint reduce = 2; reduce < SPLIT_P_GROUPS; reduce = reduce << 1 ) { + const uint stride = (reduce >> 1) * ALIGNED_P_M * TN * MAT_O_N * TM; + if ((localLinearId % reduce) == 0) { + const uint reduceBase = localLinearId * ALIGNED_P_M * TN * MAT_O_N * TM + SLM_OFFSET_O; + float sumTemp0[4]; + float sumTemp1[4]; + const uint reduceVec4Count = ALIGNED_P_M * TN * MAT_O_N * TM / 4 / WARPSIZE; + [[unroll]] for (uint totalLoads = 0; totalLoads < reduceVec4Count; totalLoads++) { + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + sumTemp0[kk] = slm_pool_f32.slm_pool_o[reduceBase + totalLoads * 4 * WARPSIZE + 4 * lane + kk]; + sumTemp1[kk] = slm_pool_f32.slm_pool_o[reduceBase + stride + totalLoads * 4 * WARPSIZE + 4 * lane + kk]; + } + + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + sumTemp0[kk] = sumTemp0[kk] + sumTemp1[kk]; + } + + [[unroll]] for (uint kk = 0; kk < 4; kk++) { + slm_pool_f32.slm_pool_o[reduceBase + totalLoads * 4 * WARPSIZE + 4 * lane + kk] = sumTemp0[kk]; + } + } + } + barrier(); + } + + if (localLinearId == 0) { + const uint slmBase0 = SLM_OFFSET_O + lane * WARP_V_DIM; + const uint slmBase1 = slmBase0 + SPLIT_P_GROUPS / 2 * ALIGNED_P_M * TN * MAT_O_N * TM; + + const uint offsetOutBase = (d * p.batchStrideO + vWarpIdx * WARP_V_DIM + v * GQA_RATIO * HEAD_DIM + outTokIdx * oDim * N_TOKS_PER_GROUP) / 4; + float fp32SoftMaxMul[ALIGNED_P_M]; + float fp32Output[ALIGNED_P_M][4]; + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + fp32SoftMaxMul[pm] = slm_pool_f32.slm_pool_o[SLM_OFFSET_SOFTMAX_SUM + pm * WARPSIZE + lane]; + } + + [[unroll]] for (uint vg = 0; vg < WARP_V_DIM / 4; vg++) { + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + [[unroll]] for (uint vc = 0; vc < 4; vc++) { + fp32Output[pm][vc] = slm_pool_f32.slm_pool_o[slmBase0 + pm * WARPSIZE * WARP_V_DIM + vg * 4 + vc] * fp32SoftMaxMul[pm]; + fp32Output[pm][vc] = fp32Output[pm][vc] + slm_pool_f32.slm_pool_o[slmBase1 + pm * WARPSIZE * WARP_V_DIM + vg * 4 + vc] * fp32SoftMaxMul[pm]; + } + } + + [[unroll]] for (uint pm = 0; pm < ALIGNED_P_M; pm++) { + if (outputMask[pm] == true) { + out_f32_vec4[offsetOutBase + outOffsets[pm] + vg] = vec4(fp32Output[pm][0], fp32Output[pm][1], fp32Output[pm][2], fp32Output[pm][3]); + } + } + } + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl index 8ba4725f..4fcf7c1f 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl @@ -121,25 +121,25 @@ layout (binding = 1) readonly buffer K_PACKED_Q5_1_P32 { block_q5_1_packed32 dat FLOAT_TYPEV4 dequantize4(uint ib, uint iqs, uint a_offset, uint binding_idx) { if (binding_idx == BINDING_IDX_K) { switch (FaTypeK) { - case FA_TYPE_F32: FA_DEQUANT4_F32 (k_packed_f32) - case FA_TYPE_Q4_0: FA_DEQUANT4_Q4_0(k_packed_q4_0) - case FA_TYPE_Q4_1: FA_DEQUANT4_Q4_1(k_packed_q4_1) - case FA_TYPE_Q5_0: FA_DEQUANT4_Q5_0(k_packed_q5_0) - case FA_TYPE_Q5_1: FA_DEQUANT4_Q5_1(k_packed_q5_1) - case FA_TYPE_Q8_0: FA_DEQUANT4_Q8_0(k_packed_q8_0) - case FA_TYPE_IQ4_NL: FA_DEQUANT4_IQ4_NL(k_packed_iq4_nl) - case FA_TYPE_BF16: FA_DEQUANT4_BF16(k_packed_bf16) + case GGML_TYPE_F32: FA_DEQUANT4_F32 (k_packed_f32) + case GGML_TYPE_Q4_0: FA_DEQUANT4_Q4_0(k_packed_q4_0) + case GGML_TYPE_Q4_1: FA_DEQUANT4_Q4_1(k_packed_q4_1) + case GGML_TYPE_Q5_0: FA_DEQUANT4_Q5_0(k_packed_q5_0) + case GGML_TYPE_Q5_1: FA_DEQUANT4_Q5_1(k_packed_q5_1) + case GGML_TYPE_Q8_0: FA_DEQUANT4_Q8_0(k_packed_q8_0) + case GGML_TYPE_IQ4_NL: FA_DEQUANT4_IQ4_NL(k_packed_iq4_nl) + case GGML_TYPE_BF16: FA_DEQUANT4_BF16(k_packed_bf16) } } else { switch (FaTypeV) { - case FA_TYPE_F32: FA_DEQUANT4_F32 (v_packed_f32) - case FA_TYPE_Q4_0: FA_DEQUANT4_Q4_0(v_packed_q4_0) - case FA_TYPE_Q4_1: FA_DEQUANT4_Q4_1(v_packed_q4_1) - case FA_TYPE_Q5_0: FA_DEQUANT4_Q5_0(v_packed_q5_0) - case FA_TYPE_Q5_1: FA_DEQUANT4_Q5_1(v_packed_q5_1) - case FA_TYPE_Q8_0: FA_DEQUANT4_Q8_0(v_packed_q8_0) - case FA_TYPE_IQ4_NL: FA_DEQUANT4_IQ4_NL(v_packed_iq4_nl) - case FA_TYPE_BF16: FA_DEQUANT4_BF16(v_packed_bf16) + case GGML_TYPE_F32: FA_DEQUANT4_F32 (v_packed_f32) + case GGML_TYPE_Q4_0: FA_DEQUANT4_Q4_0(v_packed_q4_0) + case GGML_TYPE_Q4_1: FA_DEQUANT4_Q4_1(v_packed_q4_1) + case GGML_TYPE_Q5_0: FA_DEQUANT4_Q5_0(v_packed_q5_0) + case GGML_TYPE_Q5_1: FA_DEQUANT4_Q5_1(v_packed_q5_1) + case GGML_TYPE_Q8_0: FA_DEQUANT4_Q8_0(v_packed_q8_0) + case GGML_TYPE_IQ4_NL: FA_DEQUANT4_IQ4_NL(v_packed_iq4_nl) + case GGML_TYPE_BF16: FA_DEQUANT4_BF16(v_packed_bf16) } } return FLOAT_TYPEV4(0); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl index 6bf10a7c..49900aa5 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl @@ -4,20 +4,20 @@ int32_t get_k_qs(uint ib, uint iqs, uint a_offset) { switch (FaTypeK) { - case FA_TYPE_Q4_0: { + case GGML_TYPE_Q4_0: { uint vui = pack32(u16vec2(k_packed_q4_0.data[a_offset + ib].qs[(iqs & 0xF) / 2 + 0], k_packed_q4_0.data[a_offset + ib].qs[(iqs & 0xF) / 2 + 1])); uint shift = (iqs & 0x10) >> 2; vui >>= shift; return int32_t(vui & 0x0F0F0F0F); } - case FA_TYPE_Q4_1: { // uses packed32 alias + case GGML_TYPE_Q4_1: { // uses packed32 alias uint vui = k_packed_q4_1_p32.data[a_offset + ib].qs[(iqs & 0xF) / 4]; uint shift = (iqs & 0x10) >> 2; vui >>= shift; return int32_t(vui & 0x0F0F0F0F); } - case FA_TYPE_Q5_0: { + case GGML_TYPE_Q5_0: { uint vui = pack32(u16vec2(k_packed_q5_0.data[a_offset + ib].qs[(iqs & 0xF) / 2 + 0], k_packed_q5_0.data[a_offset + ib].qs[(iqs & 0xF) / 2 + 1])); uint qh = pack32(u16vec2(k_packed_q5_0.data[a_offset + ib].qh[0], @@ -27,7 +27,7 @@ int32_t get_k_qs(uint ib, uint iqs, uint a_offset) { uint qh_bits = (qh >> iqs) & 0xF; return int32_t(vui & 0x0F0F0F0F) | int32_t((qh_bits * 0x02040810u) & 0x10101010u); } - case FA_TYPE_Q5_1: { // qs via packed32, qh via packed16 + case GGML_TYPE_Q5_1: { // qs via packed32, qh via packed16 uint vui = k_packed_q5_1_p32.data[a_offset + ib].qs[(iqs & 0xF) / 4]; uint qh = k_packed_q5_1.data[a_offset + ib].qh; uint shift = (iqs & 0x10) >> 2; @@ -35,7 +35,7 @@ int32_t get_k_qs(uint ib, uint iqs, uint a_offset) { uint qh_bits = (qh >> iqs) & 0xF; return int32_t(vui & 0x0F0F0F0F) | int32_t((qh_bits * 0x02040810u) & 0x10101010u); } - case FA_TYPE_Q8_0: { + case GGML_TYPE_Q8_0: { return pack32(i16vec2(k_packed_q8_0.data[a_offset + ib].qs[iqs / 2], k_packed_q8_0.data[a_offset + ib].qs[iqs / 2 + 1])); } @@ -47,11 +47,11 @@ int32_t get_k_qs(uint ib, uint iqs, uint a_offset) { // return (d, 0) so call sites always see the same shape. FLOAT_TYPEV2 get_k_scale(uint ib, uint a_offset) { switch (FaTypeK) { - case FA_TYPE_Q4_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q4_0.data[a_offset + ib].d), 0.0); - case FA_TYPE_Q4_1: return FLOAT_TYPEV2(k_packed_q4_1_p32.data[a_offset + ib].dm); - case FA_TYPE_Q5_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q5_0.data[a_offset + ib].d), 0.0); - case FA_TYPE_Q5_1: return FLOAT_TYPEV2(k_packed_q5_1_p32.data[a_offset + ib].dm); - case FA_TYPE_Q8_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q8_0.data[a_offset + ib].d), 0.0); + case GGML_TYPE_Q4_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q4_0.data[a_offset + ib].d), 0.0); + case GGML_TYPE_Q4_1: return FLOAT_TYPEV2(k_packed_q4_1_p32.data[a_offset + ib].dm); + case GGML_TYPE_Q5_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q5_0.data[a_offset + ib].d), 0.0); + case GGML_TYPE_Q5_1: return FLOAT_TYPEV2(k_packed_q5_1_p32.data[a_offset + ib].dm); + case GGML_TYPE_Q8_0: return FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q8_0.data[a_offset + ib].d), 0.0); default: return FLOAT_TYPEV2(0); } } @@ -61,16 +61,16 @@ void k_block_to_shmem(const uint buf_ib, const uint global_ib, const uint iqs, c // explicit casts. The bit pattern is what we care about here -- the actual // signed/unsigned interpretation happens downstream in the dot product. switch (FaTypeK) { - case FA_TYPE_Q4_0: { + case GGML_TYPE_Q4_0: { kblocksh[buf_ib].qs[iqs] = int32_t(pack32(u16vec2(k_packed_q4_0.data[a_offset + global_ib].qs[iqs * 2], k_packed_q4_0.data[a_offset + global_ib].qs[iqs * 2 + 1]))); break; } - case FA_TYPE_Q4_1: { + case GGML_TYPE_Q4_1: { kblocksh[buf_ib].qs[iqs] = int32_t(k_packed_q4_1_p32.data[a_offset + global_ib].qs[iqs]); break; } - case FA_TYPE_Q5_0: { + case GGML_TYPE_Q5_0: { kblocksh[buf_ib].qs[iqs] = int32_t(pack32(u16vec2(k_packed_q5_0.data[a_offset + global_ib].qs[iqs * 2], k_packed_q5_0.data[a_offset + global_ib].qs[iqs * 2 + 1]))); if (iqs == 0) { @@ -79,14 +79,14 @@ void k_block_to_shmem(const uint buf_ib, const uint global_ib, const uint iqs, c } break; } - case FA_TYPE_Q5_1: { + case GGML_TYPE_Q5_1: { kblocksh[buf_ib].qs[iqs] = int32_t(k_packed_q5_1_p32.data[a_offset + global_ib].qs[iqs]); if (iqs == 0) { kblocksh[buf_ib].qh = k_packed_q5_1.data[a_offset + global_ib].qh; } break; } - case FA_TYPE_Q8_0: { + case GGML_TYPE_Q8_0: { kblocksh[buf_ib].qs[iqs] = pack32(i16vec2(k_packed_q8_0.data[a_offset + global_ib].qs[iqs * 2], k_packed_q8_0.data[a_offset + global_ib].qs[iqs * 2 + 1])); break; @@ -96,11 +96,11 @@ void k_block_to_shmem(const uint buf_ib, const uint global_ib, const uint iqs, c if (iqs == 0) { // Q4_0/Q5_0/Q8_0 store dm.x = d; Q4_1/Q5_1 store dm = (d, m) pair. switch (FaTypeK) { - case FA_TYPE_Q4_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q4_0.data[a_offset + global_ib].d), 0.0); break; - case FA_TYPE_Q4_1: kblocksh[buf_ib].dm = FLOAT_TYPEV2(k_packed_q4_1_p32.data[a_offset + global_ib].dm); break; - case FA_TYPE_Q5_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q5_0.data[a_offset + global_ib].d), 0.0); break; - case FA_TYPE_Q5_1: kblocksh[buf_ib].dm = FLOAT_TYPEV2(k_packed_q5_1_p32.data[a_offset + global_ib].dm); break; - case FA_TYPE_Q8_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q8_0.data[a_offset + global_ib].d), 0.0); break; + case GGML_TYPE_Q4_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q4_0.data[a_offset + global_ib].d), 0.0); break; + case GGML_TYPE_Q4_1: kblocksh[buf_ib].dm = FLOAT_TYPEV2(k_packed_q4_1_p32.data[a_offset + global_ib].dm); break; + case GGML_TYPE_Q5_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q5_0.data[a_offset + global_ib].d), 0.0); break; + case GGML_TYPE_Q5_1: kblocksh[buf_ib].dm = FLOAT_TYPEV2(k_packed_q5_1_p32.data[a_offset + global_ib].dm); break; + case GGML_TYPE_Q8_0: kblocksh[buf_ib].dm = FLOAT_TYPEV2(FLOAT_TYPE(k_packed_q8_0.data[a_offset + global_ib].d), 0.0); break; } } } @@ -121,31 +121,31 @@ struct fa_k_qs_block8 { fa_k_qs_block8 get_k_qs_block8(uint ib, uint a_offset) { fa_k_qs_block8 r; uint qh = 0; - if (FaTypeK == FA_TYPE_Q5_0) { + if (FaTypeK == GGML_TYPE_Q5_0) { qh = pack32(u16vec2(k_packed_q5_0.data[a_offset + ib].qh[0], k_packed_q5_0.data[a_offset + ib].qh[1])); - } else if (FaTypeK == FA_TYPE_Q5_1) { + } else if (FaTypeK == GGML_TYPE_Q5_1) { qh = k_packed_q5_1.data[a_offset + ib].qh; } - const bool has_qh = (FaTypeK == FA_TYPE_Q5_0) || (FaTypeK == FA_TYPE_Q5_1); + const bool has_qh = (FaTypeK == GGML_TYPE_Q5_0) || (FaTypeK == GGML_TYPE_Q5_1); [[unroll]] for (uint32_t d = 0; d < 4; d++) { uint vui = 0; switch (FaTypeK) { - case FA_TYPE_Q4_0: { // packed16 + case GGML_TYPE_Q4_0: { // packed16 vui = pack32(u16vec2(k_packed_q4_0.data[a_offset + ib].qs[d * 2 + 0], k_packed_q4_0.data[a_offset + ib].qs[d * 2 + 1])); break; } - case FA_TYPE_Q4_1: { // packed32 alias + case GGML_TYPE_Q4_1: { // packed32 alias vui = k_packed_q4_1_p32.data[a_offset + ib].qs[d]; break; } - case FA_TYPE_Q5_0: { // packed16 + case GGML_TYPE_Q5_0: { // packed16 vui = pack32(u16vec2(k_packed_q5_0.data[a_offset + ib].qs[d * 2 + 0], k_packed_q5_0.data[a_offset + ib].qs[d * 2 + 1])); break; } - case FA_TYPE_Q5_1: { // packed32 alias + case GGML_TYPE_Q5_1: { // packed32 alias vui = k_packed_q5_1_p32.data[a_offset + ib].qs[d]; break; } @@ -164,21 +164,21 @@ fa_k_qs_block8 get_k_qs_block8(uint ib, uint a_offset) { int32_t get_k_qs_shmem(const uint buf_ib, const uint pos) { switch (FaTypeK) { - case FA_TYPE_Q4_0: - case FA_TYPE_Q4_1: { + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: { uint sub = pos % 4; uint shift = ((pos % 8) >= 4) ? 4u : 0u; return int32_t((uint(kblocksh[buf_ib].qs[sub]) >> shift) & 0x0F0F0F0Fu); } - case FA_TYPE_Q5_0: - case FA_TYPE_Q5_1: { + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q5_1: { uint sub = pos % 4; uint shift = ((pos % 8) >= 4) ? 4u : 0u; int32_t result = int32_t((uint(kblocksh[buf_ib].qs[sub]) >> shift) & 0x0F0F0F0Fu); uint qh_bits = (kblocksh[buf_ib].qh >> (pos * 4u)) & 0xFu; return result | int32_t((qh_bits * 0x02040810u) & 0x10101010u); } - case FA_TYPE_Q8_0: { + case GGML_TYPE_Q8_0: { return kblocksh[buf_ib].qs[pos]; } default: return 0; @@ -187,10 +187,10 @@ int32_t get_k_qs_shmem(const uint buf_ib, const uint pos) { ACC_TYPE k_dot_correction(const uint qib, const ACC_TYPEV2 k_dm) { switch (FaTypeK) { - case FA_TYPE_Q4_0: return -ACC_TYPE(8.0) * ACC_TYPE(Qf[qib].ds.y) * k_dm.x; - case FA_TYPE_Q5_0: return -ACC_TYPE(16.0) * ACC_TYPE(Qf[qib].ds.y) * k_dm.x; - case FA_TYPE_Q4_1: - case FA_TYPE_Q5_1: return ACC_TYPE(Qf[qib].ds.y) * k_dm.y; + case GGML_TYPE_Q4_0: return -ACC_TYPE(8.0) * ACC_TYPE(Qf[qib].ds.y) * k_dm.x; + case GGML_TYPE_Q5_0: return -ACC_TYPE(16.0) * ACC_TYPE(Qf[qib].ds.y) * k_dm.x; + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_1: return ACC_TYPE(Qf[qib].ds.y) * k_dm.y; default: return ACC_TYPE(0.0); } } diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_sparse_compact.comp b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_sparse_compact.comp new file mode 100644 index 00000000..3d313626 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_sparse_compact.comp @@ -0,0 +1,102 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : enable +#extension GL_EXT_shader_16bit_storage : require +#extension GL_EXT_shader_explicit_arithmetic_types_int32 : require +#ifdef USE_SUBGROUPS +#extension GL_KHR_shader_subgroup_basic : require +#extension GL_KHR_shader_subgroup_ballot : require +#endif + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; +layout(constant_id = 0) const uint BLOCK_SIZE = 128; +layout(constant_id = 1) const uint NUM_SUBGROUPS = 1; + +layout (binding = 0) readonly buffer M {float16_t data_m[];}; +layout (binding = 1) writeonly buffer I {int32_t data_i[];}; + +layout (push_constant) uniform parameter { + uint KV; + uint nem1; + uint nem2; + uint nbm1; + uint nbm2; + uint nbm3; + uint n_kv_max; +} p; + +#ifdef USE_SUBGROUPS +shared uvec4 ballots_sh[NUM_SUBGROUPS]; +#else +shared uint scan[BLOCK_SIZE]; +#endif + +// One workgroup per mask row: compact the finite-mask KV positions into a +// per-row index list of length n_kv_max, -1 padded. Emitted in ascending KV +// order so the downstream attention accumulation is deterministic. +void main() { + const uint i1 = gl_WorkGroupID.x; + const uint i2 = gl_WorkGroupID.y; + const uint i3 = gl_WorkGroupID.z; + const uint tid = gl_LocalInvocationIndex; + + const uint m_base = i3 * p.nbm3 + i2 * p.nbm2 + i1 * p.nbm1; + const uint out_base = ((i3 * p.nem2 + i2) * p.nem1 + i1) * p.n_kv_max; + + uint base = 0; + for (uint chunk = 0; chunk < p.KV; chunk += BLOCK_SIZE) { + const uint k = chunk + tid; + bool selected = false; + if (k < p.KV) { + const float v = float(data_m[m_base + k]); + selected = !isinf(v) && !isnan(v); + } + +#ifdef USE_SUBGROUPS + const uvec4 ballot = subgroupBallot(selected); + if (subgroupElect()) { + ballots_sh[gl_SubgroupID] = ballot; + } + barrier(); + + uint subgroup_base = 0; + uint total = 0; + [[unroll]] for (uint s = 0; s < gl_NumSubgroups; ++s) { + if (s == gl_SubgroupID) { + subgroup_base = total; + } + total += subgroupBallotBitCount(ballots_sh[s]); + } + barrier(); + + const uint slot = base + subgroup_base + subgroupBallotExclusiveBitCount(ballot); +#else + // Hillis-Steele inclusive prefix sum over the workgroup. + scan[tid] = selected ? 1u : 0u; + barrier(); + for (uint off = 1; off < BLOCK_SIZE; off <<= 1) { + uint add = 0; + if (tid >= off) { + add = scan[tid - off]; + } + barrier(); + scan[tid] += add; + barrier(); + } + + const uint inclusive = scan[tid]; + const uint total = scan[BLOCK_SIZE - 1]; + const uint slot = base + inclusive - 1u; +#endif + + if (selected && slot < p.n_kv_max) { + data_i[out_base + slot] = int32_t(k); + } + base += total; + barrier(); + } + + for (uint s = min(base, p.n_kv_max) + tid; s < p.n_kv_max; s += BLOCK_SIZE) { + data_i[out_base + s] = int32_t(-1); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.glsl index 9d4176f3..e13de9a0 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.glsl @@ -1,6 +1,8 @@ #extension GL_EXT_shader_16bit_storage : require #extension GL_EXT_control_flow_attributes : require +#include "utils.glsl" + layout (push_constant) uniform parameter { uint ne; @@ -32,18 +34,6 @@ uint get_idx() { uint get_aoffset() { return p.misalign_offsets >> 16; } uint get_doffset() { return p.misalign_offsets & 0xFFFF; } -// see init_fastdiv_values in ggml-vulkan.cpp -uint fastdiv(uint n, uint mp, uint L) { - uint msbs, lsbs; - // msbs = mulhi(n, mp) - umulExtended(n, mp, msbs, lsbs); - return (msbs + n) >> L; -} - -uint fastdiv_L(uint packed, uint slot) { - return (packed >> (slot * 8)) & 0x3Fu; -} - uint src0_idx(uint idx) { const uint i03 = fastdiv(idx, p.ne0_012mp, fastdiv_L(p.ne0_Ls, 0)); const uint i03_offset = i03 * p.ne02*p.ne01*p.ne00; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_quant.comp b/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_quant.comp index 9dba437e..19af30ac 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_quant.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_quant.comp @@ -27,10 +27,10 @@ void main() { const uint i11 = gid_z / p.ne12; const uint i12 = gid_z % p.ne12; - const uint i01 = data_b[i10*p.nb10 + i11*p.nb11 + i12*p.nb12]; + const uint i01 = data_b[get_boffset() + i10*p.nb10 + i11*p.nb11 + i12*p.nb12]; - const uint a_offset = i01*p.nb01 + i11*p.nb02 + i12*p.nb03; - const uint d_offset = i10*p.nb21 + i11*p.nb22 + i12*p.nb23; + const uint a_offset = get_aoffset() + i01*p.nb01 + i11*p.nb02 + i12*p.nb03; + const uint d_offset = get_doffset() + i10*p.nb21 + i11*p.nb22 + i12*p.nb23; const uint ib = a_offset + i00/QUANT_K; // block index const uint iqs = (i00%QUANT_K)/QUANT_R; // quant index diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/ggml_type_ids.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/ggml_type_ids.glsl new file mode 100644 index 00000000..0f10c733 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/ggml_type_ids.glsl @@ -0,0 +1,34 @@ +#if !defined(GGML_TYPE_IDS_COMP) +#define GGML_TYPE_IDS_COMP + +// ggml_type enum values — must match ggml.h +#define GGML_TYPE_F32 0u +#define GGML_TYPE_F16 1u +#define GGML_TYPE_Q4_0 2u +#define GGML_TYPE_Q4_1 3u +#define GGML_TYPE_Q5_0 6u +#define GGML_TYPE_Q5_1 7u +#define GGML_TYPE_Q8_0 8u +#define GGML_TYPE_Q2_K 10u +#define GGML_TYPE_Q3_K 11u +#define GGML_TYPE_Q4_K 12u +#define GGML_TYPE_Q5_K 13u +#define GGML_TYPE_Q6_K 14u +#define GGML_TYPE_IQ2_XXS 16u +#define GGML_TYPE_IQ2_XS 17u +#define GGML_TYPE_IQ3_XXS 18u +#define GGML_TYPE_IQ1_S 19u +#define GGML_TYPE_IQ4_NL 20u +#define GGML_TYPE_IQ3_S 21u +#define GGML_TYPE_IQ2_S 22u +#define GGML_TYPE_IQ4_XS 23u +#define GGML_TYPE_IQ1_M 29u +#define GGML_TYPE_BF16 30u +#define GGML_TYPE_TQ1_0 34u +#define GGML_TYPE_TQ2_0 35u +#define GGML_TYPE_MXFP4 39u +#define GGML_TYPE_NVFP4 40u +#define GGML_TYPE_Q1_0 41u +#define GGML_TYPE_Q2_0 42u + +#endif // !defined(GGML_TYPE_IDS_COMP) diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl index c3cae736..fc2951ec 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl @@ -1,5 +1,7 @@ #extension GL_EXT_shader_16bit_storage : require +#include "utils.glsl" + layout(local_size_x = 512, local_size_y = 1, local_size_z = 1) in; @@ -39,9 +41,3 @@ uint get_aoffset() { return p.misalign_offsets >> 16; } uint get_boffset() { return (p.misalign_offsets >> 8) & 0xFF; } uint get_doffset() { return p.misalign_offsets & 0xFF; } -// see init_fastdiv_values in ggml-vulkan.cpp -uint fastdiv(uint n, uint mp, uint L) { - uint msbs, lsbs; - umulExtended(n, mp, msbs, lsbs); - return (msbs + n) >> L; -} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp b/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp index f4130d22..ea77a3d7 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp @@ -31,7 +31,7 @@ layout (binding = 0) readonly buffer X {A_TYPE data_a[];}; layout (binding = 1) writeonly buffer D {D_TYPE data_d[];}; #if BDA -layout (buffer_reference) buffer D_ptr {D_TYPE d;}; +layout (buffer_reference, buffer_reference_align = D_SIZE) buffer D_ptr {D_TYPE d;}; #endif void im2col(const uint ow, const uint z_idx) { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp b/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp index 93f61fd8..64ae7e4f 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp @@ -50,7 +50,7 @@ layout (binding = 0) readonly buffer X {A_TYPE data_a[];}; layout (binding = 1) writeonly buffer D {D_TYPE data_d[];}; #if BDA -layout (buffer_reference) buffer D_ptr {D_TYPE d;}; +layout (buffer_reference, buffer_reference_align = D_SIZE) buffer D_ptr {D_TYPE d;}; #endif void main() { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/iq_shmem_init.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/iq_shmem_init.glsl new file mode 100644 index 00000000..12e50ee9 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/iq_shmem_init.glsl @@ -0,0 +1,2 @@ +void init_iq_shmem(uvec3 wgsize) { +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/lightning_indexer.comp b/ggml/src/ggml-vulkan/vulkan-shaders/lightning_indexer.comp new file mode 100644 index 00000000..9b34d836 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/lightning_indexer.comp @@ -0,0 +1,151 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : require +#extension GL_EXT_shader_16bit_storage : require +#extension GL_EXT_shader_explicit_arithmetic_types_float16 : require +#extension GL_KHR_shader_subgroup_basic : enable +#if USE_SUBGROUP_ADD +#extension GL_KHR_shader_subgroup_arithmetic : enable +#endif + +#define BINDING_IDX_K 0u + +#include "types.glsl" +#include "fa_types.glsl" +#define FaTypeV GGML_TYPE_F32 + +layout(constant_id = 0) const uint FaTypeK = GGML_TYPE_F32; +layout(constant_id = 1) const uint FaBlockBytesK = 4; +layout(constant_id = 2) const uint SUBGROUP_SIZE = 32; + +#include "flash_attn_dequant.glsl" + +// one workgroup computes one output element, one invocation per head element +#define HEAD_SIZE 128 + +layout(local_size_x = HEAD_SIZE, local_size_y = 1, local_size_z = 1) in; + +layout(binding = 0) readonly buffer QBuf { float q[]; }; +layout(binding = 1) readonly buffer KBufF16 { float16_t k_f16[]; }; +layout(binding = 1) readonly buffer KBufF32 { float k_f32[]; }; +layout(binding = 1) readonly buffer KBufBF16 { uint16_t k_bf16[]; }; +layout(binding = 2) readonly buffer WBuf { float weights[]; }; +layout(binding = 3) readonly buffer MBuf { float16_t mask[]; }; +layout(binding = 4) writeonly buffer DstBuf { float dst[]; }; + +layout(push_constant) uniform PushConstants { + uint n_kv; + uint n_heads; + uint n_tokens; + uint n_streams; + uint n_masks; + uint dispatch_x; + uint q_nb1; + uint q_nb2; + uint q_nb3; + uint k_nb2; + uint k_nb3; + uint w_nb1; + uint w_nb3; + uint m_nb1; + uint m_nb3; + uint d_nb1; + uint d_nb3; +}; + +shared float k_row[HEAD_SIZE]; + +#if USE_SUBGROUP_ADD +shared float sg_partials[HEAD_SIZE / SUBGROUP_SIZE]; +#else +shared float partials[HEAD_SIZE]; +#endif + +void main() { + const uint tid = gl_LocalInvocationID.x; + const uint output_idx = gl_WorkGroupID.y * dispatch_x + gl_WorkGroupID.x; + const uint n_outputs = n_kv * n_tokens * n_streams; + + if (fa_type_needs_shmem(FaTypeK)) { + init_iq_shmem(gl_WorkGroupSize); + } + + if (output_idx >= n_outputs) { + return; + } + + const uint ik = output_idx % n_kv; + const uint ts = output_idx / n_kv; + const uint t = ts % n_tokens; + const uint s = ts / n_tokens; + const uint k_offset = ik * k_nb2 + s * k_nb3; + + // k strides come in as bytes, so scale them down to the view being indexed + const uint k_block_elems = fa_block_elems(FaTypeK); + const uint k_elem_bytes = FaBlockBytesK / k_block_elems; + + if (FaTypeK == GGML_TYPE_F16) { + k_row[tid] = float(k_f16[k_offset / k_elem_bytes + tid]); + } else if (FaTypeK == GGML_TYPE_F32) { + k_row[tid] = k_f32[k_offset / k_elem_bytes + tid]; + } else if (FaTypeK == GGML_TYPE_BF16) { + k_row[tid] = bf16_to_fp32(uint(k_bf16[k_offset / k_elem_bytes + tid])); + } else if (4 * tid < HEAD_SIZE) { + const uint coord = 4 * tid; + const uint ib = coord / k_block_elems; + const uint iqs = coord % k_block_elems; + const vec4 values = dequantize4(ib, iqs, k_offset / FaBlockBytesK, BINDING_IDX_K); + k_row[coord + 0] = values.x; + k_row[coord + 1] = values.y; + k_row[coord + 2] = values.z; + k_row[coord + 3] = values.w; + } + barrier(); + + const float k_val = k_row[tid]; + + float score = 0.0; + for (uint h = 0; h < n_heads; ++h) { + const float prod = q[h * q_nb1 + t * q_nb2 + s * q_nb3 + tid] * k_val; + +#if USE_SUBGROUP_ADD + const float sg_sum = subgroupAdd(prod); + if (gl_SubgroupInvocationID == 0) { + sg_partials[gl_SubgroupID] = sg_sum; + } + barrier(); + + if (tid == 0) { + float sum = 0.0; + [[unroll]] for (uint i = 0; i < HEAD_SIZE / SUBGROUP_SIZE; ++i) { + sum += sg_partials[i]; + } + score += max(sum, 0.0) * weights[h + t * w_nb1 + s * w_nb3]; + } + // the reads above must complete before the next iteration overwrites sg_partials + barrier(); +#else + partials[tid] = prod; + barrier(); + + [[unroll]] for (uint stride = HEAD_SIZE / 2; stride > 0; stride >>= 1) { + if (tid < stride) { + partials[tid] += partials[tid + stride]; + } + barrier(); + } + + if (tid == 0) { + score += max(partials[0], 0.0) * weights[h + t * w_nb1 + s * w_nb3]; + } + // the read of partials[0] above must complete before the next iteration + // overwrites partials[tid] + barrier(); +#endif + } + + if (tid == 0) { + const uint mask_offset = ik + t * m_nb1 + (s % n_masks) * m_nb3; + dst[ik + t * d_nb1 + s * d_nb3] = score + float(mask[mask_offset]); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_s.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_s.comp index 5cdf2a89..42f52b4a 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_s.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_s.comp @@ -7,7 +7,14 @@ layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; FLOAT_TYPE temp[NUM_COLS][NUM_ROWS]; -void calc_superblock(const uint a_offset, const uint b_offset, const uint ib32, const uint i, const uint num_blocks_per_row, const uint first_row, const uint num_rows) { +// invocations per superblock. with many columns, 8 invocations need too many +// registers and spill, so use 16 to halve the per-invocation B working set +const uint TPB = NUM_COLS <= 4 ? 8 : 16; +const uint NL = 32 / TPB; // l steps per invocation + +void calc_superblock(const uint a_offset, const uint b_offset, const uint itid, const uint i, const uint num_blocks_per_row, const uint first_row, const uint num_rows) { + const uint ib32 = itid / (TPB / 8); + const uint l0 = (itid % (TPB / 8)) * NL; const uint y_idx = i * QUANT_K + 32 * ib32; uint ibi = a_offset + first_row * num_blocks_per_row + i; @@ -16,11 +23,8 @@ void calc_superblock(const uint a_offset, const uint b_offset, const uint ib32, const uint scale = (data_a[ibi].scales[ib32/2] >> (4 * (ib32 & 1))) & 0xF; const float dscale = d * (1 + 2 * scale); const uint qh = data_a[ibi].qh[ib32]; - FLOAT_TYPE sum[NUM_COLS]; - [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { - sum[j] = 0.0; - } - [[unroll]] for (uint l = 0; l < 4; ++l) { + [[unroll]] for (uint ll = 0; ll < NL; ++ll) { + const uint l = l0 + ll; const u8vec2 qs = unpack8(uint32_t(data_a_packed16[ibi].qs[4 * ib32 + l])).xy; // vec4 used due to #12147 const uint sign = data_a[ibi].signs[4 * ib32 + l]; const vec4 grid0 = vec4(unpack8(iq3s_grid[qs.x | ((qh << (8 - 2*l)) & 0x100)])); @@ -30,7 +34,7 @@ void calc_superblock(const uint a_offset, const uint b_offset, const uint ib32, const vec4 b0 = vec4(data_b_v4[(j*p.batch_stride_b + b_offset + y_idx) / 4 + 2*l + 0]); const vec4 b4 = vec4(data_b_v4[(j*p.batch_stride_b + b_offset + y_idx) / 4 + 2*l + 1]); - sum[j] = + const FLOAT_TYPE sum = fma(FLOAT_TYPE(b0.x), FLOAT_TYPE((sign & 1) != 0 ? -grid0.x : grid0.x), fma(FLOAT_TYPE(b0.y), FLOAT_TYPE((sign & 2) != 0 ? -grid0.y : grid0.y), fma(FLOAT_TYPE(b0.z), FLOAT_TYPE((sign & 4) != 0 ? -grid0.z : grid0.z), @@ -39,12 +43,11 @@ void calc_superblock(const uint a_offset, const uint b_offset, const uint ib32, fma(FLOAT_TYPE(b4.y), FLOAT_TYPE((sign & 32) != 0 ? -grid1.y : grid1.y), fma(FLOAT_TYPE(b4.z), FLOAT_TYPE((sign & 64) != 0 ? -grid1.z : grid1.z), fma(FLOAT_TYPE(b4.w), FLOAT_TYPE((sign & 128) != 0 ? -grid1.w : grid1.w), - sum[j])))))))); + FLOAT_TYPE(0.0))))))))); + + temp[j][n] = fma(dscale, sum, temp[j][n]); } } - [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { - temp[j][n] = fma(dscale, sum[j], temp[j][n]); - } ibi += num_blocks_per_row; } } @@ -55,11 +58,11 @@ void compute_outputs(const uint32_t first_row, const uint32_t num_rows) { const uint num_blocks_per_row = p.ncols / QUANT_K; - // 8 threads are used to process each block - const uint blocks_per_wg = gl_WorkGroupSize.x/8; + // TPB invocations are used to process each block + const uint blocks_per_wg = gl_WorkGroupSize.x/TPB; const uint tid = gl_LocalInvocationID.x; - const uint itid = tid % 8; // 0...7 - const uint ix = tid / 8; + const uint itid = tid % TPB; + const uint ix = tid / TPB; [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { [[unroll]] for (uint i = 0; i < NUM_ROWS; ++i) { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq4_xs.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq4_xs.comp new file mode 100644 index 00000000..a2b99d9a --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq4_xs.comp @@ -0,0 +1,97 @@ +#version 450 + +#extension GL_EXT_shader_explicit_arithmetic_types_int32 : require + +#include "mul_mat_vec_base.glsl" + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +FLOAT_TYPE temp[NUM_COLS][NUM_ROWS]; + +// dedicated iq4_xs mat-vec, mirrors mul_mat_vec_iq3_s.comp +// one packed32 word per l, so the 6-bit subblock scale is hoisted to a single fma after register accumulation + +void calc_superblock(const uint a_offset, const uint b_offset, const uint ib32, const uint i, const uint num_blocks_per_row, const uint first_row, const uint num_rows) { + const uint y_idx = i * QUANT_K + 32 * ib32; + + uint ibi = a_offset + first_row * num_blocks_per_row + i; + [[unroll]] for (uint n = 0; n < num_rows; ++n) { + const float d = float(data_a[ibi].d); + const uint sl = (data_a[ibi].scales_l[ib32/2] >> (4 * (ib32 & 1))) & 0xF; + const uint sh = (data_a[ibi].scales_h >> (2 * ib32)) & 3; + const float dscale = d * float(int(sl | (sh << 4)) - 32); + + FLOAT_TYPE sum[NUM_COLS]; + [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { + sum[j] = FLOAT_TYPE(0); + } + + [[unroll]] for (uint l = 0; l < 4; ++l) { + const uint w = data_a_packed32[ibi].qs[4 * ib32 + l]; + const u8vec4 q0 = unpack8(w & 0x0F0F0F0F); + const u8vec4 q1 = unpack8((w >> 4) & 0x0F0F0F0F); + + [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { + const vec4 b0 = vec4(data_b_v4[(j*p.batch_stride_b + b_offset + y_idx) / 4 + l]); + const vec4 b1 = vec4(data_b_v4[(j*p.batch_stride_b + b_offset + y_idx) / 4 + 4 + l]); + + sum[j] = fma(FLOAT_TYPE(b0.x), FLOAT_TYPE(kvalues_iq4nl[q0.x]), + fma(FLOAT_TYPE(b0.y), FLOAT_TYPE(kvalues_iq4nl[q0.y]), + fma(FLOAT_TYPE(b0.z), FLOAT_TYPE(kvalues_iq4nl[q0.z]), + fma(FLOAT_TYPE(b0.w), FLOAT_TYPE(kvalues_iq4nl[q0.w]), + fma(FLOAT_TYPE(b1.x), FLOAT_TYPE(kvalues_iq4nl[q1.x]), + fma(FLOAT_TYPE(b1.y), FLOAT_TYPE(kvalues_iq4nl[q1.y]), + fma(FLOAT_TYPE(b1.z), FLOAT_TYPE(kvalues_iq4nl[q1.z]), + fma(FLOAT_TYPE(b1.w), FLOAT_TYPE(kvalues_iq4nl[q1.w]), + sum[j])))))))); + } + } + + [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { + temp[j][n] = fma(dscale, sum[j], temp[j][n]); + } + + ibi += num_blocks_per_row; + } +} + +void compute_outputs(const uint32_t first_row, const uint32_t num_rows) { + uint a_offset, b_offset, d_offset; + + get_offsets(a_offset, b_offset, d_offset); + + const uint num_blocks_per_row = p.ncols / QUANT_K; + + // 8 threads are used to process each block + const uint blocks_per_wg = gl_WorkGroupSize.x/8; + const uint tid = gl_LocalInvocationID.x; + const uint itid = tid % 8; // 0...7 + const uint ix = tid / 8; + + [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { + [[unroll]] for (uint i = 0; i < NUM_ROWS; ++i) { + temp[j][i] = FLOAT_TYPE(0); + } + } + + [[unroll]] for (uint i = ix; i < num_blocks_per_row; i += blocks_per_wg) + calc_superblock(a_offset, b_offset, itid, i, num_blocks_per_row, first_row, num_rows); + + reduce_result(temp, d_offset, first_row, num_rows, tid); +} + +void main() { + const uint first_row = NUM_ROWS * (gl_WorkGroupID.x + gl_NumWorkGroups.x * gl_WorkGroupID.z); + + init_iq_shmem(gl_WorkGroupSize); + + // do NUM_ROWS at a time, unless there aren't enough remaining rows + if (first_row + NUM_ROWS <= p.stride_d) { + compute_outputs(first_row, NUM_ROWS); + } else { + if (first_row >= p.stride_d) { + return; + } + compute_outputs(first_row, p.stride_d - first_row); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_tq1_0.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_tq1_0.comp new file mode 100644 index 00000000..2c99a268 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_tq1_0.comp @@ -0,0 +1,85 @@ +#version 450 +#extension GL_EXT_shader_explicit_arithmetic_types : require + +#include "mul_mat_vec_base.glsl" + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +FLOAT_TYPE temp[NUM_COLS][NUM_ROWS]; + +// Walks the packed bytes directly (byte m, digit t) rather than via +// tq1_0_byte_of()/tq1_0_digit_of(): one byte per thread, expanded in place. +void compute_outputs(const uint32_t first_row, const uint32_t num_rows) { + uint a_offset, b_offset, d_offset; + get_offsets(a_offset, b_offset, d_offset); + + const uint num_blocks_per_row = p.ncols / QUANT_K; + const uint tid = gl_LocalInvocationID.x; + + [[unroll]] for (uint j = 0; j < NUM_COLS; ++j) { + [[unroll]] for (uint i = 0; i < NUM_ROWS; ++i) { + temp[j][i] = FLOAT_TYPE(0); + } + } + + for (uint nrow = 0; nrow < num_rows; ++nrow) { + const uint ib0 = a_offset + (first_row + nrow) * num_blocks_per_row; + for (uint jcol = 0; jcol < NUM_COLS; ++jcol) { + const uint b_base = (jcol * p.batch_stride_b); + for (uint i = tid/8; i < num_blocks_per_row; i += gl_WorkGroupSize.x/8) { + const FLOAT_TYPE d = float(data_a[ib0 + i].d); + + // First qs chunk: 32 bytes (5*32 elements) + [[unroll]] for (uint m = tid%8; m < 32; m += 8) { + const uint q_byte = uint(data_a[ib0 + i].qs[m]); + [[unroll]] for (uint t = 0; t < 5; ++t) { + const uint xi = tq1_0_trit(q_byte, t); + const FLOAT_TYPE dequant_val = FLOAT_TYPE(d * (float(xi) - 1.0f)); + const uint elem = t * 32u + m; + const uint b_idx = i * QUANT_K + elem; + temp[jcol][nrow] += dequant_val * FLOAT_TYPE(data_b[b_base + b_offset + b_idx]); + } + } + + // Second qs chunk: 16 bytes (5*16 elements) + [[unroll]] for (uint m = tid%8; m < 16; m += 8) { + const uint q_byte = uint(data_a[ib0 + i].qs[32u + m]); + [[unroll]] for (uint t = 0; t < 5; ++t) { + const uint xi = tq1_0_trit(q_byte, t); + const FLOAT_TYPE dequant_val = FLOAT_TYPE(d * (float(xi) - 1.0f)); + const uint elem = 160u + t * 16u + m; + const uint b_idx = i * QUANT_K + elem; + temp[jcol][nrow] += dequant_val * FLOAT_TYPE(data_b[b_base + b_offset + b_idx]); + } + } + + // qh bytes: 4 bytes (4*4 elements) + [[unroll]] for (uint j = tid%8; j < 4; j += 8) { + const uint qh_byte = uint(data_a[ib0 + i].qh[j]); + [[unroll]] for (uint t = 0; t < 4; ++t) { + const uint xi = tq1_0_trit(qh_byte, t); + const FLOAT_TYPE dequant_val = FLOAT_TYPE(d * (float(xi) - 1.0f)); + const uint elem = 240u + t * 4u + j; + const uint b_idx = i * QUANT_K + elem; + temp[jcol][nrow] += dequant_val * FLOAT_TYPE(data_b[b_base + b_offset + b_idx]); + } + } + } + } + } + + reduce_result(temp, d_offset, first_row, num_rows, tid); +} + +void main() { + const uint first_row = NUM_ROWS * (gl_WorkGroupID.x + gl_NumWorkGroups.x * gl_WorkGroupID.z); + + if (first_row + NUM_ROWS <= p.stride_d) { + compute_outputs(first_row, NUM_ROWS); + } else { + if (first_row >= p.stride_d) { + return; + } + compute_outputs(first_row, p.stride_d - first_row); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp index 18d441ea..3383aa88 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp @@ -5,6 +5,7 @@ #define MMQ #define NEEDS_IQ1S_GRID_GPU +#define KVALUES_IQ4NL_I8 #define B_TYPE block_q8_1_x4 #include "mul_mat_vec_base.glsl" @@ -15,7 +16,7 @@ layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; #define K_PER_ITER 16 #elif defined(DATA_A_QUANT_LEGACY) || defined(DATA_A_MXFP4) #define K_PER_ITER 8 -#elif defined(DATA_A_IQ1_S) || defined(DATA_A_IQ1_M) +#elif defined(DATA_A_IQ1_S) || defined(DATA_A_IQ1_M) || defined(DATA_A_IQ4_XS) #define K_PER_ITER 32 #else #error unimplemented diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl index a5403ac8..03c348bf 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl @@ -448,6 +448,28 @@ FLOAT_TYPE mmvq_dot_product(const uint ib_a, const uint iqs) { } #endif +#if defined(DATA_A_IQ4_XS) +FLOAT_TYPE mmvq_dot_product(const uint ib_a, const uint iqs) { + const uint ib = ib_a / 8; + const uint ib32 = ib_a % 8; + + int32_t q_sum = 0; + [[unroll]] for (uint j = 0; j < 4; ++j) { + const uint32_t vui = data_a_packed32[ib].qs[4 * ib32 + j]; + const i32vec2 qs_a = iq4nl_to_i8x8(vui); + + q_sum += dotPacked4x8EXT(qs_a.x, cache_b_qs[j]); + q_sum += dotPacked4x8EXT(qs_a.y, cache_b_qs[j + 4]); + } + + const uint sl = (data_a_packed32[ib].scales_l >> (4 * ib32)) & 0xF; + const uint sh = (data_a_packed32[ib].scales_h >> (2 * ib32)) & 3; + const float d = float(data_a[ib].d) * float(int(sl | (sh << 4)) - 32); + + return FLOAT_TYPE(float(cache_b_ds.x) * d * float(q_sum)); +} +#endif + #if defined(DATA_A_IQ1_S) void repack8(uint ib, uint iqs, out i32vec4 out0, out i32vec4 out1) { const uint ib32 = iqs / 32; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp index 3df88044..11098ee7 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp @@ -9,6 +9,9 @@ #if defined(DATA_A_IQ1_M) #extension GL_EXT_shader_explicit_arithmetic_types_int16 : require #endif +#if !defined(DATA_A_F32) && !defined(DATA_A_F16) && !defined(DATA_A_BF16) +#extension GL_EXT_shader_explicit_arithmetic_types_int16 : require +#endif #if defined(DATA_A_BF16) && defined(COOPMAT) #extension GL_EXT_bfloat16 : enable @@ -28,24 +31,54 @@ #extension GL_EXT_shader_explicit_arithmetic_types_int16 : require #endif +#ifdef MULMAT_QUANT +#include "ggml_type_ids.glsl" +layout (constant_id = 12) const uint MmTypeA = 0; +#endif + #include "types.glsl" #include "dot_product_funcs.glsl" +#ifndef MULMAT_QUANT #ifndef LOAD_VEC_A #define LOAD_VEC_A 1 #endif +#endif #ifndef LOAD_VEC_B #define LOAD_VEC_B 1 #endif layout (constant_id = 11) const uint ALIGNED = 0; +#ifdef MULMAT_QUANT + +uint mm_load_vec_a() { + switch (MmTypeA) { + case GGML_TYPE_Q1_0: + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q5_1: + return 8u; + case GGML_TYPE_Q2_0: + case GGML_TYPE_Q5_0: + case GGML_TYPE_Q8_0: + case GGML_TYPE_Q2_K: + case GGML_TYPE_Q4_K: + case GGML_TYPE_Q5_K: + return 4u; + default: + return 2u; + } +} +#endif + #if !defined(TO_FLOAT_TYPE) #define TO_FLOAT_TYPE FLOAT_TYPE #endif layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; +#ifndef MULMAT_QUANT layout (binding = 0) readonly buffer A {A_TYPE data_a[];}; #if defined(DATA_A_F32) layout (binding = 0) readonly buffer A_SCALAR {float data_a_scalar[];}; @@ -60,6 +93,30 @@ layout (binding = 0) readonly buffer A_PACKED16 {A_TYPE_PACKED16 data_a_packed16 #if defined(A_TYPE_PACKED32) layout (binding = 0) readonly buffer A_PACKED32 {A_TYPE_PACKED32 data_a_packed32[];}; #endif +#else +// Unpacked struct aliases +layout (binding = 0) readonly buffer BUF_Q1_0 { block_q1_0 data[]; } a_q1_0; +layout (binding = 0) readonly buffer BUF_Q2_0 { block_q2_0 data[]; } a_q2_0; +layout (binding = 0) readonly buffer BUF_Q2_K { block_q2_K data[]; } a_q2_k; +layout (binding = 0) readonly buffer BUF_Q3_K { block_q3_K data[]; } a_q3_k; +layout (binding = 0) readonly buffer BUF_Q4_K { block_q4_K data[]; } a_q4_k; +layout (binding = 0) readonly buffer BUF_Q5_K { block_q5_K data[]; } a_q5_k; +layout (binding = 0) readonly buffer BUF_Q6_K { block_q6_K data[]; } a_q6_k; +layout (binding = 0) readonly buffer BUF_TQ1_0 { block_tq1_0 data[]; } a_tq1_0; +layout (binding = 0) readonly buffer BUF_TQ2_0 { block_tq2_0 data[]; } a_tq2_0; +// Packed16 aliases +layout (binding = 0) readonly buffer BUF_Q4_0_P16 { block_q4_0_packed16 data[]; } a_q4_0_p16; +layout (binding = 0) readonly buffer BUF_Q5_0_P16 { block_q5_0_packed16 data[]; } a_q5_0_p16; +layout (binding = 0) readonly buffer BUF_Q8_0_P16 { block_q8_0_packed16 data[]; } a_q8_0_p16; +layout (binding = 0) readonly buffer BUF_Q3_K_P16 { block_q3_K_packed16 data[]; } a_q3_k_p16; +layout (binding = 0) readonly buffer BUF_Q6_K_P16 { block_q6_K_packed16 data[]; } a_q6_k_p16; +// Packed32 aliases +layout (binding = 0) readonly buffer BUF_Q4_1_P32 { block_q4_1_packed32 data[]; } a_q4_1_p32; +layout (binding = 0) readonly buffer BUF_Q5_1_P32 { block_q5_1_packed32 data[]; } a_q5_1_p32; +layout (binding = 0) readonly buffer BUF_Q2_K_P32 { block_q2_K_packed32 data[]; } a_q2_k_p32; +layout (binding = 0) readonly buffer BUF_Q4_K_P32 { block_q4_K_packed32 data[]; } a_q4_k_p32; +layout (binding = 0) readonly buffer BUF_Q5_K_P32 { block_q5_K_packed32 data[]; } a_q5_k_p32; +#endif layout (binding = 1) readonly buffer B {B_TYPE data_b[];}; layout (binding = 1) readonly buffer B_SCALAR {B_TYPE_SCALAR data_b_scalar[];}; @@ -88,6 +145,8 @@ layout (push_constant) uniform parameter uint nei1; uint nbi1; uint ne11; + uint n_experts; + uint hoist_row_ids; #else uint base_work_group_z; uint num_batches; @@ -119,8 +178,13 @@ layout (constant_id = 3) const uint BK = 16; // Assumed to be 32 if working wit #endif #ifdef COOPMAT +#ifdef MULMAT_QUANT +layout(constant_id = 13) const uint SHMEM_STRIDE_PAD = 4; +layout(constant_id = 14) const bool APPLY_SLM_A_RESHAPE = false; +#else layout(constant_id = 12) const uint SHMEM_STRIDE_PAD = 4; layout(constant_id = 13) const bool APPLY_SLM_A_RESHAPE = false; +#endif #else const uint SHMEM_STRIDE_PAD = 1; const bool APPLY_SLM_A_RESHAPE = false; @@ -139,6 +203,10 @@ shared ACC_TYPE coopmat_stage[TM * TN * NUM_WARPS]; #include "mul_mm_id_funcs.glsl" #include "mul_mm_funcs.glsl" +#ifdef MULMAT_QUANT +#include "iq_shmem_init.glsl" +#endif + void main() { const uint ic = gl_WorkGroupID.y; @@ -148,7 +216,7 @@ void main() { return; } #endif -#ifdef NEEDS_INIT_IQ_SHMEM +#if defined(NEEDS_INIT_IQ_SHMEM) || defined(MULMAT_QUANT) init_iq_shmem(gl_WorkGroupSize); #endif @@ -198,9 +266,12 @@ void main() { #if defined(DATA_A_F32) || defined(DATA_A_F16) || defined(DATA_A_BF16) const uint LOAD_VEC_A_EFF = (ALIGNED != 0) ? LOAD_VEC_A : 1; const uint LOAD_VEC_BATCH_A = (ALIGNED != 0) ? 1 : 2; -#else +#elif !defined(MULMAT_QUANT) const uint LOAD_VEC_A_EFF = LOAD_VEC_A; const uint LOAD_VEC_BATCH_A = 1; +#else + const uint LOAD_VEC_A_EFF = mm_load_vec_a(); + const uint LOAD_VEC_BATCH_A = 1; #endif const uint LOAD_VEC_B_EFF = (ALIGNED != 0) ? LOAD_VEC_B : 1; const uint LOAD_VEC_BATCH_B = (ALIGNED != 0) ? 1 : 2; @@ -214,30 +285,37 @@ void main() { const uint loadstride_b = gl_WorkGroupSize.x * LOAD_VEC_B_EFF * LOAD_VEC_BATCH_B / BK; #ifdef MUL_MAT_ID -#ifdef MUL_MAT_ID_USE_SUBGROUPS - if (bitCount(p.nei0) == 1) { - load_row_ids(expert_idx, true, ic); + if (p.hoist_row_ids != 0) { + load_row_ids_hoisted(expert_idx, ic); } else { - load_row_ids(expert_idx, false, ic); - } +#ifdef MUL_MAT_ID_USE_SUBGROUPS + if (bitCount(p.nei0) == 1) { + load_row_ids(expert_idx, true, ic); + } else { + load_row_ids(expert_idx, false, ic); + } #else - _ne1 = 0; - for (uint ii1 = 0; ii1 < p.nei1 && _ne1 < (ic + 1) * BN; ii1++) { - for (uint ii0 = 0; ii0 < p.nei0 && _ne1 < (ic + 1) * BN; ii0++) { - if (data_ids[ii1*p.nbi1 + ii0] == expert_idx) { - if (_ne1 >= ic * BN) { - row_ids[_ne1 - ic * BN] = u16vec2(ii0, ii1); + _ne1 = 0; + for (uint ii1 = 0; ii1 < p.nei1 && _ne1 < (ic + 1) * BN; ii1++) { + for (uint ii0 = 0; ii0 < p.nei0 && _ne1 < (ic + 1) * BN; ii0++) { + if (data_ids[ii1*p.nbi1 + ii0] == expert_idx) { + if (_ne1 >= ic * BN) { + row_ids[_ne1 - ic * BN] = u16vec2(ii0, ii1); + } + _ne1++; } - _ne1++; } } - } - barrier(); + barrier(); #endif + } // Workgroup has no work if (ic * BN >= _ne1) return; + + uint required_work_items = (_ne1 - ic * BN) * BK / LOAD_VEC_B_EFF / LOAD_VEC_BATCH_B; + uint required_warp_c = (_ne1 - ic * BN + WN - 1) / WN; #endif #ifdef MUL_MAT_ID @@ -288,6 +366,9 @@ void main() { [[unroll]] for (uint l = 0; l < BM; l += loadstride_a) { load_a_to_shmem(pos_a, loadr_a, loadc_a + l, ir * BM + loadc_a + l, block, end_k); } + #ifdef MUL_MAT_ID + if (gl_LocalInvocationID.x < required_work_items) { + #endif [[unroll]] for (uint l = 0; l < BN; l += loadstride_b) { #if !defined(MUL_MAT_ID) load_b_to_shmem(pos_b, loadr_b, loadc_b + l, ic * BN + loadc_b + l, block, end_k); @@ -295,6 +376,9 @@ void main() { load_b_to_shmem(pos_b, loadr_b, loadc_b + l, ic, _ne1, block, end_k); #endif } + #ifdef MUL_MAT_ID + } + #endif barrier(); @@ -302,6 +386,9 @@ void main() { pos_b += BK / LOAD_VEC_B_EFF; #ifdef COOPMAT +#ifdef MUL_MAT_ID + if (warp_c < required_warp_c) { +#endif [[unroll]] for (uint i = 0; i < BK; i += TK) { [[unroll]] for (uint cm_row = 0; cm_row < cms_per_row; cm_row++) { // Load from shared into cache @@ -314,6 +401,9 @@ void main() { } } } +#ifdef MUL_MAT_ID + } +#endif #else [[unroll]] for (uint i = 0; i < BK / BK_STEP; i++) { // Load from shared into cache diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp index a2e15f6f..9b59e8bc 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp @@ -21,6 +21,13 @@ #extension GL_EXT_bfloat16 : enable #endif +#include "ggml_type_ids.glsl" + +#ifdef MULMAT_QUANT +layout (constant_id = 7) const uint MmTypeA = 0; +layout (constant_id = 8) const uint MmABlockBytes = 2; +#endif + #include "types.glsl" #include "utils.glsl" @@ -34,9 +41,27 @@ layout (constant_id = 2) const uint BN = 64; layout (constant_id = 3) const uint BK = 16; // Assumed to be 32 if working with a quant layout (constant_id = 4) const bool enable_smaller_matrices = false; -const uint BNover2 = enable_smaller_matrices ? (BN / 2) : BN; -const uint BNover4 = enable_smaller_matrices ? (BN / 4) : BN; +const uint BNover2 = BN / 2; +const uint BNover4 = enable_smaller_matrices ? (BN / 4) : (BN / 2); layout (constant_id = 5) const uint ALIGNED = 0; +layout (constant_id = 6) const uint subgroup_size = 32; + +#ifdef MULMAT_QUANT + +uint mm_quant_k() { + switch (MmTypeA) { + case GGML_TYPE_Q4_0: case GGML_TYPE_Q4_1: case GGML_TYPE_Q5_0: case GGML_TYPE_Q5_1: + case GGML_TYPE_Q8_0: + return 32u; + case GGML_TYPE_Q1_0: + return 128u; + case GGML_TYPE_Q2_0: + return 64u; + default: + return 256u; + } +} +#endif layout (push_constant) uniform parameter { @@ -56,6 +81,8 @@ layout (push_constant) uniform parameter uint nei1; uint nbi1; uint ne11; + uint n_experts; + uint hoist_row_ids; #else uint base_work_group_z; uint num_batches; @@ -64,27 +91,79 @@ layout (push_constant) uniform parameter uint ne12; uint broadcast2; uint broadcast3; -#endif // N dimension for the B matrix can be >= p.N uint padded_N; +#endif } p; +#ifndef MULMAT_QUANT layout (binding = 0) readonly buffer A {A_TYPE data_a[];}; +#else +layout (binding = 0) readonly buffer A {uint8_t data_a[];}; +#endif layout (binding = 1) readonly buffer B {B_TYPE data_b[];}; layout (binding = 2) writeonly buffer D {D_TYPE data_d[];}; #if defined(MUL_MAT_ID) && defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR) layout (binding = 1) readonly buffer B4 {B_TYPEV4 data_b_v4[];}; #endif -#if QUANT_K > 1 +#if defined(MULMAT_QUANT) || QUANT_K > 1 #include "dequant_funcs_cm2.glsl" +#ifndef MULMAT_QUANT +// Per-type path: use the alias set by dequant_funcs_cm2.glsl #if defined(dequantFuncA_v) && defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR) #define DECODEFUNCA , dequantFuncA, dequantFuncA_v #else #define DECODEFUNCA , dequantFuncA #endif #else +layout(buffer_reference, std430, buffer_reference_align = 1) buffer decodeBufA { + uint8_t raw[MmABlockBytes]; +}; + +float16_t mmDecodeA(const in decodeBufA bl_in, const in uint blockCoords[2], const in uint coordInBlock[2]) { + switch (MmTypeA) { + case GGML_TYPE_Q1_0: return dequantFuncQ1_0 (decodeBufQ1_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q2_0: return dequantFuncQ2_0 (decodeBufQ2_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0 (decodeBufQ4_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1 (decodeBufQ4_1 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0 (decodeBufQ5_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1 (decodeBufQ5_1 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0 (decodeBufQ8_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q2_K: return dequantFuncQ2_K (decodeBufQ2_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q3_K: return dequantFuncQ3_K (decodeBufQ3_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q6_K: return dequantFuncQ6_K (decodeBufQ6_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_TQ1_0: return dequantFuncTQ1_0(decodeBufTQ1_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_TQ2_0: return dequantFuncTQ2_0(decodeBufTQ2_0(bl_in), blockCoords, coordInBlock); + default: return float16_t(0); + } +} + +#ifdef GGML_VULKAN_COOPMAT2_DECODE_VECTOR +f16vec4 mmDecodeA_v(const in decodeBufA bl_in, const in uint blockCoords[2], const in uint coordInBlock[2]) { + switch (MmTypeA) { + case GGML_TYPE_Q1_0: return dequantFuncQ1_0_v (decodeBufQ1_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q2_0: return dequantFuncQ2_0_v (decodeBufQ2_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_0: return dequantFuncQ4_0_v (decodeBufQ4_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q4_1: return dequantFuncQ4_1_v (decodeBufQ4_1 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_0: return dequantFuncQ5_0_v (decodeBufQ5_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q5_1: return dequantFuncQ5_1_v (decodeBufQ5_1 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q8_0: return dequantFuncQ8_0_v (decodeBufQ8_0 (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q2_K: return dequantFuncQ2_K_v (decodeBufQ2_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q3_K: return dequantFuncQ3_K_v (decodeBufQ3_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_Q6_K: return dequantFuncQ6_K_v (decodeBufQ6_K (bl_in), blockCoords, coordInBlock); + case GGML_TYPE_TQ1_0: return dequantFuncTQ1_0_v(decodeBufTQ1_0(bl_in), blockCoords, coordInBlock); + case GGML_TYPE_TQ2_0: return dequantFuncTQ2_0_v(decodeBufTQ2_0(bl_in), blockCoords, coordInBlock); + default: return f16vec4(0); + } +} +#define DECODEFUNCA , mmDecodeA, mmDecodeA_v +#else +#define DECODEFUNCA , mmDecodeA +#endif +#endif +#else #define DECODEFUNCA #endif @@ -112,7 +191,6 @@ layout(buffer_reference, std430, buffer_reference_align = 2) buffer decodeBufB { }; uint _ne1; -layout (constant_id = 6) const uint subgroup_size = 32; shared uvec4 ballots_sh[BLOCK_SIZE / subgroup_size]; B_TYPE decodeFuncB(const in decodeBufB bl, const in uint blockCoords[2], const in uint coordInBlock[2]) @@ -225,6 +303,27 @@ void load_row_ids(uint expert_idx, bool nei0_is_pow2, uint ic) { } barrier(); } + +void load_row_ids_hoisted(uint expert_idx, uint ic) { + _ne1 = uint(data_expert_count[expert_idx]); + + const uint tile_begin = ic * BN; + const uint tile_count = tile_begin < _ne1 ? min(BN, _ne1 - tile_begin) : 0; + const uint expert_offset = uint(data_expert_count[p.n_experts + expert_idx]); + const uint row_ids_offset = 2 * p.n_experts + 1 + expert_offset + tile_begin; + + for (uint i = gl_LocalInvocationIndex; i < tile_count; i += BLOCK_SIZE) { + const uint packed_row_id = uint(data_expert_count[row_ids_offset + i]); + const uint ii0 = packed_row_id & 0xffffu; + const uint ii1 = packed_row_id >> 16; + row_ids[i] = u16vec4(fastmod(ii0, p.ne11), ii1, ii0, 0); + } + barrier(); +} +#endif + +#ifdef MULMAT_QUANT +#include "iq_shmem_init.glsl" #endif void main() { @@ -245,7 +344,7 @@ void main() { #endif #endif -#ifdef NEEDS_INIT_IQ_SHMEM +#if defined(NEEDS_INIT_IQ_SHMEM) || defined(MULMAT_QUANT) init_iq_shmem(gl_WorkGroupSize); #endif @@ -266,7 +365,9 @@ void main() { const uint ik = gl_WorkGroupID.x / blocks_m; #ifdef MUL_MAT_ID - if (bitCount(p.nei0) == 1) { + if (p.hoist_row_ids != 0) { + load_row_ids_hoisted(expert_idx, ic); + } else if (bitCount(p.nei0) == 1) { load_row_ids(expert_idx, true, ic); } else { load_row_ids(expert_idx, false, ic); @@ -284,22 +385,33 @@ void main() { const uint end_k = min(p.K, (ik + 1) * p.k_split); #endif +#ifdef MULMAT_QUANT + const uint qk = mm_quant_k(); +#else + const uint qk = QUANT_K; +#endif + #ifdef MUL_MAT_ID - uint pos_a = expert_idx * (p.batch_stride_a / QUANT_K); + uint pos_a = expert_idx * (p.batch_stride_a / qk); uint pos_b = 0; #else - uint pos_a = batch_idx_a * (p.batch_stride_a / QUANT_K); + uint pos_a = batch_idx_a * (p.batch_stride_a / qk); uint pos_b = batch_idx * p.batch_stride_b; uint pos_d = batch_idx * p.batch_stride_d + ik * p.batch_stride_d * p.num_batches; #endif - uint stride_a = p.stride_a / QUANT_K; +#ifdef MULMAT_QUANT + // pos_a is a byte offset into the raw buffer; strides stay in block units + pos_a *= MmABlockBytes; +#endif + + uint stride_a = p.stride_a / qk; uint stride_b = p.stride_b; // Hint to the compiler that values are aligned (want 16B alignment). // Quants are always block-aligned, no alignment needed. if (ALIGNED != 0) { -#if QUANT_K == 1 +#if !defined(MULMAT_QUANT) && QUANT_K == 1 stride_a &= ~7; #endif stride_b &= ~7; @@ -309,25 +421,32 @@ void main() { tensorLayoutNV<2> tensorLayoutA = createTensorLayoutNV(2); tensorLayoutNV<2, gl_CooperativeMatrixClampModeConstantNV> tensorLayoutAClamp = createTensorLayoutNV(2, gl_CooperativeMatrixClampModeConstantNV); tensorLayoutNV<2> tensorLayoutB = createTensorLayoutNV(2); +#ifndef MUL_MAT_ID tensorLayoutNV<2, gl_CooperativeMatrixClampModeConstantNV> tensorLayoutBClamp = createTensorLayoutNV(2, gl_CooperativeMatrixClampModeConstantNV); +#endif tensorLayoutNV<2, gl_CooperativeMatrixClampModeConstantNV> tensorLayoutD = createTensorLayoutNV(2, gl_CooperativeMatrixClampModeConstantNV); -#if QUANT_K > 1 - tensorLayoutA = setTensorLayoutBlockSizeNV(tensorLayoutA, 1, QUANT_K); - tensorLayoutAClamp = setTensorLayoutBlockSizeNV(tensorLayoutAClamp, 1, QUANT_K); -#endif + tensorLayoutA = setTensorLayoutBlockSizeNV(tensorLayoutA, 1, qk); + tensorLayoutAClamp = setTensorLayoutBlockSizeNV(tensorLayoutAClamp, 1, qk); #if defined(MUL_MAT_ID) && defined(GGML_VULKAN_COOPMAT2_DECODE_VECTOR) tensorLayoutB = setTensorLayoutBlockSizeNV(tensorLayoutB, 1, BK); #endif // Use end_k rather than p.K as the dimension because that's what // we need to bound check against when using split_k. - // Bounds check B against padded_N, but bounds check D against N. tensorLayoutA = setTensorLayoutDimensionNV(tensorLayoutA, p.M, end_k); +#ifdef MUL_MAT_ID + // MUL_MAT_ID pads each B row to stride_b so partial K tiles read zeros without clamping. + tensorLayoutB = setTensorLayoutDimensionNV(tensorLayoutB, BN, p.stride_b); +#else + // Bounds check B against padded_N, but bounds check D against N. tensorLayoutB = setTensorLayoutDimensionNV(tensorLayoutB, p.padded_N, end_k); +#endif tensorLayoutD = setTensorLayoutDimensionNV(tensorLayoutD, p.N, p.M); tensorLayoutAClamp = setTensorLayoutDimensionNV(tensorLayoutAClamp, p.M, end_k); +#ifndef MUL_MAT_ID tensorLayoutBClamp = setTensorLayoutDimensionNV(tensorLayoutBClamp, p.padded_N, end_k); +#endif tensorLayoutD = setTensorLayoutStrideNV(tensorLayoutD, p.stride_d, 1); @@ -338,19 +457,19 @@ void main() { const uint START_ALIGN_K = 256; // For Qi_K (block size 256), unroll whole 256 element tiles. // For legacy quants (block size 32), unroll 8x. - const uint UNROLL_K = (QUANT_K == 256) ? 256 : (BK * 8); + const uint UNROLL_K = (qk == 256) ? 256 : (BK * 8); const uint unroll_count = UNROLL_K / BK; // Detect a fast path where all loads are entirely in bounds and no clamping is required if ((ir + 1) * BM <= p.M && (ic + 1) * BN <= p.padded_N && (start_k % START_ALIGN_K) == 0 && (end_k % BK) == 0 && -#if QUANT_K == 1 +#if !defined(MULMAT_QUANT) && QUANT_K == 1 (stride_a % 8) == 0 && #endif (stride_b % 8) == 0) { // Hint to the compiler that values are aligned (want 16B alignment) start_k &= ~(START_ALIGN_K-1); stride_b &= ~7; -#if QUANT_K == 1 +#if !defined(MULMAT_QUANT) && QUANT_K == 1 stride_a &= ~7; #endif @@ -504,7 +623,9 @@ void main() { tensorLayoutB = setTensorLayoutStrideNV(tensorLayoutB, stride_b, 1); +#ifndef MUL_MAT_ID tensorLayoutBClamp = setTensorLayoutStrideNV(tensorLayoutBClamp, stride_b, 1); +#endif uint k_iters = (end_k - start_k + BK - 1) / BK; @@ -519,10 +640,10 @@ void main() { [[dont_unroll]] for (uint block_k = start_k, i = 0; i < k_iters; block_k += BK, ++i) { - if ((block_k % QUANT_K) == 0) { + if ((block_k % qk) == 0) { store_scales(tid); } - if (block_k + BK < end_k && ((block_k + BK) % QUANT_K) == 0) { + if (block_k + BK < end_k && ((block_k + BK) % qk) == 0) { fetch_scales(ir * BM, pos_a, stride_a, block_k + BK, tid, false); } @@ -556,17 +677,17 @@ void main() { coopMatPerElementNV(mat_d, mat_d, perElemOpD, ir, ic); return; } - if (enable_smaller_matrices && ic * BN + BNover2 >= _ne1) { + if (ic * BN + BNover2 >= _ne1) { coopmat sum; sum = coopmat(0.0); [[dont_unroll]] for (uint block_k = start_k, i = 0; i < k_iters; block_k += BK, ++i) { - if ((block_k % QUANT_K) == 0) { + if ((block_k % qk) == 0) { store_scales(tid); } - if (block_k + BK < end_k && ((block_k + BK) % QUANT_K) == 0) { + if (block_k + BK < end_k && ((block_k + BK) % qk) == 0) { fetch_scales(ir * BM, pos_a, stride_a, block_k + BK, tid, false); } @@ -607,10 +728,10 @@ void main() { [[dont_unroll]] for (uint block_k = start_k, i = 0; i < k_iters; block_k += BK, ++i) { - if ((block_k % QUANT_K) == 0) { + if ((block_k % qk) == 0) { store_scales(tid); } - if (block_k + BK < end_k && ((block_k + BK) % QUANT_K) == 0) { + if (block_k + BK < end_k && ((block_k + BK) % qk) == 0) { fetch_scales(ir * BM, pos_a, stride_a, block_k + BK, tid, false); } diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl index 7d852dce..588fb735 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl @@ -20,651 +20,704 @@ void store_a(uint m, uint k_pair, FLOAT_TYPEV2 value) { void load_a_to_shmem(const uint pos_a, const uint row, const uint col, const uint idx_m, const uint block, const uint end_k) { #if defined(DATA_A_F32) || defined(DATA_A_F16) #if LOAD_VEC_A == 8 - if (ALIGNED != 0) { - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - const uint k_pair = row * LOAD_VEC_A / 2; - FLOAT_TYPEV8 aa = FLOAT_TYPEV8(data_a[idx]); - store_a(col, k_pair, aa[0].xy); - store_a(col, k_pair + 1, aa[0].zw); - store_a(col, k_pair + 2, aa[1].xy); - store_a(col, k_pair + 3, aa[1].zw); - return; - } + if (ALIGNED != 0) { + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + FLOAT_TYPEV8 aa = FLOAT_TYPEV8(data_a[idx]); + store_a(col, k_pair, aa[0].xy); + store_a(col, k_pair + 1, aa[0].zw); + store_a(col, k_pair + 2, aa[1].xy); + store_a(col, k_pair + 3, aa[1].zw); + return; + } #elif LOAD_VEC_A == 4 - if (ALIGNED != 0) { - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - const uint k_pair = row * LOAD_VEC_A / 2; - FLOAT_TYPEV4 aa = FLOAT_TYPEV4(data_a[idx]); - store_a(col, k_pair, aa.xy); - store_a(col, k_pair + 1, aa.zw); - return; - } + if (ALIGNED != 0) { + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + FLOAT_TYPEV4 aa = FLOAT_TYPEV4(data_a[idx]); + store_a(col, k_pair, aa.xy); + store_a(col, k_pair + 1, aa.zw); + return; + } #endif - const uint idx = pos_a + col * p.stride_a + row * 2; - if (idx_m < p.M && block + row * 2 + 1 < end_k) { - store_a(col, row, FLOAT_TYPEV2(data_a_scalar[idx], - data_a_scalar[idx + 1])); - } else if (idx_m < p.M && block + row * 2 < end_k) { - store_a(col, row, FLOAT_TYPEV2(data_a_scalar[idx], 0.0f)); - } else { - store_a(col, row, FLOAT_TYPEV2(0.0f)); - } + const uint idx = pos_a + col * p.stride_a + row * 2; + if (idx_m < p.M && block + row * 2 + 1 < end_k) { + store_a(col, row, FLOAT_TYPEV2(data_a_scalar[idx], + data_a_scalar[idx + 1])); + } else if (idx_m < p.M && block + row * 2 < end_k) { + store_a(col, row, FLOAT_TYPEV2(data_a_scalar[idx], 0.0f)); + } else { + store_a(col, row, FLOAT_TYPEV2(0.0f)); + } #elif defined(DATA_A_BF16) #if LOAD_VEC_A == 4 - if (ALIGNED != 0) { - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - const uint k_pair = row * LOAD_VEC_A / 2; - FLOAT_TYPEV4 aa = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_a[idx])); - store_a(col, k_pair, aa.xy); - store_a(col, k_pair + 1, aa.zw); - return; - } + if (ALIGNED != 0) { + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + FLOAT_TYPEV4 aa = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_a[idx])); + store_a(col, k_pair, aa.xy); + store_a(col, k_pair + 1, aa.zw); + return; + } #endif - const uint idx = pos_a + col * p.stride_a + row * 2; - if (idx_m < p.M && block + row * 2 + 1 < end_k) { - store_a(col, row, FLOAT_TYPEV2(TO_FLOAT_TYPE(data_a_scalar[idx]), - TO_FLOAT_TYPE(data_a_scalar[idx + 1]))); - } else if (idx_m < p.M && block + row * 2 < end_k) { - store_a(col, row, FLOAT_TYPEV2(TO_FLOAT_TYPE(data_a_scalar[idx]), 0.0f)); - } else { - store_a(col, row, FLOAT_TYPEV2(0.0f)); - } -#elif defined(DATA_A_Q4_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 4; - const uint iqs = idx & 0x03; - - const float d = float(data_a_packed16[ib].d); - const uint vui = uint(data_a_packed16[ib].qs[2*iqs]) | (uint(data_a_packed16[ib].qs[2*iqs + 1]) << 16); - const vec4 v0 = (vec4(unpack8(vui & 0x0F0F0F0F)) - 8.0f) * d; - const vec4 v1 = (vec4(unpack8((vui >> 4) & 0x0F0F0F0F)) - 8.0f) * d; - - const uint k_pair = row * LOAD_VEC_A / 4; - store_a(col, k_pair, FLOAT_TYPEV2(v0.xy)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v0.zw)); - store_a(col, k_pair + 8, FLOAT_TYPEV2(v1.xy)); - store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.zw)); -#elif defined(DATA_A_Q4_1) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 4; - const uint iqs = idx & 0x03; - - const vec2 dm = vec2(data_a_packed32[ib].dm); - const uint vui = data_a_packed32[ib].qs[iqs]; - const vec4 v0 = vec4(unpack8(vui & 0x0F0F0F0F)) * dm.x + dm.y; - const vec4 v1 = vec4(unpack8((vui >> 4) & 0x0F0F0F0F)) * dm.x + dm.y; - - const uint k_pair = row * LOAD_VEC_A / 4; - store_a(col, k_pair, FLOAT_TYPEV2(v0.xy)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v0.zw)); - store_a(col, k_pair + 8, FLOAT_TYPEV2(v1.xy)); - store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.zw)); -#elif defined(DATA_A_Q5_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 8; - const uint iqs = idx & 0x07; - - const float d = float(data_a_packed16[ib].d); - const uint uint_qh = uint(data_a_packed16[ib].qh[1]) << 16 | uint(data_a_packed16[ib].qh[0]); - const ivec2 qh0 = ivec2(((uint_qh >> 2*iqs) << 4) & 0x10, (uint_qh >> (2*iqs + 12)) & 0x10); - const ivec2 qh1 = ivec2(((uint_qh >> (2*iqs + 1)) << 4) & 0x10, (uint_qh >> (2*iqs + 13)) & 0x10); - - const uint vui = uint(data_a_packed16[ib].qs[iqs]); - const vec4 v = (vec4((vui & 0xF) | qh0.x, ((vui >> 4) & 0xF) | qh0.y, ((vui >> 8) & 0xF) | qh1.x, (vui >> 12) | qh1.y) - 16.0f) * d; - store_a(col, row, FLOAT_TYPEV2(v.xz)); - store_a(col, row + 8, FLOAT_TYPEV2(v.yw)); -#elif defined(DATA_A_Q5_1) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 4; - const uint iqs = idx & 0x03; - - const vec2 dm = vec2(data_a_packed32[ib].dm); - const uint uint_qh = data_a_packed32[ib].qh; - const uvec2 qh0 = uvec2(((uint_qh >> 4*iqs) << 4) & 0x10, (uint_qh >> (4*iqs + 12)) & 0x10); - const uvec2 qh1 = uvec2(((uint_qh >> (4*iqs + 1)) << 4) & 0x10, (uint_qh >> (4*iqs + 13)) & 0x10); - const uvec2 qh2 = uvec2(((uint_qh >> (4*iqs + 2)) << 4) & 0x10, (uint_qh >> (4*iqs + 14)) & 0x10); - const uvec2 qh3 = uvec2(((uint_qh >> (4*iqs + 3)) << 4) & 0x10, (uint_qh >> (4*iqs + 15)) & 0x10); - - const uint vui = data_a_packed32[ib].qs[iqs]; - const vec4 v0 = vec4((vui & 0xF) | qh0.x, ((vui >> 4) & 0xF) | qh0.y, ((vui >> 8) & 0xF) | qh1.x, ((vui >> 12) & 0xF) | qh1.y) * dm.x + dm.y; - const vec4 v1 = vec4(((vui >> 16) & 0xF) | qh2.x, ((vui >> 20) & 0xF) | qh2.y, ((vui >> 24) & 0xF) | qh3.x, ((vui >> 28) & 0xF) | qh3.y) * dm.x + dm.y; - - const uint k_pair = row * LOAD_VEC_A / 4; - store_a(col, k_pair, FLOAT_TYPEV2(v0.xz)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v1.xz)); - store_a(col, k_pair + 8, FLOAT_TYPEV2(v0.yw)); - store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.yw)); -#elif defined(DATA_A_Q8_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 8; - const uint iqs = idx & 0x07; - - const float d = float(data_a_packed16[ib].d); - const i8vec2 v0 = unpack8(int32_t(data_a_packed16[ib].qs[2*iqs])).xy; // vec4 used due to #12147 - const i8vec2 v1 = unpack8(int32_t(data_a_packed16[ib].qs[2*iqs + 1])).xy; - const vec4 v = vec4(v0.x, v0.y, v1.x, v1.y) * d; - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v.zw)); -#elif defined(DATA_A_Q1_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 16; - const uint iqs = idx & 0xfu; - - const float d = float(data_a[ib].d); - const uint bits = uint(data_a[ib].qs[iqs]); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2((bits & 0x01u) != 0u ? d : -d, (bits & 0x02u) != 0u ? d : -d)); - store_a(col, k_pair + 1, FLOAT_TYPEV2((bits & 0x04u) != 0u ? d : -d, (bits & 0x08u) != 0u ? d : -d)); - store_a(col, k_pair + 2, FLOAT_TYPEV2((bits & 0x10u) != 0u ? d : -d, (bits & 0x20u) != 0u ? d : -d)); - store_a(col, k_pair + 3, FLOAT_TYPEV2((bits & 0x40u) != 0u ? d : -d, (bits & 0x80u) != 0u ? d : -d)); -#elif defined(DATA_A_Q2_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 16; - const uint iqs = idx & 0xfu; - - const FLOAT_TYPE d = FLOAT_TYPE(data_a[ib].d); - const uint bits = uint(data_a[ib].qs[iqs]); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, d * (FLOAT_TYPEV2(bits & 3u, (bits >> 2u) & 3u) - FLOAT_TYPEV2(1.0f))); - store_a(col, k_pair + 1, d * (FLOAT_TYPEV2((bits >> 4u) & 3u, bits >> 6u) - FLOAT_TYPEV2(1.0f))); -#elif defined(DATA_A_Q2_K) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 64; // 4 values per idx - const uint iqs = (idx % 64) * 2; // 0,2,4..126 - - const uint qsi = (iqs / 64) * 16 + (iqs % 16); // 0..15 - const uint scalesi = iqs / 8; // 0..15 - const uint qsshift = ((iqs % 64) / 16) * 2; // 0,2,4,6 - - const vec4 qs = vec4(unpack8((data_a_packed32[ib].qs[qsi / 2] >> qsshift) & 0x03030303)); - const uint scales = data_a[ib].scales[scalesi]; - const vec2 dm = vec2(data_a[ib].dm); - - const vec4 v = dm.x * float(scales & 0xF) * qs - dm.y * float(scales >> 4); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v.zw)); -#elif defined(DATA_A_TQ2_0) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 128; // 2 values per idx - const uint iqs = (idx % 128) * 2; // elem 0,2,4..254 - - const uint qsi = (iqs / 128) * 32 + (iqs % 32); // byte pair start - const uint shift = 2 * ((iqs % 128) / 32); // 0,2,4,6 - - const uvec2 qs = uvec2(data_a[ib].qs[qsi], data_a[ib].qs[qsi + 1]); - const float d = float(data_a[ib].d); - - const vec2 v = d * (vec2((qs >> shift) & 3) - 1.0); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); -#elif defined(DATA_A_Q3_K) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 128; // 2 values per idx - const uint iqs = idx % 128; // 0..127 - - const uint n = iqs / 64; // 0,1 - const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..62 - const uint hmi = (iqs % 16) * 2; // 0,2,4..30 - const uint j = (iqs % 64) / 4; // 0..3 - const uint is = iqs / 8; // 0..15 - const uint halfsplit = ((iqs % 64) / 16); // 0,1,2,3 - const uint qsshift = halfsplit * 2; // 0,2,4,6 - - const int8_t us = int8_t(((data_a[ib].scales[is % 8] >> (4 * int(is / 8))) & 0xF) - | (((data_a[ib].scales[8 + (is % 4)] >> (2 * int(is / 4))) & 3) << 4)); - const float dl = float(data_a[ib].d) * float(us - 32); - - const vec2 qs = vec2(unpack8((uint(data_a_packed16[ib].qs[qsi / 2]) >> qsshift) & 0x0303).xy); - const vec2 hm = vec2(unpack8(((uint(data_a_packed16[ib].hmask[hmi / 2]) >> (4 * n + halfsplit)) & 0x0101 ^ 0x0101) << 2).xy); - - store_a(col, row * LOAD_VEC_A / 2, FLOAT_TYPEV2(dl * (qs.x - hm.x), - dl * (qs.y - hm.y))); -#elif defined(DATA_A_Q4_K) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 64; // 4 values per idx - const uint iqs = (idx % 64) * 2; // 0,2,4..126 - - const uint n = iqs / 32; // 0,1,2,3 - const uint b = (iqs % 32) / 16; // 0,1 - const uint is = 2 * n + b; // 0..7 - const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..126 - - const vec2 loadd = vec2(data_a[ib].dm); - - const uvec3 scales = uvec3(data_a_packed32[ib].scales[0], - data_a_packed32[ib].scales[1], - data_a_packed32[ib].scales[2]); - const uint scalesoffs = (is & 3) * 8; + const uint idx = pos_a + col * p.stride_a + row * 2; + if (idx_m < p.M && block + row * 2 + 1 < end_k) { + store_a(col, row, FLOAT_TYPEV2(TO_FLOAT_TYPE(data_a_scalar[idx]), + TO_FLOAT_TYPE(data_a_scalar[idx + 1]))); + } else if (idx_m < p.M && block + row * 2 < end_k) { + store_a(col, row, FLOAT_TYPEV2(TO_FLOAT_TYPE(data_a_scalar[idx]), 0.0f)); + } else { + store_a(col, row, FLOAT_TYPEV2(0.0f)); + } +#elif defined(DATA_A_IQ1_S) + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; - const uint scidx0 = (is < 4) ? 0 : 2; - const uint scidxshift0 = scalesoffs; - const uint scidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; - const uint mbidx0 = (is < 4) ? 1 : 2; - const uint mbidxshift0 = (is < 4) ? scalesoffs : scalesoffs + 4; - const uint mbidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; - - const uint8_t sc = uint8_t(((scales[scidx0] >> scidxshift0) & 0xF) | ((scales[0] >> scidxshift1) & 0x30)); - const uint8_t mbyte = uint8_t(((scales[mbidx0] >> mbidxshift0) & 0xF) | ((scales[1] >> mbidxshift1) & 0x30)); - - const float d = loadd.x * sc; - const float m = -loadd.y * mbyte; + const uint ib = idx / 32; + const uint ib32 = (idx % 32) / 4; + const uint ib8 = idx % 32; - const vec4 q = vec4(unpack8((data_a_packed32[ib].qs[qsi / 4] >> (b * 4)) & 0x0F0F0F0F)); + const float d = float(data_a[ib].d); + const uint qh = data_a[ib].qh[ib32]; + const uint qs = data_a[ib].qs[ib8]; + const float dl = d * (2 * bitfieldExtract(qh, 12, 3) + 1); + const float delta = ((qh & 0x8000) != 0) ? -IQ1S_DELTA : IQ1S_DELTA; + const int16_t grid = int16_t(iq1s_grid[qs | (bitfieldExtract(qh, 3 * int(ib8 & 3), 3) << 8)]); - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(fma(d, q.x, m), fma(d, q.y, m))); - store_a(col, k_pair + 1, FLOAT_TYPEV2(fma(d, q.z, m), fma(d, q.w, m))); -#elif defined(DATA_A_Q5_K) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + [[unroll]] for (int k = 0; k < 4; ++k) { + store_a(col, k_pair + k, FLOAT_TYPEV2(dl * (bitfieldExtract(grid, 4 * k , 2) + delta), + dl * (bitfieldExtract(grid, 4 * k + 2, 2) + delta))); - const uint ib = idx / 64; // 4 values per idx - const uint iqs = (idx % 64) * 2; // 0,2,4..126 + } +#elif defined(DATA_A_IQ1_M) + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + + const uint ib = idx / 32; + const uint ib8 = idx % 32; + const uint ib16 = ib8 / 2; + + const uint16_t[4] scales = data_a[ib].scales; + const u16vec4 s = u16vec4(scales[0], scales[1], scales[2], scales[3]) >> 12; + const float d = float(unpackHalf2x16(s.x | (s.y << 4) | (s.z << 8) | (s.w << 12)).x); + const uint sc = scales[ib8 / 8]; + const uint qs = data_a[ib].qs[ib8]; + const uint qh = data_a[ib].qh[ib16] >> (4 * (ib8 & 1)); + const float dl = d * (2 * bitfieldExtract(sc, 3 * int(ib16 & 3), 3) + 1); + const float delta = ((qh & 8) != 0) ? -IQ1M_DELTA : IQ1M_DELTA; + const int16_t grid = int16_t(iq1s_grid[qs | ((qh & 7) << 8)]); + + [[unroll]] for (int k = 0; k < 4; ++k) { + store_a(col, k_pair + k, FLOAT_TYPEV2(dl * (bitfieldExtract(grid, 4 * k , 2) + delta), + dl * (bitfieldExtract(grid, 4 * k + 2, 2) + delta))); - const uint n = iqs / 32; // 0,1,2,3 - const uint b = (iqs % 32) / 16; // 0,1 - const uint is = 2 * n + b; // 0..7 - const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..126 - const uint qhi = (iqs % 16) * 2; // 0,2,4..30 + } +#elif defined(DATA_A_IQ2_XXS) + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + + const uint ib = idx / 32; + const uint ib32 = (idx % 32) / 4; + const uint ib8 = idx % 4; + + const float d = float(data_a[ib].d); + const uint qs = data_a[ib].qs[8 * ib32 + ib8]; + const uint signs = pack32(u8vec4( + data_a[ib].qs[8*ib32 + 4], + data_a[ib].qs[8*ib32 + 5], + data_a[ib].qs[8*ib32 + 6], + data_a[ib].qs[8*ib32 + 7] + )); + const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + (signs >> 28))); + const uint32_t sign7 = bitfieldExtract(signs, 7 * int(ib8), 7); + const uint sign = sign7 | (bitCount(sign7) << 7); + const uvec2 grid = iq2xxs_grid[qs]; + const vec4 grid0 = vec4(unpack8(grid.x)); + const vec4 grid1 = vec4(unpack8(grid.y)); + + store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, + (sign & 2) != 0 ? -grid0.y : grid0.y)); + + store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, + (sign & 8) != 0 ? -grid0.w : grid0.w)); + + store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, + (sign & 32) != 0 ? -grid1.y : grid1.y)); + + store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, + (sign & 128) != 0 ? -grid1.w : grid1.w)); - const vec2 loadd = vec2(data_a[ib].dm); +#elif defined(DATA_A_IQ2_XS) + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; - const uvec3 scales = uvec3(data_a_packed32[ib].scales[0], - data_a_packed32[ib].scales[1], - data_a_packed32[ib].scales[2]); - const uint scalesoffs = (is & 3) * 8; + const uint ib = idx / 32; + const uint ib32 = (idx % 32) / 4; + const uint ib8 = idx % 4; - const uint scidx0 = (is < 4) ? 0 : 2; - const uint scidxshift0 = scalesoffs; - const uint scidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; - const uint mbidx0 = (is < 4) ? 1 : 2; - const uint mbidxshift0 = (is < 4) ? scalesoffs : scalesoffs + 4; - const uint mbidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; + const float d = float(data_a[ib].d); + const uint scale = (data_a[ib].scales[ib32] >> (2 * (ib8 & 2))) & 0xf; + const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + scale)); + const uint qs = data_a[ib].qs[4 * ib32 + ib8]; + const uint sign7 = qs >> 9; + const uint sign = sign7 | (bitCount(sign7) << 7); + const uvec2 grid = iq2xs_grid[qs & 511]; + const vec4 grid0 = vec4(unpack8(grid.x)); + const vec4 grid1 = vec4(unpack8(grid.y)); - const uint8_t sc = uint8_t(((scales[scidx0] >> scidxshift0) & 0xF) | ((scales[0] >> scidxshift1) & 0x30)); - const uint8_t mbyte = uint8_t(((scales[mbidx0] >> mbidxshift0) & 0xF) | ((scales[1] >> mbidxshift1) & 0x30)); - - const float d = loadd.x * sc; - const float m = -loadd.y * mbyte; - - const uint qs = (data_a_packed32[ib].qs[qsi / 4] >> (b * 4)) & 0x0F0F0F0F; - const uint qh = ((data_a_packed32[ib].qh[qhi / 4] >> (iqs / 16)) & 0x01010101) << 4; - const vec4 q = vec4(unpack8(qs | qh)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(fma(d, q.x, m), fma(d, q.y, m))); - store_a(col, k_pair + 1, FLOAT_TYPEV2(fma(d, q.z, m), fma(d, q.w, m))); -#elif defined(DATA_A_Q6_K) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 128; // 2 values per idx - const uint iqs = idx % 128; // 0..127 - - const uint n = iqs / 64; // 0,1 - const uint b = ((iqs % 64) / 32) * 4; // 0,4 - const uint is_b = (iqs % 16) / 8; // 0,1 - const uint qhshift = ((iqs % 64) / 16) * 2; // 0,2,4,6 - const uint is = 8 * n + qhshift + is_b; // 0..15 - const uint qsi = n * 32 + (iqs % 32); // 0..63 - const uint qhi = n * 16 + (iqs % 16); // 0..31 - - const float dscale = float(data_a[ib].d) * float(data_a[ib].scales[is]); + store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, + (sign & 2) != 0 ? -grid0.y : grid0.y)); - const uint ql = (uint(data_a_packed16[ib].ql[qsi]) >> b) & 0x0F0F; - const uint qh = (uint(data_a_packed16[ib].qh[qhi]) >> qhshift) & 0x0303; - const vec2 q = (vec2(unpack8(ql | (qh << 4)).xy) - 32) * dscale; + store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, + (sign & 8) != 0 ? -grid0.w : grid0.w)); + + store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, + (sign & 32) != 0 ? -grid1.y : grid1.y)); + + store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, + (sign & 128) != 0 ? -grid1.w : grid1.w)); - store_a(col, row * LOAD_VEC_A / 2, FLOAT_TYPEV2(q.x, q.y)); -#elif defined(DATA_A_IQ1_S) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 32; // 8 values per idx - const uint ib32 = (idx % 32) / 4; // 0..7 - const uint ib8 = idx % 32; - - const float d = float(data_a[ib].d); - const uint qh = data_a[ib].qh[ib32]; - const uint qs = data_a[ib].qs[ib8]; - const float dl = d * (2 * bitfieldExtract(qh, 12, 3) + 1); - const float delta = ((qh & 0x8000) != 0) ? -IQ1S_DELTA : IQ1S_DELTA; - const int16_t grid = int16_t(iq1s_grid[qs | (bitfieldExtract(qh, 3 * int(ib8 & 3), 3) << 8)]); - - const uint k_pair = row * LOAD_VEC_A / 2; - [[unroll]] for (int k = 0; k < 4; ++k) { - store_a(col, k_pair + k, FLOAT_TYPEV2(dl * (bitfieldExtract(grid, 4 * k , 2) + delta), - dl * (bitfieldExtract(grid, 4 * k + 2, 2) + delta))); - } -#elif defined(DATA_A_IQ1_M) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 32; // 8 values per idx - const uint ib8 = idx % 32; - const uint ib16 = ib8 / 2; - - const uint16_t[4] scales = data_a[ib].scales; - const u16vec4 s = u16vec4(scales[0], scales[1], scales[2], scales[3]) >> 12; - const float d = float(unpackHalf2x16(s.x | (s.y << 4) | (s.z << 8) | (s.w << 12)).x); - const uint sc = scales[ib8 / 8]; - const uint qs = data_a[ib].qs[ib8]; - const uint qh = data_a[ib].qh[ib16] >> (4 * (ib8 & 1)); - const float dl = d * (2 * bitfieldExtract(sc, 3 * int(ib16 & 3), 3) + 1); - const float delta = ((qh & 8) != 0) ? -IQ1M_DELTA : IQ1M_DELTA; - const int16_t grid = int16_t(iq1s_grid[qs | ((qh & 7) << 8)]); - - const uint k_pair = row * LOAD_VEC_A / 2; - [[unroll]] for (int k = 0; k < 4; ++k) { - store_a(col, k_pair + k, FLOAT_TYPEV2(dl * (bitfieldExtract(grid, 4 * k , 2) + delta), - dl * (bitfieldExtract(grid, 4 * k + 2, 2) + delta))); - } -#elif defined(DATA_A_IQ2_XXS) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 32; // 8 values per idx - const uint ib32 = (idx % 32) / 4; // 0..7 - const uint ib8 = idx % 4; - - const float d = float(data_a[ib].d); - const uint qs = data_a[ib].qs[8 * ib32 + ib8]; - const uint signs = pack32(u8vec4( - data_a[ib].qs[8*ib32 + 4], - data_a[ib].qs[8*ib32 + 5], - data_a[ib].qs[8*ib32 + 6], - data_a[ib].qs[8*ib32 + 7] - )); - const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + (signs >> 28))); - const uint32_t sign7 = bitfieldExtract(signs, 7 * int(ib8), 7); - const uint sign = sign7 | (bitCount(sign7) << 7); - const uvec2 grid = iq2xxs_grid[qs]; - const vec4 grid0 = vec4(unpack8(grid.x)); - const vec4 grid1 = vec4(unpack8(grid.y)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, - (sign & 2) != 0 ? -grid0.y : grid0.y)); - store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, - (sign & 8) != 0 ? -grid0.w : grid0.w)); - store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, - (sign & 32) != 0 ? -grid1.y : grid1.y)); - store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, - (sign & 128) != 0 ? -grid1.w : grid1.w)); -#elif defined(DATA_A_IQ2_XS) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 32; // 8 values per idx - const uint ib32 = (idx % 32) / 4; // 0..7 - const uint ib8 = idx % 4; // 0..3 - - const float d = float(data_a[ib].d); - const uint scale = (data_a[ib].scales[ib32] >> (2 * (ib8 & 2))) & 0xf; - const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + scale)); - const uint qs = data_a[ib].qs[4 * ib32 + ib8]; - const uint sign7 = qs >> 9; - const uint sign = sign7 | (bitCount(sign7) << 7); - const uvec2 grid = iq2xs_grid[qs & 511]; - const vec4 grid0 = vec4(unpack8(grid.x)); - const vec4 grid1 = vec4(unpack8(grid.y)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, - (sign & 2) != 0 ? -grid0.y : grid0.y)); - store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, - (sign & 8) != 0 ? -grid0.w : grid0.w)); - store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, - (sign & 32) != 0 ? -grid1.y : grid1.y)); - store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, - (sign & 128) != 0 ? -grid1.w : grid1.w)); #elif defined(DATA_A_IQ2_S) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 32; // 8 values per idx - const uint ib8 = idx % 32; // 0..31 - const uint ib32 = ib8 / 4; // 0..7 - - const uint scale = (data_a[ib].scales[ib32] >> (2 * (ib8 & 2))) & 0xf; - const uint qs = data_a[ib].qs[ib8]; - const uint qh = data_a[ib].qh[ib32]; - const uint qhshift = 2 * (ib8 % 4); - const uint sign = data_a[ib].qs[QUANT_K / 8 + ib8]; - - const float d = float(data_a[ib].d); - const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + scale)); - const uvec2 grid = iq2s_grid[qs | ((qh << (8 - qhshift)) & 0x300)]; - const vec4 grid0 = vec4(unpack8(grid.x)); - const vec4 grid1 = vec4(unpack8(grid.y)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, - (sign & 2) != 0 ? -grid0.y : grid0.y)); - store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, - (sign & 8) != 0 ? -grid0.w : grid0.w)); - store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, - (sign & 32) != 0 ? -grid1.y : grid1.y)); - store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, - (sign & 128) != 0 ? -grid1.w : grid1.w)); + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + + const uint ib = idx / 32; + const uint ib8 = idx % 32; + const uint ib32 = ib8 / 4; + + const uint scale = (data_a[ib].scales[ib32] >> (2 * (ib8 & 2))) & 0xf; + const uint qs = data_a[ib].qs[ib8]; + const uint qh = data_a[ib].qh[ib32]; + const uint qhshift = 2 * (ib8 % 4); + const uint sign = data_a[ib].qs[QUANT_K_IQ2_S / 8 + ib8]; + + const float d = float(data_a[ib].d); + const FLOAT_TYPE db = FLOAT_TYPE(d * 0.25 * (0.5 + scale)); + const uvec2 grid = iq2s_grid[qs | ((qh << (8 - qhshift)) & 0x300)]; + const vec4 grid0 = vec4(unpack8(grid.x)); + const vec4 grid1 = vec4(unpack8(grid.y)); + + store_a(col, k_pair, db * FLOAT_TYPEV2((sign & 1) != 0 ? -grid0.x : grid0.x, + (sign & 2) != 0 ? -grid0.y : grid0.y)); + + store_a(col, k_pair + 1, db * FLOAT_TYPEV2((sign & 4) != 0 ? -grid0.z : grid0.z, + (sign & 8) != 0 ? -grid0.w : grid0.w)); + + store_a(col, k_pair + 2, db * FLOAT_TYPEV2((sign & 16) != 0 ? -grid1.x : grid1.x, + (sign & 32) != 0 ? -grid1.y : grid1.y)); + + store_a(col, k_pair + 3, db * FLOAT_TYPEV2((sign & 64) != 0 ? -grid1.z : grid1.z, + (sign & 128) != 0 ? -grid1.w : grid1.w)); + #elif defined(DATA_A_IQ3_XXS) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 64; // 4 values per idx - const uint iqs = idx % 64; // 0..63 - const uint is = QUANT_K / 4 + 4 * (iqs / 8); // 8 values - - const float d = float(data_a[ib].d); - const uint qs = data_a[ib].qs[iqs]; - const uint signs = pack32(u16vec2( - data_a_packed16[ib].qs[is/2], - data_a_packed16[ib].qs[is/2+1] - )); - const float db = d * 0.5 * (0.5 + (signs >> 28)); - const uint32_t sign7 = bitfieldExtract(signs, 7 * (int(iqs / 2) % 4), 7); - const uint sign = (sign7 | (bitCount(sign7) << 7)) >> (4 * (idx % 2)); - const uint grid = iq3xxs_grid[qs]; - const vec4 v = db * vec4(unpack8(grid)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2((sign & 1) != 0 ? -v.x : v.x, - (sign & 2) != 0 ? -v.y : v.y)); - store_a(col, k_pair + 1, FLOAT_TYPEV2((sign & 4) != 0 ? -v.z : v.z, - (sign & 8) != 0 ? -v.w : v.w)); + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + + const uint ib = idx / 64; + const uint iqs = idx % 64; + const uint is = QUANT_K_IQ3_XXS / 4 + 4 * (iqs / 8); + + const float d = float(data_a[ib].d); + const uint qs = data_a[ib].qs[iqs]; + const uint signs = pack32(u16vec2( + data_a_packed16[ib].qs[is/2], + data_a_packed16[ib].qs[is/2+1] + )); + const float db = d * 0.5 * (0.5 + (signs >> 28)); + const uint32_t sign7 = bitfieldExtract(signs, 7 * (int(iqs / 2) % 4), 7); + const uint sign = (sign7 | (bitCount(sign7) << 7)) >> (4 * (idx % 2)); + const uint grid = iq3xxs_grid[qs]; + const vec4 v = db * vec4(unpack8(grid)); + + store_a(col, k_pair, FLOAT_TYPEV2((sign & 1) != 0 ? -v.x : v.x, + (sign & 2) != 0 ? -v.y : v.y)); + + store_a(col, k_pair + 1, FLOAT_TYPEV2((sign & 4) != 0 ? -v.z : v.z, + (sign & 8) != 0 ? -v.w : v.w)); + #elif defined(DATA_A_IQ3_S) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - - const uint ib = idx / 64; // 4 values per idx - const uint iqs = idx % 64; // 0..63 - const uint iqh = iqs / 8; - - const float d = float(data_a[ib].d); - const uint qs = data_a[ib].qs[iqs]; - const uint qh = data_a[ib].qh[iqh]; - const int8_t sign = int8_t(data_a[ib].signs[iqs / 2] >> (4 * (idx % 2))); - const uint scale = data_a[ib].scales[iqs / 16]; - const i8vec2 sign01 = i8vec2(1 - (2 & i8vec2(sign << 1, sign))); - const float db = d * (1 + 2 * ((scale >> (4 * (iqh & 1))) & 0xf)); - const uint32_t grid = iq3s_grid[qs | ((qh << (8 - (iqs % 8))) & 256)]; - const vec4 v = db * vec4(unpack8(grid)); - - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2((sign & 1) != 0 ? -v.x : v.x, - (sign & 2) != 0 ? -v.y : v.y)); - store_a(col, k_pair + 1, FLOAT_TYPEV2((sign & 4) != 0 ? -v.z : v.z, - (sign & 8) != 0 ? -v.w : v.w)); + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 2; + + const uint ib = idx / 64; + const uint iqs = idx % 64; + const uint iqh = iqs / 8; + + const float d = float(data_a[ib].d); + const uint qs = data_a[ib].qs[iqs]; + const uint qh = data_a[ib].qh[iqh]; + const int8_t sign = int8_t(data_a[ib].signs[iqs / 2] >> (4 * (idx % 2))); + const uint scale = data_a[ib].scales[iqs / 16]; + const i8vec2 sign01 = i8vec2(1 - (2 & i8vec2(sign << 1, sign))); + const float db = d * (1 + 2 * ((scale >> (4 * (iqh & 1))) & 0xf)); + const uint32_t grid = iq3s_grid[qs | ((qh << (8 - (iqs % 8))) & 256)]; + const vec4 v = db * vec4(unpack8(grid)); + + store_a(col, k_pair, FLOAT_TYPEV2((sign & 1) != 0 ? -v.x : v.x, + (sign & 2) != 0 ? -v.y : v.y)); + + store_a(col, k_pair + 1, FLOAT_TYPEV2((sign & 4) != 0 ? -v.z : v.z, + (sign & 8) != 0 ? -v.w : v.w)); + #elif defined(DATA_A_IQ4_XS) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 4; + + const uint ib = idx / 32; + const uint ib32 = (idx % 32) / 4; + const uint iq = 4 * ib32 + (idx % 4); - const uint ib = idx / 64; // 4 values per idx - const uint ib32 = (idx % 64) / 8; // 0..7 - const uint iq = 4 * ib32 + (idx % 4); + const uint sl = (data_a[ib].scales_l[ib32/2] >> (4 * (ib32 & 1))) & 0xF; + const uint sh = ((data_a[ib].scales_h) >> (2 * ib32)) & 3; + const float d = float(data_a[ib].d); + const float dl = d * float(int(sl | (sh << 4)) - 32); + const uint vui = uint(data_a_packed32[ib].qs[iq]); - const uint sl = (data_a[ib].scales_l[ib32/2] >> (4 * (ib32 & 1))) & 0xF; - const uint sh = ((data_a[ib].scales_h) >> (2 * ib32)) & 3; - const uint qshift = idx & 4; - u8vec4 qs = unpack8((uint(data_a_packed32[ib].qs[iq]) >> qshift) & 0x0F0F0F0F); + const u8vec4 qs0 = unpack8( vui & 0x0F0F0F0F); + const u8vec4 qs1 = unpack8((vui >> 4) & 0x0F0F0F0F); + const vec4 v0 = dl * vec4(kvalues_iq4nl[qs0.x], kvalues_iq4nl[qs0.y], kvalues_iq4nl[qs0.z], kvalues_iq4nl[qs0.w]); + const vec4 v1 = dl * vec4(kvalues_iq4nl[qs1.x], kvalues_iq4nl[qs1.y], kvalues_iq4nl[qs1.z], kvalues_iq4nl[qs1.w]); - const float d = float(data_a[ib].d); - const vec4 v = d * float(int(sl | (sh << 4)) - 32) * vec4(kvalues_iq4nl[qs.x], kvalues_iq4nl[qs.y], kvalues_iq4nl[qs.z], kvalues_iq4nl[qs.w]); + store_a(col, k_pair, FLOAT_TYPEV2(v0.xy)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v0.zw)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(v1.xy)); + store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.zw)); - const uint k_pair = row * LOAD_VEC_A / 2; - store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); - store_a(col, k_pair + 1, FLOAT_TYPEV2(v.zw)); #elif defined(DATA_A_IQ4_NL) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 4; + + const uint ib = idx / 8; + const uint iqs = idx & 0x07; - const uint ib = idx / 8; - const uint iqs = idx & 0x07; + const FLOAT_TYPE d = FLOAT_TYPE(data_a_packed16[ib].d); + const uint vui = uint(data_a_packed16[ib].qs[iqs]); - const FLOAT_TYPE d = FLOAT_TYPE(data_a_packed16[ib].d); - const uint vui = uint(data_a_packed16[ib].qs[iqs]); + store_a(col, k_pair, d * FLOAT_TYPEV2(kvalues_iq4nl[vui & 0xF], + kvalues_iq4nl[bitfieldExtract(vui, 8, 4)])); + + store_a(col, k_pair + 8, d * FLOAT_TYPEV2(kvalues_iq4nl[bitfieldExtract(vui, 4, 4)], + kvalues_iq4nl[vui >> 12])); - const uint k_pair = row * LOAD_VEC_A / 4; - store_a(col, k_pair, d * FLOAT_TYPEV2(kvalues_iq4nl[vui & 0xF], - kvalues_iq4nl[bitfieldExtract(vui, 8, 4)])); - store_a(col, k_pair + 8, d * FLOAT_TYPEV2(kvalues_iq4nl[bitfieldExtract(vui, 4, 4)], - kvalues_iq4nl[vui >> 12])); #elif defined(DATA_A_MXFP4) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint k_pair = row * LOAD_VEC_A / 4; - const uint ib = idx / 8; - const uint iqs = (idx & 0x07) * 2; + const uint ib = idx / 8; + const uint iqs = (idx & 0x07) * 2; - const uint vui = uint(data_a[ib].qs[iqs]); - const uint vui2 = uint(data_a[ib].qs[iqs+1]); + const uint vui = uint(data_a[ib].qs[iqs]); + const uint vui2 = uint(data_a[ib].qs[iqs+1]); #ifdef USE_OCP_FP4 - const float d = e8m0_to_fp32(data_a[ib].e); - const u8vec2 packed = u8vec2(vui, vui2); - store_a(col, row, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 0u)) * FLOAT_TYPE(d)); - store_a(col, row + 8, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 4u)) * FLOAT_TYPE(d)); + const float d = e8m0_to_fp32(data_a[ib].e); + const u8vec2 packed = u8vec2(vui, vui2); + store_a(col, k_pair, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 0u)) * FLOAT_TYPE(d)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 4u)) * FLOAT_TYPE(d)); #else - const float d = e8m0_to_fp32(data_a[ib].e) * 0.5; - store_a(col, row, FLOAT_TYPEV2(kvalues_mxfp4[vui & 0xF] * d, - kvalues_mxfp4[vui2 & 0xF] * d)); - store_a(col, row + 8, FLOAT_TYPEV2(kvalues_mxfp4[vui >> 4] * d, - kvalues_mxfp4[vui2 >> 4] * d)); + const float d = e8m0_to_fp32(data_a[ib].e) * 0.5; + store_a(col, k_pair, FLOAT_TYPEV2(kvalues_mxfp4[vui & 0xF] * d, + kvalues_mxfp4[vui2 & 0xF] * d)); + + store_a(col, k_pair + 8, FLOAT_TYPEV2(kvalues_mxfp4[vui >> 4] * d, + kvalues_mxfp4[vui2 >> 4] * d)); + #endif #elif defined(DATA_A_NVFP4) - const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; - const uint ib = idx / 16u; - const uint sub = (idx & 0xC) >> 2; - const uint iqs = (idx & 0xF) * 2; - const uint vui = uint(data_a[ib].qs[iqs]); - const uint vui2 = uint(data_a[ib].qs[iqs+1]); - - // lo and hi nibbles are 8 elements apart, which doesn't quite line up with - // how the thread mapping and buf_idx calculation works for other types. - const uint eff_row = (row & 3) + (row & ~3) * 2; + const uint idx = pos_a + col * p.stride_a / LOAD_VEC_A + row; + const uint eff_row = (row & 3) + (row & ~3) * 2; + + const uint ib = idx / 16u; + const uint sub = (idx & 0xC) >> 2; + const uint iqs = (idx & 0xF) * 2; + const uint vui = uint(data_a[ib].qs[iqs]); + const uint vui2 = uint(data_a[ib].qs[iqs+1]); + #ifdef USE_OCP_FP4 - const FLOAT_TYPE d = FLOAT_TYPE(ue4m3_from_bits(data_a[ib].d[sub])); - const u8vec2 packed = u8vec2(vui, vui2); - store_a(col, eff_row, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 0u)) * d); - store_a(col, eff_row + 4, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 4u)) * d); + const FLOAT_TYPE d = FLOAT_TYPE(ue4m3_from_bits(data_a[ib].d[sub])); + const u8vec2 packed = u8vec2(vui, vui2); + store_a(col, eff_row, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 0u)) * d); + store_a(col, eff_row + 4, FLOAT_TYPEV2(bitcastExtractfe2m1EXT(packed, 4u)) * d); #else - const float d = ue4m3_to_fp32(data_a[ib].d[sub]) * 0.5; - store_a(col, eff_row, FLOAT_TYPEV2(kvalues_mxfp4[vui & 0xF] * d, - kvalues_mxfp4[vui2 & 0xF] * d)); - store_a(col, eff_row + 4, FLOAT_TYPEV2(kvalues_mxfp4[vui >> 4] * d, - kvalues_mxfp4[vui2 >> 4] * d)); + const float d = ue4m3_to_fp32(data_a[ib].d[sub]) * 0.5; + store_a(col, eff_row, FLOAT_TYPEV2(kvalues_mxfp4[vui & 0xF] * d, + kvalues_mxfp4[vui2 & 0xF] * d)); + store_a(col, eff_row + 4, FLOAT_TYPEV2(kvalues_mxfp4[vui >> 4] * d, + kvalues_mxfp4[vui2 >> 4] * d)); #endif +#else + if (MmTypeA == GGML_TYPE_Q4_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 4; + + const uint ib = idx / 4; + const uint iqs = idx & 0x03; + + const float d = float(a_q4_0_p16.data[ib].d); + const uint vui = uint(a_q4_0_p16.data[ib].qs[2*iqs]) | (uint(a_q4_0_p16.data[ib].qs[2*iqs + 1]) << 16); + const vec4 v0 = (vec4(unpack8(vui & 0x0F0F0F0F)) - 8.0f) * d; + const vec4 v1 = (vec4(unpack8((vui >> 4) & 0x0F0F0F0F)) - 8.0f) * d; + + store_a(col, k_pair, FLOAT_TYPEV2(v0.xy)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v0.zw)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(v1.xy)); + store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.zw)); + } else if (MmTypeA == GGML_TYPE_Q4_1) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 4; + + const uint ib = idx / 4; + const uint iqs = idx & 0x03; + + const vec2 dm = vec2(a_q4_1_p32.data[ib].dm); + const uint vui = a_q4_1_p32.data[ib].qs[iqs]; + const vec4 v0 = vec4(unpack8(vui & 0x0F0F0F0F)) * dm.x + dm.y; + const vec4 v1 = vec4(unpack8((vui >> 4) & 0x0F0F0F0F)) * dm.x + dm.y; + + store_a(col, k_pair, FLOAT_TYPEV2(v0.xy)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v0.zw)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(v1.xy)); + store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.zw)); + } else if (MmTypeA == GGML_TYPE_Q5_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 4; + + const uint ib = idx / 8; + const uint iqs = idx & 0x07; + + const float d = float(a_q5_0_p16.data[ib].d); + const uint uint_qh = uint(a_q5_0_p16.data[ib].qh[1]) << 16 | uint(a_q5_0_p16.data[ib].qh[0]); + const ivec2 qh0 = ivec2(((uint_qh >> 2*iqs) << 4) & 0x10, (uint_qh >> (2*iqs + 12)) & 0x10); + const ivec2 qh1 = ivec2(((uint_qh >> (2*iqs + 1)) << 4) & 0x10, (uint_qh >> (2*iqs + 13)) & 0x10); + + const uint vui = uint(a_q5_0_p16.data[ib].qs[iqs]); + const vec4 v = (vec4((vui & 0xF) | qh0.x, ((vui >> 4) & 0xF) | qh0.y, ((vui >> 8) & 0xF) | qh1.x, (vui >> 12) | qh1.y) - 16.0f) * d; + + store_a(col, k_pair, FLOAT_TYPEV2(v.xz)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(v.yw)); + } else if (MmTypeA == GGML_TYPE_Q5_1) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 4; + + const uint ib = idx / 4; + const uint iqs = idx & 0x03; + + const vec2 dm = vec2(a_q5_1_p32.data[ib].dm); + const uint uint_qh = a_q5_1_p32.data[ib].qh; + const uvec2 qh0 = uvec2(((uint_qh >> 4*iqs) << 4) & 0x10, (uint_qh >> (4*iqs + 12)) & 0x10); + const uvec2 qh1 = uvec2(((uint_qh >> (4*iqs + 1)) << 4) & 0x10, (uint_qh >> (4*iqs + 13)) & 0x10); + const uvec2 qh2 = uvec2(((uint_qh >> (4*iqs + 2)) << 4) & 0x10, (uint_qh >> (4*iqs + 14)) & 0x10); + const uvec2 qh3 = uvec2(((uint_qh >> (4*iqs + 3)) << 4) & 0x10, (uint_qh >> (4*iqs + 15)) & 0x10); + + const uint vui = a_q5_1_p32.data[ib].qs[iqs]; + const vec4 v0 = vec4((vui & 0xF) | qh0.x, ((vui >> 4) & 0xF) | qh0.y, ((vui >> 8) & 0xF) | qh1.x, ((vui >> 12) & 0xF) | qh1.y) * dm.x + dm.y; + const vec4 v1 = vec4(((vui >> 16) & 0xF) | qh2.x, ((vui >> 20) & 0xF) | qh2.y, ((vui >> 24) & 0xF) | qh3.x, ((vui >> 28) & 0xF) | qh3.y) * dm.x + dm.y; + + store_a(col, k_pair, FLOAT_TYPEV2(v0.xz)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v1.xz)); + store_a(col, k_pair + 8, FLOAT_TYPEV2(v0.yw)); + store_a(col, k_pair + 9, FLOAT_TYPEV2(v1.yw)); + } else if (MmTypeA == GGML_TYPE_Q8_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 8; + const uint iqs = idx & 0x07; + + const float d = float(a_q8_0_p16.data[ib].d); + const i8vec2 v0 = unpack8(int32_t(a_q8_0_p16.data[ib].qs[2*iqs])).xy; // vec4 used due to #12147 + const i8vec2 v1 = unpack8(int32_t(a_q8_0_p16.data[ib].qs[2*iqs + 1])).xy; + const vec4 v = vec4(v0.x, v0.y, v1.x, v1.y) * d; + + store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v.zw)); + } else if (MmTypeA == GGML_TYPE_Q1_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 16; + const uint iqs = idx & 0xfu; + + const float d = float(a_q1_0.data[ib].d); + const uint bits = uint(a_q1_0.data[ib].qs[iqs]); + + store_a(col, k_pair, FLOAT_TYPEV2((bits & 0x01u) != 0u ? d : -d, (bits & 0x02u) != 0u ? d : -d)); + store_a(col, k_pair + 1, FLOAT_TYPEV2((bits & 0x04u) != 0u ? d : -d, (bits & 0x08u) != 0u ? d : -d)); + store_a(col, k_pair + 2, FLOAT_TYPEV2((bits & 0x10u) != 0u ? d : -d, (bits & 0x20u) != 0u ? d : -d)); + store_a(col, k_pair + 3, FLOAT_TYPEV2((bits & 0x40u) != 0u ? d : -d, (bits & 0x80u) != 0u ? d : -d)); + } else if (MmTypeA == GGML_TYPE_Q2_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 16; + const uint iqs = idx & 0xfu; + + const FLOAT_TYPE d = FLOAT_TYPE(a_q2_0.data[ib].d); + const uint bits = uint(a_q2_0.data[ib].qs[iqs]); + + store_a(col, k_pair, d * (FLOAT_TYPEV2(bits & 3u, (bits >> 2u) & 3u) - FLOAT_TYPEV2(1.0f))); + store_a(col, k_pair + 1, d * (FLOAT_TYPEV2((bits >> 4u) & 3u, bits >> 6u) - FLOAT_TYPEV2(1.0f))); + } else if (MmTypeA == GGML_TYPE_Q2_K) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 64; // 4 values per idx + const uint iqs = (idx % 64) * 2; // 0,2,4..126 + + const uint qsi = (iqs / 64) * 16 + (iqs % 16); // 0..15 + const uint scalesi = iqs / 8; // 0..15 + const uint qsshift = ((iqs % 64) / 16) * 2; // 0,2,4,6 + + const vec4 qs = vec4(unpack8((a_q2_k_p32.data[ib].qs[qsi / 2] >> qsshift) & 0x03030303)); + const uint scales = a_q2_k.data[ib].scales[scalesi]; + const vec2 dm = vec2(a_q2_k.data[ib].dm); + + const vec4 v = dm.x * float(scales & 0xF) * qs - dm.y * float(scales >> 4); + + store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); + store_a(col, k_pair + 1, FLOAT_TYPEV2(v.zw)); + } else if (MmTypeA == GGML_TYPE_Q3_K) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 128; // 2 values per idx + const uint iqs = idx % 128; // 0..127 + + const uint n = iqs / 64; // 0,1 + const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..62 + const uint hmi = (iqs % 16) * 2; // 0,2,4..30 + const uint j = (iqs % 64) / 4; // 0..3 + const uint is = iqs / 8; // 0..15 + const uint halfsplit = ((iqs % 64) / 16); // 0,1,2,3 + const uint qsshift = halfsplit * 2; // 0,2,4,6 + + const int8_t us = int8_t(((a_q3_k.data[ib].scales[is % 8] >> (4 * int(is / 8))) & 0xF) + | (((a_q3_k.data[ib].scales[8 + (is % 4)] >> (2 * int(is / 4))) & 3) << 4)); + const float dl = float(a_q3_k.data[ib].d) * float(us - 32); + + const vec2 qs = vec2(unpack8((uint(a_q3_k_p16.data[ib].qs[qsi / 2]) >> qsshift) & 0x0303).xy); + const vec2 hm = vec2(unpack8(((uint(a_q3_k_p16.data[ib].hmask[hmi / 2]) >> (4 * n + halfsplit)) & 0x0101 ^ 0x0101) << 2).xy); + + store_a(col, k_pair, FLOAT_TYPEV2(dl * (qs.x - hm.x), + dl * (qs.y - hm.y))); + + } else if (MmTypeA == GGML_TYPE_Q4_K) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 64; // 4 values per idx + const uint iqs = (idx % 64) * 2; // 0,2,4..126 + + const uint n = iqs / 32; // 0,1,2,3 + const uint b = (iqs % 32) / 16; // 0,1 + const uint is = 2 * n + b; // 0..7 + const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..126 + + const vec2 loadd = vec2(a_q4_k.data[ib].dm); + + const uvec3 scales = uvec3(a_q4_k_p32.data[ib].scales[0], + a_q4_k_p32.data[ib].scales[1], + a_q4_k_p32.data[ib].scales[2]); + const uint scalesoffs = (is & 3) * 8; + + const uint scidx0 = (is < 4) ? 0 : 2; + const uint scidxshift0 = scalesoffs; + const uint scidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; + const uint mbidx0 = (is < 4) ? 1 : 2; + const uint mbidxshift0 = (is < 4) ? scalesoffs : scalesoffs + 4; + const uint mbidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; + + const uint8_t sc = uint8_t(((scales[scidx0] >> scidxshift0) & 0xF) | ((scales[0] >> scidxshift1) & 0x30)); + const uint8_t mbyte = uint8_t(((scales[mbidx0] >> mbidxshift0) & 0xF) | ((scales[1] >> mbidxshift1) & 0x30)); + + const float d = loadd.x * sc; + const float m = -loadd.y * mbyte; + + const vec4 q = vec4(unpack8((a_q4_k_p32.data[ib].qs[qsi / 4] >> (b * 4)) & 0x0F0F0F0F)); + + store_a(col, k_pair, FLOAT_TYPEV2(fma(d, q.x, m), fma(d, q.y, m))); + store_a(col, k_pair + 1, FLOAT_TYPEV2(fma(d, q.z, m), fma(d, q.w, m))); + } else if (MmTypeA == GGML_TYPE_Q5_K) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 64; // 4 values per idx + const uint iqs = (idx % 64) * 2; // 0,2,4..126 + + const uint n = iqs / 32; // 0,1,2,3 + const uint b = (iqs % 32) / 16; // 0,1 + const uint is = 2 * n + b; // 0..7 + const uint qsi = n * 32 + (iqs % 16) * 2; // 0,2,4..126 + const uint qhi = (iqs % 16) * 2; // 0,2,4..30 + + const vec2 loadd = vec2(a_q5_k.data[ib].dm); + + const uvec3 scales = uvec3(a_q5_k_p32.data[ib].scales[0], + a_q5_k_p32.data[ib].scales[1], + a_q5_k_p32.data[ib].scales[2]); + const uint scalesoffs = (is & 3) * 8; + + const uint scidx0 = (is < 4) ? 0 : 2; + const uint scidxshift0 = scalesoffs; + const uint scidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; + const uint mbidx0 = (is < 4) ? 1 : 2; + const uint mbidxshift0 = (is < 4) ? scalesoffs : scalesoffs + 4; + const uint mbidxshift1 = (is < 4) ? scalesoffs : scalesoffs + 2; + + const uint8_t sc = uint8_t(((scales[scidx0] >> scidxshift0) & 0xF) | ((scales[0] >> scidxshift1) & 0x30)); + const uint8_t mbyte = uint8_t(((scales[mbidx0] >> mbidxshift0) & 0xF) | ((scales[1] >> mbidxshift1) & 0x30)); + + const float d = loadd.x * sc; + const float m = -loadd.y * mbyte; + + const uint qs = (a_q5_k_p32.data[ib].qs[qsi / 4] >> (b * 4)) & 0x0F0F0F0F; + const uint qh = ((a_q5_k_p32.data[ib].qh[qhi / 4] >> (iqs / 16)) & 0x01010101) << 4; + const vec4 q = vec4(unpack8(qs | qh)); + + store_a(col, k_pair, FLOAT_TYPEV2(fma(d, q.x, m), fma(d, q.y, m))); + store_a(col, k_pair + 1, FLOAT_TYPEV2(fma(d, q.z, m), fma(d, q.w, m))); + } else if (MmTypeA == GGML_TYPE_Q6_K) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + const uint k_pair = row * mm_load_vec_a() / 2; + + const uint ib = idx / 128; // 2 values per idx + const uint iqs = idx % 128; // 0..127 + + const uint n = iqs / 64; // 0,1 + const uint b = ((iqs % 64) / 32) * 4; // 0,4 + const uint is_b = (iqs % 16) / 8; // 0,1 + const uint qhshift = ((iqs % 64) / 16) * 2; // 0,2,4,6 + const uint is = 8 * n + qhshift + is_b; // 0..15 + const uint qsi = n * 32 + (iqs % 32); // 0..63 + const uint qhi = n * 16 + (iqs % 16); // 0..31 + + const float dscale = float(a_q6_k.data[ib].d) * float(a_q6_k.data[ib].scales[is]); + + const uint ql = (uint(a_q6_k_p16.data[ib].ql[qsi]) >> b) & 0x0F0F; + const uint qh = (uint(a_q6_k_p16.data[ib].qh[qhi]) >> qhshift) & 0x0303; + const vec2 q = (vec2(unpack8(ql | (qh << 4)).xy) - 32) * dscale; + + store_a(col, k_pair, FLOAT_TYPEV2(q.x, q.y)); + } else if (MmTypeA == GGML_TYPE_TQ1_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + + const uint ib = idx / 128; // 2 values per idx + const uint iqs = (idx % 128) * 2; // elem 0,2,4..254 + + const float d = float(a_tq1_0.data[ib].d); + vec2 v; + for (uint kk = 0u; kk < 2u; ++kk) { + const uint e = iqs + kk; + const uint bidx = tq1_0_byte_of(e); + const uint qbyte = uint(bidx < 48u ? a_tq1_0.data[ib].qs[bidx] + : a_tq1_0.data[ib].qh[bidx - 48u]); + v[kk] = d * (float(tq1_0_trit(qbyte, tq1_0_digit_of(e))) - 1.0); + } + + const uint k_pair = row * mm_load_vec_a() / 2; + store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); + } else if (MmTypeA == GGML_TYPE_TQ2_0) { + const uint idx = pos_a + col * p.stride_a / mm_load_vec_a() + row; + + const uint ib = idx / 128; // 2 values per idx + const uint iqs = (idx % 128) * 2; // elem 0,2,4..254 + + const uint qsi = (iqs / 128) * 32 + (iqs % 32); // byte pair start + const uint shift = 2 * ((iqs % 128) / 32); // 0,2,4,6 + + const uvec2 qs = uvec2(a_tq2_0.data[ib].qs[qsi], a_tq2_0.data[ib].qs[qsi + 1]); + const float d = float(a_tq2_0.data[ib].d); + + const vec2 v = d * (vec2((qs >> shift) & 3) - 1.0); + + const uint k_pair = row * mm_load_vec_a() / 2; + store_a(col, k_pair, FLOAT_TYPEV2(v.xy)); + } #endif } #if !defined(MUL_MAT_ID) void load_b_to_shmem(const uint pos_b, const uint row, const uint col, const uint idx_n, const uint block, const uint end_k) { #if LOAD_VEC_B == 8 - if (ALIGNED != 0) { - // Not supported for b_type bf16 because bf16mat2x4 does not exist - const uint idx = pos_b + col * p.stride_b / LOAD_VEC_B + row; - const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; - FLOAT_TYPEV8 bb = FLOAT_TYPEV8(data_b[idx]); - buf_b[buf_idx + 0] = bb[0].xy; - buf_b[buf_idx + 1] = bb[0].zw; - buf_b[buf_idx + 2] = bb[1].xy; - buf_b[buf_idx + 3] = bb[1].zw; - return; - } + if (ALIGNED != 0) { + // Not supported for b_type bf16 because bf16mat2x4 does not exist + const uint idx = pos_b + col * p.stride_b / LOAD_VEC_B + row; + const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; + FLOAT_TYPEV8 bb = FLOAT_TYPEV8(data_b[idx]); + buf_b[buf_idx + 0] = bb[0].xy; + buf_b[buf_idx + 1] = bb[0].zw; + buf_b[buf_idx + 2] = bb[1].xy; + buf_b[buf_idx + 3] = bb[1].zw; + return; + } #elif LOAD_VEC_B == 4 - if (ALIGNED != 0) { - const uint idx = pos_b + col * p.stride_b / LOAD_VEC_B + row; - const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; + if (ALIGNED != 0) { + const uint idx = pos_b + col * p.stride_b / LOAD_VEC_B + row; + const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; #if defined(DATA_B_BF16) - FLOAT_TYPEV4 bb = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_b[idx])); + FLOAT_TYPEV4 bb = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_b[idx])); #else - FLOAT_TYPEV4 bb = FLOAT_TYPEV4(data_b[idx]); + FLOAT_TYPEV4 bb = FLOAT_TYPEV4(data_b[idx]); #endif - buf_b[buf_idx + 0] = bb.xy; - buf_b[buf_idx + 1] = bb.zw; - return; - } + buf_b[buf_idx + 0] = bb.xy; + buf_b[buf_idx + 1] = bb.zw; + return; + } #endif - const uint idx = pos_b + col * p.stride_b + row * 2; - const uint buf_idx = col * SHMEM_STRIDE + row; - if (idx_n < p.N && block + row * 2 + 1 < end_k) { - buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), - TO_FLOAT_TYPE(data_b_scalar[idx + 1])); - } else if (idx_n < p.N && block + row * 2 < end_k) { - buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), 0.0f); - } else { - buf_b[buf_idx] = FLOAT_TYPEV2(0.0f); - } + const uint idx = pos_b + col * p.stride_b + row * 2; + const uint buf_idx = col * SHMEM_STRIDE + row; + if (idx_n < p.N && block + row * 2 + 1 < end_k) { + buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), + TO_FLOAT_TYPE(data_b_scalar[idx + 1])); + } else if (idx_n < p.N && block + row * 2 < end_k) { + buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), 0.0f); + } else { + buf_b[buf_idx] = FLOAT_TYPEV2(0.0f); + } } #else void load_b_to_shmem(const uint pos_b, const uint row, const uint col, const uint ic, const uint _ne1, const uint block, const uint end_k) { #if LOAD_VEC_B == 8 - if (ALIGNED != 0) { - // Not supported for b_type bf16 because bf16mat2x4 does not exist - const u16vec2 row_idx = row_ids[col]; - const uint idx = pos_b + row_idx.y * p.batch_stride_b / LOAD_VEC_B + (row_idx.x % p.ne11) * p.stride_b / LOAD_VEC_B + row; - const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; - FLOAT_TYPEV8 bb = FLOAT_TYPEV8(data_b[idx]); - buf_b[buf_idx + 0] = bb[0].xy; - buf_b[buf_idx + 1] = bb[0].zw; - buf_b[buf_idx + 2] = bb[1].xy; - buf_b[buf_idx + 3] = bb[1].zw; - return; - } + if (ALIGNED != 0) { + // Not supported for b_type bf16 because bf16mat2x4 does not exist + const u16vec2 row_idx = row_ids[col]; + const uint idx = pos_b + row_idx.y * p.batch_stride_b / LOAD_VEC_B + (row_idx.x % p.ne11) * p.stride_b / LOAD_VEC_B + row; + const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; + FLOAT_TYPEV8 bb = FLOAT_TYPEV8(data_b[idx]); + buf_b[buf_idx + 0] = bb[0].xy; + buf_b[buf_idx + 1] = bb[0].zw; + buf_b[buf_idx + 2] = bb[1].xy; + buf_b[buf_idx + 3] = bb[1].zw; + return; + } #elif LOAD_VEC_B == 4 - if (ALIGNED != 0) { - const u16vec2 row_idx = row_ids[col]; - const uint idx = pos_b + row_idx.y * p.batch_stride_b / LOAD_VEC_B + (row_idx.x % p.ne11) * p.stride_b / LOAD_VEC_B + row; - const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; + if (ALIGNED != 0) { + const u16vec2 row_idx = row_ids[col]; + const uint idx = pos_b + row_idx.y * p.batch_stride_b / LOAD_VEC_B + (row_idx.x % p.ne11) * p.stride_b / LOAD_VEC_B + row; + const uint buf_idx = col * SHMEM_STRIDE + row * LOAD_VEC_B / 2; #if defined(DATA_B_BF16) - FLOAT_TYPEV4 bb = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_b[idx])); + FLOAT_TYPEV4 bb = FLOAT_TYPEV4(TO_FLOAT_TYPE(data_b[idx])); #else - FLOAT_TYPEV4 bb = FLOAT_TYPEV4(data_b[idx]); + FLOAT_TYPEV4 bb = FLOAT_TYPEV4(data_b[idx]); #endif - buf_b[buf_idx + 0] = bb.xy; - buf_b[buf_idx + 1] = bb.zw; - return; - } + buf_b[buf_idx + 0] = bb.xy; + buf_b[buf_idx + 1] = bb.zw; + return; + } #endif - const uint row_i = ic * BN + col; - const uint buf_idx = col * SHMEM_STRIDE + row; - if (row_i < _ne1 && block + row * 2 + 1 < end_k) { - const u16vec2 row_idx = row_ids[col]; - const uint idx = pos_b + row_idx.y * p.batch_stride_b + (row_idx.x % p.ne11) * p.stride_b + row * 2; - buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), - TO_FLOAT_TYPE(data_b_scalar[idx + 1])); - } else if (row_i < _ne1 && block + row * 2 < end_k) { - const u16vec2 row_idx = row_ids[col]; - const uint idx = pos_b + row_idx.y * p.batch_stride_b + (row_idx.x % p.ne11) * p.stride_b + row * 2; - buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), 0.0f); - } else { - buf_b[buf_idx] = FLOAT_TYPEV2(0.0f); - } + const uint row_i = ic * BN + col; + const uint buf_idx = col * SHMEM_STRIDE + row; + if (row_i < _ne1 && block + row * 2 + 1 < end_k) { + const u16vec2 row_idx = row_ids[col]; + const uint idx = pos_b + row_idx.y * p.batch_stride_b + (row_idx.x % p.ne11) * p.stride_b + row * 2; + buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), + TO_FLOAT_TYPE(data_b_scalar[idx + 1])); + } else if (row_i < _ne1 && block + row * 2 < end_k) { + const u16vec2 row_idx = row_ids[col]; + const uint idx = pos_b + row_idx.y * p.batch_stride_b + (row_idx.x % p.ne11) * p.stride_b + row * 2; + buf_b[buf_idx] = FLOAT_TYPEV2(TO_FLOAT_TYPE(data_b_scalar[idx]), 0.0f); + } else { + buf_b[buf_idx] = FLOAT_TYPEV2(0.0f); + } } #endif diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_id_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_id_funcs.glsl index 26c5c12a..54ad60b2 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_id_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_id_funcs.glsl @@ -71,4 +71,19 @@ void load_row_ids(uint expert_idx, bool nei0_is_pow2, uint ic) { barrier(); } #endif // MUL_MAT_ID_USE_SUBGROUPS + +void load_row_ids_hoisted(uint expert_idx, uint ic) { + _ne1 = uint(data_expert_count[expert_idx]); + + const uint tile_begin = ic * BN; + const uint tile_count = tile_begin < _ne1 ? min(BN, _ne1 - tile_begin) : 0; + const uint expert_offset = uint(data_expert_count[p.n_experts + expert_idx]); + const uint row_ids_offset = 2 * p.n_experts + 1 + expert_offset + tile_begin; + + for (uint i = gl_LocalInvocationIndex; i < tile_count; i += BLOCK_SIZE) { + const uint packed_row_id = uint(data_expert_count[row_ids_offset + i]); + row_ids[i] = u16vec2(packed_row_id & 0xffffu, packed_row_id >> 16); + } + barrier(); +} #endif // MUL_MAT_ID diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp index aae1c2e8..67908e70 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp @@ -6,6 +6,8 @@ #extension GL_EXT_integer_dot_product : require +#define KVALUES_IQ4NL_I8 + #ifdef FLOAT16 #extension GL_EXT_shader_explicit_arithmetic_types_float16 : require #endif @@ -56,6 +58,8 @@ layout (push_constant) uniform parameter uint nei1; uint nbi1; uint ne11; + uint n_experts; + uint hoist_row_ids; #else uint base_work_group_z; uint num_batches; @@ -157,27 +161,31 @@ void main() { const uint loadstride_b = BLOCK_SIZE * LOAD_VEC_B / BK; #ifdef MUL_MAT_ID -#ifdef MUL_MAT_ID_USE_SUBGROUPS - if (bitCount(p.nei0) == 1) { - load_row_ids(expert_idx, true, ic); + if (p.hoist_row_ids != 0) { + load_row_ids_hoisted(expert_idx, ic); } else { - load_row_ids(expert_idx, false, ic); - } +#ifdef MUL_MAT_ID_USE_SUBGROUPS + if (bitCount(p.nei0) == 1) { + load_row_ids(expert_idx, true, ic); + } else { + load_row_ids(expert_idx, false, ic); + } #else - _ne1 = 0; - for (uint ii1 = 0; ii1 < p.nei1 && _ne1 < (ic + 1) * BN; ii1++) { - for (uint ii0 = 0; ii0 < p.nei0 && _ne1 < (ic + 1) * BN; ii0++) { - if (data_ids[ii1*p.nbi1 + ii0] == expert_idx) { - if (_ne1 >= ic * BN) { - row_ids[_ne1 - ic * BN] = u16vec2(ii0, ii1); + _ne1 = 0; + for (uint ii1 = 0; ii1 < p.nei1 && _ne1 < (ic + 1) * BN; ii1++) { + for (uint ii0 = 0; ii0 < p.nei0 && _ne1 < (ic + 1) * BN; ii0++) { + if (data_ids[ii1*p.nbi1 + ii0] == expert_idx) { + if (_ne1 >= ic * BN) { + row_ids[_ne1 - ic * BN] = u16vec2(ii0, ii1); + } + _ne1++; } - _ne1++; } } - } - barrier(); + barrier(); #endif + } // Workgroup has no work if (ic * BN >= _ne1) return; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl index 24da4f71..136d5b74 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl @@ -217,6 +217,41 @@ ACC_TYPE mmq_dot_product(const uint ib_a) { } #endif +#if defined(DATA_A_IQ4_XS) +void block_a_to_shmem(const uint buf_ib, const uint ib, const uint iqs) { + const uint ib_k = ib / 8; + const uint ib32 = ib % 8; + const uint32_t vui = data_a_packed32[ib_k].qs[4 * ib32 + iqs]; + const i32vec2 qs = iq4nl_to_i8x8(vui); + + buf_a[buf_ib].qs[iqs ] = qs.x; + buf_a[buf_ib].qs[iqs + 4] = qs.y; + + if (iqs == 0) { + const uint sl = (data_a_packed32[ib_k].scales_l >> (4 * ib32)) & 0xF; + const uint sh = (data_a_packed32[ib_k].scales_h >> (2 * ib32)) & 3; + buf_a[buf_ib].d = FLOAT_TYPE(float(data_a[ib_k].d) * float(int(sl | (sh << 4)) - 32)); + } +} + +void block_a_to_registers(const uint reg_ib, const uint buf_ib) { + cache_a[reg_ib].d = buf_a[buf_ib].d; + + [[unroll]] for (uint iqs = 0; iqs < 8; iqs++) { + cache_a[reg_ib].qs[iqs] = buf_a[buf_ib].qs[iqs]; + } +} + +ACC_TYPE mmq_dot_product(const uint ib_a) { + int32_t q_sum = 0; + [[unroll]] for (uint iqs = 0; iqs < 8; iqs++) { + q_sum += dotPacked4x8EXT(cache_a[ib_a].qs[iqs], cache_b.qs[iqs]); + } + + return ACC_TYPE(float(cache_a[ib_a].d) * float(cache_b.ds.x) * float(q_sum)); +} +#endif + // For k-quants, ib and iqs still assume 32-wide blocks, but k-quants are 256-wide // iqs still refers to a 32-bit integer, meaning 0..7 for 32-wide quants #if defined(DATA_A_Q2_K) @@ -454,6 +489,55 @@ ACC_TYPE mmq_dot_product(const uint ib_a) { } #endif +#if defined(DATA_A_IQ3_S) +// 2-byte loads for IQ3_S blocks (110 bytes) +void block_a_to_shmem(const uint buf_ib, const uint ib, const uint iqs) { + const uint ib_k = ib / 8; + const uint ib32 = ib % 8; + + // grid indices for qs[2 * iqs] and qs[2 * iqs + 1] + const uint qs = uint(data_a_packed16[ib_k].qs[ib32 * 4 + iqs]); + // their two high index bits + const uint qh = uint(data_a_packed16[ib_k].qh[ib32 / 2]) >> ((ib32 & 1) * 8 + 2 * iqs); + // one sign bit per value, 8 values + const uint signs = uint(data_a_packed16[ib_k].signs[ib32 * 2 + iqs / 2]) >> ((iqs & 1) * 8); + + // grid holds 4 values of 1..15, one per byte + const ivec4 vals0 = ivec4(unpack8(iq3s_grid[( qs & 0xFF) | ((qh & 1) << 8)])); + const ivec4 vals1 = ivec4(unpack8(iq3s_grid[((qs >> 8) & 0xFF) | ((qh & 2) << 7)])); + + // negate with (v ^ -s) - -s to avoid branches + const ivec4 m0 = -(ivec4(signs, signs >> 1, signs >> 2, signs >> 3) & 1); + const ivec4 m1 = -(ivec4(signs >> 4, signs >> 5, signs >> 6, signs >> 7) & 1); + + buf_a[buf_ib].qs[2 * iqs ] = pack32(i8vec4((vals0 ^ m0) - m0)); + buf_a[buf_ib].qs[2 * iqs + 1] = pack32(i8vec4((vals1 ^ m1) - m1)); + + if (iqs == 0) { + const uint scale = (uint(data_a_packed16[ib_k].scales[ib32 / 4]) >> ((ib32 & 3) * 4)) & 0xF; + + buf_a[buf_ib].d = FLOAT_TYPE(float(data_a_packed16[ib_k].d) * float(1 + 2 * scale)); + } +} + +void block_a_to_registers(const uint reg_ib, const uint buf_ib) { + cache_a[reg_ib].d = buf_a[buf_ib].d; + + [[unroll]] for (uint iqs = 0; iqs < 8; iqs++) { + cache_a[reg_ib].qs[iqs] = buf_a[buf_ib].qs[iqs]; + } +} + +ACC_TYPE mmq_dot_product(const uint ib_a) { + int32_t q_sum = 0; + [[unroll]] for (uint iqs = 0; iqs < 8; iqs++) { + q_sum += dotPacked4x8EXT(cache_a[ib_a].qs[iqs], cache_b.qs[iqs]); + } + + return ACC_TYPE(float(cache_a[ib_a].d) * float(cache_b.ds.x) * float(q_sum)); +} +#endif + void block_b_to_shmem(const uint buf_ib, const uint ib, const uint iqs, const bool is_in_bounds) { if (is_in_bounds) { const uint ib_outer = ib / 4; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl index 2b7adcb6..56784cff 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl @@ -53,12 +53,24 @@ struct block_a_cache { int32_t qs[8]; FLOAT_TYPE dm; }; +#elif defined(DATA_A_IQ4_XS) +#define QUANT_R_MMQ 2 +struct block_a_cache { + int32_t qs[8]; + FLOAT_TYPE d; +}; #elif defined(DATA_A_MXFP4) #define QUANT_R_MMQ 2 struct block_a_cache { int32_t qs[8]; FLOAT_TYPE d; }; +#elif defined(DATA_A_IQ3_S) +#define QUANT_R_MMQ 2 +struct block_a_cache { + int32_t qs[8]; + FLOAT_TYPE d; +}; #elif defined(DATA_A_Q2_K) #define QUANT_R_MMQ 4 struct block_a_cache { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/pad_reflect_1d.comp b/ggml/src/ggml-vulkan/vulkan-shaders/pad_reflect_1d.comp new file mode 100644 index 00000000..2389020f --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/pad_reflect_1d.comp @@ -0,0 +1,43 @@ +#version 450 + +#include "types.glsl" +#include "generic_unary_head.glsl" // included to use functions like fastdiv etc. + +layout(local_size_x = 512, local_size_y = 1, local_size_z = 1) in; + +void main() { + + const uint idx = get_idx(); + + if (idx >= p.ne) { + return; + } + + const uint p0 = floatBitsToUint(p.param1); + const uint p1 = floatBitsToUint(p.param2); + + const uint i3 = fastdiv(idx, p.ne1_012mp, fastdiv_L(p.ne1_Ls, 0)); + const uint i3_offset = i3 * p.ne12 * p.ne11 * p.ne10; + + const uint i2 = fastdiv(idx - i3_offset, p.ne1_01mp, fastdiv_L(p.ne1_Ls, 1)); + const uint i2_offset = i2 * p.ne11 * p.ne10; + + const uint i1 = fastdiv(idx - i3_offset - i2_offset, p.ne1_0mp, fastdiv_L(p.ne1_Ls, 2)); + const uint i0 = idx - i3_offset - i2_offset - i1 * p.ne10; + + uint src_col; + + if (i0 < p0) { + src_col = p0 - i0; // left pad area + } else if (i0 < p0 + p.ne00) { + src_col = i0 - p0; // center area + } else { + src_col = 2u * p.ne00 - 2u - (i0 - p0); // right pad area + } + + const uint src_idx = i3 * p.nb03 + i2 * p.nb02 + i1 * p.nb01 + src_col * p.nb00; + const uint d_idx = i3 * p.nb13 + i2 * p.nb12 + i1 * p.nb11 + i0 * p.nb10; + + // copy the computed value to the destination tensor + data_d[get_doffset() + d_idx] = D_TYPE(data_a[get_aoffset() + src_idx]); +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp b/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp index 55b89f19..ee813842 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp @@ -27,12 +27,24 @@ layout (binding = 6) readonly buffer R_I {uvec2 rope_data_i[];}; // indices for #define GGML_ROPE_TYPE_MROPE 8 #define GGML_ROPE_TYPE_VISION 24 +#elif RMS_NORM_ADD_FUSION + +layout (binding = 3) readonly buffer C {float data_c[];}; +layout (binding = 4) readonly buffer E {float data_e[];}; + +#elif RMS_NORM_SET_ROWS_FUSION + +layout (binding = 3) readonly buffer I {uvec2 data_i[];}; + #endif #extension GL_EXT_control_flow_attributes : enable #define BLOCK_SIZE 512 layout (constant_id = 1) const bool do_multiply = false; +#if RMS_NORM_ADD_FUSION +layout (constant_id = 2) const bool do_post_multiply = false; +#endif layout(local_size_x = BLOCK_SIZE, local_size_y = 1, local_size_z = 1) in; @@ -57,6 +69,8 @@ void rms_norm(uint num_iters) { #if RMS_NORM_ROPE_FUSION // Per-row offset in shared memory uint32_t d_offset = 0; +#elif RMS_NORM_SET_ROWS_FUSION + uint32_t d_offset = data_i[channel].x*p.nb21 + row*ncols + get_doffset(); #else uint32_t d_offset = ((samp*nchannels + channel)*nrows + row)*ncols + get_doffset(); #endif @@ -91,14 +105,28 @@ void rms_norm(uint num_iters) { if (col >= ncols) { continue; } - data_d[d_offset + col] = D_TYPE(scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + fastmod(col, p.ne10)])); + FLOAT_TYPE value = scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + fastmod(col, p.ne10)]); +#if RMS_NORM_ADD_FUSION + value += FLOAT_TYPE(data_c[d_offset + col]); + if (do_post_multiply) { + value *= FLOAT_TYPE(data_e[0]); + } +#endif + data_d[d_offset + col] = D_TYPE(value); } } else { [[unroll]] for (uint col = tid, idx = 0; idx < num_iters; col += BLOCK_SIZE, ++idx) { if (col >= ncols) { continue; } - data_d[d_offset + col] = D_TYPE(scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + col])); + FLOAT_TYPE value = scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + col]); +#if RMS_NORM_ADD_FUSION + value += FLOAT_TYPE(data_c[d_offset + col]); + if (do_post_multiply) { + value *= FLOAT_TYPE(data_e[0]); + } +#endif + data_d[d_offset + col] = D_TYPE(value); } } } else { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm_partials.comp b/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm_partials.comp index 4618b2c7..cf7ab21f 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm_partials.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm_partials.comp @@ -10,11 +10,19 @@ #define BLOCK_SIZE 128 layout (constant_id = 1) const bool do_multiply = false; +#if RMS_NORM_ADD_FUSION +layout (constant_id = 2) const bool do_post_multiply = false; +#endif layout(local_size_x = BLOCK_SIZE, local_size_y = 1, local_size_z = 1) in; layout (binding = 3, std430) readonly buffer PartialsBuf {float partial_sums[];}; +#if RMS_NORM_ADD_FUSION +layout (binding = 4) readonly buffer C {float data_c[];}; +layout (binding = 5) readonly buffer E {float data_e[];}; +#endif + shared FLOAT_TYPE sumsh[BLOCK_SIZE]; void main() { @@ -55,9 +63,23 @@ void main() { if (do_multiply) { if (ncols > p.ne10) { - data_d[d_offset + col] = D_TYPE(scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + fastmod(col, p.ne10)])); + FLOAT_TYPE value = scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + fastmod(col, p.ne10)]); +#if RMS_NORM_ADD_FUSION + value += FLOAT_TYPE(data_c[d_offset + col]); + if (do_post_multiply) { + value *= FLOAT_TYPE(data_e[0]); + } +#endif + data_d[d_offset + col] = D_TYPE(value); } else { - data_d[d_offset + col] = D_TYPE(scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + col])); + FLOAT_TYPE value = scale * FLOAT_TYPE(data_a[a_offset + col]) * FLOAT_TYPE(data_b[b_offset + col]); +#if RMS_NORM_ADD_FUSION + value += FLOAT_TYPE(data_c[d_offset + col]); + if (do_post_multiply) { + value *= FLOAT_TYPE(data_e[0]); + } +#endif + data_d[d_offset + col] = D_TYPE(value); } } else { data_d[d_offset + col] = D_TYPE(scale * FLOAT_TYPE(data_a[a_offset + col])); diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl index 03358793..feb55b20 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl @@ -50,19 +50,21 @@ void rope_norm(const uint i0, const uint i1, const uint i2, const uint i3, rope_ } idst += p.d_offset; - if (i0 >= p.n_dims) { + if (i0 < p.n_offs || i0 >= p.n_offs + p.n_dims) { rope_data_d[idst + 0] = ROPE_D_TYPE(rope_data_a[ix + 0]); rope_data_d[idst + 1] = ROPE_D_TYPE(rope_data_a[ix + 1]); return; } - const float theta_base = rope_data_pos[i2] * pow(p.theta_scale, i0/2.0f); + const uint iw = i0 - p.n_offs; // relative idx - const float freq_factor = p.has_ff != 0 ? rope_data_ff[i0/2] : 1.0f; + const float theta_base = rope_data_pos[i2] * pow(p.theta_scale, iw/2.0f); + + const float freq_factor = p.has_ff != 0 ? rope_data_ff[iw/2] : 1.0f; float cos_theta, sin_theta; - rope_yarn(theta_base / freq_factor, i0, cos_theta, sin_theta, p); + rope_yarn(theta_base / freq_factor, iw, cos_theta, sin_theta, p); const float x0 = float(rope_data_a[ix + 0]); const float x1 = float(rope_data_a[ix + 1]); @@ -87,25 +89,28 @@ void rope_neox(const uint i0, const uint i1, const uint i2, const uint i3, rope_ } idst += p.d_offset; - if (i0 >= p.n_dims) { + if (i0 < p.n_offs || i0 >= p.n_offs + p.n_dims) { rope_data_d[idst + i0/2 + 0] = ROPE_D_TYPE(rope_data_a[ix + i0/2 + 0]); rope_data_d[idst + i0/2 + 1] = ROPE_D_TYPE(rope_data_a[ix + i0/2 + 1]); return; } - const float theta_base = rope_data_pos[i2] * pow(p.theta_scale, i0/2.0f); + const uint iw = i0 - p.n_offs; // relative idx - const float freq_factor = p.has_ff != 0 ? rope_data_ff[i0/2] : 1.0f; + const float theta_base = rope_data_pos[i2] * pow(p.theta_scale, iw/2.0f); + + const float freq_factor = p.has_ff != 0 ? rope_data_ff[iw/2] : 1.0f; float cos_theta, sin_theta; - rope_yarn(theta_base / freq_factor, i0, cos_theta, sin_theta, p); + rope_yarn(theta_base / freq_factor, iw, cos_theta, sin_theta, p); - const float x0 = float(rope_data_a[ix + 0]); - const float x1 = float(rope_data_a[ix + p.n_dims/2]); + // idst/ix point at channel i0/2; the first channel of the rotated pair is p.n_offs + iw/2 = i0/2 + p.n_offs/2 + const float x0 = float(rope_data_a[ix + p.n_offs/2 + 0]); + const float x1 = float(rope_data_a[ix + p.n_offs/2 + p.n_dims/2]); - rope_data_d[idst + 0] = ROPE_D_TYPE(x0*cos_theta - x1*sin_theta); - rope_data_d[idst + p.n_dims/2] = ROPE_D_TYPE(x0*sin_theta + x1*cos_theta); + rope_data_d[idst + p.n_offs/2 + 0] = ROPE_D_TYPE(x0*cos_theta - x1*sin_theta); + rope_data_d[idst + p.n_offs/2 + p.n_dims/2] = ROPE_D_TYPE(x0*sin_theta + x1*cos_theta); } @@ -125,53 +130,56 @@ void rope_multi(const uint i0, const uint i1, const uint i2, const uint i3, rope } idst += p.d_offset; - if (i0 >= p.n_dims) { + if (i0 < p.n_offs || i0 >= p.n_offs + p.n_dims) { rope_data_d[idst + i0/2 + 0] = ROPE_D_TYPE(rope_data_a[ix + i0/2 + 0]); rope_data_d[idst + i0/2 + 1] = ROPE_D_TYPE(rope_data_a[ix + i0/2 + 1]); return; } + const uint iw = i0 - p.n_offs; // relative idx + const int sect_dims = p.sections[0] + p.sections[1] + p.sections[2] + p.sections[3]; const int sec_w = p.sections[1] + p.sections[0]; - const uint sector = (i0 / 2) % sect_dims; + const uint sector = (iw / 2) % sect_dims; float theta_base = 0.0; if (p.is_imrope != 0) { if (sector % 3 == 1 && sector < 3 * p.sections[1]) { - theta_base = rope_data_pos[i2 + p.ne02 * 1]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 1]*pow(p.theta_scale, iw/2.0f); } else if (sector % 3 == 2 && sector < 3 * p.sections[2]) { - theta_base = rope_data_pos[i2 + p.ne02 * 2]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 2]*pow(p.theta_scale, iw/2.0f); } else if (sector % 3 == 0 && sector < 3 * p.sections[0]) { - theta_base = rope_data_pos[i2]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2]*pow(p.theta_scale, iw/2.0f); } else { - theta_base = rope_data_pos[i2 + p.ne02 * 3]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 3]*pow(p.theta_scale, iw/2.0f); } } else { if (sector < p.sections[0]) { - theta_base = rope_data_pos[i2]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2]*pow(p.theta_scale, iw/2.0f); } else if (sector >= p.sections[0] && sector < sec_w) { - theta_base = rope_data_pos[i2 + p.ne02 * 1]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 1]*pow(p.theta_scale, iw/2.0f); } else if (sector >= sec_w && sector < sec_w + p.sections[2]) { - theta_base = rope_data_pos[i2 + p.ne02 * 2]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 2]*pow(p.theta_scale, iw/2.0f); } else if (sector >= sec_w + p.sections[2]) { - theta_base = rope_data_pos[i2 + p.ne02 * 3]*pow(p.theta_scale, i0/2.0f); + theta_base = rope_data_pos[i2 + p.ne02 * 3]*pow(p.theta_scale, iw/2.0f); } } - const float freq_factor = p.has_ff != 0 ? rope_data_ff[i0/2] : 1.0f; + const float freq_factor = p.has_ff != 0 ? rope_data_ff[iw/2] : 1.0f; float cos_theta, sin_theta; - rope_yarn(theta_base / freq_factor, i0, cos_theta, sin_theta, p); + rope_yarn(theta_base / freq_factor, iw, cos_theta, sin_theta, p); - const float x0 = float(rope_data_a[ix + 0]); - const float x1 = float(rope_data_a[ix + p.n_dims/2]); + // idst/ix point at channel i0/2; the first channel of the rotated pair is p.n_offs + iw/2 = i0/2 + p.n_offs/2 + const float x0 = float(rope_data_a[ix + p.n_offs/2 + 0]); + const float x1 = float(rope_data_a[ix + p.n_offs/2 + p.n_dims/2]); - rope_data_d[idst + 0] = ROPE_D_TYPE(x0*cos_theta - x1*sin_theta); - rope_data_d[idst + p.n_dims/2] = ROPE_D_TYPE(x0*sin_theta + x1*cos_theta); + rope_data_d[idst + p.n_offs/2 + 0] = ROPE_D_TYPE(x0*cos_theta - x1*sin_theta); + rope_data_d[idst + p.n_offs/2 + p.n_dims/2] = ROPE_D_TYPE(x0*sin_theta + x1*cos_theta); } void rope_vision(const uint i0, const uint i1, const uint i2, const uint i3, rope_params p) { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl index 3602485b..b88a73fc 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl @@ -5,6 +5,7 @@ struct rope_params { uint rope_mode; uint nrows; uint n_dims; + uint n_offs; float freq_scale; float freq_base; float ext_factor; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/sum_rows.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/sum_rows.glsl index 2b841baa..1cb0f782 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/sum_rows.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/sum_rows.glsl @@ -1,4 +1,6 @@ +#include "utils.glsl" + // vk_op_sum_rows_push_constants layout (push_constant) uniform parameter { @@ -15,11 +17,3 @@ layout (push_constant) uniform parameter uint get_aoffset() { return p.misalign_offsets >> 16; } uint get_doffset() { return p.misalign_offsets & 0xFFFF; } -// see init_fastdiv_values in ggml-vulkan.cpp -uint fastdiv(uint n, uint mp, uint L) { - uint msbs, lsbs; - // msbs = mulhi(n, mp) - umulExtended(n, mp, msbs, lsbs); - return (msbs + n) >> L; -} - diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/swiglu_clamp.comp b/ggml/src/ggml-vulkan/vulkan-shaders/swiglu_clamp.comp new file mode 100644 index 00000000..dfe32975 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/swiglu_clamp.comp @@ -0,0 +1,12 @@ +#version 450 + +#include "glu_head.glsl" + +float op(float a, float b) { + float gate = min(a, p.limit); + float up = clamp(b, -p.limit, p.limit); + + return gate / (1.0f + exp(-gate)) * up; +} + +#include "glu_main.glsl" diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/topk_radix_select.comp b/ggml/src/ggml-vulkan/vulkan-shaders/topk_radix_select.comp new file mode 100644 index 00000000..8e14b2e9 --- /dev/null +++ b/ggml/src/ggml-vulkan/vulkan-shaders/topk_radix_select.comp @@ -0,0 +1,144 @@ +#version 450 + +#extension GL_EXT_control_flow_attributes : enable +#extension GL_EXT_shader_16bit_storage : require + +#include "types.glsl" + +layout(constant_id = 0) const int BLOCK_SIZE = 1024; +layout(constant_id = 1) const int QSA = 0; // 1: fuse the qwen4 QSA indexer gather + f16 mask + +layout(local_size_x_id = 0, local_size_y = 1, local_size_z = 1) in; + +layout (binding = 0) readonly buffer A {float data_a[];}; // input values, or QSA block scores [n_tps, n_blocks, n_stream] +layout (binding = 1) writeonly buffer D {int data_d[];}; // [k, ...] +layout (binding = 2) readonly buffer CB {int cell_blk[];}; // QSA: cell->block map [n_kv, n_stream] +layout (binding = 3) readonly buffer M {float16_t mask[];}; // QSA: raw f16 kq_mask [n_kv, n_tps, n_stream] +layout (binding = 4) buffer S {float scratch[];}; // QSA: [nrows, n_kv] gathered inputs + +layout (push_constant) uniform parameter { + uint ncols; + uint k; + uint nrows; + uint n_tps; // QSA only + uint n_blocks; // QSA only + uint n_stream; // QSA only +} p; + +#define RADIX_BITS 8 +#define RADIX_SIZE (1 << RADIX_BITS) + +shared uint histo[RADIX_SIZE]; +shared uint sh_bucket; +shared uint sh_above; +shared uint out_count; + +// order-preserving float -> uint mapping +uint f2ui(float x) { + uint y = floatBitsToUint(x); + if ((y & 0x80000000u) != 0u) { + y ^= 0xFFFFFFFFu; + } else { + y |= 0x80000000u; + } + return y; +} + +// QSA element i of row (t,s): score[cell_blk[i,s], t, s] + mask[i,t,s] +float gather(uint row, uint i) { + const uint t = row % p.n_tps; + const uint s = row / p.n_tps; + const uint block = uint(cell_blk[s * p.ncols + i]); + const float a = data_a[(s * p.n_blocks + block) * p.n_tps + t]; + const float m = float(mask[(s * p.n_tps + t) * p.ncols + i]); + return a + m; +} + +float load(uint row, uint i, bool first) { + if (QSA == 0) { + return data_a[row * p.ncols + i]; + } + // materialize the scattered gather on the first pass and reuse it after; each + // invocation only touches its own scratch entries, so no barrier is needed + const uint off = row * p.ncols + i; + if (first) { + const float v = gather(row, i); + scratch[off] = v; + return v; + } + return scratch[off]; +} + +// one workgroup per row: radix-select the K-th largest, then compact it plus enough ties +void topk(const uint row) { + const uint tid = gl_LocalInvocationID.x; + const uint ncols = p.ncols; + const uint row_out = row * p.k; + + uint prefix = 0; // fixed high bits of the threshold key + uint desired = p.k; // count still needed from the candidate range + + [[unroll]] for (int shift = 32 - RADIX_BITS; shift >= 0; shift -= RADIX_BITS) { + for (uint i = tid; i < RADIX_SIZE; i += BLOCK_SIZE) { + histo[i] = 0; + } + barrier(); + + const bool first = (shift == 32 - RADIX_BITS); + const uint hi_mask = (shift + RADIX_BITS >= 32) ? 0u : (0xFFFFFFFFu << uint(shift + RADIX_BITS)); + const uint prefix_hi = prefix & hi_mask; + for (uint i = tid; i < ncols; i += BLOCK_SIZE) { + const uint key = f2ui(load(row, i, first)); + if ((key & hi_mask) == prefix_hi) { + atomicAdd(histo[(key >> uint(shift)) & (RADIX_SIZE - 1)], 1u); + } + } + barrier(); + + // top-down scan for the bucket holding the K-th value + if (tid == 0) { + uint acc = 0; + uint b = 0; + for (int bb = RADIX_SIZE - 1; bb >= 0; --bb) { + const uint c = histo[bb]; + if (acc + c >= desired) { b = uint(bb); break; } + acc += c; + } + sh_bucket = b; + sh_above = acc; + } + barrier(); + + prefix |= sh_bucket << uint(shift); + desired -= sh_above; + barrier(); + } + + if (tid == 0) { + out_count = 0; + } + barrier(); + + // emit everything above the threshold, then fill the rest from ties + const uint threshold = prefix; + for (uint i = tid; i < ncols; i += BLOCK_SIZE) { + if (f2ui(load(row, i, false)) > threshold) { + data_d[row_out + atomicAdd(out_count, 1u)] = int(i); + } + } + barrier(); + for (uint i = tid; i < ncols; i += BLOCK_SIZE) { + if (f2ui(load(row, i, false)) == threshold) { + const uint pos = atomicAdd(out_count, 1u); + if (pos < p.k) { + data_d[row_out + pos] = int(i); + } + } + } +} + +void main() { + for (uint row = gl_WorkGroupID.y; row < p.nrows; row += gl_NumWorkGroups.y) { + topk(row); + } +} diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl index adb1bb8b..7d62e92e 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl @@ -303,6 +303,41 @@ struct block_q2_K_packed32 #define DATA_A_QUANT_K #endif +#define QUANT_K_TQ1_0 256 + +// TQ1_0: base-3 packed trits, 5 per byte in `qs` (48B) and 4 in `qh` (4B). +struct block_tq1_0 +{ + uint8_t qs[(QUANT_K_TQ1_0 - 4 * QUANT_K_TQ1_0 / 64) / 5]; + uint8_t qh[QUANT_K_TQ1_0 / 64]; + float16_t d; +}; + +// Element e in [0,255] -> its packed byte (0..47 qs, 48..51 qh) and digit. +uint tq1_0_byte_of(uint e) { + return e < 160u ? (e % 32u) + : e < 240u ? 32u + ((e - 160u) % 16u) + : 48u + ((e - 240u) % 4u); +} +uint tq1_0_digit_of(uint e) { + return e < 160u ? (e / 32u) + : e < 240u ? ((e - 160u) / 16u) + : ((e - 240u) / 4u); +} +// The 8-bit truncation below is part of the format, not an optimisation: +// the C reference does `uint8_t q = qs[..] * pow3[n]`. +uint tq1_0_trit(uint qbyte, uint t) { + const uint POW3_PACKED = (1u << 28) | (3u << 21) | (9u << 14) | (27u << 7) | 81u; + return ((((qbyte * ((POW3_PACKED >> (7u * (4u - t))) & 0x7Fu)) & 255u) * 3u) >> 8); +} + +#if defined(DATA_A_TQ1_0) +#define QUANT_K QUANT_K_TQ1_0 +#define QUANT_R 1 +#define A_TYPE block_tq1_0 +#define DATA_A_QUANT_K +#endif + #define QUANT_K_TQ2_0 256 // ternary (BitNet): 2-bit codes, w = (q - 1) * d; qs layout matches q2_K's @@ -920,6 +955,7 @@ shared uint16_t iq1s_grid[2048]; shared uint32_t iq1s_grid_gpu[2048]; #endif +#if defined(DATA_A_IQ1_S) || defined(DATA_A_IQ1_M) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -943,6 +979,17 @@ void init_iq_shmem(uvec3 wgsize) barrier(); } #endif +#endif + +#if defined(DATA_A_IQ2_XXS) || defined(DATA_A_IQ2_XS) || defined(DATA_A_IQ2_S) +#if defined(DATA_A_IQ2_S) +shared uvec2 iq2s_grid[1024]; +#elif defined(DATA_A_IQ2_XS) +shared uvec2 iq2xs_grid[512]; +#else +shared uvec2 iq2xxs_grid[256]; +#endif +#endif #define QUANT_K_IQ2_XXS 256 #define QUANT_R_IQ2_XXS 1 @@ -1028,8 +1075,7 @@ const uvec2[256] iq2xxs_grid_const = { uvec2(0x08080808, 0x2b2b082b), uvec2(0x08192b08, 0x2b2b1908), uvec2(0x19190808, 0x2b2b2b08), uvec2(0x08081908, 0x2b2b2b19) }; -shared uvec2 iq2xxs_grid[256]; - +#if defined(DATA_A_IQ2_XXS) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -1041,12 +1087,15 @@ void init_iq_shmem(uvec3 wgsize) } barrier(); } +#endif +#if defined(DATA_A_IQ2_XXS) #define QUANT_K QUANT_K_IQ2_XXS #define QUANT_R QUANT_R_IQ2_XXS #define A_TYPE block_iq2_xxs #define A_TYPE_PACKED16 block_iq2_xxs_packed16 #endif +#endif #define QUANT_K_IQ2_XS 256 #define QUANT_R_IQ2_XS 1 @@ -1198,8 +1247,7 @@ const uvec2 iq2xs_grid_const[512] = { uvec2(0x082b2b08, 0x2b2b2b2b), uvec2(0x082b2b2b, 0x2b2b2b2b), uvec2(0x2b190819, 0x2b2b2b2b), uvec2(0x2b2b2b2b, 0x2b2b2b2b), }; -shared uvec2 iq2xs_grid[512]; - +#if defined(DATA_A_IQ2_XS) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -1211,12 +1259,15 @@ void init_iq_shmem(uvec3 wgsize) } barrier(); } +#endif +#if defined(DATA_A_IQ2_XS) #define QUANT_K QUANT_K_IQ2_XS #define QUANT_R QUANT_R_IQ2_XS #define A_TYPE block_iq2_xs #define A_TYPE_PACKED16 block_iq2_xs_packed16 #endif +#endif #define QUANT_K_IQ2_S 256 #define QUANT_R_IQ2_S 1 @@ -1498,8 +1549,7 @@ const uvec2 iq2s_grid_const[1024] = { uvec2(0x082b082b, 0x2b2b2b2b), uvec2(0x082b2b08, 0x2b2b2b2b), uvec2(0x2b082b08, 0x2b2b2b2b), uvec2(0x2b2b2b2b, 0x2b2b2b2b) }; -shared uvec2 iq2s_grid[1024]; - +#if defined(DATA_A_IQ2_S) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -1511,12 +1561,23 @@ void init_iq_shmem(uvec3 wgsize) } barrier(); } +#endif +#if defined(DATA_A_IQ2_S) #define QUANT_K QUANT_K_IQ2_S #define QUANT_R QUANT_R_IQ2_S #define A_TYPE block_iq2_s #define A_TYPE_PACKED16 block_iq2_s_packed16 #endif +#endif + +#if defined(DATA_A_IQ3_XXS) || defined(DATA_A_IQ3_S) +#if defined(DATA_A_IQ3_S) +shared uint32_t iq3s_grid[512]; +#else +shared uint32_t iq3xxs_grid[256]; +#endif +#endif #define QUANT_K_IQ3_XXS 256 #define QUANT_R_IQ3_XXS 1 @@ -1570,8 +1631,7 @@ const uint32_t iq3xxs_grid_const[256] = { 0x3e1c1c1c, 0x3e1c3404, 0x3e24140c, 0x3e24240c, 0x3e2c0404, 0x3e2c0414, 0x3e2c1424, 0x3e341c04, }; -shared uint32_t iq3xxs_grid[256]; - +#if defined(DATA_A_IQ3_XXS) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -1583,12 +1643,15 @@ void init_iq_shmem(uvec3 wgsize) } barrier(); } +#endif +#if defined(DATA_A_IQ3_XXS) #define QUANT_K QUANT_K_IQ3_XXS #define QUANT_R QUANT_R_IQ3_XXS #define A_TYPE block_iq3_xxs #define A_TYPE_PACKED16 block_iq3_xxs_packed16 #endif +#endif #define QUANT_K_IQ3_S 256 #define QUANT_R_IQ3_S 1 @@ -1680,8 +1743,7 @@ const uint32_t iq3s_grid_const[512] = { 0x0f090307, 0x0f090501, 0x0f090b01, 0x0f0b0505, 0x0f0b0905, 0x0f0d0105, 0x0f0d0703, 0x0f0f0101, }; -shared uint32_t iq3s_grid[512]; - +#if defined(DATA_A_IQ3_S) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { @@ -1693,12 +1755,15 @@ void init_iq_shmem(uvec3 wgsize) } barrier(); } +#endif +#if defined(DATA_A_IQ3_S) #define QUANT_K QUANT_K_IQ3_S #define QUANT_R QUANT_R_IQ3_S #define A_TYPE block_iq3_s #define A_TYPE_PACKED16 block_iq3_s_packed16 #endif +#endif #define QUANT_K_IQ4_XS 256 #define QUANT_R_IQ4_XS 1 @@ -1810,17 +1875,38 @@ const int8_t kvalues_iq4nl_const[16] = { int8_t(1), int8_t(13), int8_t(25), int8_t(38), int8_t(53), int8_t(69), int8_t(89), int8_t(113) }; +#ifdef KVALUES_IQ4NL_I8 +shared int8_t kvalues_iq4nl[16]; +#else shared FLOAT_TYPE kvalues_iq4nl[16]; +#endif +#if defined(DATA_A_IQ4_NL) || defined(DATA_A_IQ4_XS) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { // copy the table into shared memory and sync for (uint i = gl_LocalInvocationIndex.x; i < kvalues_iq4nl.length(); i += wgsize.x) { +#ifdef KVALUES_IQ4NL_I8 + kvalues_iq4nl[i] = kvalues_iq4nl_const[i]; +#else kvalues_iq4nl[i] = FLOAT_TYPE(kvalues_iq4nl_const[i]); +#endif } barrier(); } + +#ifdef KVALUES_IQ4NL_I8 +i32vec2 iq4nl_to_i8x8(uint32_t vui) { + const u8vec4 i0 = unpack8( vui & 0x0F0F0F0F); + const u8vec4 i1 = unpack8((vui >> 4) & 0x0F0F0F0F); + + return i32vec2( + pack32(i8vec4(kvalues_iq4nl[i0.x], kvalues_iq4nl[i0.y], kvalues_iq4nl[i0.z], kvalues_iq4nl[i0.w])), + pack32(i8vec4(kvalues_iq4nl[i1.x], kvalues_iq4nl[i1.y], kvalues_iq4nl[i1.z], kvalues_iq4nl[i1.w]))); +} +#endif +#endif #endif #if defined(DATA_A_MXFP4) || defined(DATA_A_NVFP4) @@ -1851,7 +1937,7 @@ float ue4m3_to_fp32_build(uint u) { } #endif -#if !defined(USE_OCP_FP4) +#if (defined(DATA_A_MXFP4) || defined(DATA_A_NVFP4)) && !defined(USE_OCP_FP4) #define NEEDS_INIT_IQ_SHMEM void init_iq_shmem(uvec3 wgsize) { diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/unary.comp b/ggml/src/ggml-vulkan/vulkan-shaders/unary.comp index 5ee5275d..9ee7769b 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/unary.comp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/unary.comp @@ -1,9 +1,23 @@ #version 450 #include "types.glsl" +#if defined(UNARY_MUL_FUSION) +#include "generic_binary_head.glsl" +#else #include "generic_unary_head.glsl" +#endif +#if defined(UNARY_MUL_FUSION) +// OP on src1 +layout(constant_id = 1) const bool op_on_b = false; +#endif + +#if defined(UNARY_MUL_FUSION) +layout(local_size_x = 256, local_size_y = 1, local_size_z = 1) in; +const uint num_threads = 256; +#else layout(local_size_x = 512, local_size_y = 1, local_size_z = 1) in; +#endif float op_abs(float x) { return abs(x); @@ -123,6 +137,7 @@ float op_gelu_erf(float a) { return 0.5f * a * (1.0f + sign_x * y); } +#if !defined(UNARY_MUL_FUSION) float op_xielu(float x) { const float alpha_n = p.param1; const float alpha_p = p.param2; @@ -136,6 +151,7 @@ float op_xielu(float x) { const float min_x_eps = min(x, eps); return (op_expm1(min_x_eps) - x) * alpha_n + beta * x; } +#endif float op_floor(float x) { return floor(x); @@ -155,8 +171,28 @@ float op_trunc(float x) { } void main() { - const uint idx = get_idx(); - + uint idx = get_idx(); + +#if defined(UNARY_MUL_FUSION) + // keep total threads at 512 + [[unroll]] for (uint iter = 0; iter < 2; ++iter) { + if (idx >= p.ne) { + continue; + } + uint i00, i01, i02, i03; + get_indices(idx, i00, i01, i02, i03); + + if (op_on_b) { + data_d[get_doffset() + dst_idx(i00, i01, i02, i03)] = + D_TYPE(FLOAT_TYPE(OP(float(data_b[get_boffset() + src1_idx(i00, i01, i02, i03)]))) * FLOAT_TYPE(data_a[get_aoffset() + src0_idx(i00, i01, i02, i03)])); + } else { + data_d[get_doffset() + dst_idx(i00, i01, i02, i03)] = + D_TYPE(FLOAT_TYPE(OP(float(data_a[get_aoffset() + src0_idx(i00, i01, i02, i03)]))) * FLOAT_TYPE(data_b[get_boffset() + src1_idx(i00, i01, i02, i03)])); + } + + idx += num_threads; + } +#else if (idx >= p.ne) { return; } @@ -165,4 +201,5 @@ void main() { const uint d_idx = get_doffset() + dst_idx(idx); data_d[d_idx] = D_TYPE(OP(float(data_a[a_idx]))); +#endif } diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/utils.glsl b/ggml/src/ggml-vulkan/vulkan-shaders/utils.glsl index dc4a1e6d..8aac64d7 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/utils.glsl +++ b/ggml/src/ggml-vulkan/vulkan-shaders/utils.glsl @@ -9,14 +9,26 @@ uint fastmod(uint a, uint b) { return a % b; } -uint fastdiv(uint a, uint b) { +// see init_fastdiv_values in ggml-vulkan.cpp +uint fastdiv(uint n, uint mp, uint L) { + uint msbs, lsbs; + // msbs = mulhi(n, mp) + umulExtended(n, mp, msbs, lsbs); + return (msbs + n) >> L; +} + +uint fastdiv_L(uint packed, uint slot) { + return (packed >> (slot * 8)) & 0x3Fu; +} + +uint fastdiv_small(uint a, uint b) { return (a < b) ? 0 : (a / b); } void get_indices(uint idx, out uint i00, out uint i01, out uint i02, out uint i03, uint ne00, uint ne01, uint ne02, uint ne03) { - i03 = fastdiv(idx, (ne02*ne01*ne00)); + i03 = fastdiv_small(idx, (ne02*ne01*ne00)); const uint i03_offset = i03 * ne02*ne01*ne00; - i02 = fastdiv((idx - i03_offset), (ne01*ne00)); + i02 = fastdiv_small((idx - i03_offset), (ne01*ne00)); const uint i02_offset = i02*ne01*ne00; i01 = (idx - i03_offset - i02_offset) / ne00; i00 = idx - i03_offset - i02_offset - i01*ne00; diff --git a/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp b/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp index 6c9f76af..12f9b3f5 100644 --- a/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +++ b/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp @@ -72,6 +72,7 @@ const std::vector type_names = { "iq4_nl", "mxfp4", "nvfp4", + "tq1_0", "tq2_0", "bf16", }; @@ -245,6 +246,17 @@ bool is_iq_quant(const std::string& type_name) { return string_starts_with(type_name, "iq"); } +bool is_lut_quant(const std::string& type_name) { + return is_iq_quant(type_name) || type_name == "mxfp4" || type_name == "nvfp4"; +} + +std::string lut_load_vec_a(const std::string& type_name) { + if (type_name == "iq1_s" || type_name == "iq1_m" || type_name == "iq2_xxs" || type_name == "iq2_xs" || type_name == "iq2_s" || type_name == "iq4_xs") { + return "8"; + } + return "4"; +} + static const char path_separator = '/'; std::string join_paths(const std::string& path1, const std::string& path2) { @@ -582,20 +594,28 @@ void matmul_shaders(bool fp16, MatMulIdType matmul_id_type, bool coopmat, bool c } for (const auto& tname : type_names) { - std::string load_vec_quant = "2"; - if ((tname == "q1_0") || (tname == "q4_0") || (tname == "q4_1") || (tname == "q5_1") || (tname == "iq1_s") || (tname == "iq1_m") || (tname == "iq2_xxs") || (tname == "iq2_xs") || (tname == "iq2_s")) - load_vec_quant = "8"; - else if ((tname == "q2_0") || (tname == "q5_0") || (tname == "q8_0") || (tname == "q2_k") || (tname == "q4_k") || (tname == "q5_k") || (tname == "iq3_xxs") || (tname == "iq3_s") || (tname == "iq4_xs") || (tname == "iq4_nl") || (tname == "mxfp4") || (tname == "nvfp4")) - load_vec_quant = "4"; - if (tname == "bf16") { continue; } - std::string data_a_key = "DATA_A_" + to_uppercase(tname); - // For aligned matmul loads - std::string load_vec_a = (coopmat2 || tname == "f32" || tname == "f16" || tname == "bf16") ? load_vec : load_vec_quant; + // Float types keep per-type compilation (different accumulation loop structure) + if (tname == "f32" || tname == "f16") { + std::string data_a_key = "DATA_A_" + to_uppercase(tname); + + const std::map float_type_dict = { + {"FLOAT_TYPE", FLOAT_TYPE(1, tname)}, + {"FLOAT_TYPEV2", FLOAT_TYPE(2, tname)}, + {"FLOAT_TYPEV4", FLOAT_TYPE(4, tname)}, + {"FLOAT_TYPEV8", FLOAT_TYPE(8, tname)}, + }; + if (!coopmat2) { + string_to_spv(shader_name + "_" + tname + "_f32" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", load_vec}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + } + continue; + } + + std::string data_a_key = "DATA_A_" + to_uppercase(tname); const std::map float_type_dict = { {"FLOAT_TYPE", FLOAT_TYPE(1, tname)}, {"FLOAT_TYPEV2", FLOAT_TYPE(2, tname)}, @@ -603,30 +623,52 @@ void matmul_shaders(bool fp16, MatMulIdType matmul_id_type, bool coopmat, bool c {"FLOAT_TYPEV8", FLOAT_TYPE(8, tname)}, }; - // don't generate f32 variants for coopmat2 - if (!coopmat2) { - string_to_spv(shader_name + "_" + tname + "_f32" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", load_vec_a}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); +#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) + if (!f16acc && !coopmat && !coopmat2 && !dot2 && (is_legacy_quant(tname) || is_k_quant(tname) || tname == "mxfp4" || tname == "iq3_s" || tname == "iq4_xs")) { + string_to_spv(shader_name + "_" + tname + "_q8_1", "mul_mmq.comp", merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"D_TYPE", "float"},}), fp16, coopmat, coopmat2, f16acc); } +#endif - if (tname != "f16" && tname != "f32") { - string_to_spv(shader_name + "_" + tname + "_f16" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", load_vec_a}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); - } + if (is_lut_quant(tname)) { + std::string lva = lut_load_vec_a(tname); + + string_to_spv(shader_name + "_" + tname + "_f16" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", lva}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); -#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) - if ((coopmat || coopmat2) && (tname == "mxfp4" || tname == "nvfp4")) { if (!coopmat2) { - string_to_spv(shader_name + "_" + tname + "_f32_ocp" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"USE_OCP_FP4", "1"}, {"LOAD_VEC_A", load_vec_a}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + string_to_spv(shader_name + "_" + tname + "_f32" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", lva}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + } + +#if defined(GGML_VULKAN_FLOAT_E2M1_GLSLC_SUPPORT) && defined(GGML_VULKAN_FLOAT_E4M3_GLSLC_SUPPORT) + if ((tname == "mxfp4" || tname == "nvfp4") && (coopmat || coopmat2)) { + string_to_spv(shader_name + "_" + tname + "_f16_ocp" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"USE_OCP_FP4", "1"}, {"LOAD_VEC_A", lva}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + if (!coopmat2) { + string_to_spv(shader_name + "_" + tname + "_f32_ocp" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"USE_OCP_FP4", "1"}, {"LOAD_VEC_A", lva}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + } } - string_to_spv(shader_name + "_" + tname + "_f16_ocp" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"USE_OCP_FP4", "1"}, {"LOAD_VEC_A", load_vec_a}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); - } #endif + continue; + } -#if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - // Integer dot mmq performs better with f32 accumulators (different shader, skip for dot2) - if (!f16acc && !coopmat && !coopmat2 && !dot2 && (is_legacy_quant(tname) || is_k_quant(tname) || tname == "mxfp4")) { - string_to_spv(shader_name + "_" + tname + "_q8_1", "mul_mmq.comp", merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"D_TYPE", "float"},}), fp16, coopmat, coopmat2, f16acc); + // dedicated shader needed due to regression on Ampere + if (coopmat2 && (tname == "q4_k" || tname == "q5_k")) { + string_to_spv(shader_name + "_" + tname + "_f16" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, float_type_dict), {{data_a_key, "1"}, {"LOAD_VEC_A", load_vec}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + } + } + + // Quant shader: one SPIR-V for all quant types, selected via MmTypeA spec constant + { + const std::map quant_float_type_dict = { + {"FLOAT_TYPE", FLOAT_TYPE(1, "q4_0")}, + {"FLOAT_TYPEV2", FLOAT_TYPE(2, "q4_0")}, + {"FLOAT_TYPEV4", FLOAT_TYPE(4, "q4_0")}, + {"FLOAT_TYPEV8", FLOAT_TYPE(8, "q4_0")}, + }; + + string_to_spv(shader_name + "_quant_f16" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, quant_float_type_dict), {{"MULMAT_QUANT", "1"}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f16}, {"B_TYPE_SCALAR", "float16_t"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); + + if (!coopmat2) { + string_to_spv(shader_name + "_quant_f32" + dot2_sfx, source_name, merge_maps(merge_maps(base_dict, quant_float_type_dict), {{"MULMAT_QUANT", "1"}, {"LOAD_VEC_B", load_vec}, {"B_TYPE", aligned_b_type_f32}, {"B_TYPE_SCALAR", "float"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}}), fp16, coopmat, coopmat2, f16acc); } -#endif } } @@ -734,7 +776,7 @@ void process_shaders() { for (const auto& tname : type_names) { // mul mat vec std::string data_a_key = "DATA_A_" + to_uppercase(tname); - std::string shader = (string_ends_with(tname, "_k") || string_starts_with(tname, "iq1_") || string_starts_with(tname, "iq2_") || string_starts_with(tname, "iq3_") || tname == "tq2_0") ? "mul_mat_vec_" + tname + ".comp" : "mul_mat_vec.comp"; + std::string shader = (string_ends_with(tname, "_k") || string_starts_with(tname, "iq1_") || string_starts_with(tname, "iq2_") || string_starts_with(tname, "iq3_") || tname == "iq4_xs" || tname == "tq2_0" || tname == "tq1_0") ? "mul_mat_vec_" + tname + ".comp" : "mul_mat_vec.comp"; string_to_spv("mul_mat_vec_" + tname + "_f32_f32", shader, merge_maps(base_dict, {{data_a_key, "1"}, {"B_TYPE", "float"}, {"B_TYPEV2", "vec2"}, {"B_TYPEV4", "vec4"}, {"D_TYPE", "float"}})); string_to_spv("mul_mat_vec_" + tname + "_f16_f32", shader, merge_maps(base_dict, {{data_a_key, "1"}, {"B_TYPE", "float16_t"}, {"B_TYPEV2", "f16vec2"}, {"B_TYPEV4", "f16vec4"}, {"D_TYPE", "float"}})); @@ -765,7 +807,7 @@ void process_shaders() { // mul mat vec with integer dot product #if defined(GGML_VULKAN_INTEGER_DOT_GLSLC_SUPPORT) - if (is_legacy_quant(tname) || tname == "mxfp4" || is_k_quant(tname) || tname == "iq1_s" || tname == "iq1_m") { + if (is_legacy_quant(tname) || tname == "mxfp4" || is_k_quant(tname) || tname == "iq1_s" || tname == "iq1_m" || tname == "iq4_xs") { string_to_spv("mul_mat_vec_" + tname + "_q8_1_f32", "mul_mat_vecq.comp", merge_maps(base_dict, {{data_a_key, "1"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"FLOAT_TYPEV2", "vec2"}, {"ACC_TYPE", "float"}})); string_to_spv("mul_mat_vec_" + tname + "_q8_1_f32_subgroup", "mul_mat_vecq.comp", merge_maps(base_dict, {{data_a_key, "1"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"FLOAT_TYPEV2", "vec2"}, {"ACC_TYPE", "float"}, {"USE_SUBGROUP_ADD", "1"}})); string_to_spv("mul_mat_vec_" + tname + "_q8_1_f32_subgroup_no_shmem", "mul_mat_vecq.comp", merge_maps(base_dict, {{data_a_key, "1"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"FLOAT_TYPEV2", "vec2"}, {"ACC_TYPE", "float"}, {"USE_SUBGROUP_ADD_NO_SHMEM", "1"}})); @@ -780,6 +822,10 @@ void process_shaders() { if (tname != "f16" && tname != "bf16") { string_to_spv("dequant_" + tname, "dequant_" + tname + ".comp", merge_maps(base_dict, {{data_a_key, "1"}, {"D_TYPE", "float16_t"}})); } + // Fused dequant+transpose variant for FA quant-KV (per-head-contiguous f16 scratch). + if (tname == "q8_0") { + string_to_spv("dequant_" + tname + "_transpose", "dequant_" + tname + ".comp", merge_maps(base_dict, {{data_a_key, "1"}, {"D_TYPE", "float16_t"}, {"DEQUANT_TRANSPOSE", "1"}})); + } shader = (tname == "f32" || tname == "f16" || tname == "bf16") ? "get_rows.comp" : "get_rows_quant.comp"; @@ -801,6 +847,10 @@ void process_shaders() { string_to_spv("norm_f32", "norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("group_norm_f32", "group_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("rms_norm_f32", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}})); + string_to_spv("rms_norm_mul_add_f32", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"RMS_NORM_ADD_FUSION", "1"}})); + string_to_spv("rms_norm_mul_add_partials_f32", "rms_norm_partials.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"RMS_NORM_ADD_FUSION", "1"}})); + string_to_spv("rms_norm_set_rows_f32_f32", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"RMS_NORM_SET_ROWS_FUSION", "1"}})); + string_to_spv("rms_norm_set_rows_f32_f16", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float16_t"}, {"RMS_NORM_SET_ROWS_FUSION", "1"}})); string_to_spv("rms_norm_partials_f32", "rms_norm_partials.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("rms_norm_mul_rope_f32_f32", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"ROPE_D_TYPE", "float"}, {"RMS_NORM_ROPE_FUSION", "1"}})); string_to_spv("rms_norm_mul_rope_f32_f16", "rms_norm.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"ROPE_D_TYPE", "float16_t"}, {"RMS_NORM_ROPE_FUSION", "1"}})); @@ -826,6 +876,8 @@ void process_shaders() { string_to_spv("cpy_transpose_16", "copy_transpose.comp", {{"A_TYPE", "uint16_t"}, {"D_TYPE", "uint16_t"}}); string_to_spv("cpy_transpose_32", "copy_transpose.comp", {{"A_TYPE", "uint"}, {"D_TYPE", "uint"}}); + string_to_spv("cpy_transpose_02_16", "copy_transpose_02.comp", {{"A_TYPE", "uint16_t"}, {"D_TYPE", "uint16_t"}}); + string_to_spv("cpy_transpose_02_32", "copy_transpose_02.comp", {{"A_TYPE", "uint"}, {"D_TYPE", "uint"}}); for (std::string t : {"q1_0", "q2_0", "q4_0", "q4_1", "q5_0", "q5_1", "q8_0", "iq4_nl"}) { string_to_spv("cpy_f32_" + t, "copy_to_quant.comp", {{"DATA_A_" + to_uppercase(t), "1"}, {"S_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}}); @@ -871,6 +923,12 @@ void process_shaders() { string_to_spv("fa_mask_opt", "flash_attn_mask_opt.comp", {}); + string_to_spv("fa_decode_ph1", "flash_attn_decode_phase_1.comp", {}, true, true, false, false); + string_to_spv("fa_decode_ph2", "flash_attn_decode_phase_2.comp", {}, true, true, false, false); + + string_to_spv("fa_sparse_compact", "flash_attn_sparse_compact.comp", {}); + string_to_spv("fa_sparse_compact_subgroup", "flash_attn_sparse_compact.comp", {{"USE_SUBGROUPS", "1"}}); + string_to_spv("quantize_q8_1", "quantize_q8_1.comp", {}); string_to_spv("quantize_q8_1_subgroup", "quantize_q8_1.comp", {{"USE_SUBGROUPS", "1"}}); @@ -890,6 +948,7 @@ void process_shaders() { string_to_spv("scale_f32", "scale.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}}); string_to_spv("pad_f32", "pad.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); + string_to_spv("pad_reflect_1d_f32", "pad_reflect_1d.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); string_to_spv("concat_i8", "concat.comp", {{"A_TYPE", "uint8_t"}, {"B_TYPE", "uint8_t"}, {"D_TYPE", "uint8_t"}}); string_to_spv("concat_i16", "concat.comp", {{"A_TYPE", "uint16_t"}, {"B_TYPE", "uint16_t"}, {"D_TYPE", "uint16_t"}}); @@ -954,6 +1013,15 @@ void process_shaders() { string_to_spv("softplus_f16", "unary.comp", {{"A_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"OP", "op_softplus"}}); string_to_spv("softplus_f32", "unary.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}, {"OP", "op_softplus"}}); + string_to_spv("gelu_mul_f32", "unary.comp", {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"OP", "op_gelu"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("gelu_mul_f16", "unary.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}, {"OP", "op_gelu"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("sigmoid_mul_f32", "unary.comp", {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"OP", "op_sigmoid"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("sigmoid_mul_f16", "unary.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}, {"OP", "op_sigmoid"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("silu_mul_f32", "unary.comp", {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"OP", "op_silu"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("silu_mul_f16", "unary.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}, {"OP", "op_silu"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("softplus_mul_f32","unary.comp", {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}, {"OP", "op_softplus"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("softplus_mul_f16","unary.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}, {"OP", "op_softplus"}, {"UNARY_MUL_FUSION", "1"}}); + string_to_spv("add1_f16_f16", "add1.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}}); string_to_spv("add1_f16_f32", "add1.comp", {{"A_TYPE", "float16_t"}, {"B_TYPE", "float"}, {"D_TYPE", "float16_t"}, {"FLOAT_TYPE", "float"}}); string_to_spv("add1_f32_f32", "add1.comp", {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}, {"FLOAT_TYPE", "float"}}); @@ -979,6 +1047,8 @@ void process_shaders() { string_to_spv("swiglu_f32", "swiglu.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); string_to_spv("swiglu_oai_f16", "swiglu_oai.comp", {{"A_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}}); string_to_spv("swiglu_oai_f32", "swiglu_oai.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); + string_to_spv("swiglu_clamp_f16", "swiglu_clamp.comp", {{"A_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}}); + string_to_spv("swiglu_clamp_f32", "swiglu_clamp.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); string_to_spv("geglu_erf_f16", "geglu_erf.comp", {{"A_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}}); string_to_spv("geglu_erf_f32", "geglu_erf.comp", {{"A_TYPE", "float"}, {"D_TYPE", "float"}}); string_to_spv("geglu_quick_f16","geglu_quick.comp", {{"A_TYPE", "float16_t"}, {"D_TYPE", "float16_t"}}); @@ -1019,17 +1089,24 @@ void process_shaders() { string_to_spv("topk_argsort_f32", "topk_argsort.comp", {{"A_TYPE", "float"}}); string_to_spv("topk_nary_search_f32", "topk_nary_search.comp", {{"A_TYPE", "float"}}); + string_to_spv("topk_radix_select_f32", "topk_radix_select.comp", {{"A_TYPE", "float"}}); string_to_spv("argmax_f32", "argmax.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "int"}})); string_to_spv("sum_rows_f32", "sum_rows.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); + string_to_spv("cross_entropy_loss_f32", "cross_entropy_loss.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}})); + string_to_spv("cross_entropy_loss_back_f32", "cross_entropy_loss_back.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"B_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("fwht_f32", "fwht.comp", {}); string_to_spv("fwht_shmem_f32", "fwht.comp", {{"FWHT_SHMEM", "1"}}); string_to_spv("count_equal_i32", "count_equal.comp", merge_maps(base_dict, {{"A_TYPE", "int"}, {"B_TYPE", "int"}, {"D_TYPE", "int"}})); + string_to_spv("dsv4_hc_comb_f32", "dsv4_hc_comb.comp", {}); + string_to_spv("dsv4_hc_pre_f32", "dsv4_hc_pre.comp", {}); + string_to_spv("dsv4_hc_post_f32", "dsv4_hc_post.comp", {}); string_to_spv("cumsum_f32", "cumsum.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("cumsum_multipass1_f32", "cumsum_multipass1.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("cumsum_multipass2_f32", "cumsum_multipass2.comp", merge_maps(base_dict, {{"A_TYPE", "float"}, {"D_TYPE", "float"}})); string_to_spv("count_experts", "count_experts.comp", merge_maps(base_dict, {{"A_TYPE", "uint"}, {"D_TYPE", "uint"}})); + string_to_spv("count_experts_subgroup", "count_experts.comp", merge_maps(base_dict, {{"A_TYPE", "uint"}, {"D_TYPE", "uint"}, {"USE_SUBGROUPS", "1"}})); for (std::string dim_str : {"", "_3d"}) { for (bool bda : {false, true}) { @@ -1060,6 +1137,12 @@ void process_shaders() { string_to_spv("gated_linear_attn_f32", "gla.comp", merge_maps(base_dict, {{"A_TYPE", "float"}})); + // Compile IQ4_NL support in so its shared LUT is available when K uses it. + // K quant type is selected at runtime via the FaTypeK spec constant. + std::map li_dict = {{"FLOAT_TYPE", "float"}, {"FLOAT_TYPEV4", "vec4"}, {"DATA_A_IQ4_NL", "1"}}; + string_to_spv("lightning_indexer_f32", "lightning_indexer.comp", li_dict); + string_to_spv("lightning_indexer_subgroup_f32", "lightning_indexer.comp", merge_maps(li_dict, {{"USE_SUBGROUP_ADD", "1"}})); + string_to_spv("rwkv_wkv7_f32", "wkv7.comp", merge_maps(base_dict, {{"A_TYPE", "float"}})); string_to_spv("gated_delta_net_f32", "gated_delta_net.comp", merge_maps(base_dict, {{"FLOAT_TYPE", "float"}, {"USE_SUBGROUP_ADD", "1"}, {"USE_SUBGROUP_CLUSTERED", "1"}})); @@ -1245,7 +1328,7 @@ void write_output_files() { for (const std::string& btype : btypes) { for (const auto& tname : type_names) { - if (btype == "q8_1" && !is_legacy_quant(tname) && tname != "mxfp4" && !is_k_quant(tname) && tname != "iq1_s" && tname != "iq1_m") { + if (btype == "q8_1" && !is_legacy_quant(tname) && tname != "mxfp4" && !is_k_quant(tname) && tname != "iq1_s" && tname != "iq1_m" && tname != "iq4_xs") { continue; } hdr << "extern const void * arr_dmmv_" << tname << "_" << btype << "_f32_data[3];\n"; diff --git a/ggml/src/ggml-webgpu/CMakeLists.txt b/ggml/src/ggml-webgpu/CMakeLists.txt index 1503a1ef..2eacca7f 100644 --- a/ggml/src/ggml-webgpu/CMakeLists.txt +++ b/ggml/src/ggml-webgpu/CMakeLists.txt @@ -39,6 +39,12 @@ ggml_add_backend_library(ggml-webgpu add_dependencies(ggml-webgpu generate_shaders) +# Dawn needs C++20 (https://dawn.googlesource.com/dawn/+/refs/heads/main/docs/quickstart-cmake.md#prerequisites) +target_compile_features(ggml-webgpu PRIVATE cxx_std_20) + +# Disable C++20 module scanning since emscan-deps fails to find webgpu_cpp.h +set_target_properties(ggml-webgpu PROPERTIES CXX_SCAN_FOR_MODULES OFF) + if(EMSCRIPTEN) set(EMDAWNWEBGPU_DIR "" CACHE PATH "Path to emdawnwebgpu_pkg") diff --git a/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp b/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp index 0604e1c2..47a266d7 100644 --- a/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp +++ b/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp @@ -81,6 +81,7 @@ struct ggml_webgpu_shader_lib_context { ggml_tensor * src4; ggml_tensor * src5; ggml_tensor * dst; + ggml_tensor * dst_fuse; uint32_t max_wg_size; size_t wg_mem_limit_bytes = 0; @@ -106,6 +107,11 @@ struct ggml_webgpu_generic_shader_decisions { bool inplace = false; }; +struct ggml_webgpu_get_rows_shader_decisions { + uint32_t wg_size = 0; + bool vectorized = false; +}; + struct ggml_webgpu_binary_shader_decisions { uint32_t wg_size = 0; bool inplace = false; @@ -407,12 +413,13 @@ struct ggml_webgpu_im2col_pipeline_key_hash { /** Gated Delta Net **/ struct ggml_webgpu_gated_delta_net_pipeline_key { - int type; - int s_v; - int kda; + int type; + int s_v; + int kda; + bool fused_cache; bool operator==(const ggml_webgpu_gated_delta_net_pipeline_key & other) const { - return type == other.type && s_v == other.s_v && kda == other.kda; + return type == other.type && s_v == other.s_v && kda == other.kda && fused_cache == other.fused_cache; } }; @@ -954,10 +961,11 @@ struct ggml_webgpu_mul_mat_vec_pipeline_key { int vectorized; uint32_t num_cols; bool use_mmvq; + bool src_overlap; bool operator==(const ggml_webgpu_mul_mat_vec_pipeline_key & other) const { return src0_type == other.src0_type && src1_type == other.src1_type && vectorized == other.vectorized && - num_cols == other.num_cols && use_mmvq == other.use_mmvq; + num_cols == other.num_cols && use_mmvq == other.use_mmvq && src_overlap == other.src_overlap; } }; @@ -969,6 +977,7 @@ struct ggml_webgpu_mul_mat_vec_pipeline_key_hash { ggml_webgpu_hash_combine(seed, key.vectorized); ggml_webgpu_hash_combine(seed, key.num_cols); ggml_webgpu_hash_combine(seed, key.use_mmvq); + ggml_webgpu_hash_combine(seed, key.src_overlap); return seed; } }; @@ -977,6 +986,7 @@ struct ggml_webgpu_mul_mat_vec_shader_decisions { uint32_t wg_size; uint32_t outputs_per_wg; uint32_t vec_size; + bool src_overlap = false; }; struct ggml_webgpu_quantize_q8_pipeline_key { @@ -998,10 +1008,11 @@ struct ggml_webgpu_mul_mat_pipeline_key { ggml_type src1_type; int vectorized; int use_subgroup_matrix; + bool src_overlap; bool operator==(const ggml_webgpu_mul_mat_pipeline_key & other) const { return src0_type == other.src0_type && src1_type == other.src1_type && vectorized == other.vectorized && - use_subgroup_matrix == other.use_subgroup_matrix; + use_subgroup_matrix == other.use_subgroup_matrix && src_overlap == other.src_overlap; } }; @@ -1012,6 +1023,7 @@ struct ggml_webgpu_mul_mat_pipeline_key_hash { ggml_webgpu_hash_combine(seed, key.src1_type); ggml_webgpu_hash_combine(seed, key.vectorized); ggml_webgpu_hash_combine(seed, key.use_subgroup_matrix); + ggml_webgpu_hash_combine(seed, key.src_overlap); return seed; } }; @@ -1034,6 +1046,7 @@ struct ggml_webgpu_mul_mat_shader_decisions { uint32_t subgroup_matrix_n; uint32_t mul_mat_wg_size; + bool src_overlap = false; }; /** MUL_MAT_ID **/ @@ -1545,8 +1558,8 @@ class ggml_webgpu_shader_lib { return argsort_merge_pipelines[order]; } - webgpu_pipeline get_get_rows_pipeline(const ggml_webgpu_shader_lib_context & context) { - const bool vectorized = context.src0->type == GGML_TYPE_F32 && context.dst->ne[0] % 4 == 0; + webgpu_pipeline get_get_rows_pipeline(const ggml_webgpu_shader_lib_context & context, bool vec4_aligned) { + const bool vectorized = context.src0->type == GGML_TYPE_F32 && context.dst->ne[0] % 4 == 0 && vec4_aligned; ggml_webgpu_get_rows_pipeline_key key = {}; key.src_type = context.src0->type; key.vectorized = (int) vectorized; @@ -1663,8 +1676,9 @@ class ggml_webgpu_shader_lib { defines.push_back("WG_SIZE=" + std::to_string(context.max_wg_size)); auto processed = preprocessor.preprocess(wgsl_get_rows, defines); - auto decisions = std::make_shared(); + auto decisions = std::make_shared(); decisions->wg_size = context.max_wg_size; + decisions->vectorized = vectorized; webgpu_pipeline pipeline = ggml_webgpu_create_pipeline(device, processed, variant); pipeline.context = decisions; get_rows_pipelines[key] = pipeline; @@ -1853,6 +1867,7 @@ class ggml_webgpu_shader_lib { key.type = context.dst->type; key.s_v = (int) context.src2->ne[0]; key.kda = context.src3->ne[0] == context.src2->ne[0]; + key.fused_cache = context.dst_fuse != nullptr; auto it = gated_delta_net_pipelines.find(key); if (it != gated_delta_net_pipelines.end()) { @@ -1875,6 +1890,11 @@ class ggml_webgpu_shader_lib { variant += "_kda"; } + if (key.fused_cache) { + defines.push_back("FUSED_CACHE"); + variant += "_fused_cache"; + } + defines.push_back("S_V=" + std::to_string(key.s_v) + "u"); defines.push_back("WG_SIZE=" + std::to_string(key.s_v) + "u"); @@ -1950,7 +1970,7 @@ class ggml_webgpu_shader_lib { return quantize_q8_pipelines[key]; } - webgpu_pipeline get_mul_mat_vec_pipeline(const ggml_webgpu_shader_lib_context & context) { + webgpu_pipeline get_mul_mat_vec_pipeline(const ggml_webgpu_shader_lib_context & context, bool src_overlap) { ggml_webgpu_mul_mat_vec_pipeline_key key = {}; key.src0_type = context.src0->type; key.src1_type = context.src1->type; @@ -1961,6 +1981,7 @@ class ggml_webgpu_shader_lib { key.num_cols = context.dst->ne[1]; key.use_mmvq = ggml_webgpu_can_use_mmvq(context.src0, context.src1, context.supports_dot_product, context.vendor); + key.src_overlap = src_overlap; auto it = mul_mat_vec_pipelines.find(key); if (it != mul_mat_vec_pipelines.end()) { @@ -2068,6 +2089,11 @@ class ggml_webgpu_shader_lib { defines.push_back("Q8_1_T"); } + if (key.src_overlap) { + defines.push_back("SRC_OVERLAP"); + variant += "_src_overlap"; + } + defines.push_back(std::string("WG_SIZE=") + std::to_string(wg_size)); defines.push_back(std::string("OUTPUTS_PER_WG=") + std::to_string(outputs_per_wg)); defines.push_back(context.supports_subgroups ? "USE_SUBGROUP_REDUCTION" : "USE_WORKGROUP_REDUCTION"); @@ -2089,7 +2115,7 @@ class ggml_webgpu_shader_lib { return mul_mat_vec_pipelines[key]; } - webgpu_pipeline get_mul_mat_fast_pipeline(const ggml_webgpu_shader_lib_context & context) { + webgpu_pipeline get_mul_mat_fast_pipeline(const ggml_webgpu_shader_lib_context & context, bool src_overlap) { ggml_webgpu_mul_mat_pipeline_key key = {}; key.src0_type = context.src0->type; key.src1_type = context.src1->type; @@ -2098,6 +2124,7 @@ class ggml_webgpu_shader_lib { 1 : 0; key.use_subgroup_matrix = context.supports_subgroup_matrix; + key.src_overlap = src_overlap; auto it = mul_mat_fast_pipelines.find(key); if (it != mul_mat_fast_pipelines.end()) { @@ -2216,6 +2243,11 @@ class ggml_webgpu_shader_lib { variant += "_vectorized"; } + if (key.src_overlap) { + defines.push_back("SRC_OVERLAP"); + variant += "_src_overlap"; + } + if (!key.use_subgroup_matrix) { defines.push_back("WORKGROUP_SIZE_M=" + std::to_string(WEBGPU_MUL_MAT_WG_SIZE_M) + "u"); defines.push_back("WORKGROUP_SIZE_N=" + std::to_string(WEBGPU_MUL_MAT_WG_SIZE_N) + "u"); @@ -3083,6 +3115,10 @@ class ggml_webgpu_shader_lib { defines.push_back("OP_GEGLU_QUICK"); variant += "_geglu_quick"; break; + case GGML_GLU_OP_SWIGLU_CLAMP: + defines.push_back("OP_SWIGLU_CLAMP"); + variant += "_swiglu_clamp"; + break; default: GGML_ABORT("Unsupported GLU op"); } diff --git a/ggml/src/ggml-webgpu/ggml-webgpu.cpp b/ggml/src/ggml-webgpu/ggml-webgpu.cpp index 394aeeda..9c5dc768 100644 --- a/ggml/src/ggml-webgpu/ggml-webgpu.cpp +++ b/ggml/src/ggml-webgpu/ggml-webgpu.cpp @@ -374,20 +374,28 @@ static wgpu::Buffer ggml_webgpu_tensor_buf(const ggml_tensor * tensor) { return ctx->buffer; } +// Binding offset for a tensor: the largest aligned offset at or before the tensor whose +// distance to the tensor is a whole number of type blocks, so shaders can index the +// misalignment in elements even for block quantized types. +static size_t ggml_webgpu_tensor_align_offset(const ggml_tensor * t, size_t alignment) { + const size_t offset = ggml_webgpu_tensor_offset(t); + const size_t type_size = ggml_type_size(t->type); + size_t aligned = offset & ~(alignment - 1); + while ((offset - aligned) % type_size != 0) { + GGML_ASSERT(aligned >= alignment); + aligned -= alignment; + } + return aligned; +} + static size_t ggml_webgpu_tensor_misalignment(const ggml_tensor * t, size_t alignment) { - size_t offset = ggml_webgpu_tensor_offset(t); - return offset & (alignment - 1); + return ggml_webgpu_tensor_offset(t) - ggml_webgpu_tensor_align_offset(t, alignment); } static size_t ggml_webgpu_tensor_misalignment(webgpu_context & ctx, const ggml_tensor * t) { return ggml_webgpu_tensor_misalignment(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment); } -static size_t ggml_webgpu_tensor_align_offset(const ggml_tensor * t, size_t alignment) { - size_t offset = ggml_webgpu_tensor_offset(t); - return offset & ~(alignment - 1); -} - static size_t ggml_webgpu_tensor_align_offset(webgpu_context & ctx, const ggml_tensor * t) { return ggml_webgpu_tensor_align_offset(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment); } @@ -1375,7 +1383,8 @@ static webgpu_encoded_op ggml_webgpu_gated_delta_net(webgpu_context & ctx, ggml_tensor * src3, ggml_tensor * src4, ggml_tensor * src5, - ggml_tensor * dst) { + ggml_tensor * dst, + ggml_tensor * dst_fuse) { ggml_webgpu_shader_lib_context shader_lib_ctx = {}; shader_lib_ctx.src0 = src0; shader_lib_ctx.src1 = src1; @@ -1383,6 +1392,7 @@ static webgpu_encoded_op ggml_webgpu_gated_delta_net(webgpu_context & ctx, shader_lib_ctx.src3 = src3; shader_lib_ctx.src4 = src4; shader_lib_ctx.dst = dst; + shader_lib_ctx.dst_fuse = dst_fuse; shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup; webgpu_pipeline pipeline = ctx->shader_lib->get_gated_delta_net_pipeline(shader_lib_ctx); @@ -1418,6 +1428,8 @@ static webgpu_encoded_op ggml_webgpu_gated_delta_net(webgpu_context & ctx, (uint32_t) (src2->ne[3] / src0->ne[3]), K, scale_u32, + dst_fuse ? (uint32_t) (dst_fuse->nb[2] / ggml_type_size(dst_fuse->type)) : 0, + dst_fuse ? (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst_fuse) / ggml_type_size(dst_fuse->type)) : 0, }; std::vector entries = { @@ -1427,6 +1439,10 @@ static webgpu_encoded_op ggml_webgpu_gated_delta_net(webgpu_context & ctx, ggml_webgpu_make_tensor_bind_group_entry(ctx, 6, dst), }; + if (dst_fuse) { + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 7, dst_fuse)); + } + return ggml_backend_webgpu_build(ctx, pipeline, params, entries, h, n_seqs); } @@ -1510,15 +1526,24 @@ static webgpu_encoded_op ggml_webgpu_get_rows(webgpu_context & ctx, shader_lib_ctx.dst = dst; shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup; - webgpu_pipeline pipeline = ctx->shader_lib->get_get_rows_pipeline(shader_lib_ctx); - auto * decisions = static_cast(pipeline.context.get()); + const uint32_t offset_src = (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)); + const uint32_t offset_dst = (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)); + const uint32_t stride_src1 = (uint32_t) (src->nb[1] / ggml_type_size(src->type)); + const uint32_t stride_src2 = (uint32_t) (src->nb[2] / ggml_type_size(src->type)); + const uint32_t stride_src3 = (uint32_t) (src->nb[3] / ggml_type_size(src->type)); - std::vector params = { (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)), + const bool vec4_aligned = offset_src % 4 == 0 && offset_dst % 4 == 0 && stride_src1 % 4 == 0 && + stride_src2 % 4 == 0 && stride_src3 % 4 == 0; + + webgpu_pipeline pipeline = ctx->shader_lib->get_get_rows_pipeline(shader_lib_ctx, vec4_aligned); + auto * decisions = static_cast(pipeline.context.get()); + + std::vector params = { offset_src, (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, idx) / ggml_type_size(idx->type)), - (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)), - (uint32_t) (src->nb[1] / ggml_type_size(src->type)), - (uint32_t) (src->nb[2] / ggml_type_size(src->type)), - (uint32_t) (src->nb[3] / ggml_type_size(src->type)), + offset_dst, + stride_src1, + stride_src2, + stride_src3, (uint32_t) (idx->nb[0] / ggml_type_size(idx->type)), (uint32_t) (idx->nb[1] / ggml_type_size(idx->type)), (uint32_t) (idx->nb[2] / ggml_type_size(idx->type)), @@ -1536,7 +1561,7 @@ static webgpu_encoded_op ggml_webgpu_get_rows(webgpu_context & ctx, ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, idx), ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst) }; - uint32_t blocks_per_row = (uint32_t) (dst->ne[0] / (src->type == GGML_TYPE_F32 && dst->ne[0] % 4 == 0 ? 4 : 1)); + uint32_t blocks_per_row = (uint32_t) (dst->ne[0] / (decisions->vectorized ? 4 : 1)); uint32_t total_rows = (uint32_t) (dst->ne[1] * dst->ne[2] * dst->ne[3]); uint32_t total_threads = float_parallel ? blocks_per_row * total_rows : total_rows; uint32_t wg_x = CEIL_DIV(total_threads, decisions->wg_size); @@ -1628,48 +1653,65 @@ static webgpu_encoded_op ggml_webgpu_mul_mat(webgpu_context & ctx, // Get or create pipeline webgpu_pipeline pipeline; std::vector dispatches; + const bool src_overlap = ggml_webgpu_tensor_binding_overlap(ctx->global_ctx, src0, src1) && !use_mmvq; if (use_mat_vec) { if (use_mmvq) { ggml_webgpu_quantize_q8_dispatch(ctx, src0, src1, dst, dispatches); } - pipeline = ctx->shader_lib->get_mul_mat_vec_pipeline(shader_lib_ctx); + pipeline = ctx->shader_lib->get_mul_mat_vec_pipeline(shader_lib_ctx, src_overlap); } else { - pipeline = ctx->shader_lib->get_mul_mat_fast_pipeline(shader_lib_ctx); + pipeline = ctx->shader_lib->get_mul_mat_fast_pipeline(shader_lib_ctx, src_overlap); + } + + uint32_t offset_src0 = (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)); + uint32_t offset_src1 = (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)); + size_t merged_offset = 0; + size_t merged_size = 0; + if (src_overlap) { + const ggml_webgpu_merged_binding_range merged_range = + ggml_webgpu_tensor_merged_binding_range(ctx, { src0, src1 }); + merged_offset = merged_range.offset; + merged_size = merged_range.size; + offset_src0 = ggml_webgpu_tensor_merged_element_offset(src0, merged_range); + offset_src1 = ggml_webgpu_tensor_merged_element_offset(src1, merged_range); } // Build params - std::vector params = { - (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)), - (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)), - (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)), - (uint32_t) dst->ne[0], - (uint32_t) dst->ne[1], - (uint32_t) src0->ne[0], - (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)), - (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)), - (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)), - (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)), - (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)), - (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)), - (uint32_t) src0->ne[2], - (uint32_t) src0->ne[3], - (uint32_t) (src1->ne[2] / src0->ne[2]), - (uint32_t) (src1->ne[3] / src0->ne[3]) - }; + std::vector params = { offset_src0, + offset_src1, + (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)), + (uint32_t) dst->ne[0], + (uint32_t) dst->ne[1], + (uint32_t) src0->ne[0], + (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)), + (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)), + (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)), + (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)), + (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)), + (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)), + (uint32_t) src0->ne[2], + (uint32_t) src0->ne[3], + (uint32_t) (src1->ne[2] / src0->ne[2]), + (uint32_t) (src1->ne[3] / src0->ne[3]) }; // Build bind group entries std::vector entries = {}; - - entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0)); if (use_mmvq) { + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0)); auto & mmvq_qq8_entry = dispatches[0].bind_group_entries[1]; entries.push_back(ggml_webgpu_make_bind_group_entry(1, ggml_webgpu_tensor_buf(dst), mmvq_qq8_entry.offset, mmvq_qq8_entry.size)); + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst)); + } else if (src_overlap) { + entries.push_back( + ggml_webgpu_make_bind_group_entry(0, ggml_webgpu_tensor_buf(src0), merged_offset, merged_size)); + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst)); } else { + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0)); entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1)); + entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst)); } - entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst)); // Calculate workgroup dimensions uint32_t wg_x = 1; @@ -2697,6 +2739,7 @@ static webgpu_encoded_op ggml_webgpu_rope(webgpu_context & ctx, const int n_dims = ((int32_t *) dst->op_params)[1]; const int mode = ((int32_t *) dst->op_params)[2]; const int n_ctx_orig = ((int32_t *) dst->op_params)[4]; + const int n_offs = ((int32_t *) dst->op_params)[15]; float freq_base; float freq_scale; @@ -2745,7 +2788,8 @@ static webgpu_encoded_op ggml_webgpu_rope(webgpu_context & ctx, (uint32_t) sections[0], (uint32_t) sections[1], (uint32_t) sections[2], - (uint32_t) sections[3] + (uint32_t) sections[3], + (uint32_t) n_offs }; std::vector entries = { ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0), @@ -2816,7 +2860,7 @@ static webgpu_encoded_op ggml_webgpu_glu(webgpu_context & ctx, (uint32_t) dst->ne[2], (uint32_t) ((int32_t *) dst->op_params)[1], // swapped ggml_webgpu_u32_from_f32(ggml_get_op_params_f32(dst, 2)), // alpha, for swiglu_oai - ggml_webgpu_u32_from_f32(ggml_get_op_params_f32(dst, 3)), // limit, for swiglu_oai + ggml_webgpu_u32_from_f32(ggml_get_op_params_f32(dst, 3)), // limit }; std::vector entries; @@ -3184,6 +3228,67 @@ static bool ggml_webgpu_can_fuse_rms_norm_mul(const struct ggml_cgraph * cgraph, return true; } +static bool ggml_webgpu_can_fuse_gdn_cache(const struct ggml_cgraph * cgraph, int node_idx, int & num_encoded_ops) { + const ggml_tensor * gdn = cgraph->nodes[node_idx]; + + // the kernel skips the snapshot tail, so the gdn output must not be a graph output + if (gdn->op != GGML_OP_GATED_DELTA_NET || gdn->type != GGML_TYPE_F32 || (gdn->flags & GGML_TENSOR_FLAG_OUTPUT)) { + return false; + } + + const ggml_tensor * src_v = gdn->src[2]; + const int64_t S_v = src_v->ne[0]; + const int64_t H = src_v->ne[1]; + const int64_t n_tokens = src_v->ne[2]; + const int64_t n_seqs = src_v->ne[3]; + const int64_t D = S_v * S_v * H; + const int64_t K = ggml_get_op_params_i32(gdn, 0); // snapshot slot count + const int64_t n_written = std::min(n_tokens, K); // newest n_written slots are written + + // snapshot tail starts right after the attention scores + const size_t tail_off = ggml_row_size(GGML_TYPE_F32, S_v * H * n_tokens * n_seqs); + + // snapshot cpy is the first real node after the gdn (skip views/no-ops) + const ggml_tensor * cpy = nullptr; + int cpy_idx = 0; + for (int j = node_idx + 1; j < cgraph->n_nodes && cpy == nullptr; ++j) { + const ggml_tensor * n = cgraph->nodes[j]; + if (ggml_op_is_empty(n->op) || ggml_is_empty(n)) { + continue; + } + if (n->op != GGML_OP_CPY || (n->flags & GGML_TENSOR_FLAG_OUTPUT)) { + return false; + } + cpy = n; + cpy_idx = j; + } + if (cpy == nullptr) { + return false; + } + + const ggml_tensor * cpy_src = cpy->src[0]; // view of the gdn snapshot tail + const ggml_tensor * cpy_dst = cpy->src[1]; // cache view the kernel writes to + + // src must be this gdn's snapshot tail (contiguous, at the tail offset) + if (cpy_src->op != GGML_OP_VIEW || cpy_src->view_src != gdn || cpy_src->view_offs != tail_off || + !ggml_is_contiguous(cpy_src)) { + return false; + } + + // dst is the [D, n_seqs, n_written] cache view; require nb[1] == D (the per-seq stride the kernel + // assumes). ggml_cpy pins src to the same element count. + const std::array expected_ne = { D, n_seqs, n_written, 1 }; + if (cpy_dst->op != GGML_OP_VIEW || cpy_dst->type != GGML_TYPE_F32 || cpy_dst->data == nullptr || + !std::equal(expected_ne.begin(), expected_ne.end(), cpy_dst->ne) || + cpy_dst->nb[0] != ggml_type_size(GGML_TYPE_F32) || cpy_dst->nb[1] != (size_t) ggml_row_size(GGML_TYPE_F32, D)) { + return false; + } + + num_encoded_ops = cpy_idx - node_idx + 1; + + return true; +} + static webgpu_encoded_op ggml_webgpu_upscale(webgpu_context ctx, ggml_tensor * src, ggml_tensor * dst) { const uint32_t mode_flags = (uint32_t) ggml_get_op_params_i32(dst, 0); std::vector params = { (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)), @@ -3322,7 +3427,14 @@ static std::optional ggml_webgpu_encode(webgpu_context ctx, return ggml_webgpu_ssm_scan(ctx, src0, src1, src2, node->src[3], node->src[4], node->src[5], node->src[6], node); case GGML_OP_GATED_DELTA_NET: - return ggml_webgpu_gated_delta_net(ctx, src0, src1, src2, node->src[3], node->src[4], node->src[5], node); + if (ggml_webgpu_can_fuse_gdn_cache(cgraph, node_idx, num_encoded_ops)) { + ggml_tensor * dst_fuse = cgraph->nodes[node_idx + num_encoded_ops - 1]->src[1]; + return ggml_webgpu_gated_delta_net(ctx, src0, src1, src2, node->src[3], node->src[4], node->src[5], + node, dst_fuse); + } else { + return ggml_webgpu_gated_delta_net(ctx, src0, src1, src2, node->src[3], node->src[4], node->src[5], + node, nullptr); + } case GGML_OP_PAD: return ggml_webgpu_pad(ctx, src0, node); case GGML_OP_ARGMAX: @@ -3694,11 +3806,18 @@ static void ggml_backend_webgpu_buffer_get_tensor(ggml_backend_buffer_t buffer, size_t total_offset = ggml_webgpu_tensor_offset(tensor) + offset; - size_t final_size = size; - if (size % 4 != 0) { + size_t local_offset = total_offset % 4; + if (local_offset != 0) { + // If offset is not a multiple of 4, we need to round it down to the previous + // multiple of 4 + total_offset -= local_offset; + } + + size_t final_size = size + local_offset; + if (final_size % 4 != 0) { // If size is not a multiple of 4, we need to round it up to the next // multiple of 4 - final_size = size + (4 - (size % 4)); + final_size += 4 - (final_size % 4); } std::lock_guard lock(buf_ctx->global_ctx->mutex); @@ -3729,7 +3848,7 @@ static void ggml_backend_webgpu_buffer_get_tensor(ggml_backend_buffer_t buffer, const void * mapped_range = buf_ctx->global_ctx->get_tensor_staging_buf.GetConstMappedRange(0, final_size); // Copy the data from the mapped range to the output buffer - std::memcpy(data, mapped_range, size); + std::memcpy(data, (const void *) ((const char *) mapped_range + local_offset), size); buf_ctx->global_ctx->get_tensor_staging_buf.Unmap(); WEBGPU_CPU_PROFILE_TOTAL_END(get_tensor, buf_ctx->global_ctx); } @@ -3980,16 +4099,17 @@ static void ggml_backend_webgpu_request_adapter(wgpu::Instance & instance, wgpu: options.nextInChain = &adapterTogglesDesc; #endif - instance.WaitAny(instance.RequestAdapter( - &options, wgpu::CallbackMode::AllowSpontaneous, - [&adapter](wgpu::RequestAdapterStatus status, wgpu::Adapter _adapter, const char * message) { - if (status != wgpu::RequestAdapterStatus::Success) { - GGML_LOG_ERROR("ggml_webgpu: Failed to get an adapter: %s\n", message); - return; - } - adapter = std::move(_adapter); - }), - UINT64_MAX); + instance.WaitAny( + instance.RequestAdapter( + &options, wgpu::CallbackMode::AllowSpontaneous, + [&adapter](wgpu::RequestAdapterStatus status, wgpu::Adapter _adapter, wgpu::StringView message) { + if (status != wgpu::RequestAdapterStatus::Success) { + GGML_LOG_ERROR("ggml_webgpu: Failed to get an adapter: %s\n", std::string(message).c_str()); + return; + } + adapter = std::move(_adapter); + }), + UINT64_MAX); } static void create_webgpu_device(ggml_backend_webgpu_reg_context * ctx) { @@ -4386,6 +4506,9 @@ static bool ggml_backend_webgpu_device_supports_op(ggml_backend_dev_t dev, const default: break; } + if (ggml_get_op_params_i32(op, 3) == GGML_PREC_F32) { + supports_op = false; + } break; case GGML_OP_FLASH_ATTN_EXT: { @@ -4464,6 +4587,7 @@ static bool ggml_backend_webgpu_device_supports_op(ggml_backend_dev_t dev, const case GGML_GLU_OP_SWIGLU: case GGML_GLU_OP_GEGLU_ERF: case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: supports_op = op->type == GGML_TYPE_F32 || op->type == GGML_TYPE_F16; break; case GGML_GLU_OP_SWIGLU_OAI: diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/argsort.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/argsort.wgsl index 46ed19fc..fa5d9535 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/argsort.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/argsort.wgsl @@ -34,11 +34,9 @@ var params: Params; var shmem_idx: array; #if ORDER == 0 -#define EXTREME_VALUE 1e30 #define SWAP_COMPARE_UP > #define SWAP_COMPARE_DOWN < #else -#define EXTREME_VALUE -1e30 #define SWAP_COMPARE_UP < #define SWAP_COMPARE_DOWN > #endif @@ -78,11 +76,9 @@ fn main(@builtin(workgroup_id) wid: vec3, let dir_up = (lid.x & k) == 0; let a_idx = shmem_idx[lid.x]; let b_idx = shmem_idx[ixj]; - let a_val = select(EXTREME_VALUE, src[row_base + a_idx], a_idx < params.src_ne0); - let b_val = select(EXTREME_VALUE, src[row_base + b_idx], b_idx < params.src_ne0); let should_swap = select( - (a_val SWAP_COMPARE_DOWN b_val), - (a_val SWAP_COMPARE_UP b_val), + b_idx >= params.src_ne0 || (a_idx < params.src_ne0 && src[row_base + a_idx] SWAP_COMPARE_DOWN src[row_base + b_idx]), + a_idx >= params.src_ne0 || (b_idx < params.src_ne0 && src[row_base + a_idx] SWAP_COMPARE_UP src[row_base + b_idx]), dir_up); if (should_swap) { shmem_idx[lid.x] = b_idx; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl b/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl index b0cf2853..4a500e4e 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl @@ -1,3 +1,7 @@ +#ifndef SRC0 +#define SRC0 src0 +#endif + #ifdef BYTE_HELPERS fn get_byte(value: u32, index: u32) -> u32 { return (value >> (index * 8)) & 0xFF; @@ -46,7 +50,7 @@ fn load_f16_as_f32_at_src(byte_offset: u32) -> f32 { #ifdef DECLARE_BYTE_LOADERS_SRC0 fn load_u16_at_src0(byte_offset: u32) -> u32 { - let word = src0[byte_offset / 4u]; + let word = SRC0[byte_offset / 4u]; let shift = (byte_offset & 0x2u) * 8u; return (word >> shift) & 0xFFFFu; } @@ -55,14 +59,14 @@ fn load_u16_at_src0(byte_offset: u32) -> u32 { // Caller extracts the 16-bit half it needs via & 0xFFFFu or >> 16u. // this is used in k-quants for better performance fn load_u32_at_src0_aligned(byte_offset: u32) -> u32 { - return src0[(byte_offset & ~3u) / 4u]; + return SRC0[(byte_offset & ~3u) / 4u]; } fn load_u32_at_src0(byte_offset: u32) -> u32 { let word_idx = byte_offset / 4u; let shift = (byte_offset & 0x3u) * 8u; - let lo = src0[word_idx]; - let hi = src0[word_idx + 1u]; + let lo = SRC0[word_idx]; + let hi = SRC0[word_idx + 1u]; let shifted = (lo >> shift) | (hi << (32u - shift)); return select(shifted, lo, shift == 0u); } @@ -73,7 +77,7 @@ fn load_f16_at_src0(byte_offset: u32) -> f16 { } fn load_f16_as_f32_at_src0(byte_offset: u32) -> f32 { - let word = src0[byte_offset / 4u]; + let word = SRC0[byte_offset / 4u]; let shift = (byte_offset & 0x2u) * 8u; let d_bits = (word >> shift) & 0xFFFFu; return unpack2x16float(d_bits)[0]; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl index d5bf2af8..a7dee651 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl @@ -5,10 +5,9 @@ enable subgroups; enable chromium_experimental_subgroup_matrix; #define BYTE_HELPERS -#include "common_decls.tmpl" - #define FLASH_ATTN_SCALAR_KV #include "flash_attn_decls.tmpl" +#include "common_decls.tmpl" // Default values // The actual values are defined in shader-lib. diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl index 8cd18b92..7edca84f 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl @@ -2,8 +2,8 @@ enable f16; enable subgroups; #define BYTE_HELPERS -#include "common_decls.tmpl" #include "flash_attn_decls.tmpl" +#include "common_decls.tmpl" // Default values // The actual values are defined in shader-lib. diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl index 42f3b108..ae941245 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl @@ -3,9 +3,9 @@ enable f16; enable subgroups; #define BYTE_HELPERS -#include "common_decls.tmpl" #define FLASH_ATTN_VEC_SPLIT #include "flash_attn_decls.tmpl" +#include "common_decls.tmpl" // Default values // The actual values are defined in shader-lib. diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl index 7d7b3475..6f4b5a31 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl @@ -19,6 +19,16 @@ var src_state: array; @group(0) @binding(6) var dst: array; +#ifdef FUSED_CACHE +@group(0) @binding(7) +var dst_fuse: array; +#define DST_SNAP dst_fuse +#define PARAMS_BINDING 8 +#else +#define DST_SNAP dst +#define PARAMS_BINDING 7 +#endif + struct Params { h: u32, n_tokens: u32, @@ -41,9 +51,11 @@ struct Params { rq3: u32, K: u32, scale: f32, + dst_fuse_nb2: u32, + dst_fuse_off: u32, }; -@group(0) @binding(7) +@group(0) @binding(PARAMS_BINDING) var params: Params; var sh_k: array; @@ -66,7 +78,14 @@ fn main( // input state holds s0 only [S_v, S_v, H, n_seqs]: per-seq stride is H*D. let state_in_base = (seq_id * params.h + head_id) * state_size; let state_out_base = (seq_id * params.h + head_id) * state_size; + +#ifdef FUSED_CACHE + let state_size_per_snap = params.dst_fuse_nb2; + let snap_off = params.dst_fuse_off; +#else let state_size_per_snap = state_size * params.h * params.n_seqs; + let snap_off = params.s_off; +#endif var state: array; for (var i = 0u; i < S_V; i++) { @@ -131,9 +150,9 @@ fn main( // snapshot slot mapping: slot 0 = most recent state, slot s = s tokens back. let target_slot = i32(params.n_tokens) - 1 - i32(t); if (target_slot >= 0 && target_slot < i32(params.K)) { - let slot_base = params.s_off + u32(target_slot) * state_size_per_snap + state_out_base; + let slot_base = snap_off + u32(target_slot) * state_size_per_snap + state_out_base; for (var i = 0u; i < S_V; i++) { - dst[slot_base + col * S_V + i] = state[i]; + DST_SNAP[slot_base + col * S_V + i] = state[i]; } } } @@ -143,7 +162,7 @@ fn main( if (params.K == 1u) { for (var i = 0u; i < S_V; i++) { - dst[params.s_off + state_out_base + col * S_V + i] = state[i]; + DST_SNAP[snap_off + state_out_base + col * S_V + i] = state[i]; } } } diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl index d03f1c20..6bbed5d3 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl @@ -37,6 +37,14 @@ fn op(a: f32, b: f32) -> f32 { return out_glu; } #endif +#ifdef OP_SWIGLU_CLAMP +fn op(a: DataType, b: DataType) -> DataType { + let limit = DataType(params.limit); + let gate = min(a, limit); + let up = clamp(b, -limit, limit); + return gate / (1.0 + exp(-gate)) * up; +} +#endif #ifdef OP_GEGLU_ERF const p_erf: DataType = 0.3275911; const a1_erf: DataType = 0.254829592; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl index 13996ab5..44b6bb71 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl @@ -1,3 +1,10 @@ +#ifndef SRC0 +#define SRC0 src0 +#endif +#ifndef SRC1 +#define SRC1 src1 +#endif + #ifdef VEC #define VEC_SIZE 4 #define SHMEM_TYPE vec4 @@ -39,7 +46,7 @@ fn init_shmem_src0(thread_id: u32, batch_offset: u32, offset_m: u32, k_outer: u3 let src0_idx = batch_offset + global_m * params.stride_01 + global_k; let src0_val = select( // taking a slight performance hit to avoid oob SRC0_TYPE(0.0), - src0[src0_idx/VEC_SIZE], + SRC0[src0_idx/VEC_SIZE], global_m < params.m && global_k < params.k); store_shmem(SHMEM_TYPE(src0_val), elem_idx); } @@ -57,7 +64,7 @@ fn init_shmem_src1(thread_id: u32, batch_offset: u32, offset_n: u32, k_outer: u3 let src1_idx = batch_offset + global_n * params.stride_11 + global_k; let src1_val = select( SRC1_TYPE(0.0), - src1[src1_idx/VEC_SIZE], + SRC1[src1_idx/VEC_SIZE], global_n < params.n && global_k < params.k); store_shmem(SHMEM_TYPE(src1_val), TILE_SRC0_SHMEM + elem_idx); } diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl index 98bbdeb8..0e17fae1 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl @@ -1,8 +1,12 @@ enable f16; #define DECLARE_BYTE_LOADERS_SRC0 -#include "common_decls.tmpl" +#ifdef SRC_OVERLAP +#define SRC0 merged_src +#define SRC1 merged_src +#endif +#include "common_decls.tmpl" #include "mul_mat_decls.tmpl" #ifdef VEC @@ -36,11 +40,17 @@ struct MulMatParams { broadcast3: u32 }; +#ifdef SRC_OVERLAP +@group(0) @binding(0) var merged_src: array; +#define DST_BINDING 1 +#else @group(0) @binding(0) var src0: array; // M rows, K columns @group(0) @binding(1) var src1: array; // K rows, N columns (transposed) -@group(0) @binding(2) var dst: array; // M rows, N columns (transposed) +#define DST_BINDING 2 +#endif -@group(0) @binding(3) var params: MulMatParams; +@group(0) @binding(DST_BINDING) var dst: array; // M rows, N columns (transposed) +@group(0) @binding(DST_BINDING + 1) var params: MulMatParams; fn get_local_n(thread_id: u32) -> u32 { return thread_id / WORKGROUP_SIZE_M; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_subgroup_matrix.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_subgroup_matrix.wgsl index d86a72ce..35998a9b 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_subgroup_matrix.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_subgroup_matrix.wgsl @@ -4,6 +4,10 @@ enable subgroups; enable chromium_experimental_subgroup_matrix; #define DECLARE_BYTE_LOADERS_SRC0 +#ifdef SRC_OVERLAP +#define SRC0 merged_src +#define SRC1 merged_src +#endif #include "common_decls.tmpl" #include "mul_mat_decls.tmpl" @@ -48,11 +52,17 @@ struct MulMatParams { }; // SRC0_TYPE and SRC1_TYPE are defined in mul_mat_decls, which is included +#ifdef SRC_OVERLAP +@group(0) @binding(0) var merged_src: array; +#define DST_BINDING 1 +#else @group(0) @binding(0) var src0: array; // M rows, K columns @group(0) @binding(1) var src1: array; // K rows, N columns (transposed) -@group(0) @binding(2) var dst: array; // M rows, N columns (transposed) +#define DST_BINDING 2 +#endif -@group(0) @binding(3) var params: MulMatParams; +@group(0) @binding(DST_BINDING) var dst: array; // M rows, N columns (transposed) +@group(0) @binding(DST_BINDING + 1) var params: MulMatParams; const WG_M_SG_TILE_SIZE = SUBGROUP_M * SUBGROUP_MATRIX_M * SUBGROUP_MATRIX_M_SIZE; const WG_N_SG_TILE_SIZE = SUBGROUP_N * SUBGROUP_MATRIX_N * SUBGROUP_MATRIX_N_SIZE; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl index ebdf0951..1781a6c7 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl @@ -7,6 +7,11 @@ enable f16; requires packed_4x8_integer_dot_product; #endif +#ifdef SRC_OVERLAP +#define SRC0 merged_src +#define SRC1 merged_src +#endif + #define DECLARE_BYTE_LOADERS_SRC0 #include "common_decls.tmpl" @@ -35,17 +40,22 @@ struct MulMatParams { broadcast3: u32 }; +#if defined(MMVQ) @group(0) @binding(0) var src0: array; - -#ifdef MMVQ @group(0) @binding(1) var src1q: array; +#define DST_BINDING 2 +#elif defined(SRC_OVERLAP) +@group(0) @binding(0) var merged_src: array; +#define DST_BINDING 1 #else +@group(0) @binding(0) var src0: array; @group(0) @binding(1) var src1: array; +#define DST_BINDING 2 #endif -@group(0) @binding(2) var dst: array; +@group(0) @binding(DST_BINDING) var dst: array; // "mul_mat_vec_acc.tmpl" requires params.k, params.m, params.stride_01 -@group(0) @binding(3) var params: MulMatParams; +@group(0) @binding(DST_BINDING + 1) var params: MulMatParams; // Flattened as [row][thread] to keep each row's reduction contiguous in memory. var partial_sums: array; diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl index 8fd0d190..864b4bd2 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl @@ -1,3 +1,10 @@ +#ifndef SRC0 +#define SRC0 src0 +#endif +#ifndef SRC1 +#define SRC1 src1 +#endif + #ifdef U32_DEQUANT_HELPERS #define SRC0_TYPE u32 @@ -43,13 +50,13 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src for (var k = thread_id; k < k_vec; k += WG_SIZE) { var x_vals: array; for (var col = 0u;col < NUM_COLS;col += 1) { - x_vals[col] = src1[src1_idx_base_vec + col * (params.stride_11 / VEC_SIZE) + k]; + x_vals[col] = SRC1[src1_idx_base_vec + col * (params.stride_11 / VEC_SIZE) + k]; } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { let output_row = row_base + row; if (output_row < params.m) { let src0_idx = (src0_batch_offset + output_row * params.stride_01) / VEC_SIZE + k; - let w = src0[src0_idx]; + let w = SRC0[src0_idx]; for (var col = 0u;col < NUM_COLS;col += 1) { acc[col][row] += inner_dot(w, x_vals[col]); } @@ -76,7 +83,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -116,8 +123,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4] = f32(src1[x_base + col * params.stride_11 + i + 16]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4] = f32(SRC1[x_base + col * params.stride_11 + i + 16]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -160,8 +167,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4] = f32(src1[x_base + col * params.stride_11 + i + 16]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4] = f32(SRC1[x_base + col * params.stride_11 + i + 16]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -205,8 +212,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4] = f32(src1[x_base + col * params.stride_11 + i + 16]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4] = f32(SRC1[x_base + col * params.stride_11 + i + 16]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -253,8 +260,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4] = f32(src1[x_base + col * params.stride_11 + i + 16]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4] = f32(SRC1[x_base + col * params.stride_11 + i + 16]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -302,7 +309,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -347,7 +354,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -409,10 +416,10 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 4u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4u] = f32(src1[x_base + col * params.stride_11 + 32u + i]); - x_block[col][i + 8u] = f32(src1[x_base + col * params.stride_11 + 64u + i]); - x_block[col][i + 12u] = f32(src1[x_base + col * params.stride_11 + 96u + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4u] = f32(SRC1[x_base + col * params.stride_11 + 32u + i]); + x_block[col][i + 8u] = f32(SRC1[x_base + col * params.stride_11 + 64u + i]); + x_block[col][i + 12u] = f32(SRC1[x_base + col * params.stride_11 + 96u + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -518,8 +525,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 8u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 8u] = f32(src1[x_base + col * params.stride_11 + 32u + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 8u] = f32(SRC1[x_base + col * params.stride_11 + 32u + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -610,10 +617,10 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src for (var col = 0u; col < NUM_COLS;col += 1) { let col_base = x_base + col * params.stride_11; for (var i = 0u; i < 4u; i++) { - x_block[col][i] = f32(src1[col_base + i]); - x_block[col][i + 4u] = f32(src1[col_base + 32u + i]); - x_block[col][i + 8u] = f32(src1[col_base + 128u + i]); - x_block[col][i + 12u] = f32(src1[col_base + 160u + i]); + x_block[col][i] = f32(SRC1[col_base + i]); + x_block[col][i + 4u] = f32(SRC1[col_base + 32u + i]); + x_block[col][i + 8u] = f32(SRC1[col_base + 128u + i]); + x_block[col][i + 12u] = f32(SRC1[col_base + 160u + i]); } } @@ -713,10 +720,10 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src for (var col = 0u; col < NUM_COLS;col += 1) { let col_base = x_base + col * params.stride_11; for (var i = 0u; i < 4u; i++) { - x_block[col][i] = f32(src1[col_base + i]); - x_block[col][i + 4u] = f32(src1[col_base + 32u + i]); - x_block[col][i + 8u] = f32(src1[col_base + 128u + i]); - x_block[col][i + 12u] = f32(src1[col_base + 160u + i]); + x_block[col][i] = f32(SRC1[col_base + i]); + x_block[col][i + 4u] = f32(SRC1[col_base + 32u + i]); + x_block[col][i + 8u] = f32(SRC1[col_base + 128u + i]); + x_block[col][i + 12u] = f32(SRC1[col_base + 160u + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -823,10 +830,10 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src for (var col = 0u; col < NUM_COLS;col += 1) { let col_base = x_base + col * params.stride_11; for (var l = 0u; l < 4u; l++) { - x_block[col][l] = f32(src1[col_base + l]); - x_block[col][l + 4u] = f32(src1[col_base + 32u + l]); - x_block[col][l + 8u] = f32(src1[col_base + 64u + l]); - x_block[col][l + 12u] = f32(src1[col_base + 96u + l]); + x_block[col][l] = f32(SRC1[col_base + l]); + x_block[col][l + 4u] = f32(SRC1[col_base + 32u + l]); + x_block[col][l + 8u] = f32(SRC1[col_base + 64u + l]); + x_block[col][l + 12u] = f32(SRC1[col_base + 96u + l]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -899,7 +906,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -960,7 +967,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1039,7 +1046,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1101,7 +1108,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1168,7 +1175,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1234,7 +1241,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1302,7 +1309,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1367,8 +1374,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4u] = f32(src1[x_base + col * params.stride_11 + i + 16u]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4u] = f32(SRC1[x_base + col * params.stride_11 + i + 16u]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1418,7 +1425,7 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < 16u; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1476,8 +1483,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 4] = f32(src1[x_base + col * params.stride_11 + i + 16]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 4] = f32(SRC1[x_base + col * params.stride_11 + i + 16]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { @@ -1521,8 +1528,8 @@ fn accumulate_vec_dot(thread_id: u32, row_base: u32, src0_batch_offset: u32, src var x_block: array, NUM_COLS>; for (var col = 0u; col < NUM_COLS;col += 1) { for (var i = 0u; i < ELEMS_PER_THREAD / 2; i++) { - x_block[col][i] = f32(src1[x_base + col * params.stride_11 + i]); - x_block[col][i + 8] = f32(src1[x_base + col * params.stride_11 + i + 8]); + x_block[col][i] = f32(SRC1[x_base + col * params.stride_11 + i]); + x_block[col][i + 8] = f32(SRC1[x_base + col * params.stride_11 + i + 8]); } } for (var row = 0u; row < OUTPUTS_PER_WG; row++) { diff --git a/ggml/src/ggml-webgpu/wgsl-shaders/rope.wgsl b/ggml/src/ggml-webgpu/wgsl-shaders/rope.wgsl index 1c874e14..6ff53088 100644 --- a/ggml/src/ggml-webgpu/wgsl-shaders/rope.wgsl +++ b/ggml/src/ggml-webgpu/wgsl-shaders/rope.wgsl @@ -38,7 +38,8 @@ struct Params { sections0: u32, sections1: u32, sections2: u32, - sections3: u32 + sections3: u32, + n_offs: u32 }; @group(0) @binding(0) @@ -126,7 +127,8 @@ fn rope_yarn(theta_extrap: f32, i: u32) -> vec2 { fn pair_base(i0: u32, div_2: bool) -> u32 { if (div_2) { - return i0 / 2; + // first channel of the rotated pair: n_offs + (i0 - n_offs)/2 + return i0 / 2 + params.n_offs / 2; } else { return i0; } @@ -165,20 +167,22 @@ fn main(@builtin(global_invocation_id) gid: vec3) { let i_src_row = params.offset_src0 + i3 * params.stride_src03 + i2 * params.stride_src02 + i1 * params.stride_src01; let i_dst_row = params.offset_dst + i3 * params.stride_dst3 + i2 * params.stride_dst2 + i1 * params.stride_dst1; - if (i0 >= params.n_dims && !is_vision) { + if ((i0 < params.n_offs || i0 >= params.n_offs + params.n_dims) && !is_vision) { let i_src = i_src_row + i0; let i_dst = i_dst_row + i0; rotate(i_dst, i_dst + 1, f32(src0[i_src]), f32(src0[i_src + 1])); return; } + let iw = i0 - params.n_offs; // relative idx + var theta_base_mult: u32 = 0; - var theta_scale_pwr: u32 = i0 / 2; + var theta_scale_pwr: u32 = iw / 2; if (is_mrope) { let sect_dims = params.sections0 + params.sections1 + params.sections2 + params.sections3; let sec_w = params.sections1 + params.sections0; let sec_e = params.sections2 + sec_w; - let sector = (i0 / 2) % sect_dims; + let sector = (iw / 2) % sect_dims; if (is_imrope) { if (sector % 3 == 1 && sector < 3 * params.sections1) { theta_base_mult = 1; @@ -203,7 +207,7 @@ fn main(@builtin(global_invocation_id) gid: vec3) { } else if (sector >= sec_e) { if (is_vision) { theta_scale_pwr = sector - sec_e; - theta_scale_pwr = (i0 / 2) % sec_e; + theta_scale_pwr = (iw / 2) % sec_e; } theta_base_mult = 3; } else if (is_vision) { @@ -212,7 +216,7 @@ fn main(@builtin(global_invocation_id) gid: vec3) { } } let theta_base = f32(src1[params.offset_src1 + i2 + params.ne2 * theta_base_mult]) * pow(params.theta_scale, f32(theta_scale_pwr)); - let thetas = rope_yarn(theta_base/freq_factor(i0), i0); + let thetas = rope_yarn(theta_base/freq_factor(iw), iw); let i_src = i_src_row + pair_base(i0, is_neox || is_mrope || is_vision); let i_dst = i_dst_row + pair_base(i0, is_neox || is_mrope || is_vision); diff --git a/ggml/src/ggml-zendnn/CMakeLists.txt b/ggml/src/ggml-zendnn/CMakeLists.txt index 87d721f6..6e393d6b 100644 --- a/ggml/src/ggml-zendnn/CMakeLists.txt +++ b/ggml/src/ggml-zendnn/CMakeLists.txt @@ -86,6 +86,6 @@ endif() target_link_libraries(ggml-zendnn PRIVATE m pthread) -if (GGML_OPENMP) - target_link_libraries(ggml-zendnn PRIVATE OpenMP::OpenMP_CXX) +if (GGML_OPENMP_ENABLED) + target_link_libraries(ggml-zendnn PRIVATE ${GGML_OPENMP_TARGET_CXX}) endif() diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c index d0d369c4..a286083b 100644 --- a/ggml/src/ggml.c +++ b/ggml/src/ggml.c @@ -1,6 +1,7 @@ #define _CRT_SECURE_NO_DEPRECATE // Disables "unsafe" warnings on Windows #define _USE_MATH_DEFINES // For M_PI on MSVC +#include "ggml-version.h" #include "ggml-backend.h" #include "ggml-impl.h" #include "ggml-threading.h" @@ -1253,10 +1254,10 @@ static const char * GGML_GLU_OP_NAME[GGML_GLU_OP_COUNT] = { "SWIGLU_OAI", "GEGLU_ERF", "GEGLU_QUICK", + "SWIGLU_CLAMP", }; -static_assert(GGML_GLU_OP_COUNT == 6, "GGML_GLU_OP_COUNT != 6"); - +static_assert(GGML_GLU_OP_COUNT == 7, "GGML_GLU_OP_COUNT != 7"); static_assert(sizeof(struct ggml_object)%GGML_MEM_ALIGN == 0, "ggml_object size must be a multiple of GGML_MEM_ALIGN"); static_assert(sizeof(struct ggml_tensor)%GGML_MEM_ALIGN == 0, "ggml_tensor size must be a multiple of GGML_MEM_ALIGN"); @@ -3119,6 +3120,17 @@ struct ggml_tensor * ggml_swiglu_oai( return result; } +struct ggml_tensor * ggml_swiglu_clamp( + struct ggml_context * ctx, + struct ggml_tensor * a, + struct ggml_tensor * b, + float limit) { + struct ggml_tensor * result = ggml_glu_impl(ctx, a, b, GGML_GLU_OP_SWIGLU_CLAMP, false); + ggml_set_op_params_f32(result, 3, limit); + + return result; +} + // ggml_norm static struct ggml_tensor * ggml_norm_impl( @@ -3265,6 +3277,57 @@ struct ggml_tensor * ggml_l2_norm_inplace( return ggml_l2_norm_impl(ctx, a, eps, true); } +// ggml_prec + +bool ggml_prec_set_acc( + struct ggml_tensor * a, + enum ggml_prec prec) { + switch (a->op) { + case GGML_OP_MUL_MAT: + case GGML_OP_MUL_MAT_ID: + { + const int32_t prec_i32 = (int32_t) prec; + ggml_set_op_params_i32(a, 0, prec_i32); + } + break; + case GGML_OP_FLASH_ATTN_EXT: + { + const int32_t prec_i32 = (int32_t) prec; + ggml_set_op_params_i32(a, 3, prec_i32); + } + break; + default: + return false; + }; + + return true; +} + +bool ggml_prec_set_src( + struct ggml_tensor * a, + enum ggml_prec prec, + int idx) { + GGML_ASSERT(idx >= 0 && idx < GGML_MAX_SRC); + + switch (a->op) { + case GGML_OP_MUL_MAT: + case GGML_OP_MUL_MAT_ID: + { + if (idx != 1) { + return false; + } + + const int32_t prec_i32 = (int32_t) prec; + ggml_set_op_params_i32(a, 2 + idx, prec_i32); + } + break; + default: + return false; + }; + + return true; +} + // ggml_mul_mat static inline bool ggml_can_mul_mat(const struct ggml_tensor * t0, const struct ggml_tensor * t1) { @@ -3834,8 +3897,8 @@ struct ggml_tensor * ggml_permute( struct ggml_tensor * result = ggml_view_tensor(ctx, a); ggml_format_name(result, "%s (permuted)", a->name); - int ne[GGML_MAX_DIMS]; - int nb[GGML_MAX_DIMS]; + int64_t ne[GGML_MAX_DIMS]; + size_t nb[GGML_MAX_DIMS]; ne[axis0] = a->ne[0]; ne[axis1] = a->ne[1]; @@ -4042,6 +4105,41 @@ struct ggml_tensor * ggml_diag_mask_zero_inplace( return ggml_diag_mask_zero_impl(ctx, a, n_past, true); } +// ggml_clamp + +static struct ggml_tensor * ggml_clamp_impl( + struct ggml_context * ctx, + struct ggml_tensor * a, + float min, + float max, + bool inplace) { + struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a); + + float params[] = { min, max }; + ggml_set_op_params(result, params, sizeof(params)); + + result->op = GGML_OP_CLAMP; + result->src[0] = a; + + return result; +} + +struct ggml_tensor * ggml_clamp( + struct ggml_context * ctx, + struct ggml_tensor * a, + float min, + float max) { + return ggml_clamp_impl(ctx, a, min, max, false); +} + +struct ggml_tensor * ggml_clamp_inplace( + struct ggml_context * ctx, + struct ggml_tensor * a, + float min, + float max) { + return ggml_clamp_impl(ctx, a, min, max, true); +} + // ggml_soft_max static struct ggml_tensor * ggml_soft_max_impl( @@ -4200,7 +4298,7 @@ static struct ggml_tensor * ggml_rope_impl( struct ggml_tensor * result = inplace ? ggml_view_tensor(ctx, a) : ggml_dup_tensor(ctx, a); - int32_t params[15] = { /*n_past*/ 0, n_dims, mode, /*n_ctx*/ 0, n_ctx_orig }; + int32_t params[16] = { /*n_past*/ 0, n_dims, mode, /*n_ctx*/ 0, n_ctx_orig }; memcpy(params + 5, &freq_base, sizeof(float)); memcpy(params + 6, &freq_scale, sizeof(float)); memcpy(params + 7, &ext_factor, sizeof(float)); @@ -4212,6 +4310,8 @@ static struct ggml_tensor * ggml_rope_impl( } else { memset(params + 11, 0, sizeof(int32_t) * GGML_MROPE_SECTIONS); } + params[15] = 0; // n_offs, set via ggml_rope_set_offset() + ggml_set_op_params(result, params, sizeof(params)); result->op = GGML_OP_ROPE; @@ -4422,23 +4522,18 @@ struct ggml_tensor * ggml_rope_multi_back( result->op = GGML_OP_ROPE_BACK; return result; } -// ggml_clamp -struct ggml_tensor * ggml_clamp( - struct ggml_context * ctx, +struct ggml_tensor * ggml_rope_set_offset( struct ggml_tensor * a, - float min, - float max) { - // TODO: when implement backward, fix this: - struct ggml_tensor * result = ggml_view_tensor(ctx, a); - - float params[] = { min, max }; - ggml_set_op_params(result, params, sizeof(params)); + int n_offs) { + GGML_ASSERT(a->op == GGML_OP_ROPE || a->op == GGML_OP_ROPE_BACK); + GGML_ASSERT(n_offs >= 0); - result->op = GGML_OP_CLAMP; - result->src[0] = a; + const int32_t mode = ggml_get_op_params_i32(a, 2); + GGML_ASSERT(mode != GGML_ROPE_TYPE_VISION); - return result; + ggml_set_op_params_i32(a, 15, n_offs); + return a; } static int64_t ggml_calc_conv_output_size(int64_t ins, int64_t ks, int s, int p, int d) { @@ -5463,6 +5558,15 @@ enum ggml_prec ggml_flash_attn_ext_get_prec( return (enum ggml_prec) prec_i32; } +void ggml_flash_attn_ext_set_n_kv_max( + struct ggml_tensor * a, + int32_t n_kv_max) { + GGML_ASSERT(a->op == GGML_OP_FLASH_ATTN_EXT); + GGML_ASSERT(n_kv_max >= 0); + + ggml_set_op_params_i32(a, 4, n_kv_max); +} + void ggml_flash_attn_ext_add_sinks( struct ggml_tensor * a, struct ggml_tensor * sinks) { @@ -6403,10 +6507,12 @@ struct ggml_tensor * ggml_dsv4_hc_comb( // ggml_dsv4_hc_pre -struct ggml_tensor * ggml_dsv4_hc_pre( +static struct ggml_tensor * ggml_dsv4_hc_pre_impl( struct ggml_context * ctx, struct ggml_tensor * x, - struct ggml_tensor * weights) { + struct ggml_tensor * weights, + float scale, + bool gated) { GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(weights->type == GGML_TYPE_F32); @@ -6416,13 +6522,22 @@ struct ggml_tensor * ggml_dsv4_hc_pre( GGML_ASSERT(hc > 0); GGML_ASSERT(x->ne[3] == 1); - GGML_ASSERT(weights->ne[0] == hc); - GGML_ASSERT(weights->ne[1] == n_tokens); - GGML_ASSERT(weights->ne[2] == 1); + if (gated) { + GGML_ASSERT(weights->ne[0] == n_embd); + GGML_ASSERT(weights->ne[1] == hc); + GGML_ASSERT(weights->ne[2] == n_tokens); + } else { + GGML_ASSERT(weights->ne[0] == hc); + GGML_ASSERT(weights->ne[1] == n_tokens); + GGML_ASSERT(weights->ne[2] == 1); + } GGML_ASSERT(weights->ne[3] == 1); struct ggml_tensor * result = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_embd, n_tokens); + ggml_set_op_params_f32(result, 0, scale); + ggml_set_op_params_i32(result, 1, gated ? 1 : 0); + result->op = GGML_OP_DSV4_HC_PRE; result->src[0] = x; result->src[1] = weights; @@ -6430,6 +6545,21 @@ struct ggml_tensor * ggml_dsv4_hc_pre( return result; } +struct ggml_tensor * ggml_dsv4_hc_pre( + struct ggml_context * ctx, + struct ggml_tensor * x, + struct ggml_tensor * weights) { + return ggml_dsv4_hc_pre_impl(ctx, x, weights, 1.0f, false); +} + +struct ggml_tensor * ggml_dsv4_hc_pre_gated( + struct ggml_context * ctx, + struct ggml_tensor * x, + struct ggml_tensor * gate, + float scale) { + return ggml_dsv4_hc_pre_impl(ctx, x, gate, scale, true); +} + // ggml_dsv4_hc_post struct ggml_tensor * ggml_dsv4_hc_post( @@ -6441,7 +6571,6 @@ struct ggml_tensor * ggml_dsv4_hc_post( GGML_ASSERT(x->type == GGML_TYPE_F32); GGML_ASSERT(residual->type == GGML_TYPE_F32); GGML_ASSERT(post->type == GGML_TYPE_F32); - GGML_ASSERT(comb->type == GGML_TYPE_F32); const int64_t n_embd = x->ne[0]; const int64_t n_tokens = x->ne[1]; @@ -6460,10 +6589,13 @@ struct ggml_tensor * ggml_dsv4_hc_post( GGML_ASSERT(post->ne[2] == 1); GGML_ASSERT(post->ne[3] == 1); - GGML_ASSERT(comb->ne[0] == hc); - GGML_ASSERT(comb->ne[1] == hc); - GGML_ASSERT(comb->ne[2] == n_tokens); - GGML_ASSERT(comb->ne[3] == 1); + if (comb) { + GGML_ASSERT(comb->type == GGML_TYPE_F32); + GGML_ASSERT(comb->ne[0] == hc); + GGML_ASSERT(comb->ne[1] == hc); + GGML_ASSERT(comb->ne[2] == n_tokens); + GGML_ASSERT(comb->ne[3] == 1); + } struct ggml_tensor * result = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_embd, hc, n_tokens); @@ -7283,7 +7415,7 @@ void ggml_build_backward_expand( } // inplace operations are currently not supported - GGML_ASSERT(!node->view_src || node->op == GGML_OP_CPY || node->op == GGML_OP_VIEW || + GGML_ASSERT(!node->view_src || node->op == GGML_OP_CPY || node->op == GGML_OP_SET_ROWS || node->op == GGML_OP_VIEW || node->op == GGML_OP_RESHAPE || node->op == GGML_OP_PERMUTE || node->op == GGML_OP_TRANSPOSE); const size_t ihash = ggml_hash_find(&cgraph->visited_hash_set, node); diff --git a/ggml/src/gguf.cpp b/ggml/src/gguf.cpp index 6c7b5817..0eb9fb74 100644 --- a/ggml/src/gguf.cpp +++ b/ggml/src/gguf.cpp @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -237,6 +238,7 @@ struct gguf_reader { : callback(callback), userdata(userdata), max_chunk_read(max_chunk_read), + start_offset(data_offset), data_offset(data_offset), nbytes_remain(nbytes_remain) { GGML_ASSERT(max_chunk_read > 0); @@ -365,6 +367,11 @@ struct gguf_reader { return data_offset; } + // position in the file where the GGUF data starts, alignment is relative to it, not to the file + uint64_t start() const { + return start_offset; + } + bool seek(uint64_t absolute_offset) const { const uint64_t end_offset = uint64_t(data_offset) + nbytes_remain; if (absolute_offset > end_offset) { @@ -414,6 +421,7 @@ struct gguf_reader { gguf_reader_callback_t callback = nullptr; void * userdata = nullptr; size_t max_chunk_read = 0; + uint64_t start_offset = 0; mutable uint64_t data_offset = 0; mutable uint64_t nbytes_remain = 0; }; @@ -762,7 +770,7 @@ static struct gguf_context * gguf_init_from_reader(const struct gguf_reader & gr GGML_ASSERT(int64_t(ctx->info.size()) == n_tensors); // we require the data section to be aligned, so take into account any padding - if (n_tensors > 0 && !gr.seek(GGML_PAD(gr.tell(), ctx->alignment))) { + if (n_tensors > 0 && !gr.seek(gr.start() + GGML_PAD(gr.tell() - gr.start(), ctx->alignment))) { GGML_LOG_ERROR("%s: failed to seek to beginning of data section\n", __func__); gguf_free(ctx); return nullptr; diff --git a/ggml/tests/test-backend-ops.cpp b/ggml/tests/test-backend-ops.cpp index 3349a64b..8402fcfe 100644 --- a/ggml/tests/test-backend-ops.cpp +++ b/ggml/tests/test-backend-ops.cpp @@ -56,7 +56,7 @@ static void init_tensor_uniform(ggml_tensor * tensor, float min = -1.0f, float m std::vector data(nels); { // parallel initialization - static const size_t n_threads = N_THREADS; + static const size_t n_threads = std::max(1, std::min(nels/1024, std::min(4, N_THREADS/2))); auto init_thread = [&](size_t start, size_t end) { thread_local std::default_random_engine gen(std::random_device{}()); @@ -189,6 +189,33 @@ static void init_tensor_kq_mask(ggml_tensor * tensor, float min = -1.0f, float m ggml_backend_tensor_set(tensor, data_f16.data(), 0, data_f16.size()*sizeof(ggml_fp16_t)); } +static void init_tensor_kq_mask_sparse(ggml_tensor * tensor, int64_t n_kv_max) { + GGML_ASSERT(tensor->type == GGML_TYPE_F16); + GGML_ASSERT(n_kv_max > 0 && n_kv_max <= tensor->ne[0]); + + const int64_t ne0 = tensor->ne[0]; + const int64_t nrows = ggml_nrows(tensor); + std::vector data_f32(ggml_nelements(tensor), -INFINITY); + std::vector data_f16(ggml_nelements(tensor)); + std::vector order(ne0); + for (int64_t i = 0; i < ne0; ++i) { + order[i] = i; + } + + std::mt19937 gen(0x5A17); + for (int64_t row = 0; row < nrows; ++row) { + std::shuffle(order.begin(), order.end(), gen); + const int64_t count = n_kv_max - row % std::min(n_kv_max, 17); + std::sort(order.begin(), order.begin() + count); + for (int64_t i = 0; i < count; ++i) { + data_f32[row*ne0 + order[i]] = -0.03125f * (1 + (i + row) % 7); + } + } + + ggml_fp32_to_fp16_row(data_f32.data(), data_f16.data(), data_f16.size()); + ggml_backend_tensor_set(tensor, data_f16.data(), 0, data_f16.size()*sizeof(ggml_fp16_t)); +} + // generate a lower triangular matrix static void init_tensor_tril(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) { GGML_ASSERT(tensor->type == GGML_TYPE_F32); @@ -433,18 +460,11 @@ static std::string var_to_str(ggml_scale_mode mode) { #define VARS_TO_STR14(a, b, c, d, e, f, g, h, i, j, k, l, m, n) VAR_TO_STR(a) + "," + VARS_TO_STR13(b, c, d, e, f, g, h, i, j, k, l, m, n) #define VARS_TO_STR15(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o) VAR_TO_STR(a) + "," + VARS_TO_STR14(b, c, d, e, f, g, h, i, j, k, l, m, n, o) #define VARS_TO_STR16(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o, p) VAR_TO_STR(a) + "," + VARS_TO_STR15(b, c, d, e, f, g, h, i, j, k, l, m, n, o, p) - -#ifdef GGML_USE_SYCL -static bool inline _isinf(float f) { - return (*(uint32_t *)&f & 0x7fffffff) == 0x7f800000; -} -#else -static bool inline _isinf(float f) { return std::isinf(f); } -#endif +#define VARS_TO_STR17(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o, p, q) VAR_TO_STR(a) + "," + VARS_TO_STR16(b, c, d, e, f, g, h, i, j, k, l, m, n, o, p, q) // accept FLT_MAX as infinity static bool isinf_or_max(float f) { - return _isinf(f) || f == FLT_MAX || f == -FLT_MAX; + return std::isinf(f) || f == FLT_MAX || f == -FLT_MAX; } static bool ggml_is_view_op(enum ggml_op op) { @@ -1125,6 +1145,47 @@ static void print_test_result_locked(printer * output_printer, const test_result output_printer->print_test_result(result); } +// Splits the -o filter into comma separated entries. Commas inside parentheses +// (i.e. inside a full test case string) are not treated as separators. +static std::vector op_filter_entries(const char * op_names_filter) { + std::vector entries; + if (op_names_filter == nullptr) { + return entries; + } + std::string_view filter(op_names_filter); + while (!filter.empty()) { + auto comma_pos = filter.find_first_of(','); + const auto lparen_pos = filter.find_first_of('('); + if (lparen_pos < comma_pos) { + const auto rparen_pos = filter.find_first_of(')'); + comma_pos = filter.find_first_of(',', rparen_pos); + } + entries.push_back(filter.substr(0, comma_pos)); + filter = comma_pos != std::string_view::npos ? filter.substr(comma_pos + 1) : ""; + } + return entries; +} + +// An entry from the -o filter matches an op if it is either +// * an exact op name as given by ggml_op_desc() (e.g. "ADD"), or +// * a regex that matches the op name (e.g. "DSV4.*") +static bool op_filter_entry_matches(std::string_view entry, std::string_view op_name) { + if (entry == op_name) { + return true; + } + // plain op names are matched exactly, anything else is treated as a regex + if (std::regex_match(std::string(entry), std::regex("[A-Z0-9_]+"))) { + return false; + } + std::regex re; + try { + re = std::regex(std::string(entry)); + } catch (const std::regex_error &) { + return false; + } + return std::regex_search(op_name.data(), op_name.data() + op_name.size(), re); +} + struct test_case { virtual ~test_case() {} @@ -1199,6 +1260,11 @@ struct test_case { } } + // re-draw data-dependent inputs between timed perf iterations + virtual void reinit_perf_iter(ggml_context * ctx) { + GGML_UNUSED(ctx); + } + virtual size_t op_size(ggml_tensor * t) { size_t size = ggml_nbytes(t); // add source tensors @@ -1283,34 +1349,24 @@ struct test_case { return t; } - // Checks an op against the test filter, which is a comma separated list of OP names or specific variations + // Checks an op against the test filter, which is a comma separated list of OP names, regexes, or specific variations bool matches_filter(ggml_tensor * op, const char * op_names_filter) { - if (op_names_filter) { - const auto op_name = op_desc(op); - const auto op_full_name = op_name + "(" + vars() + ")"; - std::string_view filter(op_names_filter); - while (!filter.empty()) { - auto comma_pos = filter.find_first_of(','); - const auto lparen_pos = filter.find_first_of('('); - if (lparen_pos < comma_pos) { - auto rparen_pos = filter.find_first_of(')'); - comma_pos = filter.find_first_of(',', rparen_pos); - const auto op_filter = filter.substr(0, comma_pos); - if (op_filter == op_full_name) { - return true; - } - } else { - const auto op_filter = filter.substr(0, comma_pos); - if (op_filter == op_name) { - return true; - } + if (op_names_filter == nullptr) { + return true; + } + const auto op_name = op_desc(op); + const auto op_full_name = op_name + "(" + vars() + ")"; + for (const auto & entry : op_filter_entries(op_names_filter)) { + if (entry.find_first_of('(') != std::string_view::npos) { + // a full test case string, matched exactly + if (entry == op_full_name) { + return true; } - filter = comma_pos != std::string_view::npos ? filter.substr(comma_pos + 1) : ""; + } else if (op_filter_entry_matches(entry, op_name)) { + return true; } - return false; - } else { - return true; } + return false; } test_status_t eval(ggml_backend_t backend1, @@ -1633,6 +1689,9 @@ struct test_case { total_time_us += end_time - start_time; total_mem += mem; total_runs += n_runs; + + // re-draw any data-dependent inputs (expert ids) outside the timed region + reinit_perf_iter(ctx.get()); } while (total_time_us < 1000*1000); // run for at least 1 second // Create test result @@ -2243,6 +2302,63 @@ struct test_swiglu_oai : public test_case { } }; +struct test_swiglu_clamp : public test_case { + const ggml_type type; + const std::array ne_a; + int v; // view (1 : non-contiguous a) + float limit; + + std::string vars() override { + return VARS_TO_STR4(type, ne_a, v, limit); + } + + test_swiglu_clamp(ggml_type type = GGML_TYPE_F32, + std::array ne_a = {128, 2, 2, 2}, + int v = 0, + float limit = 7.0f) + : type(type), ne_a(ne_a), v(v), limit(limit) {} + + ggml_tensor * build_graph(ggml_context * ctx) override { + ggml_tensor * a; + ggml_tensor * b; + if (v & 1) { + auto ne = ne_a; ne[0] *= 3; + a = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_set_param(a); + ggml_set_name(a, "a"); + + a = ggml_view_4d(ctx, a, ne_a[0], ne_a[1], ne_a[2], ne_a[3], a->nb[1], a->nb[2], a->nb[3], 0); + ggml_set_name(a, "view_of_a"); + + b = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_set_param(b); + ggml_set_name(b, "b"); + + b = ggml_view_4d(ctx, b, ne_a[0], ne_a[1], ne_a[2], ne_a[3], b->nb[1], b->nb[2], b->nb[3], 0); + ggml_set_name(b, "view_of_b"); + } else { + a = ggml_new_tensor(ctx, type, 4, ne_a.data()); + ggml_set_param(a); + ggml_set_name(a, "a"); + + b = ggml_new_tensor(ctx, type, 4, ne_a.data()); + ggml_set_param(b); + ggml_set_name(b, "b"); + } + + ggml_tensor * out = ggml_swiglu_clamp(ctx, a, b, limit); + ggml_set_name(out, "out"); + + return out; + } + + void initialize_tensors(ggml_context * ctx) override { + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { + init_tensor_uniform(t, -150.f, 150.f); + } + } +}; + // GGML_OP_GET_ROWS struct test_get_rows : public test_case { const ggml_type type; @@ -2251,27 +2367,42 @@ struct test_get_rows : public test_case { const int r; // rows to get const int be1; // batch size const int be2; // batch size - const bool v; // view (non-contiguous src1) + const bool v; // view src1 + const bool vs0; // view src0 + const int offset_cols; // // column offset of the view src0 std::string vars() override { - return VARS_TO_STR7(type, n, m, r, be1, be2, v); + return VARS_TO_STR9(type, n, m, r, be1, be2, v, vs0, offset_cols); } - test_get_rows(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int be1 = 1, int be2 = 1, bool v = false) - : type(type), n(n), m(m), r(r), be1(be1), be2(be2), v(v) {} + test_get_rows(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int be1 = 1, int be2 = 1, bool v = false, bool vs0 = false, int offset_cols = 0) + : type(type), n(n), m(m), r(r), be1(be1), be2(be2), v(v), vs0(vs0), offset_cols(offset_cols) {} ggml_tensor * build_graph(ggml_context * ctx) override { - ggml_tensor * in = ggml_new_tensor_4d(ctx, type, n, m, be1, be2); - ggml_set_name(in, "in"); + ggml_tensor * in; + if (vs0) { + const int offset_rows = 3; + const int padded_m = m + offset_rows; + const int padded_n = n + offset_cols; + ggml_tensor * in_padded = ggml_new_tensor_4d(ctx, type, padded_n, padded_m, be1, be2); + ggml_set_name(in_padded, "in_padded"); + in = ggml_view_4d(ctx, in_padded, n, m, be1, be2, + in_padded->nb[1], in_padded->nb[2], in_padded->nb[3], + offset_cols * in_padded->nb[0] + offset_rows * in_padded->nb[1]); + ggml_set_name(in, "in_view"); + } else { + in = ggml_new_tensor_4d(ctx, type, n, m, be1, be2); + ggml_set_name(in, "in"); + } - ggml_tensor * rows = ggml_new_tensor_3d(ctx, GGML_TYPE_I32, r, be1, be2); + ggml_tensor * rows = ggml_new_tensor_3d(ctx, GGML_TYPE_I32, v ? r + 1 : r, be1, be2); ggml_set_name(rows, "rows"); if (v) { - rows = ggml_view_3d(ctx, rows, r/2, be1, be2, rows->nb[1], rows->nb[2], 0); + rows = ggml_view_3d(ctx, rows, r/2, be1, be2, rows->nb[1], rows->nb[2], rows->nb[0]); ggml_set_name(rows, "view_of_rows"); } - const bool grad_supported = ggml_is_matrix(in) && ggml_is_vector(rows); + const bool grad_supported = !vs0 && ggml_is_matrix(in) && ggml_is_vector(rows); if (grad_supported) { ggml_set_param(in); // rows is a constant input -> no gradients @@ -2285,14 +2416,16 @@ struct test_get_rows : public test_case { void initialize_tensors(ggml_context * ctx) override { for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { + if (ggml_is_view_op(t->op)) { + continue; + } if (t->type == GGML_TYPE_I32) { - if (ggml_is_view_op(t->op)) { continue; } // rows - std::vector data(r*be1*be2); - for (int i = 0; i < r*be1*be2; i++) { + std::vector data(ggml_nelements(t)); + for (size_t i = 0; i < data.size(); i++) { data[i] = rand() % m; } - ggml_backend_tensor_set(t, data.data(), 0, r * be1 * be2 * sizeof(int)); + ggml_backend_tensor_set(t, data.data(), 0, data.size() * sizeof(int)); } else { init_tensor_uniform(t); } @@ -2469,8 +2602,13 @@ struct test_set_rows : public test_case { // See dicussion here: https://github.com/ggml-org/llama.cpp/pull/23760#issuecomment-4566312209 double max_nmse_err(ggml_backend_t backend) override { ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(ggml_backend_get_device(backend)); - if (type_dst == GGML_TYPE_Q8_0 && strcmp(ggml_backend_reg_name(reg), "WebGPU") == 0) { - return std::max(test_case::max_nmse_err(backend), 2e-7); + if (type_dst == GGML_TYPE_Q8_0) { + if (strcmp(ggml_backend_reg_name(reg), "WebGPU") == 0) { + return std::max(test_case::max_nmse_err(backend), 2e-7); + } + if (strcmp(ggml_backend_reg_name(reg), "HTP") == 0) { + return std::max(test_case::max_nmse_err(backend), 5e-6); + } } return test_case::max_nmse_err(backend); } @@ -2578,13 +2716,16 @@ struct test_rope_set_rows : public test_case { } }; -// GGML_OP_RMS_NORM + GGML_OP_MUL + GGML_OP_ROPE (+ GGML_OP_VIEW + GGML_OP_SET_ROWS) +// GGML_OP_RMS_NORM with optional GGML_OP_MUL, GGML_OP_ROPE, GGML_OP_VIEW and GGML_OP_SET_ROWS struct test_rms_norm_mul_rope : public test_case { const std::array ne; const float eps; const bool multi_add; // test a sequence of adds feeding into rms_norm + const bool mul; + const bool rope; const bool set_rows; const bool broadcast; // multiply by a 1D [ne0] weight, as model norm weights are + const ggml_type set_rows_type; int mode; std::string op_desc(ggml_tensor * t) override { @@ -2595,63 +2736,90 @@ struct test_rms_norm_mul_rope : public test_case { bool run_whole_graph() override { return true; } std::string vars() override { - return VARS_TO_STR6(ne, eps, multi_add, set_rows, broadcast, mode); + return VARS_TO_STR9(ne, eps, multi_add, mul, rope, set_rows, broadcast, mode, set_rows_type); } test_rms_norm_mul_rope(std::array ne, float eps = 1e-6f, bool multi_add = false, - bool set_rows = false, bool broadcast = false, int mode = GGML_ROPE_TYPE_NORMAL) - : ne(ne), eps(eps), multi_add(multi_add), set_rows(set_rows), broadcast(broadcast), mode(mode) {} + bool set_rows = false, bool broadcast = false, int mode = GGML_ROPE_TYPE_NORMAL, + bool mul = true, bool rope = true, ggml_type set_rows_type = GGML_TYPE_F16) + : ne(ne), eps(eps), multi_add(multi_add), mul(mul), rope(rope), set_rows(set_rows), broadcast(broadcast), + set_rows_type(set_rows_type), mode(mode) {} ggml_tensor * build_graph(ggml_context * ctx) override { - ggml_tensor * a = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], 1); - ggml_tensor * b = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], 1); - ggml_tensor * c = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], 1); + ggml_tensor * a = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], ne[3]); + ggml_tensor * b = nullptr; + ggml_tensor * c = nullptr; + ggml_tensor * w = nullptr; + + if (multi_add || (mul && !broadcast)) { + b = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], 1); + } if (multi_add) { - a = ggml_add(ctx, ggml_add(ctx, a, b), c); + c = ggml_new_tensor_4d(ctx, GGML_TYPE_F32, ne[0], ne[1], ne[2], 1); + } + if (mul) { + w = broadcast ? ggml_new_tensor_1d(ctx, GGML_TYPE_F32, ne[0]) : b; } - ggml_tensor * w = broadcast ? ggml_new_tensor_1d(ctx, GGML_TYPE_F32, ne[0]) : b; + if (multi_add) { + a = ggml_add(ctx, ggml_add(ctx, a, b), c); + } - a = ggml_mul(ctx, ggml_rms_norm(ctx, a, eps), w); + a = ggml_rms_norm(ctx, a, eps); - ggml_tensor * pos = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, ne[2]); + if (mul) { + a = ggml_mul(ctx, a, w); + } - ggml_tensor * rope = ggml_rope(ctx, a, pos, ne[0], mode); + if (rope) { + const bool is_mrope = mode & GGML_ROPE_TYPE_MROPE; + ggml_tensor * pos = ggml_new_tensor_1d(ctx, GGML_TYPE_I32, ne[2] * (is_mrope ? 4 : 1)); - ggml_tensor * out; + if (is_mrope) { + const int n_dims = ne[0]; + int sections[4] = { n_dims/3, n_dims/3, n_dims/3, 0 }; + a = ggml_rope_multi(ctx, a, pos, nullptr, n_dims, sections, mode, 0, 10000.0f, 1.0f, 0.0f, 1.0f, 32.0f, 1.0f); + } else { + a = ggml_rope(ctx, a, pos, ne[0], mode); + } + } if (set_rows) { - ggml_tensor * view = ggml_view_2d(ctx, rope, ne[0] * ne[1], ne[2], rope->nb[2], 0); + ggml_tensor * view = ggml_view_2d(ctx, a, ne[0] * ne[1], ne[2], a->nb[2], 0); - ggml_tensor * dst = ggml_new_tensor_4d(ctx, GGML_TYPE_F16, ne[0] * ne[1], ne[2] * ne[3], 1, 1); + ggml_tensor * dst = ggml_new_tensor_2d(ctx, set_rows_type, ne[0] * ne[1], ne[2] * 2); ggml_set_name(dst, "dst"); - ggml_tensor * row_idxs = ggml_new_tensor_3d(ctx, GGML_TYPE_I64, ne[2], 1, 1); + ggml_tensor * row_idxs = ggml_new_tensor_1d(ctx, GGML_TYPE_I64, ne[2]); ggml_set_name(row_idxs, "row_idxs"); - out = ggml_set_rows(ctx, dst, view, row_idxs); - ggml_set_name(out, "out"); - } else { - out = rope; + a = ggml_set_rows(ctx, dst, view, row_idxs); } - return out; + ggml_set_name(a, "out"); + return a; } void initialize_tensors(ggml_context * ctx) override { for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { - if (t->type == GGML_TYPE_I64 || t->type == GGML_TYPE_I32) { - if (ggml_is_view_op(t->op)) { - continue; + if (t->type == GGML_TYPE_I64) { + init_set_rows_row_ids(t, ne[2] * 2); + } else if (t->type == GGML_TYPE_I32) { + std::vector data(ggml_nelements(t)); + for (int32_t & value : data) { + value = rand() % 512; } - - init_set_rows_row_ids(t, ne[2]); + ggml_backend_tensor_set(t, data.data(), 0, ggml_nbytes(t)); } else { init_tensor_uniform(t); } } } + + double max_nmse_err() override { + return ne[0] == 8192 ? 5e-6 : test_case::max_nmse_err(); + } }; // GGML_OP_ARGMAX @@ -3061,28 +3229,36 @@ struct test_cpy : public test_case { }; // GGML_OP_CONT +// permute = {0, 0, 0, 0} means no permutation: the source is transposed (or +// view-sliced). A non-identity permute applies ggml_permute before ggml_cont. struct test_cont : public test_case { const ggml_type type; const std::array ne; bool use_view_slice; + const std::array permute; std::string vars() override { - return VARS_TO_STR3(type, ne, use_view_slice); + return VARS_TO_STR4(type, ne, use_view_slice, permute); } test_cont(ggml_type type = GGML_TYPE_F32, std::array ne = {10, 10, 10, 1}, - bool use_view_slice = false) - : type(type), ne(ne), use_view_slice(use_view_slice) {} + bool use_view_slice = false, + std::array permute = {0, 0, 0, 0}) + : type(type), ne(ne), use_view_slice(use_view_slice), permute(permute) {} ggml_tensor * build_graph(ggml_context * ctx) override { ggml_tensor * src = ggml_new_tensor(ctx, type, 4, ne.data()); ggml_set_param(src); ggml_set_name(src, "src"); + const bool permuted = permute[0] != 0 || permute[1] != 0 || permute[2] != 0 || permute[3] != 0; ggml_tensor * dst; - if (use_view_slice) { + if (permuted) { + dst = ggml_permute(ctx, src, permute[0], permute[1], permute[2], permute[3]); + ggml_set_name(dst, "src_permuted"); + } else if (use_view_slice) { dst = ggml_view_4d(ctx, src, src->ne[0], 1, src->ne[2], src->ne[3], src->nb[1], src->nb[2], src->nb[3], src->nb[0] * (src->ne[1] - 1)); ggml_set_name(dst, "src_view_slice"); @@ -3193,6 +3369,14 @@ struct test_bin_bcast : public test_case { return op == ggml_div; } + double max_nmse_err() override { + if (op == ggml_add && type == GGML_TYPE_F16 && nf > 1) { + // Fused ADDs can keep FP32 intermediates while the CPU rounds each ADD to FP16. + return 1e-6; + } + return test_case::max_nmse_err(); + } + double max_maa_err() override { return op == ggml_add ? 1e-4 : 1e-3; } @@ -3448,6 +3632,44 @@ struct test_norm_mul_add : public test_case { return out; } }; +// GGML_OP_NORM/RMS_NORM + GGML_OP_SCALE +struct test_norm_scale : public test_case { + const ggml_type type; + const std::array ne; + const float eps; + const bool rms; + const float scale; + + std::string vars() override { + return VARS_TO_STR5(type, ne, eps, rms, scale); + } + + test_norm_scale(ggml_type type = GGML_TYPE_F32, + std::array ne = {64, 5, 4, 3}, + float eps = 1e-6f, + bool rms = false, + float scale = 1.5f) + : type(type), ne(ne), eps(eps), rms(rms), scale(scale) {} + + std::string op_desc(ggml_tensor * t) override { + GGML_UNUSED(t); + return rms ? "RMS_NORM_SCALE" : "NORM_SCALE"; + } + + bool run_whole_graph() override { return true; } + + ggml_tensor * build_graph(ggml_context * ctx) override { + ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_set_name(a, "a"); + + ggml_tensor * n = rms ? ggml_rms_norm(ctx, a, eps) : ggml_norm(ctx, a, eps); + ggml_tensor * out = ggml_scale(ctx, n, scale); + ggml_set_name(out, "out"); + + return out; + } +}; + // GGML_OP_RMS_NORM struct test_rms_norm : public test_case { const ggml_type type; @@ -3538,13 +3760,16 @@ struct test_rms_norm_back : public test_case { } }; -// GGML_OP_RMS_NORM + GGML_OP_MUL + GGML_OP_ADD +// GGML_OP_RMS_NORM + GGML_OP_MUL + GGML_OP_ADD (+ GGML_OP_MUL) struct test_rms_norm_mul_add : public test_case { const ggml_type type; const std::array ne; const float eps; const bool broadcast; const bool multi_add; // test a sequence of adds feeding into rms_norm + const bool post_mul; + const bool alias_rms_input; + const bool weight_broadcast; std::string op_desc(ggml_tensor * t) override { GGML_UNUSED(t); @@ -3554,20 +3779,23 @@ struct test_rms_norm_mul_add : public test_case { bool run_whole_graph() override { return true; } std::string vars() override { - return VARS_TO_STR5(type, ne, eps, broadcast, multi_add); + return VARS_TO_STR8(type, ne, eps, broadcast, multi_add, post_mul, alias_rms_input, weight_broadcast); } test_rms_norm_mul_add(ggml_type type = GGML_TYPE_F32, std::array ne = {64, 5, 4, 3}, - float eps = 1e-6f, bool broadcast = false, bool multi_add = false) - : type(type), ne(ne), eps(eps), broadcast(broadcast), multi_add(multi_add) {} + float eps = 1e-6f, bool broadcast = false, bool multi_add = false, bool post_mul = false, + bool alias_rms_input = false, bool weight_broadcast = false) + : type(type), ne(ne), eps(eps), broadcast(broadcast), multi_add(multi_add), post_mul(post_mul), + alias_rms_input(alias_rms_input), weight_broadcast(weight_broadcast) {} ggml_tensor * build_graph(ggml_context * ctx) override { std::array broadcast_dims = {ne[0]*2, ne[1]*3, ne[2]*3, ne[3]*4}; ggml_tensor * a = ggml_new_tensor(ctx, type, 4, broadcast ? broadcast_dims.data() : ne.data()); - ggml_tensor * b = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_tensor * b = weight_broadcast ? ggml_new_tensor_1d(ctx, type, ne[0]) : ggml_new_tensor(ctx, type, 4, ne.data()); ggml_tensor * c = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_tensor * d = nullptr; ggml_set_param(a); ggml_set_name(a, "a"); @@ -3578,10 +3806,20 @@ struct test_rms_norm_mul_add : public test_case { // Use a, b and c early, so we don't end up with an OP_NONE between rms_norm and mul a = ggml_add(ctx, ggml_add(ctx, a, b), c); + if (post_mul) { + d = ggml_new_tensor_1d(ctx, type, 1); + ggml_set_param(d); + ggml_set_name(d, "d"); + a = ggml_add(ctx, a, d); + } if (multi_add) { a = ggml_add(ctx, ggml_add(ctx, a, b), c); } - ggml_tensor * out = ggml_add(ctx, ggml_mul(ctx, ggml_rms_norm(ctx, a, eps), b), c); + ggml_tensor * mul = ggml_mul(ctx, ggml_rms_norm(ctx, a, eps), b); + ggml_tensor * out = alias_rms_input ? ggml_add_inplace(ctx, a, mul) : ggml_add(ctx, mul, c); + if (post_mul) { + out = ggml_mul(ctx, out, d); + } ggml_set_name(out, "out"); return out; @@ -3602,6 +3840,60 @@ struct test_rms_norm_mul_add : public test_case { } }; +// GGML_OP_ADD + GGML_OP_ADD (fused residual chain) +struct test_add_add : public test_case { + const ggml_type type; + const ggml_type type_addend; + const std::array ne; + const bool broadcast; + const bool view; // non-contiguous a via view_4d + + std::string op_desc(ggml_tensor * t) override { + GGML_UNUSED(t); + return "ADD_ADD"; + } + + bool run_whole_graph() override { return true; } + + std::string vars() override { + return VARS_TO_STR5(type, type_addend, ne, broadcast, view); + } + + test_add_add(ggml_type type = GGML_TYPE_F32, + ggml_type type_addend = GGML_TYPE_F32, + std::array ne = {64, 5, 4, 3}, + bool broadcast = false, + bool view = false) + : type(type), type_addend(type_addend), ne(ne), broadcast(broadcast), view(view) {} + + ggml_tensor * build_graph(ggml_context * ctx) override { + std::array broadcast_dims = {ne[0], 1, 1, 1}; + + ggml_tensor * a; + if (view) { + std::array parent = { ne[0] * 3, ne[1] * 2, ne[2], ne[3] }; + a = ggml_new_tensor(ctx, type, 4, parent.data()); + ggml_set_name(a, "a_parent"); + a = ggml_view_4d(ctx, a, ne[0], ne[1], ne[2], ne[3], a->nb[1], a->nb[2], a->nb[3], 0); + ggml_set_name(a, "a"); + } else { + a = ggml_new_tensor(ctx, type, 4, ne.data()); + ggml_set_name(a, "a"); + } + + ggml_tensor * b = ggml_new_tensor(ctx, type_addend, 4, ne.data()); + ggml_tensor * c = ggml_new_tensor(ctx, type_addend, 4, broadcast ? broadcast_dims.data() : ne.data()); + + ggml_set_name(b, "b"); + ggml_set_name(c, "c"); + + ggml_tensor * out = ggml_add(ctx, ggml_add(ctx, a, b), c); + ggml_set_name(out, "out"); + + return out; + } +}; + // GGML_OP_ADD + GGML_OP_RMS_NORM (fused operation) struct test_add_rms_norm : public test_case { const ggml_type type; @@ -3695,8 +3987,7 @@ struct test_relu_sqr : public test_case { } }; -// GGML_OP_UNARY(SILU|SIGMOID|SOFTPLUS) + GGML_OP_MUL (fused operation). -// `layout` and `tail` are used for fallback cases where fusion must be skipped +// GGML_OP_UNARY(GELU|SILU|SIGMOID|SOFTPLUS) + GGML_OP_MUL (fused operation). struct test_unary_mul : public test_case { const ggml_unary_op op; const ggml_type type; @@ -3717,7 +4008,8 @@ struct test_unary_mul : public test_case { // performs; relax the tolerance to match that drift switch (type) { case GGML_TYPE_F16: return 5e-5; - default: return 1e-7; + // gelu shader uses exp form, CPU uses tanhf + default: return op == GGML_UNARY_OP_GELU ? 5e-7 : 1e-7; } } @@ -3776,17 +4068,45 @@ struct test_unary_mul : public test_case { } else if (layout == "bcast") { a = ggml_new_tensor(ctx, type, 4, ne.data()); b = ggml_new_tensor_4d(ctx, type, ne[0], 1, 1, 1); + } else if (layout == "rep_ne0") { + // repeat on dim 0 + a = ggml_new_tensor(ctx, type, 4, ne.data()); + std::array ne_b = ne; + ne_b[0] /= 4; + b = ggml_new_tensor(ctx, type, 4, ne_b.data()); + } else if (layout == "view_mid") { + // VIEW between UNARY and MUL + a = ggml_new_tensor(ctx, type, 4, ne.data()); + b = nullptr; + } else if (layout == "gate") { + // small gate on src1 + const std::array ne_gate = { 1, ne[1], ne[2], ne[3] }; + a = ggml_new_tensor(ctx, type, 4, ne_gate.data()); + b = ggml_new_tensor(ctx, type, 4, ne.data()); } else { GGML_ABORT("unknown layout %s", layout.c_str()); } - ggml_set_name(a, "a"); - ggml_set_name(b, "b"); + if (a != nullptr) { + ggml_set_name(a, "a"); + } + if (b != nullptr) { + ggml_set_name(b, "b"); + } ggml_tensor * u = ggml_unary(ctx, a, op); ggml_set_name(u, "unary"); // a broadcasting operand can only be the second one - const bool second = swap && layout != "bcast"; + const bool second = layout == "gate" || (swap && layout != "bcast" && layout != "view_mid"); + if (layout == "view_mid") { + std::array ne_base = ne; + ne_base[0] *= 2; + ggml_tensor * base = ggml_new_tensor(ctx, type, 4, ne_base.data()); + ggml_set_name(base, "base"); + b = ggml_view_4d(ctx, base, ne[0], ne[1], ne[2], ne[3], + base->nb[1], base->nb[2], base->nb[3], 0); + ggml_set_name(b, "b"); + } ggml_tensor * out = second ? ggml_mul(ctx, b, u) : ggml_mul(ctx, u, b); if (tail == "reuse") { @@ -3908,6 +4228,9 @@ struct test_dsv4_hc : public test_case { if (name == "post") { lo = 0.0f; hi = 2.0f; return true; } + if (name == "gate") { + lo = -4.0f; hi = 4.0f; return true; + } if (name == "x" || name == "residual") { lo = -1.0f; hi = 1.0f; return true; } @@ -3971,7 +4294,9 @@ struct test_dsv4_hc_comb : public test_dsv4_hc { struct test_dsv4_hc_pre : public test_dsv4_hc { const int64_t n_embd; + const int64_t n_hc; const int64_t n_tokens; + const bool gated; std::string op_desc(ggml_tensor * t) override { GGML_UNUSED(t); @@ -3979,20 +4304,27 @@ struct test_dsv4_hc_pre : public test_dsv4_hc { } std::string vars() override { - return VARS_TO_STR2(n_embd, n_tokens); + return VARS_TO_STR4(n_embd, n_hc, n_tokens, gated); } - test_dsv4_hc_pre(int64_t n_embd = 31, int64_t n_tokens = 17) - : n_embd(n_embd), n_tokens(n_tokens) {} + test_dsv4_hc_pre(int64_t n_embd = 31, int64_t n_hc = 4, int64_t n_tokens = 17, bool gated = false) + : n_embd(n_embd), n_hc(n_hc), n_tokens(n_tokens), gated(gated) {} ggml_tensor * build_graph(ggml_context * ctx) override { - ggml_tensor * x = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_embd, hc, n_tokens); + ggml_tensor * x = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_embd, n_hc, n_tokens); ggml_set_name(x, "x"); - ggml_tensor * weights = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hc, n_tokens); - ggml_set_name(weights, "weights"); + if (gated) { + ggml_tensor * gate = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_embd, n_hc, n_tokens); + ggml_set_name(gate, "gate"); + + out = ggml_dsv4_hc_pre_gated(ctx, x, gate, 1.0f/n_hc); + } else { + ggml_tensor * weights = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_hc, n_tokens); + ggml_set_name(weights, "weights"); - out = ggml_dsv4_hc_pre(ctx, x, weights); + out = ggml_dsv4_hc_pre(ctx, x, weights); + } ggml_set_name(out, "out"); return out; } @@ -4001,6 +4333,7 @@ struct test_dsv4_hc_pre : public test_dsv4_hc { struct test_dsv4_hc_post : public test_dsv4_hc { const int64_t n_embd; const int64_t n_tokens; + const bool identity; std::string op_desc(ggml_tensor * t) override { GGML_UNUSED(t); @@ -4008,11 +4341,11 @@ struct test_dsv4_hc_post : public test_dsv4_hc { } std::string vars() override { - return VARS_TO_STR2(n_embd, n_tokens); + return VARS_TO_STR3(n_embd, n_tokens, identity); } - test_dsv4_hc_post(int64_t n_embd = 31, int64_t n_tokens = 17) - : n_embd(n_embd), n_tokens(n_tokens) {} + test_dsv4_hc_post(int64_t n_embd = 31, int64_t n_tokens = 17, bool identity = false) + : n_embd(n_embd), n_tokens(n_tokens), identity(identity) {} ggml_tensor * build_graph(ggml_context * ctx) override { ggml_tensor * x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, n_embd, n_tokens); @@ -4024,8 +4357,11 @@ struct test_dsv4_hc_post : public test_dsv4_hc { ggml_tensor * post = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, hc, n_tokens); ggml_set_name(post, "post"); - ggml_tensor * comb = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, hc, hc, n_tokens); - ggml_set_name(comb, "comb"); + ggml_tensor * comb = nullptr; + if (!identity) { + comb = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, hc, hc, n_tokens); + ggml_set_name(comb, "comb"); + } out = ggml_dsv4_hc_post(ctx, x, residual, post, comb); ggml_set_name(out, "out"); @@ -4112,9 +4448,10 @@ struct test_ssm_scan : public test_case { const int64_t n_seqs; const bool xbc_overlap; const int64_t K; + const bool weak_decay; std::string vars() override { - return VARS_TO_STR9(type, d_state, head_dim, n_head, n_group, n_seq_tokens, n_seqs, xbc_overlap, K); + return VARS_TO_STR10(type, d_state, head_dim, n_head, n_group, n_seq_tokens, n_seqs, xbc_overlap, K, weak_decay); } test_ssm_scan(ggml_type type = GGML_TYPE_F32, @@ -4125,8 +4462,9 @@ struct test_ssm_scan : public test_case { int64_t n_seq_tokens = 32, int64_t n_seqs = 32, bool xbc_overlap = false, - int64_t K = 1) - : type(type), d_state(d_state), head_dim(head_dim), n_head(n_head), n_group(n_group), n_seq_tokens(n_seq_tokens), n_seqs(n_seqs), xbc_overlap(xbc_overlap), K(K) {} + int64_t K = 1, + bool weak_decay = false) + : type(type), d_state(d_state), head_dim(head_dim), n_head(n_head), n_group(n_group), n_seq_tokens(n_seq_tokens), n_seqs(n_seqs), xbc_overlap(xbc_overlap), K(K), weak_decay(weak_decay) {} double max_nmse_err() override { // SSD path (head_dim > 1) uses FP16 intermediates (M matrix, X_dt); Mamba-1 is pure FP32. @@ -4179,7 +4517,7 @@ struct test_ssm_scan : public test_case { continue; } else if (t->ne[1] == n_head && t->ne[2] == 1) { // A {1 or d_state, n_head}: negative decay (2-D tensor, ne[2]==1 distinguishes from 3-D/4-D tensors) - init_tensor_uniform(t, -1.0f, -0.5f); + init_tensor_uniform(t, weak_decay ? -0.02f : -1.0f, weak_decay ? -0.005f : -0.5f); } else { init_tensor_uniform(t); } @@ -4395,6 +4733,122 @@ struct test_gated_delta_net : public test_case { } }; +// GGML_OP_GATED_DELTA_NET + GGML_OP_CPY (recurrent cache fusion) +struct test_gated_delta_net_cache_fusion : public test_case { + const ggml_type type; + + const int64_t head_count; + const int64_t head_size; + const int64_t n_seq_tokens; + const int64_t n_seqs; + const int64_t K; // snapshot slot count (>1) + + ggml_tensor * cpy_node = nullptr; + + std::string vars() override { + return VARS_TO_STR6(type, head_count, head_size, n_seq_tokens, n_seqs, K); + } + + test_gated_delta_net_cache_fusion(ggml_type type = GGML_TYPE_F32, + int64_t head_count = 4, int64_t head_size = 32, int64_t n_seq_tokens = 2, int64_t n_seqs = 1, + int64_t K = 2) + : type(type), head_count(head_count), head_size(head_size), n_seq_tokens(n_seq_tokens), n_seqs(n_seqs), K(K) {} + + ggml_tensor * build_graph(ggml_context * ctx) override { + const int64_t S_v = head_size; + const int64_t H_v = head_count; + const int64_t H_k = head_count; + const int64_t D = S_v * S_v * H_v; + const int64_t n_written = std::min(n_seq_tokens, K); + + ggml_tensor * q = ggml_new_tensor_4d(ctx, type, head_size, H_k, n_seq_tokens, n_seqs); + ggml_tensor * k = ggml_new_tensor_4d(ctx, type, head_size, H_k, n_seq_tokens, n_seqs); + ggml_tensor * v = ggml_new_tensor_4d(ctx, type, head_size, H_v, n_seq_tokens, n_seqs); + ggml_set_name(q, "q"); + ggml_set_name(k, "k"); + ggml_set_name(v, "v"); + ggml_tensor * g = ggml_new_tensor_4d(ctx, type, 1, H_v, n_seq_tokens, n_seqs); + ggml_tensor * beta = ggml_new_tensor_4d(ctx, type, 1, H_v, n_seq_tokens, n_seqs); + ggml_tensor * state = ggml_new_tensor_4d(ctx, type, head_size, head_size, H_v, n_seqs); + ggml_set_name(g, "g"); + ggml_set_name(beta, "beta"); + ggml_set_name(state, "state"); + + q = ggml_l2_norm(ctx, q, 1e-6f); + k = ggml_l2_norm(ctx, k, 1e-6f); + + ggml_tensor * gdn_out = ggml_gated_delta_net(ctx, q, k, v, g, beta, state, K); + ggml_set_name(gdn_out, "gdn_out"); + + // attn scores view (first part of the gdn output) + ggml_tensor * attn = ggml_view_4d(ctx, gdn_out, + S_v, H_v, n_seq_tokens, n_seqs, + ggml_row_size(gdn_out->type, S_v), + ggml_row_size(gdn_out->type, S_v * H_v), + ggml_row_size(gdn_out->type, S_v * H_v * n_seq_tokens), 0); + ggml_set_name(attn, "attn"); + + // snapshot tail view [D, n_seqs, n_written] + const int64_t attn_score_elems = S_v * H_v * n_seq_tokens * n_seqs; + ggml_tensor * src = ggml_view_3d(ctx, gdn_out, + D, n_seqs, n_written, + ggml_row_size(gdn_out->type, D), + ggml_row_size(gdn_out->type, D * n_seqs), + ggml_row_size(gdn_out->type, attn_score_elems)); + + // recurrent cache view [D, n_seqs, n_written] + ggml_tensor * cache = ggml_new_tensor_3d(ctx, type, D, n_seqs, n_written); + ggml_set_name(cache, "cache"); + ggml_tensor * dst = ggml_view_3d(ctx, cache, + D, n_seqs, n_written, + ggml_row_size(cache->type, D), + ggml_row_size(cache->type, D * n_seqs), 0); + + ggml_tensor * cpy = ggml_cpy(ctx, src, dst); + ggml_set_name(cpy, "gdn_cache_cpy"); + cpy_node = cpy; + + // read the cpy output (not the plain dst view, which would not pull the cpy into the graph) + // so that neither the gdn nor the cpy is the graph output + ggml_tensor * out = ggml_sum(ctx, cpy); + return out; + } + + std::string op_desc(ggml_tensor * t) override { + GGML_UNUSED(t); + return "GATED_DELTA_NET_CACHE_FUSION"; + } + + bool run_whole_graph() override { return true; } + std::vector fusion_test_nodes() override { return { cpy_node }; } + + uint64_t op_flops(ggml_tensor * t) override { + GGML_UNUSED(t); + const uint64_t S_v = head_size; + const uint64_t H_v = head_count; + const uint64_t T = n_seq_tokens; + const uint64_t B = n_seqs; + return (4ull*S_v + 2ull*S_v*S_v) * H_v * T * B; + } + + void initialize_tensors(ggml_context * ctx) override { + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) { + if (ggml_is_view_op(t->op)) { continue; } + if (strcmp(t->name, "g") == 0) { + init_tensor_uniform(t, -20.0f, -1e-4f); + } else if (strcmp(t->name, "beta") == 0) { + init_tensor_uniform(t, 0.0f, 1.0f); + } else if (strcmp(t->name, "v") == 0) { + init_tensor_uniform(t, -0.3f, 5.0f); + } else if (strcmp(t->name, "cache") == 0) { + init_tensor_uniform(t, 0.0f, 0.0f); + } else { + init_tensor_uniform(t); + } + } + } +}; + // GGML_OP_GATED_LINEAR_ATTN struct test_gla : public test_case { const ggml_type type; @@ -4470,9 +4924,10 @@ struct test_mul_mat : public test_case { const std::array per; // permutation of dimensions const int64_t k_v; // size of k in memory, resulting in a non-contiguous view for k_v > k, no view for k_v == 0 const uint32_t o; // number of outputs + const bool src_overlap; // a and b are overlapping views of the same tensor std::string vars() override { - return VARS_TO_STR10(type_a, type_b, m, n, k, bs, nr, per, k_v, o); + return VARS_TO_STR11(type_a, type_b, m, n, k, bs, nr, per, k_v, o, src_overlap); } double max_nmse_err() override { @@ -4501,8 +4956,8 @@ struct test_mul_mat : public test_case { std::array bs = {10, 10}, std::array nr = {2, 2}, std::array per = {0, 1, 2, 3}, - int64_t k_v = 0, uint32_t o = 1) - : type_a(type_a), type_b(type_b), m(m), n(n), k(k), bs(bs), nr(nr), per(per), k_v(k_v), o(o) {} + int64_t k_v = 0, uint32_t o = 1, bool src_overlap = false) + : type_a(type_a), type_b(type_b), m(m), n(n), k(k), bs(bs), nr(nr), per(per), k_v(k_v), o(o), src_overlap(src_overlap) {} ggml_tensor * build_graph(ggml_context * ctx) override { // C^T = A * B^T: (k, m) * (k, n) => (m, n) @@ -4535,6 +4990,18 @@ struct test_mul_mat : public test_case { b = ggml_permute(ctx, b, per[0], per[1], per[2], per[3]); ggml_set_name(a, "a_permuted"); ggml_set_name(b, "b_permuted"); + } else if (src_overlap) { + GGML_ASSERT(type_a == type_b); + GGML_ASSERT(k_v == 0); + + // a and b are interleaved views of the same tensor: (e.g. fused QKV in MiniMax-01) + ggml_tensor * base = ggml_new_tensor_4d(ctx, type_a, 2*k, std::max(m, n), bs[0]*nr[0], bs[1]*nr[1]); + ggml_set_name(base, "base"); + + a = ggml_view_4d(ctx, base, k, m, bs[0], bs[1], base->nb[1], base->nb[2], base->nb[3], 0); + b = ggml_view_4d(ctx, base, k, n, bs[0]*nr[0], bs[1]*nr[1], base->nb[1], base->nb[2], base->nb[3], k*ggml_type_size(type_a)); + ggml_set_name(a, "a"); + ggml_set_name(b, "b"); } else { const int64_t k_physical = k_v == 0 ? k : k_v; a = ggml_new_tensor_4d(ctx, type_a, k_physical, m, bs[0], bs[1]); @@ -4626,27 +5093,37 @@ struct test_mul_mat_hadamard : public test_mul_mat { GGML_UNUSED(t); return "MUL_MAT_HADAMARD"; } -}; +}; + +static void init_mul_mat_id_ids(ggml_context * ctx, int n_mats) { + std::random_device rd; + std::default_random_engine rng(rd()); + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { + if (t->type != GGML_TYPE_I32 || ggml_is_view_op(t->op)) { + continue; + } + for (int64_t r = 0; r < ggml_nrows(t); r++) { + std::vector data(t->ne[0]); + for (int i = 0; i < t->ne[0]; i++) { + data[i] = i % n_mats; + } + std::shuffle(data.begin(), data.end(), rng); + ggml_backend_tensor_set(t, data.data(), r * t->nb[1], t->ne[0] * sizeof(int32_t)); + } + } +} -static void init_mul_mat_id_tensors(ggml_context * ctx, int n_mats) { - std::random_device rd; - std::default_random_engine rng(rd()); +static void init_mul_mat_id_tensors(ggml_context * ctx, int n_mats, float amax = 1.0f) { for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { if (t->type == GGML_TYPE_I32) { - if (ggml_is_view_op(t->op)) { continue; } - // ids - for (int64_t r = 0; r < ggml_nrows(t); r++) { - std::vector data(t->ne[0]); - for (int i = 0; i < t->ne[0]; i++) { - data[i] = i % n_mats; - } - std::shuffle(data.begin(), data.end(), rng); - ggml_backend_tensor_set(t, data.data(), r * t->nb[1], t->ne[0] * sizeof(int32_t)); - } + continue; + } else if (amax != 1.0f && t->type == GGML_TYPE_F32) { + init_tensor_uniform(t, -amax, amax); } else { init_tensor_uniform(t); } } + init_mul_mat_id_ids(ctx, n_mats); } // GGML_OP_MUL_MAT_ID @@ -4659,9 +5136,10 @@ struct test_mul_mat_id : public test_case { const int64_t m; const int64_t n; const int64_t k; + const float amax; // magnitude of src1 std::string vars() override { - return VARS_TO_STR8(type_a, type_b, n_mats, n_used, b, m, n, k); + return VARS_TO_STR9(type_a, type_b, n_mats, n_used, b, m, n, k, amax); } double max_nmse_err() override { @@ -4683,9 +5161,10 @@ struct test_mul_mat_id : public test_case { test_mul_mat_id(ggml_type type_a = GGML_TYPE_F32, ggml_type type_b = GGML_TYPE_F32, int n_mats = 8, int n_used = 2, bool b = false, - int64_t m = 32, int64_t n = 32, int64_t k = 32) + int64_t m = 32, int64_t n = 32, int64_t k = 32, + float amax = 1.0f) : type_a(type_a), type_b(type_b), n_mats(n_mats), n_used(n_used), b(b), - m(m), n(n), k(k) { + m(m), n(n), k(k), amax(amax) { GGML_ASSERT(n_used <= n_mats); } @@ -4707,11 +5186,20 @@ struct test_mul_mat_id : public test_case { ggml_tensor * out = ggml_mul_mat_id(ctx, as, b, ids); ggml_set_name(out, "out"); + if (amax > 65504.0f) { + // src1 exceeds F16 range + ggml_prec_set_src(out, GGML_PREC_F32, 1); + } + return out; } void initialize_tensors(ggml_context * ctx) override { - init_mul_mat_id_tensors(ctx, n_mats); + init_mul_mat_id_tensors(ctx, n_mats, amax); + } + + void reinit_perf_iter(ggml_context * ctx) override { + init_mul_mat_id_ids(ctx, n_mats); } }; @@ -5331,24 +5819,27 @@ struct test_rope : public test_case { int v; // view (1 : non-contiguous a) bool forward; bool inplace; + int n_offs; // offset of the rotated dims window, set via ggml_rope_set_offset() std::string vars() override { // forward can be inferred from the op, does not need to be printed - return VARS_TO_STR11(type, ne_a, n_dims, mode, n_ctx, fs, ef, af, ff, v, inplace); + return VARS_TO_STR12(type, ne_a, n_dims, mode, n_ctx, fs, ef, af, ff, v, inplace, n_offs); } test_rope(ggml_type type = GGML_TYPE_F32, std::array ne_a = {10, 5, 3, 1}, int n_dims = 10, int mode = GGML_ROPE_TYPE_NORMAL, int n_ctx = 512, float fs = 1.0f, - float ef = 0.0f, float af = 0.0f, bool ff = false, int v = 0, bool forward = true, bool inplace = false) - : type(type), ne_a(ne_a), n_dims(n_dims), mode(mode), n_ctx(n_ctx), fs(fs), ef(ef), af(af), ff(ff), v(v), forward(forward), inplace(inplace) {} + float ef = 0.0f, float af = 0.0f, bool ff = false, int v = 0, bool forward = true, bool inplace = false, + int n_offs = 0) + : type(type), ne_a(ne_a), n_dims(n_dims), mode(mode), n_ctx(n_ctx), fs(fs), ef(ef), af(af), ff(ff), v(v), forward(forward), inplace(inplace), n_offs(n_offs) {} ggml_tensor * build_graph(ggml_context * ctx) override { ggml_tensor * a; if (v & 1) { auto ne = ne_a; ne[0] *= 2; ne[1] *= 4; ne[2] *= 3; a = ggml_new_tensor(ctx, type, 4, ne.data()); - if (forward) { + if (forward && n_offs == 0) { + // FIXME: support gradients with n_offs > 0 ggml_set_param(a); } ggml_set_name(a, "a"); @@ -5361,7 +5852,8 @@ struct test_rope : public test_case { // non-aligned buffer offset, which exercises backends' alignment paths. auto ne = ne_a; ne[0] *= 2; a = ggml_new_tensor(ctx, type, 4, ne.data()); - if (forward) { + if (forward && n_offs == 0) { + // FIXME: support gradients with n_offs > 0 ggml_set_param(a); } ggml_set_name(a, "a"); @@ -5372,7 +5864,8 @@ struct test_rope : public test_case { ggml_set_name(a, "view_of_a"); } else { a = ggml_new_tensor(ctx, type, 4, ne_a.data()); - if (forward) { + if (forward && n_offs == 0) { + // FIXME: support gradients with n_offs > 0 ggml_set_param(a); } ggml_set_name(a, "a"); @@ -5433,6 +5926,9 @@ struct test_rope : public test_case { out = ggml_rope_ext_back(ctx, a, pos, freq, n_dims, mode, 0, 10000.0f, fs, ef, af, 1.0f, 1.0f); } } + if (n_offs != 0) { + out = ggml_rope_set_offset(out, n_offs); + } ggml_set_name(out, "out"); return out; @@ -5759,13 +6255,6 @@ struct test_conv_2d : public test_case { // Whether the inputs are contiguous in the channel dim or the width dim const bool cwhn; - // If true, the direct CONV_2D will be used in the graph, otherwise it - // uses ggml_conv_2d: - // * if the program is called with -o CONV_2D_DIRECT_IMPL, the - // CONV_2D graph will be built, while - // * if the program is called with -o CONV_2D_INDIRECT_IMPL, the - // IM2COL -> MUL_MM graph will be built. - std::string vars() override { return VARS_TO_STR10(ne_input, ne_kernel, type_kernel, stride0, stride1, padding0, padding1, dilation0, dilation1, cwhn); } @@ -6194,6 +6683,87 @@ struct test_top_k : public test_case { } }; +// qwen4exp QSA indexer top-k fusion: expand per-block scores to cells, add the f16 mask, top-k. +struct test_topk_qsa : public test_case { + const int64_t n_blocks; + const int64_t n_kv; + const int64_t n_tps; + const int64_t n_stream; + const int width; + ggml_tensor * out {}; + + std::string op_desc(ggml_tensor * t) override { + GGML_UNUSED(t); + return "TOPK_QSA"; + } + + std::string vars() override { + return VARS_TO_STR5(n_blocks, n_kv, n_tps, n_stream, width); + } + + test_topk_qsa(int64_t n_blocks = 512, int64_t n_kv = 2048, int64_t n_tps = 2, int64_t n_stream = 1, int width = 1500) + : n_blocks(n_blocks), n_kv(n_kv), n_tps(n_tps), n_stream(n_stream), width(width) {} + + double max_err() override { return 0.0; } + bool run_whole_graph() override { return true; } + + ggml_tensor * build_graph(ggml_context * ctx) override { + ggml_tensor * score = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_blocks, n_tps, n_stream); + ggml_set_name(score, "score"); + ggml_tensor * cell_blk = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, n_kv, n_stream); + ggml_set_name(cell_blk, "cell_blk"); + ggml_tensor * kq_mask = ggml_new_tensor_3d(ctx, GGML_TYPE_F16, n_kv, n_tps, n_stream); + ggml_set_name(kq_mask, "kq_mask"); + + ggml_tensor * a = ggml_cont(ctx, ggml_permute(ctx, score, 1, 0, 2, 3)); + ggml_tensor * e = ggml_get_rows(ctx, a, cell_blk); + e = ggml_cont(ctx, ggml_permute(ctx, e, 1, 0, 2, 3)); + ggml_tensor * m = ggml_cast(ctx, kq_mask, GGML_TYPE_F32); + e = ggml_add(ctx, e, ggml_reshape_3d(ctx, m, n_kv, n_tps, n_stream)); + out = ggml_top_k(ctx, e, width); + ggml_set_name(out, "out"); + return out; + } + + std::vector fusion_test_nodes() override { return { out }; } + + // distinct mask ramp + small scores keep every cell value unique, so no top-k ties + void initialize_tensors(ggml_context * ctx) override { + for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) { + if (t->op != GGML_OP_NONE) { + continue; + } + if (t->type == GGML_TYPE_I32) { + std::vector data(ggml_nelements(t)); + for (auto & v : data) { v = rand() % n_blocks; } + ggml_backend_tensor_set(t, data.data(), 0, data.size() * sizeof(int32_t)); + } else if (t->type == GGML_TYPE_F16) { + std::vector data(ggml_nelements(t)); + for (int64_t r = 0; r < ggml_nrows(t); r++) { + for (int64_t i = 0; i < n_kv; i++) { + data[r * n_kv + i] = ggml_fp32_to_fp16((float) i); + } + } + ggml_backend_tensor_set(t, data.data(), 0, data.size() * sizeof(ggml_fp16_t)); + } else { + init_tensor_uniform(t, 0.0f, 0.5f); + } + } + } + + // top-k output order is unspecified; compare as a set of indices + double err(const float * a, const float * b, size_t n) override { + std::vector ia(n), ib(n); + double diff = 0.0; + for (size_t i = 0; i < n; i++) { + ia[i] = (int32_t) a[i]; + ib[i] = (int32_t) b[i]; + diff += std::fabs(a[i] - ia[i]) + std::fabs(b[i] - ib[i]); + } + return diff + jdst(ia.data(), ib.data(), n); + } +}; + enum MoeGatingFunc { GATING_FUNC_SOFTMAX, GATING_FUNC_SIGMOID, @@ -6300,6 +6870,79 @@ struct test_topk_moe : public test_case { } }; +struct test_moe_reduce : public test_case { + const int64_t n_embd; + const int64_t n_expert_used; + const int64_t n_tokens; + const bool unaligned_experts; + const bool with_expert_scale; + const bool interleaved_views_adds; + + test_moe_reduce( + int64_t n_embd, int64_t n_expert_used, int64_t n_tokens, + bool unaligned_experts = false, bool with_expert_scale = false, bool interleaved_views_adds = false) : + n_embd(n_embd), n_expert_used(n_expert_used), n_tokens(n_tokens), + unaligned_experts(unaligned_experts), with_expert_scale(with_expert_scale), + interleaved_views_adds(interleaved_views_adds) {} + + std::string vars() override { + return VARS_TO_STR6(n_embd, n_expert_used, n_tokens, unaligned_experts, with_expert_scale, interleaved_views_adds); + } + + std::string op_desc(ggml_tensor * t) override { + GGML_UNUSED(t); + return "MOE_REDUCE"; + } + + bool run_whole_graph() override { return true; } + + ggml_tensor * build_graph(ggml_context * ctx) override { + ggml_tensor * experts; + if (unaligned_experts) { + ggml_tensor * storage = ggml_new_tensor_1d( + ctx, GGML_TYPE_F32, n_embd * n_expert_used * n_tokens + 1); + ggml_set_name(storage, "experts_storage"); + experts = ggml_view_3d(ctx, storage, n_embd, n_expert_used, n_tokens, + n_embd * sizeof(float), n_embd * n_expert_used * sizeof(float), sizeof(float)); + } else { + experts = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, n_embd, n_expert_used, n_tokens); + } + ggml_set_name(experts, "experts"); + ggml_tensor * weights = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 1, n_expert_used, n_tokens); + ggml_set_name(weights, "weights"); + + ggml_tensor * scaled = experts; + if (with_expert_scale) { + ggml_tensor * expert_scale = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, 1, n_expert_used, n_tokens); + ggml_set_name(expert_scale, "expert_scale"); + scaled = ggml_mul(ctx, experts, expert_scale); + ggml_set_name(scaled, "scaled_experts"); + } + + ggml_tensor * weighted = ggml_mul(ctx, scaled, weights); + ggml_set_name(weighted, "weighted_experts"); + + std::vector views(n_expert_used); + for (int64_t expert = 0; expert < n_expert_used; ++expert) { + views[expert] = ggml_view_2d( + ctx, weighted, n_embd, n_tokens, weighted->nb[2], expert * weighted->nb[1]); + if (!interleaved_views_adds && mode == MODE_TEST) { + ggml_build_forward_expand(gf, views[expert]); + } + } + + ggml_tensor * out = views[0]; + for (int64_t expert = 1; expert < n_expert_used; ++expert) { + out = ggml_add(ctx, out, views[expert]); + if (!interleaved_views_adds && mode == MODE_TEST) { + ggml_build_forward_expand(gf, out); + } + } + ggml_set_name(out, "moe_reduce"); + return out; + } +}; + struct test_mul_mat_vec_fusion : public test_case { const ggml_type type; const ggml_glu_op glu_op; @@ -6344,6 +6987,9 @@ struct test_mul_mat_vec_fusion : public test_case { constexpr float alpha = 1.702f; constexpr float limit = 7.0f; out = ggml_swiglu_oai(ctx, ffn_gate, ffn_up, alpha, limit); + } else if (glu_op == GGML_GLU_OP_SWIGLU_CLAMP) { + constexpr float limit = 10.0f; + out = ggml_swiglu_clamp(ctx, ffn_gate, ffn_up, limit); } else { out = ggml_glu_split(ctx, ffn_gate, ffn_up, glu_op); } @@ -6571,20 +7217,32 @@ struct test_sum_rows : public test_case { struct test_mean : public test_case { const ggml_type type; const std::array ne; + const bool permute; + const bool slice; std::string vars() override { - return VARS_TO_STR2(type, ne); + return VARS_TO_STR4(type, ne, permute, slice); } test_mean(ggml_type type = GGML_TYPE_F32, - std::array ne = {10, 5, 4, 3}) - : type(type), ne(ne) {} + std::array ne = {10, 5, 4, 3}, + bool permute = false, bool slice = false) + : type(type), ne(ne), permute(permute), slice(slice) {} ggml_tensor * build_graph(ggml_context * ctx) override { ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data()); ggml_set_param(a); ggml_set_name(a, "a"); + if (slice) { + a = ggml_view_4d(ctx, a, + ne[0], ne[1], ne[2] / 2, ne[3] - 1, + a->nb[1], a->nb[2] * 2, a->nb[3], /*offset=*/a->nb[3]); + } + if (permute) { + a = ggml_permute(ctx, a, 0, 2, 3, 1); + } + ggml_tensor * out = ggml_mean(ctx, a); ggml_set_name(out, "out"); @@ -6730,6 +7388,49 @@ struct test_group_norm_mul_add : public test_case { } }; +// GGML_OP_L2_NORM x N: independent same-shape norms in one graph (strided qkv views or +// contiguous), consuming adds nested so the norms stay adjacent in the graph. +struct test_l2_norm_batch : public test_case { + const ggml_type type; + const std::array ne; + const int n_norms; + const float eps; + const bool strided; + + std::string vars() override { return VARS_TO_STR5(type, ne, n_norms, eps, strided); } + std::string op_desc(ggml_tensor * t) override { GGML_UNUSED(t); return "L2_NORM_BATCH"; } + bool run_whole_graph() override { return true; } + + test_l2_norm_batch(ggml_type type = GGML_TYPE_F32, std::array ne = { 128, 16, 16, 1 }, + int n_norms = 4, float eps = 1e-12f, bool strided = true) + : type(type), ne(ne), n_norms(n_norms), eps(eps), strided(strided) {} + + ggml_tensor * build_graph(ggml_context * ctx) override { + GGML_ASSERT(n_norms >= 2 && n_norms <= 8); + ggml_tensor * parent = nullptr; + if (strided) { + parent = ggml_new_tensor_4d(ctx, type, ne[0], ne[1] * n_norms, ne[2], ne[3]); // qkv buffer + } + ggml_tensor * norms[8] = {}; + for (int t = 0; t < n_norms; ++t) { + ggml_tensor * src; + if (strided) { + src = ggml_view_4d(ctx, parent, ne[0], ne[1], ne[2], ne[3], parent->nb[1], parent->nb[2], + parent->nb[3], t * ne[1] * parent->nb[1]); + } else { + src = ggml_new_tensor(ctx, type, 4, ne.data()); + } + norms[t] = ggml_l2_norm(ctx, src, eps); + } + ggml_tensor * out = norms[n_norms - 1]; + for (int t = n_norms - 2; t >= 0; --t) { + out = ggml_add(ctx, norms[t], out); + } + ggml_set_name(out, "out"); + return out; + } +}; + // GGML_OP_L2_NORM struct test_l2_norm : public test_case { const ggml_type type; @@ -7055,9 +7756,12 @@ struct test_flash_attn_ext : public test_case { const ggml_type type_K; const ggml_type type_V; std::array permute; + const bool kv_view; // create K/V as views of a larger buffer (like a KV cache) + const bool v_is_view_of_k; + const int64_t n_kv_max; std::string vars() override { - return VARS_TO_STR14(hsk, hsv, nh, nr23, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_K, type_V, permute); + return VARS_TO_STR17(hsk, hsv, nh, nr23, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_K, type_V, permute, kv_view, v_is_view_of_k, n_kv_max); } double max_nmse_err() override { @@ -7073,9 +7777,10 @@ struct test_flash_attn_ext : public test_case { test_flash_attn_ext(int64_t hsk = 128, int64_t hsv = 128, int64_t nh = 32, std::array nr23 = {1, 1}, int64_t kv = 96, int64_t nb = 8, bool mask = true, bool sinks = false, float max_bias = 0.0f, float logit_softcap = 0.0f, ggml_prec prec = GGML_PREC_F32, - ggml_type type_K = GGML_TYPE_F16, ggml_type type_V = GGML_TYPE_F16, std::array permute = {0, 1, 2, 3}) + ggml_type type_K = GGML_TYPE_F16, ggml_type type_V = GGML_TYPE_F16, std::array permute = {0, 1, 2, 3}, + bool kv_view = true, bool v_is_view_of_k = false, int64_t n_kv_max = 0) : hsk(hsk), hsv(hsv), nh(nh), nr23(nr23), kv(kv), nb(nb), mask(mask), sinks(sinks), max_bias(max_bias), logit_softcap(logit_softcap), prec(prec), - type_K(type_K), type_V(type_V), permute(permute) {} + type_K(type_K), type_V(type_V), permute(permute), kv_view(kv_view), v_is_view_of_k(v_is_view_of_k), n_kv_max(n_kv_max) {} ggml_tensor * build_graph(ggml_context * ctx) override { const int64_t hsk_padded = GGML_PAD(hsk, ggml_blck_size(type_K)); @@ -7103,21 +7808,21 @@ struct test_flash_attn_ext : public test_case { ggml_tensor * q = create_permuted(GGML_TYPE_F32, hsk_padded, nb, nh*nr23[0], nr23[1], false); ggml_set_name(q, "q"); - ggml_tensor * k = create_permuted(type_K, hsk_padded, kv, nh, nr23[1], true); // the K tensor is usually a view of the K cache + ggml_tensor * k = create_permuted(type_K, hsk_padded, kv, nh, nr23[1], kv_view); // the K tensor is usually a view of the K cache ggml_set_name(k, "k"); ggml_tensor * v = nullptr; - if (type_K == type_V && hsk_padded == 576 && hsv_padded == 512) { - // TODO: this branch should become a separate test case parameter instead of hardcoding this for these head shapes - - // in this branch, the V cache is sub-view of the K cache. this is used by some MLA-based models + if (v_is_view_of_k) { + // the V cache is a sub-view of the K cache. this is used by some MLA-based models // for more info: // - https://github.com/ggml-org/llama.cpp/pull/13435 // - https://github.com/ggml-org/llama.cpp/pull/18953#issuecomment-3774948392 // - https://github.com/ggml-org/llama.cpp/pull/18986 + GGML_ASSERT(type_K == type_V && hsv_padded <= hsk_padded); + v = ggml_view_4d(ctx, k, hsv_padded, kv, nh, nr23[1], k->nb[1], k->nb[2], k->nb[3], 0); } else { - v = create_permuted(type_V, hsv_padded, kv, nh, nr23[1], true); // the V tensor is usually a view of the V cache + v = create_permuted(type_V, hsv_padded, kv, nh, nr23[1], kv_view); // the V tensor is usually a view of the V cache } ggml_set_name(v, "v"); @@ -7135,7 +7840,8 @@ struct test_flash_attn_ext : public test_case { ggml_tensor * out = ggml_flash_attn_ext(ctx, q, k, v, m, 1.0f/sqrtf(hsk), max_bias, logit_softcap); ggml_flash_attn_ext_add_sinks(out, s); - ggml_flash_attn_ext_set_prec (out, prec); + ggml_flash_attn_ext_set_n_kv_max(out, n_kv_max); + ggml_prec_set_acc(out, prec); ggml_set_name(out, "out"); return out; @@ -7147,7 +7853,11 @@ struct test_flash_attn_ext : public test_case { // make the sink values more noticeable in order to trigger a test failure when the implementation is wrong init_tensor_uniform(t, -10.0f, 10.0f); } else if (strcmp(t->name, "m") == 0) { - init_tensor_kq_mask(t); + if (n_kv_max > 0) { + init_tensor_kq_mask_sparse(t, n_kv_max); + } else { + init_tensor_kq_mask(t); + } } else { init_tensor_uniform(t); } @@ -8222,7 +8932,7 @@ static const ggml_type all_types[] = { GGML_TYPE_Q4_K, GGML_TYPE_Q5_K, GGML_TYPE_Q6_K, GGML_TYPE_TQ2_0, - // GGML_TYPE_TQ1_0, // TODO: implement for all backends + GGML_TYPE_TQ1_0, GGML_TYPE_IQ2_XXS, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ2_S, GGML_TYPE_IQ3_XXS, GGML_TYPE_IQ1_S, GGML_TYPE_IQ1_M, GGML_TYPE_IQ4_NL, GGML_TYPE_IQ3_S, GGML_TYPE_IQ4_XS, @@ -8250,7 +8960,7 @@ static const ggml_type other_types[] = { GGML_TYPE_Q5_K, GGML_TYPE_Q6_K, GGML_TYPE_TQ2_0, - // GGML_TYPE_TQ1_0, // TODO: implement for all backends + GGML_TYPE_TQ1_0, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ2_S, GGML_TYPE_IQ3_XXS, GGML_TYPE_IQ1_S, GGML_TYPE_IQ1_M, GGML_TYPE_IQ4_NL, GGML_TYPE_IQ3_S, GGML_TYPE_IQ4_XS, @@ -8287,7 +8997,7 @@ static std::vector> make_test_cases_eval() { } // fused unary + mul (gated activations that are not expressed as GGML_OP_GLU) - for (ggml_unary_op op : { GGML_UNARY_OP_SILU, GGML_UNARY_OP_SIGMOID, GGML_UNARY_OP_SOFTPLUS }) { + for (ggml_unary_op op : { GGML_UNARY_OP_GELU, GGML_UNARY_OP_SILU, GGML_UNARY_OP_SIGMOID, GGML_UNARY_OP_SOFTPLUS }) { for (ggml_type type : { GGML_TYPE_F16, GGML_TYPE_F32 }) { for (bool swap : { false, true }) { test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, swap)); @@ -8298,9 +9008,12 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, true, "pad_other")); test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, true, "halves")); test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "packed", "consumer")); + test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "bcast")); + test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "rep_ne0")); + test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "view_mid")); + test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "gate")); // must not fuse test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "strided_dim1")); - test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "bcast")); test_cases.emplace_back(new test_unary_mul(op, type, { 128, 2, 2, 2 }, false, "packed", "reuse")); } } @@ -8322,23 +9035,35 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_dsv4_hc_comb(17, 4)); test_cases.emplace_back(new test_dsv4_hc_comb(257, 8)); test_cases.emplace_back(new test_dsv4_hc_comb(17, 20)); + // production n_iter (DeepSeek-V4 uses 20) across batch sizes that cross + // subgroup and workgroup boundaries; 1 = single-token decode + for (int64_t n_tokens : {1, 256, 336, 512, 513, 1024, 2048}) { + test_cases.emplace_back(new test_dsv4_hc_comb(n_tokens, 20)); + } - test_cases.emplace_back(new test_dsv4_hc_pre(1, 1)); - test_cases.emplace_back(new test_dsv4_hc_pre(31, 17)); - test_cases.emplace_back(new test_dsv4_hc_pre(128, 257)); - test_cases.emplace_back(new test_dsv4_hc_pre(4096, 21)); + test_cases.emplace_back(new test_dsv4_hc_pre(1, 4, 1)); + test_cases.emplace_back(new test_dsv4_hc_pre(31, 4, 17)); + test_cases.emplace_back(new test_dsv4_hc_pre(128, 4, 257)); + test_cases.emplace_back(new test_dsv4_hc_pre(4096, 4, 21)); + test_cases.emplace_back(new test_dsv4_hc_pre(31, 4, 17, true)); + test_cases.emplace_back(new test_dsv4_hc_pre(4096, 4, 21, true)); + for (int64_t n_hc : {1, 2, 3, 5, 8, 65}) { + test_cases.emplace_back(new test_dsv4_hc_pre(128, n_hc, 17)); + test_cases.emplace_back(new test_dsv4_hc_pre(128, n_hc, 17, true)); + } test_cases.emplace_back(new test_dsv4_hc_post(1, 1)); test_cases.emplace_back(new test_dsv4_hc_post(31, 17)); test_cases.emplace_back(new test_dsv4_hc_post(128, 257)); test_cases.emplace_back(new test_dsv4_hc_post(4096, 21)); + test_cases.emplace_back(new test_dsv4_hc_post(31, 17, true)); + test_cases.emplace_back(new test_dsv4_hc_post(4096, 21, true)); // glu ops for (ggml_type type : {GGML_TYPE_F16, GGML_TYPE_F32}) { for (int v : {0, 1}) { for (int op = 0; op < GGML_GLU_OP_COUNT; op++) { - if (op == GGML_GLU_OP_SWIGLU_OAI) { - // SWIGLU_OAI is handled separately + if (op == GGML_GLU_OP_SWIGLU_OAI || op == GGML_GLU_OP_SWIGLU_CLAMP) { continue; } @@ -8361,6 +9086,14 @@ static std::vector> make_test_cases_eval() { } } + for (ggml_type type : {GGML_TYPE_F16, GGML_TYPE_F32}) { + for (int v : {0, 1}) { + for (float limit : {2.0f, 10.0f}) { + test_cases.emplace_back(new test_swiglu_clamp(type, { 128, 2, 2, 2 }, v, limit)); + } + } + } + for (ggml_type type : {GGML_TYPE_F32, GGML_TYPE_Q4_0}) { test_cases.emplace_back(new test_get_rows(type, 300*256, 5, 4, 1, 2, false)); test_cases.emplace_back(new test_get_rows(type, 256, 80000, 70000, 2, 1, false)); @@ -8371,15 +9104,20 @@ static std::vector> make_test_cases_eval() { for (ggml_type type : all_types) { for (int b : {1, 7}) { for (bool v : {false, true}) { - test_cases.emplace_back(new test_get_rows(type, 256, 5, 4, b, 1, v)); + for (bool vs0 : {false, true}) { + test_cases.emplace_back(new test_get_rows(type, 256, 5, 4, b, 1, v, vs0)); + } } } } for (int b : {1, 7}) { for (bool v : {false, true}) { - test_cases.emplace_back(new test_get_rows(GGML_TYPE_I32, 256, 5, 4, b, 1, v)); + for (bool vs0 : {false, true}) { + test_cases.emplace_back(new test_get_rows(GGML_TYPE_I32, 256, 5, 4, b, 1, v, vs0)); + } } } + test_cases.emplace_back(new test_get_rows(GGML_TYPE_F32, 256, 8, 2, 1, 1, false, true, 3)); test_cases.emplace_back(new test_get_rows_back(GGML_TYPE_F32, 1, 8, 2, 1, false)); test_cases.emplace_back(new test_get_rows_back(GGML_TYPE_F32, 1, 70000, 4, 1, false)); // row count > CUDA grid-y limit (65535) @@ -8417,7 +9155,7 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_set_rows(GGML_TYPE_F16, GGML_TYPE_F16, GGML_TYPE_I64, { 1, 8, 1, 3 }, { 1, 1 }, 2, true)); test_cases.emplace_back(new test_set_rows(GGML_TYPE_F16, GGML_TYPE_F16, GGML_TYPE_I32, { 1, 8, 1, 3 }, { 1, 1 }, 2, true)); - for (int mode : { GGML_ROPE_TYPE_NORMAL, GGML_ROPE_TYPE_NEOX, GGML_ROPE_TYPE_MROPE, GGML_ROPE_TYPE_VISION }) { + for (int mode : { GGML_ROPE_TYPE_NORMAL, GGML_ROPE_TYPE_NEOX, GGML_ROPE_TYPE_MROPE, GGML_ROPE_TYPE_VISION, GGML_ROPE_TYPE_IMROPE }) { for (ggml_type type : {GGML_TYPE_F16, GGML_TYPE_F32}) { for (int ne2 : {1, 8, 512}) { test_cases.emplace_back(new test_rope_set_rows(type, GGML_TYPE_I64, { 128, 32, ne2, 1 }, mode)); @@ -8425,6 +9163,7 @@ static std::vector> make_test_cases_eval() { } } } + test_cases.emplace_back(new test_rope_set_rows(GGML_TYPE_F32, GGML_TYPE_I32, { 128, 32, 8, 1 }, GGML_ROPE_TYPE_IMROPE)); for (ggml_type type_input : {GGML_TYPE_F32}) { for (ggml_op_pool pool_type : {GGML_OP_POOL_AVG, GGML_OP_POOL_MAX}) { @@ -8633,6 +9372,9 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_conv_2d({ 256, 256, 192, 1 }, { 3, 3, 192, 96 }, kernel_type, 1, 1, 1, 1, 1, 1, false)); // bool cwhn = false test_cases.emplace_back(new test_conv_2d({ 256, 256, 192, 1 }, { 3, 3, 192, 96 }, kernel_type, 1, 1, 1, 1, 1, 1, true)); // bool cwhn = true } + test_cases.emplace_back(new test_conv_2d({ 19, 17, 8, 2 }, { 3, 3, 8, 65 }, GGML_TYPE_F16, 1, 1, 1, 1, 1, 1)); + test_cases.emplace_back(new test_conv_2d({ 19, 17, 16, 3 }, { 3, 3, 16, 33 }, GGML_TYPE_F16, 2, 3, 4, 2, 2, 1)); + test_cases.emplace_back(new test_conv_2d({ 13, 11, 16, 3 }, { 1, 1, 16, 33 }, GGML_TYPE_F16, 1, 1, 0, 0, 1, 1)); // sycl backend will limit task global_range < MAX_INT // test cases for 2D im2col with large input W and H (occurs in stable-diffusion) @@ -8746,6 +9488,7 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_conv_transpose_2d({3, 2, 3, 1}, {2, 2, 1, 3}, 1, kernel_type)); test_cases.emplace_back(new test_conv_transpose_2d({10, 10, 9, 1}, {3, 3, 1, 9}, 2, kernel_type)); test_cases.emplace_back(new test_conv_transpose_2d({129, 63, 35, 1}, {3, 3, 48, 35}, 1, kernel_type)); + test_cases.emplace_back(new test_conv_transpose_2d({10, 10, 9, 2}, {3, 3, 1, 9}, 2, kernel_type)); // for multiple batches } test_cases.emplace_back(new test_count_equal(GGML_TYPE_F32, {4, 500, 1, 1})); @@ -8892,6 +9635,20 @@ static std::vector> make_test_cases_eval() { } } + for (ggml_type type_dst : { GGML_TYPE_F32, GGML_TYPE_F16 }) { + for (std::array ne : std::initializer_list>{ + {10, 10, 10, 1}, {33, 5, 7, 1}, {64, 3, 65, 1}, {2, 3, 5, 7}, + // large, tile-aligned and tile-unaligned, matching the perf cases + {1024, 64, 64, 1}, {2304, 64, 64, 1}, {1000, 33, 65, 1} }) { + for (std::array perm : std::initializer_list>{ + {2, 1, 0, 3}, // 0<->2 swap + {1, 2, 0, 3}, // 3-cycle + {0, 2, 1, 3} }) { + test_cases.emplace_back(new test_cont(type_dst, ne, false, perm)); + } + } + } + auto add_test_bin_bcast = [&](ggml_type type, std::array ne, std::array nr, bool perm1 = false, bool src_overlap = false) { for (auto op : {ggml_add, ggml_sub, ggml_mul, ggml_div}) { test_cases.emplace_back(new test_bin_bcast(op, type, ne, nr, 1, perm1, src_overlap)); @@ -8950,6 +9707,8 @@ static std::vector> make_test_cases_eval() { // fusion test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F32, {10, 5, 4, 3}, {2, 1, 1, 1}, 2)); + test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F16, {10, 5, 4, 3}, {2, 1, 1, 1}, 2)); + test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F32, {16, 5, 4, 3}, {1, 1, 1, 1}, 2, true)); test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F32, {16, 5, 4, 3}, {1, 2, 1, 1}, 3)); test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F32, {10, 5, 4, 3}, {1, 1, 2, 1}, 4)); test_cases.emplace_back(new test_bin_bcast(ggml_add, GGML_TYPE_F32, {16, 5, 4, 3}, {1, 1, 1, 2}, 5)); @@ -8972,10 +9731,16 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_rms_norm(GGML_TYPE_F32, { n, 5, 4, 3 }, v, eps)); } test_cases.emplace_back(new test_norm(GGML_TYPE_F32, { n, 5, 4, 3 }, false, eps, true)); + test_cases.emplace_back(new test_norm_scale(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, false, 1.5f)); + test_cases.emplace_back(new test_norm_scale(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, true, 1.5f)); test_cases.emplace_back(new test_rms_norm_back(GGML_TYPE_F32, { n, 5, 4, 3 }, eps)); test_cases.emplace_back(new test_l2_norm(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, false)); test_cases.emplace_back(new test_l2_norm(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, true)); test_cases.emplace_back(new test_l2_norm(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, false, true)); + // sibling batching: strided (production shape) and contiguous, 2 and 4 wide + test_cases.emplace_back(new test_l2_norm_batch(GGML_TYPE_F32, { n, 5, 4, 3 }, 2, eps, true)); + test_cases.emplace_back(new test_l2_norm_batch(GGML_TYPE_F32, { n, 5, 4, 3 }, 4, eps, true)); + test_cases.emplace_back(new test_l2_norm_batch(GGML_TYPE_F32, { n, 5, 4, 3 }, 4, eps, false)); } // row lengths that are not a multiple of 32, for the scalar (33) and float4 (132, 260) paths for (uint32_t n : { 33, 132, 260 }) { @@ -8989,6 +9754,11 @@ static std::vector> make_test_cases_eval() { // in-place tests test_cases.emplace_back(new test_rms_norm(GGML_TYPE_F32, {64, 5, 4, 3}, false, 1e-6f, true)); + for (ggml_type set_rows_type : { GGML_TYPE_F32, GGML_TYPE_F16 }) { + test_cases.emplace_back(new test_rms_norm_mul_rope({ 256, 1, 1, 1 }, 1e-6f, false, true, false, GGML_ROPE_TYPE_NORMAL, false, false, set_rows_type)); + test_cases.emplace_back(new test_rms_norm_mul_rope({ 128, 4, 3, 1 }, 1e-6f, false, true, false, GGML_ROPE_TYPE_NORMAL, false, false, set_rows_type)); + } + for (float eps : { 0.0f, 1e-6f, 1e-4f, 1e-1f, 1.0f }) { for (uint32_t n : { 64, 1025 }) { test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { n, 5, 4, 3 }, eps, false)); @@ -9005,11 +9775,29 @@ static std::vector> make_test_cases_eval() { } test_cases.emplace_back(new test_add_rms_norm(GGML_TYPE_F32, {n, 1, 1, 1}, 1e-6f, false)); } + for (uint32_t n : {64, 1025}) { + test_cases.emplace_back(new test_add_add(GGML_TYPE_F32, GGML_TYPE_F32, { n, 5, 4, 3 }, false, false)); + test_cases.emplace_back(new test_add_add(GGML_TYPE_F32, GGML_TYPE_F32, { n, 5, 4, 3 }, true, false)); + test_cases.emplace_back(new test_add_add(GGML_TYPE_F32, GGML_TYPE_F32, { n, 5, 4, 3 }, false, true)); + test_cases.emplace_back(new test_add_add(GGML_TYPE_F16, GGML_TYPE_F16, { n, 5, 4, 3 }, false, false)); + test_cases.emplace_back(new test_add_add(GGML_TYPE_F16, GGML_TYPE_F32, { n, 5, 4, 3 }, false, false)); + test_cases.emplace_back(new test_add_add(GGML_TYPE_F16, GGML_TYPE_F32, { n, 5, 4, 3 }, true, false)); + } + + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 1536, 1, 1, 1 }, 1e-6f, false, false, true)); + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 256, 4, 1, 1 }, 1e-6f, false, false, true)); + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 256, 4, 3, 2 }, 1e-6f, false, false, true)); + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 256, 4, 3, 2 }, 1e-6f, false, false, true, false, true)); + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 1536, 1, 1, 1 }, 1e-6f, false, false, false, true)); + test_cases.emplace_back(new test_rms_norm_mul_add(GGML_TYPE_F32, { 256, 4, 1, 1 }, 1e-6f, false, false, false, true)); + + test_cases.emplace_back(new test_rms_norm_mul_rope({128, 4, 7, 2})); + test_cases.emplace_back(new test_rms_norm_mul_rope({128, 4, 7, 2}, 1e-6f, false, true)); for (auto multi_add : {false, true}) { for (auto set_rows : {false, true}) { for (auto broadcast : {false, true}) { - for (auto rope : {GGML_ROPE_TYPE_NORMAL, GGML_ROPE_TYPE_NEOX}) { + for (auto rope : {GGML_ROPE_TYPE_NORMAL, GGML_ROPE_TYPE_NEOX, GGML_ROPE_TYPE_IMROPE}) { test_cases.emplace_back(new test_rms_norm_mul_rope({768, 1, 1, 1}, 1e-6f, multi_add, set_rows, broadcast, rope)); test_cases.emplace_back(new test_rms_norm_mul_rope({768, 3, 1, 1}, 1e-6f, multi_add, set_rows, broadcast, rope)); test_cases.emplace_back(new test_rms_norm_mul_rope({768, 3, 5, 1}, 1e-6f, multi_add, set_rows, broadcast, rope)); @@ -9065,6 +9853,10 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 4, 2, false, /*K=*/4)); // Mamba-2 rollback snapshots test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 8, 2, false, /*K=*/3)); // Mamba-2 rollback overflow test_cases.emplace_back(new test_ssm_scan_rollback(GGML_TYPE_F32, 128, 64, 16, 2, 8, 2, /*K=*/3)); // rollback snapshots match prefix states + test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 64, 4)); // Metal SSD one chunk MMA only, no seq tail + test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 65, 2)); // SSD one chunk + 1-token sequential tail + test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 128, 2)); // SSD multi-chunk, no tail (exercises the chunk-to-chunk state handoff) + test_cases.emplace_back(new test_ssm_scan(GGML_TYPE_F32, 128, 64, 16, 2, 128, 2, false, /*K=*/1, /*weak_decay=*/true)); // SSD multi-chunk, carried state not numerically negligible test_cases.emplace_back(new test_rwkv_wkv6(GGML_TYPE_F32, 32, 64, 1, 1)); test_cases.emplace_back(new test_rwkv_wkv6(GGML_TYPE_F32, 32, 64, 32, 1)); @@ -9092,6 +9884,13 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F32, 256, 512, 256)); // many rows test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F32, 32, 1, 32)); // too small (N<64) test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F32, 1024, 1, 1024)); // too big (N>512) + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 64, 1, 64)); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 128, 1, 128)); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 256, 1, 256)); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 512, 1, 512)); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 128, 32, 128)); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 128, 4, 128, {2, 3})); + test_cases.emplace_back(new test_mul_mat_hadamard(GGML_TYPE_F32, GGML_TYPE_F16, 256, 512, 256)); // many rows #if 0 // > 4GB A matrix. Too slow to be enabled by default. @@ -9116,6 +9915,25 @@ static std::vector> make_test_cases_eval() { //test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 18, i, 32*256, { 1, 1}, {8, 1})); //test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 19, i, 33*256, { 1, 1}, {1, 1})); } + // mat-vec shaders split k across lanes and loop over the blocks in strides. k must be + // long enough that the loop wraps, else the stride is never exercised + test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 16, 1, 16*256, { 1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 16, 8, 16*256, { 1, 1}, {1, 1})); + } + + // Multi-column MMVQ coverage for the Q4_K weight-reuse path and a Q5_K control. + for (ggml_type type_a : { GGML_TYPE_Q4_K, GGML_TYPE_Q5_K }) { + for (int n = 1; n <= 8; ++n) { + test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 4096, n, 1024, { 1, 1 }, { 1, 1 })); + test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 1023, n, 4096, { 1, 1 }, { 1, 1 })); + } + } + + // The SYCL backend picks between one and two output rows per subgroup by row count when there + // are two destination columns (Q4_K_MMVQ_ROW_PAIR_MIN_NROWS in ggml-sycl/mmvq.cpp). Cover both + // sides of that boundary, including an odd row count above it for the row-pair tail. + for (int64_t m : {6271, 6272, 6273}) { + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_K, GGML_TYPE_F32, m, 2, 1024, { 1, 1 }, { 1, 1 })); } test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 2880, 32, 2880, {1, 1}, {1, 1})); @@ -9123,9 +9941,17 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat(GGML_TYPE_MXFP4, GGML_TYPE_F32, 2880, 32, 2880, {1, 1}, {1, 1})); // m == 1, with n on both sides of MMVF_MAX_BATCH_SIZE (8): mmvf below, operand swap above - for (int64_t n : {1, 7, 8, 9, 16, 128, 512}) { + for (int64_t n : {1, 7, 8, 9, 16, 127, 128, 511, 512}) { test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 1, n, 2048, {1, 1}, {1, 1})); } + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 1, 512, 2048, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 1, 512, 2048, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 1, 509, 2051, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 1, 509, 2051, {1, 1}, {1, 1})); + + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 31, 509, 2051, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 32, 509, 2112, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q8_0, GGML_TYPE_F32, 32, 509, 2112, {1, 1}, {1, 1})); #if 0 { @@ -9206,6 +10032,19 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat(GGML_TYPE_BF16, GGML_TYPE_F32, 16, 16, 256, {2, 3}, {1, 1}, {0, 1, 3, 2})); test_cases.emplace_back(new test_mul_mat(GGML_TYPE_BF16, GGML_TYPE_F32, 16, 16, 256, {2, 3}, {1, 1}, {0, 3, 2, 1})); + // token-tile boundary coverage. With n_used == n_mats every token routes to every expert, so + // each expert receives exactly n rows, with no dependence on the random draw. mul_mm_id is used + // from 32 tokens up: n = 32, 33, 47, 48, 49 reach it, leaving a last tile of 32, 1, 15, 16 and + // 17 rows - 16 and 17 straddle the point where the upper half stops being skipped. The smaller + // n cover the same row counts on the mat-vec path. + for (ggml_type type_a : {GGML_TYPE_Q4_K, GGML_TYPE_IQ2_XS, GGML_TYPE_F16}) { + for (int n : {1, 15, 16, 17, 31, 32, 33, 47, 48, 49}) { + test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 4, 4, false, 512, n, 256)); + } + // experts that receive no rows at all + test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 8, 1, false, 512, 1, 256)); + } + for (ggml_type type_a : other_types) { for (ggml_type type_b : {GGML_TYPE_F32}) { if (ggml_blck_size(type_a) != 256) { @@ -9214,6 +10053,12 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat(type_a, type_b, 16, 1, 256, {1, 1}, {1, 1})); } } + + // Test IQP panel path for all grid IQ types + for (ggml_type type_a : {GGML_TYPE_IQ2_XXS, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ2_S, GGML_TYPE_IQ3_XXS, + GGML_TYPE_IQ3_S, GGML_TYPE_IQ1_S, GGML_TYPE_IQ1_M, GGML_TYPE_IQ4_XS}) { + test_cases.emplace_back(new test_mul_mat(type_a, GGML_TYPE_F32, 16, 10, 256, {1, 1}, {1, 1})); + } #else // m = a rows // n = b rows @@ -9243,6 +10088,7 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 1056, 1, 67, {1, 1}, {4, 1}, {0, 2, 1, 3})); test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 16, 32, 32, { 1, 1}, {1, 1}, {0, 1, 2, 3}, 64, 3)); test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 77, 77, {12,1}, {1,1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 32, 4, 96, {3, 2}, {1, 1}, {0, 1, 2, 3}, 0, 1, true)); test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 576, 512, 576, {1,1}, {1,1})); test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 1, 2048, 8192, {1, 1}, {1, 1})); @@ -9252,6 +10098,14 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q8_0, GGML_TYPE_F32, 6, 4096, 5120, {1, 1}, {1, 1})); + // K not a multiple of 32 + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 65, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1})); + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 588, {1, 1}, {1, 1})); // 14*14*3, e.g. conv_2d im2col + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {4, 1}, {1, 1})); + #if 0 // test the mat-mat path for Metal for (int k = 1; k < 512; ++k) { @@ -9300,17 +10154,47 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_BF16, GGML_TYPE_F32, 16, 16, b, 50, 200, 64)); } - test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_F16, GGML_TYPE_F32, 1, 1, false, 8, 16, 1)); + // For issue 27873 + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_IQ2_XXS, GGML_TYPE_F32, 1, 1, false, 1, 8192, 4096)); + + for (int k : {1, 63, 65}) { + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_F16, GGML_TYPE_F32, 1, 1, false, 8, 16, k)); + } test_cases.emplace_back(new test_mul_mat_id_fusion(GGML_TYPE_F16, GGML_TYPE_F32, 16, 16, false, 32, 32, 32, 3)); // gpt-oss issue with Vulkan mmq_id test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_MXFP4, GGML_TYPE_F32, 32, 2, false, 2880, 32, 2880)); + // more than 256 experts (hoisted row-id path): 512 as in Qwen3.8-Flash-Next, + // and 1024 at the LLAMA_MAX_EXPERTS limit + for (int n : {1, 5, 64, 300}) { + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_IQ3_S, GGML_TYPE_F32, 512, 10, false, 128, n, 512)); + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q4_0, GGML_TYPE_F32, 512, 10, false, 256, n, 128)); + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_IQ3_S, GGML_TYPE_F32, 1024, 10, false, 128, n, 512)); + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q4_0, GGML_TYPE_F32, 1024, 10, false, 256, n, 128)); + } test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q4_0, GGML_TYPE_F32, 32, 2, false, 2880, 32, 2880)); + // multiple blocks per row: exercises the block-stride loop and the + // per-expert base offset, which k == 256 alone leaves untested + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_TQ1_0, GGML_TYPE_F32, 28, 10, false, 1024, 1, 4096)); + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_TQ1_0, GGML_TYPE_F32, 128, 8, false, 1024, 1, 2048)); + for (ggml_type type_a : all_types) { test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 4, 2, false, 64, 16, 3*ggml_blck_size(type_a))); } + // Test IQP panel path for all grid IQ types + for (ggml_type type_a : {GGML_TYPE_IQ2_XXS, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ2_S, GGML_TYPE_IQ3_XXS, + GGML_TYPE_IQ3_S, GGML_TYPE_IQ1_S, GGML_TYPE_IQ1_M, GGML_TYPE_IQ4_XS}) { + test_cases.emplace_back(new test_mul_mat_id(type_a, GGML_TYPE_F32, 4, 4, false, 16, 10, 256)); + } + + // test src1 f16 overflow + for (int n : {16, 32, 64}) { + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q4_K, GGML_TYPE_F32, 128, 4, false, 4096, n, 2048, 1e5f)); + test_cases.emplace_back(new test_mul_mat_id(GGML_TYPE_Q8_0, GGML_TYPE_F32, 8, 2, false, 512, n, 256, 1e5f)); + } + for (ggml_type type_a : base_types) { for (ggml_type type_b : {GGML_TYPE_F32 /*, GGML_TYPE_F16 */}) { for (int n_mats : {4, 8}) { @@ -9495,6 +10379,8 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {200001, 2, 3, 1}, true, true, GGML_TYPE_F16, {1, 1}, 0.1f, 8.0f)); test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {200000, 1, 1, 1}, false, false, GGML_TYPE_F32, {1, 1}, 1.0f, 0.0f)); test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {200000, 4, 1, 1}, false, false, GGML_TYPE_F32, {1, 1}, 1.0f, 0.0f)); + test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {4, 1, 1, 1}, false, false, GGML_TYPE_F32, {1, 1}, 1.0f, 0.0f)); + test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {4, 1023, 1, 1}, false, false, GGML_TYPE_F32, {1, 1}, 1.0f, 0.0f)); test_cases.emplace_back(new test_soft_max(GGML_TYPE_F32, {643251, 3, 1, 1}, false, false, GGML_TYPE_F32, {1, 1}, 1.0f, 0.0f)); for (float max_bias : {0.0f, 8.0f}) { @@ -9585,6 +10471,26 @@ static std::vector> make_test_cases_eval() { } } + // rotated dims window at an offset (ggml_rope_set_offset), not supported for vision mode + for (ggml_type type : {GGML_TYPE_F32, GGML_TYPE_F16}) { + for (bool fw : {true, false}) { // fw == forward + for (bool ff : {false, true}) { + test_cases.emplace_back(new test_rope(type, {128, 32, 2, 1}, 32, GGML_ROPE_TYPE_NORMAL, 512, 1.4245f, 0.7465f, 1.4245f, ff, 0, fw, false, 32)); + test_cases.emplace_back(new test_rope(type, {128, 32, 2, 1}, 32, GGML_ROPE_TYPE_NEOX, 512, 1.4245f, 0.7465f, 1.4245f, ff, 0, fw, false, 32)); + test_cases.emplace_back(new test_rope(type, {128, 12, 2, 1}, 24, GGML_ROPE_TYPE_MROPE, 512, 1.4245f, 0.7465f, 1.4245f, ff, 0, fw, false, 32)); + test_cases.emplace_back(new test_rope(type, {128, 12, 2, 1}, 24, GGML_ROPE_TYPE_IMROPE, 512, 1.4245f, 0.7465f, 1.4245f, ff, 0, fw, false, 32)); + } + } + // inplace with an offset + test_cases.emplace_back(new test_rope(type, {128, 32, 2, 1}, 32, GGML_ROPE_TYPE_NEOX, 512, 1.4245f, 0.7465f, 1.4245f, false, 0, true, true, 32)); + } + + // Real-model RoPE: F32 forward, packed Q, 512-token prefill. + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 8, 512, 1}, 64, GGML_ROPE_TYPE_IMROPE, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // qwen3.5 0.8B + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 16, 512, 1}, 64, GGML_ROPE_TYPE_IMROPE, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // qwen3.5 4B + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 8, 512, 1}, 256, GGML_ROPE_TYPE_NEOX, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // gemma4 E2B sliding + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {512, 8, 512, 1}, 128, GGML_ROPE_TYPE_NEOX, 512, 1.0f, 0.0f, 1.0f, true, 0, true)); // gemma4 E4B global + for (int v : { 0, 1, 2, 3 }) { for (int dim : { 0, 1, 2, 3, }) { test_cases.emplace_back(new test_concat(GGML_TYPE_F32, {11, 12, 13, 14}, 7, dim, v)); @@ -9637,6 +10543,17 @@ static std::vector> make_test_cases_eval() { } } } + for (int k : {4, 8, 16, 32}) { + for (int nrows : {1, 8, 16}) { + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {202048, nrows, 1, 1}, k)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {151936, nrows, 1, 1}, k)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {8192, nrows, 1, 1}, k)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {8193, nrows, 1, 1}, k)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {8192, nrows, 1, 1}, k, true)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {202048, nrows, 1, 1}, k, true)); + } + } + for (int k : {1, 2, 3, 7, 15}) { test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {16, 10, 10, 10}, k)); test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {60, 10, 10, 10}, k)); @@ -9649,6 +10566,22 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {2049, 2, 1, 3}, k)); } + // Large-k, including multi-row and ties (qwen4exp) + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 1024, 1, 1, 1 }, 1024)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 2048, 2, 1, 1 }, 1024)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 4096, 1, 1, 1 }, 2048)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 8192, 2, 1, 1 }, 2051)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 33024, 1, 1, 1 }, 2051)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 33024, 4, 1, 1 }, 2051)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 8192, 2, 1, 1 }, 2051, true)); + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, { 33024, 4, 1, 1 }, 2051, true)); + + // qwen4exp QSA indexer top-k fusion (get_rows + f16 mask + top_k) + test_cases.emplace_back(new test_topk_qsa(512, 2048, 1, 1, 1500)); + test_cases.emplace_back(new test_topk_qsa(512, 2048, 2, 1, 1500)); + test_cases.emplace_back(new test_topk_qsa(256, 2048, 4, 2, 2000)); + test_cases.emplace_back(new test_topk_qsa(64, 256, 2, 1, 200)); // small k: unfused fallback + // exhaustive top_k tests //for (int i = 1; i < 9999; ++i) { // test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {i, 2, 1, 3}, rand() % i + 1)); @@ -9677,6 +10610,9 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 32, 1, 1, 1 })); test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 32, 256, 1, 1 })); test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 32768, 1, 1, 1 })); + test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 11, 5, 6, 3 }, true, false)); + test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 11, 5, 6, 3 }, false, true)); + test_cases.emplace_back(new test_mean(GGML_TYPE_F32, { 11, 5, 6, 3 }, true, true)); test_cases.emplace_back(new test_sum(GGML_TYPE_F32, { 33, 1, 1, 1 })); test_cases.emplace_back(new test_sum(GGML_TYPE_F32, { 33, 1024, 1, 1 })); test_cases.emplace_back(new test_sum(GGML_TYPE_F32, { 33, 256, 1, 1 })); @@ -9809,7 +10745,8 @@ static std::vector> make_test_cases_eval() { for (int hsk : { 40, 64, 72, 80, 96, 128, 192, 256, 320, 512, 576 }) { for (int hsv : { 40, 64, 72, 80, 96, 128, 192, 256, 512 }) { - if (hsk != 192 && hsk != 320 && hsk != 576 && hsk != hsv) continue; + if (hsk != 96 && hsk != 192 && hsk != 320 && hsk != 576 && hsk != hsv) continue; + if (hsk == 96 && (hsv != 64 && hsv != 96)) continue; // MiniCPM3 if (hsk == 192 && (hsv != 128 && hsv != 192)) continue; if (hsk == 576 && hsv != 512) continue; // DeepSeek MLA if (hsk == 320 && hsv != 256) continue; // Mistral4 MLA @@ -9838,12 +10775,14 @@ static std::vector> make_test_cases_eval() { if (hsk != 128 && prec == GGML_PREC_DEFAULT) continue; for (ggml_type type_KV : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16, GGML_TYPE_Q8_0, GGML_TYPE_Q5_1, GGML_TYPE_Q5_0, GGML_TYPE_Q4_1, GGML_TYPE_Q4_0, GGML_TYPE_IQ4_NL}) { if (type_KV != GGML_TYPE_F16 && hsk != 64 && hsk != 72) continue; + // DeepSeek MLA: the V cache is a sub-view of the K cache + const bool v_is_view_of_k = hsk == 576; test_cases.emplace_back(new test_flash_attn_ext( - hsk, hsv, nh, {nr2, nr3}, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_KV, type_KV)); + hsk, hsv, nh, {nr2, nr3}, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_KV, type_KV, {0, 1, 2, 3}, true, v_is_view_of_k)); // run fewer test cases permuted if (mask == true && max_bias == 0.0f && logit_softcap == 0 && kv == 512) { test_cases.emplace_back(new test_flash_attn_ext( - hsk, hsv, nh, {nr2, nr3}, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_KV, type_KV, {0, 2, 1, 3})); + hsk, hsv, nh, {nr2, nr3}, kv, nb, mask, sinks, max_bias, logit_softcap, prec, type_KV, type_KV, {0, 2, 1, 3}, true, v_is_view_of_k)); } } } @@ -9859,6 +10798,10 @@ static std::vector> make_test_cases_eval() { } } + // asymmetric head_dim (hsk != hsv) with one or both sides not 64-aligned + test_cases.emplace_back(new test_flash_attn_ext(72, 64, 4, {1, 1}, 256, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(64, 72, 4, {1, 1}, 256, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + // mixed quant and Q1_0 test cases test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 128, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q4_0)); test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 128, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q4_0, GGML_TYPE_F16)); @@ -9874,6 +10817,57 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_flash_attn_ext(64, 128, 4, {1, 1}, 128, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q4_0, GGML_TYPE_Q2_0)); test_cases.emplace_back(new test_flash_attn_ext(128, 64, 4, {1, 1}, 64, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q2_0, GGML_TYPE_F16)); + // q8_0 KV cases: decode and prompt batches, KV pad, permuted KV, feature flags, and long context + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 113, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 1024, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 1024, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 2, 1, 3})); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 2}, 1025, 1, true, true, 8, 30, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 1025, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 2, 1, 3})); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 16384, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + + // MLA shape: the V cache is a sub-view of the K cache, with quantized KV + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {8, 1}, 113, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, true)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {8, 1}, 1024, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, true)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {8, 1}, 1024, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, true)); + + // Sparse mask hint: supported decode/prefill layouts and dense fallbacks. + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 2}, 4096, 3, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 768)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 512)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 2}, 4096, 2, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 768)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2304)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 1, { 8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 4, true, false, 8.0f, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + + // sparse mask with large batch size + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 512)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 2048)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + + // sparse attn (qwen4 shape - gqa 12) + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {12, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {12, 1}, 8192, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 1, {12, 2}, 8192, 67, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + + // sparse mask + quantized cache + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 1, { 8, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, false, 512)); + + // Qwen QSA: 256/256, gqa 12, budget 2048. + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {12, 1}, 8192, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + + // more V-is-sub-view-of-K cases: other head shapes, and full views with equal head sizes + test_cases.emplace_back(new test_flash_attn_ext(320, 256, 1, {32, 1}, 512, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true)); + test_cases.emplace_back(new test_flash_attn_ext(192, 128, 4, {8, 1}, 512, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 8, {4, 1}, 512, 8, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true)); + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 512, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, true, true)); + // large-KV F16 cases (Qwen3.6-27B geometry and a llama-class control): the upstream matrix // stops at kv=1024, blind to long-context FA bugs (e.g. the oneDNN SDPA ordering race on BMG). for (int64_t kv : { 4096, 16384 }) { @@ -9883,6 +10877,29 @@ static std::vector> make_test_cases_eval() { GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); } + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 512, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 4096, 64, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 4096, 16, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {2, 1}, 4096, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {4, 1}, 4096, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {12, 1}, 4096, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + + // dense-allocated (non-view) quant K/V at batch >= 64, in cache and native layouts + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 512, 75, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 2, 1, 3}, false)); + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {4, 1}, 512, 75, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 2, 1, 3}, false)); + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 1024, 75, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 2, 1, 3}, false)); + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 4, {1, 1}, 512, 75, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0, {0, 1, 2, 3}, false)); + + // FLASH_ATTN_EXT MMA: non-pow2 head size and MLA K/V view. + test_cases.emplace_back(new test_flash_attn_ext(192, 128, 8, {8, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {20, 1}, 512, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true)); + + // FLASH_ATTN_EXT MMA, swizzled K/V tiles, power-of-two stride: nbatch_K2 = 32, 64, 128, 256. + test_cases.emplace_back(new test_flash_attn_ext( 64, 64, 8, {8, 1}, 4096, 4, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(128, 128, 8, {4, 1}, 4096, 8, true, true, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {2, 1}, 1024, 32, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 4, {2, 1}, 1024, 4, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_cross_entropy_loss (GGML_TYPE_F32, { 10, 5, 4, 3})); test_cases.emplace_back(new test_cross_entropy_loss (GGML_TYPE_F32, {30000, 1, 1, 1})); test_cases.emplace_back(new test_cross_entropy_loss_back(GGML_TYPE_F32, { 10, 5, 4, 3})); @@ -9902,7 +10919,7 @@ static std::vector> make_test_cases_eval() { if (!with_gate && !with_bias) { continue; } - for (ggml_glu_op glu_op : {GGML_GLU_OP_SWIGLU, GGML_GLU_OP_GEGLU}) { + for (ggml_glu_op glu_op : {GGML_GLU_OP_SWIGLU, GGML_GLU_OP_GEGLU, GGML_GLU_OP_SWIGLU_CLAMP}) { if (!with_bias && glu_op == GGML_GLU_OP_SWIGLU_OAI) { continue; } @@ -9917,12 +10934,10 @@ static std::vector> make_test_cases_eval() { use_id, 16, 8, b, with_bias, with_gate, with_lane_scale)); test_cases.emplace_back(new test_mul_mat_vec_fusion(type, glu_op, 1, 32, 256, use_id, 16, 8, b, with_bias, with_gate, with_lane_scale, {1, 1})); - if (!use_id && with_gate && !with_bias) { - // small multi-token batches (speculative decoding / MTP verify) - for (int64_t m_batch : { 2, 4, 8 }) { - test_cases.emplace_back(new test_mul_mat_vec_fusion(type, glu_op, m_batch, 32, 256, - use_id, 16, 8, b, with_bias, with_gate, with_lane_scale, {1, 1})); - } + // multi-token batches (spec decoding) + for (int64_t m_batch : { 2, 4, 8 }) { + test_cases.emplace_back(new test_mul_mat_vec_fusion(type, glu_op, m_batch, 32, 256, + use_id, 16, 8, b, with_bias, with_gate, with_lane_scale, {1, 1})); } } } @@ -9932,6 +10947,28 @@ static std::vector> make_test_cases_eval() { } } + for (bool b : {false, true}) { + test_cases.emplace_back(new test_mul_mat_vec_fusion(GGML_TYPE_IQ2_S, GGML_GLU_OP_SWIGLU_CLAMP, 1, 32, 256, + true, 16, 8, b, false, true, false)); + } + + // Fused row-pair coverage: minimum rows, an even pair, and an odd tail. + // TODO: the max_nmse_err() for these cases is not estimated correctly causing sporadic false failures. + //for (ggml_glu_op glu_op : { GGML_GLU_OP_SWIGLU, GGML_GLU_OP_GEGLU }) { + // for (int64_t m_batch : { 2, 3, 4 }) { + // for (int64_t rows : { 1, 2, 3 }) { + // test_cases.emplace_back(new test_mul_mat_vec_fusion(GGML_TYPE_Q4_K, glu_op, m_batch, rows, 256, + // false, 16, 8, false, false, true, false, { 1, 1 })); + // } + // } + //} + + // Both sides of the same row-count boundary as above, on the fused path. + for (int64_t rows : {6271, 6272, 6273}) { + test_cases.emplace_back(new test_mul_mat_vec_fusion(GGML_TYPE_Q4_K, GGML_GLU_OP_SWIGLU, 2, rows, 256, + false, 16, 8, false, false, true, false, { 1, 1 })); + } + for (auto gate : {GATING_FUNC_SOFTMAX, GATING_FUNC_SIGMOID, GATING_FUNC_SOFTMAX_WEIGHT, GATING_FUNC_SQRT_SOFTPLUS}) { for (bool with_norm : {false, true}) { for (bool bias_probs : {false, true}) { @@ -9946,11 +10983,23 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_topk_moe({160, 4, 1, 1}, 160, with_norm, bias_probs, gate, scale_w)); test_cases.emplace_back(new test_topk_moe({256, 22, 1, 1}, 6, with_norm, bias_probs, gate, scale_w)); // Used by DeepSeek-V4 test_cases.emplace_back(new test_topk_moe({288, 22, 1, 1}, 8, with_norm, bias_probs, gate, scale_w)); // Used by StepFun 3.7 + // rows at and just past the limit where one block still covers all rows + test_cases.emplace_back(new test_topk_moe({32, 8, 1, 1}, 4, with_norm, bias_probs, gate, scale_w)); + test_cases.emplace_back(new test_topk_moe({32, 8, 1, 1}, 8, with_norm, bias_probs, gate, scale_w)); + test_cases.emplace_back(new test_topk_moe({32, 9, 1, 1}, 8, with_norm, bias_probs, gate, scale_w)); } } } } + // Cover the supported boundaries, common k = 8 shapes, interleaved views and adds, and k = 16 fallback. + test_cases.emplace_back(new test_moe_reduce(63, 2, 17)); + test_cases.emplace_back(new test_moe_reduce(2048, 8, 128)); + test_cases.emplace_back(new test_moe_reduce(2048, 8, 128, false, true)); + test_cases.emplace_back(new test_moe_reduce(63, 12, 33, true, true, true)); + test_cases.emplace_back(new test_moe_reduce(2048, 15, 40, false, true)); + test_cases.emplace_back(new test_moe_reduce(2048, 16, 32, false, true)); + test_cases.emplace_back(new test_gated_delta_net(GGML_TYPE_F32, 32, 128, 1, 1)); test_cases.emplace_back(new test_gated_delta_net(GGML_TYPE_F32, 32, 16, 1, 1)); test_cases.emplace_back(new test_gated_delta_net(GGML_TYPE_F32, 32, 16, 1, 1, 1, true, true)); @@ -9995,6 +11044,13 @@ static std::vector> make_test_cases_eval() { test_cases.emplace_back(new test_gated_delta_net(GGML_TYPE_F32, 4, 32, 8, 1, 1, false, false, /*K=*/3)); test_cases.emplace_back(new test_gated_delta_net(GGML_TYPE_F32, 4, 64, 16, 2, 1, false, false, /*K=*/4)); + // gdn + cache cpy fusion (K > 1) + test_cases.emplace_back(new test_gated_delta_net_cache_fusion(GGML_TYPE_F32, 4, 32, 2, 1, 2)); + test_cases.emplace_back(new test_gated_delta_net_cache_fusion(GGML_TYPE_F32, 4, 64, 4, 1, 2)); + test_cases.emplace_back(new test_gated_delta_net_cache_fusion(GGML_TYPE_F32, 4, 32, 4, 1, 4)); + test_cases.emplace_back(new test_gated_delta_net_cache_fusion(GGML_TYPE_F32, 8, 32, 4, 2, 4)); + test_cases.emplace_back(new test_gated_delta_net_cache_fusion(GGML_TYPE_F32, 4, 32, 8, 1, 4)); + #if 0 // these tests are disabled to save execution time, sbut they can be handy for debugging test_cases.emplace_back(new test_llama(2, true)); @@ -10042,6 +11098,22 @@ static std::vector> make_test_cases_perf() { } } + // CONT of a 0<->2 permute at DeepSeek-V4 lightning-indexer shapes: + // indexer_kq is [n_kv, n_tokens, n_head=64] and gets ggml_cont(ggml_permute(.., 2,1,0,3)). + for (int64_t n_kv : { 1024, 1280, 2048, 2304 }) { + test_cases.emplace_back(new test_cont( + GGML_TYPE_F32, {n_kv, 64, 64, 1}, false, {2, 1, 0, 3})); + } + for (int64_t n_kv : { 2048, 2304 }) { + test_cases.emplace_back(new test_cont( + GGML_TYPE_F32, {n_kv, 512, 64, 1}, false, {2, 1, 0, 3})); + } + + // LEAKY_RELU at FFN activation width, for direct comparison with RELU + for (int64_t n_tokens : {512, 2048}) { + test_cases.emplace_back(new test_leaky_relu(GGML_TYPE_F32, { 17408, n_tokens, 1, 1 }, 0.1f)); + } + // Conv2d: K=CRS=NPQ=4096 matmul performance uint32_t iwh_idx = 0; uint32_t kwh_idx = 1; @@ -10196,9 +11268,16 @@ static std::vector> make_test_cases_perf() { } } + // Q4_K multi-column mat-vec + for (int64_t m : {4096, 6144, 6272, 14336}) { + for (int bs : {1, 2, 3, 4, 8}) { + test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_K, GGML_TYPE_F32, m, bs, 4096, {1, 1}, {1, 1})); + } + } + // qwen3-30b-a3b for (int bs : {1, 4, 8, 32, 64, 128, 256, 512}) { - for (ggml_type type_a : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_Q4_0, GGML_TYPE_Q8_0, GGML_TYPE_Q4_K, GGML_TYPE_Q6_K, GGML_TYPE_IQ2_XS}) { + for (ggml_type type_a : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_Q4_0, GGML_TYPE_Q8_0, GGML_TYPE_Q4_K, GGML_TYPE_Q6_K, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ4_XS}) { for (ggml_type type_b : {GGML_TYPE_F32}) { test_cases.emplace_back(new test_mul_mat_id(type_a, type_b, 128, 8, false, 768, bs, 2048)); test_cases.emplace_back(new test_mul_mat_id_fusion(type_a, type_b, 128, 8, false, 768, bs, 2048, 1)); @@ -10207,7 +11286,7 @@ static std::vector> make_test_cases_perf() { } for (int bs : {1, 4, 8, 32, 64, 128, 256, 512}) { - for (ggml_type type_a : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_Q4_0, GGML_TYPE_Q8_0, GGML_TYPE_Q4_K, GGML_TYPE_Q6_K, GGML_TYPE_IQ2_XS}) { + for (ggml_type type_a : {GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_Q4_0, GGML_TYPE_Q8_0, GGML_TYPE_Q4_K, GGML_TYPE_Q6_K, GGML_TYPE_IQ2_XS, GGML_TYPE_IQ4_XS}) { for (ggml_type type_b : {GGML_TYPE_F32}) { test_cases.emplace_back(new test_mul_mat_id(type_a, type_b, 32, 4, false, 1792, bs, 2048)); test_cases.emplace_back(new test_mul_mat_id_fusion(type_a, type_b, 32, 4, false, 1792, bs, 2048, 1)); @@ -10241,6 +11320,14 @@ static std::vector> make_test_cases_perf() { // Qwen3-VL-8B https://github.com/ggml-org/llama.cpp/issues/17012 test_cases.emplace_back(new test_flash_attn_ext(72, 72, 16, {1, 1}, 5776, 5776, false, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + // Sparse flash attention (n_kv_max hint) decode across KV depths. + // Shapes: 576/512 DeepSeek MLA, 512/512 DeepSeek-V4/GLM-5.2, 256/256 gqa12 Qwen QSA. + for (int64_t kv : {4096, 16384, 32768}) { + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, kv, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 512)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, kv, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 512)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {12, 1}, kv, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + } + test_cases.emplace_back(new test_flash_attn_ext(64, 64, 8, {8, 1}, 7680, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); test_cases.emplace_back(new test_flash_attn_ext(64, 64, 8, {8, 1}, 7680, 4, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); test_cases.emplace_back(new test_flash_attn_ext(64, 64, 8, {8, 1}, 7680, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q4_0, GGML_TYPE_Q4_0)); @@ -10248,10 +11335,47 @@ static std::vector> make_test_cases_perf() { test_cases.emplace_back(new test_flash_attn_ext(64, 64, 8, {8, 1}, 7680, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); test_cases.emplace_back(new test_flash_attn_ext(64, 64, 8, {8, 1}, 7680, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); - for (int kv : { 4096, 8192, 16384, }) { - for (int hs : { 64, 128, }) { - for (int nr : { 1, 4, }) { - test_cases.emplace_back(new test_flash_attn_ext(hs, hs, 8, {nr, 1}, kv, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + // sparse decode at long context + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 0)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 1, { 8, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + // gemma-4-26b-a4b global-attn layers: head_count_kv=2, 16 query heads (gqa_ratio=8) + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 2, { 8, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 0)); + test_cases.emplace_back(new test_flash_attn_ext(512, 512, 2, { 8, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, false, 2048)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 0)); + test_cases.emplace_back(new test_flash_attn_ext(576, 512, 1, {16, 1}, 49152, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, true, 2048)); + + // q8_0 KV cases with long context (decode and prompt) + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 128, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 512, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 1024, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 2048, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 10000, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 20000, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 10000, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 20000, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 10000, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 20000, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_Q8_0, GGML_TYPE_Q8_0)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 10000, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 2, {16, 1}, 20000, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 4096, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 4096, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 16384, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 16384, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 65536, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 65536, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 131072, 1, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + test_cases.emplace_back(new test_flash_attn_ext(256, 256, 4, {6, 1}, 131072, 512, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16)); + + for (int kv : { 4096, 8192, 16384,32768, 65536, }) { + for (int hs : { 64, 128, 256, 576, }) { + const int hsv = hs == 576 ? 512 : hs; + const bool v_view = hs == 576; + for (int nr : { 1, 4, 8, }) { + for (int nb : { 1, 4096, }) { + test_cases.emplace_back(new test_flash_attn_ext(hs, hsv, 8, {nr, 1}, kv, nb, true, false, 0, 0, GGML_PREC_F32, GGML_TYPE_F16, GGML_TYPE_F16, {0, 1, 2, 3}, true, v_view)); + } } } } @@ -10304,6 +11428,12 @@ static std::vector> make_test_cases_perf() { } } + // Real-model RoPE: F32 forward, packed Q, 512-token prefill. + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 8, 512, 1}, 64, GGML_ROPE_TYPE_IMROPE, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // qwen3.5 0.8B + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 16, 512, 1}, 64, GGML_ROPE_TYPE_IMROPE, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // qwen3.5 4B + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {256, 8, 512, 1}, 256, GGML_ROPE_TYPE_NEOX, 512, 1.0f, 0.0f, 1.0f, false, 0, true)); // gemma4 E2B sliding + test_cases.emplace_back(new test_rope(GGML_TYPE_F32, {512, 8, 512, 1}, 128, GGML_ROPE_TYPE_NEOX, 512, 1.0f, 0.0f, 1.0f, true, 0, true)); // gemma4 E4B global + std::vector> reduce_rows_cases = { { 8192, 1, 1, 1 }, { 8192, 8192, 1, 1 }, @@ -10321,7 +11451,13 @@ static std::vector> make_test_cases_perf() { test_cases.emplace_back(new test_argsort(GGML_TYPE_F32, {200000, 16, 1, 1})); test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {2, 1, 1, 1}, 1)); - for (auto k : {1, 10, 40, 400}) { + // widths around the tiling threshold + for (auto cols : {4096, 8192, 12288, 16384, 24576, 32768, 65536, 131072}) { + for (auto nrows : {1, 16}) { + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {cols, nrows, 1, 1}, 16)); + } + } + for (auto k : {1, 4, 8, 10, 16, 32, 40, 400}) { for (auto nrows : {1, 16}) { for (auto cols : {k, 1000, 65000, 200000}) { test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {cols, nrows, 1, 1}, k)); @@ -10329,6 +11465,30 @@ static std::vector> make_test_cases_perf() { } } + // qwen4exp sparse-attention indexer: nrows = n_tokens/n_stream, so tg gives nrows==1. + // Sweep nrows to expose how much of the device a single row leaves idle. + for (auto cols : {8192, 32768, 131072}) { + for (auto nrows : {1, 2, 4, 8, 16, 32}) { + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {cols, nrows, 1, 1}, 2048)); + } + } + // backend sampler: one row of the vocab (llama-sampler.cpp top_k) + for (auto k : {20, 40}) { + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {151936, 1, 1, 1}, k)); + } + + // short rows, many of them: MoE routing and group selection. The opposite corner from + // the indexer, and the one where a work-group per row is the wasteful choice. + for (auto cols : {2, 16, 128, 1024}) { + for (auto nrows : {1024, 8192}) { + for (auto k : {1, 2, 8, 16, 32}) { + if (k <= cols) { + test_cases.emplace_back(new test_top_k(GGML_TYPE_F32, {cols, nrows, 1, 1}, k)); + } + } + } + } + for (auto nrows : {1, 4, 8, 16}) { for (auto cols : {128, 1024, 4096, 8192, 16384, 32768, 65536, 131072, 200000, 2000000}) { test_cases.emplace_back(new test_cumsum(GGML_TYPE_F32, {cols, nrows, 1, 1})); @@ -10384,6 +11544,16 @@ static std::vector> make_test_cases_perf() { } } + // launch-overhead isolation: single L2_NORM launch vs batched siblings at the GDN + // production shape (strided qkv views) -- perf-mode only, the eval list has its own + // 2/4-wide coverage + for (int n : { 128, 256 }) { + test_cases.emplace_back(new test_l2_norm(GGML_TYPE_F32, { n, 16, 16, 1 }, 1e-12f, false, false)); + test_cases.emplace_back(new test_l2_norm_batch(GGML_TYPE_F32, { n, 16, 16, 1 }, 2, 1e-12f, true)); + test_cases.emplace_back(new test_l2_norm_batch(GGML_TYPE_F32, { n, 16, 16, 1 }, 4, 1e-12f, true)); + } + + return test_cases; } @@ -10451,6 +11621,100 @@ static std::vector> make_test_cases_from_file(const c return test_cases; } +// ---- FA vec (Q,NE): forced-config numerical slice (Metal only) ---- +using set_fa_vec_override_t = void (*)(int, int); +using clear_fa_vec_override_t = void (*)(void); + +// NL = 32/NE must divide both dk/4 and dv/4. +static std::vector fa_vec_legal_ne(int dk, int dv) { + std::vector r; + for (int ne : {1, 2, 4}) { + const int nl = 32 / ne; + if ((dk/4) % nl == 0 && (dv/4) % nl == 0) { + r.push_back(ne); + } + } + return r; +} + +static bool op_names_filter_selects(const char * op_names_filter, const char * op_name) { + if (op_names_filter == nullptr) { + return true; + } + for (const auto & entry : op_filter_entries(op_names_filter)) { + // a full test case string is matched by its op name prefix + const auto lparen_pos = entry.find_first_of('('); + const auto op_entry = lparen_pos != std::string_view::npos ? entry.substr(0, lparen_pos) : entry; + if (op_filter_entry_matches(op_entry, op_name)) { + return true; + } + } + return false; +} + +// Covers padded rows, sinks, kvpad, multi-SIMDgroup reduction, quantized K/V, and MLA views. +// The override is backend-global, so this runs after all parallel workers have joined. +static bool run_fa_vec_slice(ggml_backend_t backend, ggml_backend_t backend_cpu, const char * op_names_filter) { + const char * LLAMA_TEST_FA_VEC_DISABLE = getenv("LLAMA_TEST_FA_VEC_DISABLE"); + if (LLAMA_TEST_FA_VEC_DISABLE) { + return true; + } + + if (!op_names_filter_selects(op_names_filter, "FLASH_ATTN_EXT")) { + return true; + } + + auto * reg = ggml_backend_dev_backend_reg(ggml_backend_get_device(backend)); + + auto set_ov = (set_fa_vec_override_t) ggml_backend_reg_get_proc_address(reg, "ggml_backend_metal_tuning_set_fa_vec_override"); + auto clear_ov = (clear_fa_vec_override_t) ggml_backend_reg_get_proc_address(reg, "ggml_backend_metal_tuning_clear_fa_vec_override"); + if (!set_ov || !clear_ov) { + return true; // not the Metal backend: nothing to force + } + + printf("Running FA vec slice tests (env LLAMA_TEST_FA_VEC_DISABLE=1 to skip)\n"); + + struct shape_t { int dk, dv; }; + const shape_t shapes[] = { { 128, 128 }, { 576, 512 } }; // mainstream head size + MLA shared K/V view + const int ne01_pts[] = { 1, 3 }; // decode, and padded rows for Q=2 and Q=4 + const int ne11_pts[] = { 512, 4097 }; // nsg=1, and nsg>=2 together with kvpad + const ggml_type types[] = { GGML_TYPE_F16, GGML_TYPE_Q4_0 }; + + int n_run = 0; + int n_fail = 0; + for (auto s : shapes) { + for (int ne : fa_vec_legal_ne(s.dk, s.dv)) { + for (int Q : { 1, 2, 4 }) { + for (ggml_type type_kv : types) { + for (bool sinks : { false, true }) { + for (int ne01 : ne01_pts) { + for (int ne11 : ne11_pts) { + set_ov(Q, ne); + test_flash_attn_ext tc(s.dk, s.dv, /*nh=*/4, { 1, 1 }, /*kv=*/ne11, /*nb=*/ne01, + /*mask=*/true, sinks, 0.0f, 0.0f, GGML_PREC_F32, + type_kv, type_kv); + auto st = tc.eval(backend, backend_cpu, "FLASH_ATTN_EXT", nullptr); + clear_ov(); + + if (st == test_status_t::FAIL) { + printf(" FAIL fa_vec slice: dk=%d dv=%d Q=%d ne=%d type=%s ne01=%d ne11=%d sinks=%d\n", + s.dk, s.dv, Q, ne, ggml_type_name(type_kv), ne01, ne11, (int) sinks); + n_fail++; + } + n_run++; + } + } + } + } + } + } + } + + printf(" fa_vec (Q,NE) slice: %d cases run, %d failed\n", n_run, n_fail); + + return n_fail == 0; +} + static bool test_backend(ggml_backend_t backend, ggml_backend_dev_t dev, test_mode mode, const char * op_names_filter, const char * params_filter, printer * output_printer, const char * test_file_path, int parallel_workers) { auto filter_test_cases = [](std::vector> & test_cases, const char * params_filter) { @@ -10588,7 +11852,9 @@ static bool test_backend(ggml_backend_t backend, ggml_backend_dev_t dev, test_mo output_printer->print_summary(test_summary_info(n_ok, tests_run, false)); output_printer->print_failed_tests(failed_tests); - return n_ok == tests_run; + const bool slice_ok = run_fa_vec_slice(backend, backend_cpu.get(), op_names_filter); + + return n_ok == tests_run && slice_ok; } if (mode == MODE_GRAD) { @@ -10735,20 +12001,29 @@ static void show_test_coverage() { } static void usage(char ** argv) { - printf("Usage: %s [mode] [-o ] [-b ] [-p ] [--output ] [--list-ops]", argv[0]); - printf(" [--show-coverage] [--test-file ] [-j ]\n"); - printf(" valid modes:\n"); - printf(" - test (default, compare with CPU backend for correctness)\n"); - printf(" - grad (compare gradients from backpropagation with method of finite differences)\n"); - printf(" - perf (performance evaluation)\n"); - printf(" - support (probe backend operation support)\n"); - printf(" op names for -o are as given by ggml_op_desc() (e.g. ADD, MUL_MAT, etc),\n"); - printf(" optionally including the full test case string (e.g. \"ADD(type=f16,ne=[1,1,8,1],nr=[1,1,1,1],nf=1)\")\n"); - printf(" --output specifies output format (default: console, options: console, sql, csv)\n"); - printf(" --list-ops lists all available GGML operations\n"); - printf(" --show-coverage shows test coverage\n"); - printf(" --test-file reads test operators from a test file generated by test-export-graph-ops\n"); - printf(" -j runs tests using parallel worker threads (default: 1, test mode only)\n"); + printf("Usage: %s [mode] [options]\n\n", argv[0]); + printf("Valid modes:\n"); + printf(" test (default) compare with CPU backend for correctness\n"); + printf(" grad compare gradients from backpropagation with method of finite differences\n"); + printf(" perf performance evaluation\n"); + printf(" support probe backend operation support\n\n"); + printf("Options:\n"); + printf(" -o comma separated list of exact op names (as given by ggml_op_desc()),\n"); + printf(" full test case strings, and/or regexes matched against the op name\n"); + printf(" -b run tests on the given backend (e.g. CPU, MTL0, CUDA0)\n"); + printf(" -p filter test cases by a regex matched against their params\n"); + printf(" --output output format (default: console)\n"); + printf(" --list-ops list all available GGML operations\n"); + printf(" --show-coverage show test coverage\n"); + printf(" --test-file read test operators from a test file generated by test-export-graph-ops\n"); + printf(" -j run tests using parallel worker threads (default: 1, test mode only)\n\n"); + printf("Examples:\n"); + printf(" %s -j 8\n", argv[0]); + printf(" %s -o ADD,MUL_MAT\n", argv[0]); + printf(" %s -o ADD -p 'type=f16.*perm1=0'\n", argv[0]); + printf(" %s -b MTL0 -o 'DSV4.*'\n", argv[0]); + printf(" %s -b CUDA0 -o 'ADD(type=f16,ne=[1,1,1,1],nr=[32,1,1,1],nf=1,perm1=0,src_overlap=0)'\n", argv[0]); + printf(" %s perf -o 'MUL_MAT.*'\n", argv[0]); } int main(int argc, char ** argv) { @@ -10866,8 +12141,7 @@ int main(int argc, char ** argv) { ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(dev); auto ggml_backend_set_n_threads_fn = (ggml_backend_set_n_threads_t) ggml_backend_reg_get_proc_address(reg, "ggml_backend_set_n_threads"); if (ggml_backend_set_n_threads_fn) { - // TODO: better value for n_threads - ggml_backend_set_n_threads_fn(backend.get(), N_THREADS); + ggml_backend_set_n_threads_fn(backend.get(), std::max(1, N_THREADS/2)); } size_t free, total; // NOLINT diff --git a/ggml/tests/test-quantize-fns.cpp b/ggml/tests/test-quantize-fns.cpp index 9510ac14..570fca89 100644 --- a/ggml/tests/test-quantize-fns.cpp +++ b/ggml/tests/test-quantize-fns.cpp @@ -5,6 +5,8 @@ #undef NDEBUG #include +#include +#include #include #include #include @@ -32,9 +34,9 @@ static const char* RESULT_STR[] = {"ok", "FAILED"}; // Generate synthetic data -static void generate_data(float offset, size_t n, float * dst) { +static void generate_data(float offset, size_t n, float * dst, float amplitude = 2.0f) { for (size_t i = 0; i < n; i++) { - dst[i] = 0.1 + 2*cosf(i + offset); + dst[i] = 0.1 + amplitude*cosf(i + offset); } } @@ -83,23 +85,50 @@ static float dot_product(const float * a1, const float * a2, size_t test_size) { } // Total dot product error -static float dot_product_error(const ggml_type_traits * qfns, const ggml_type_traits_cpu * qfns_cpu, size_t test_size, const float * test_data1, const float * test_data2) { - GGML_UNUSED(qfns); - - std::vector tmp_q1(2*test_size); - std::vector tmp_q2(2*test_size); - +static float dot_product_error(const ggml_type_traits_cpu * qfns_cpu, ggml_type src0_type, size_t test_size, + const float * test_data1, const float * test_data2, + const float * test_data3, const float * test_data4, + const int nrc) { const auto * vdot = ggml_get_type_traits_cpu(qfns_cpu->vec_dot_type); + const size_t pad = 64; + const size_t bx = ggml_row_size(src0_type, test_size) + pad; + const size_t by = ggml_row_size(qfns_cpu->vec_dot_type, test_size) + pad; + + std::vector tmp_q1(bx * nrc); + std::vector tmp_q2(by * nrc); qfns_cpu->from_float(test_data1, tmp_q1.data(), test_size); vdot->from_float(test_data2, tmp_q2.data(), test_size); - float result = INFINITY; - qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1); + if (nrc == 1) { + float result = INFINITY; + qfns_cpu->vec_dot(test_size, &result, 0, tmp_q1.data(), 0, tmp_q2.data(), 0, 1); + + const float dot_ref = dot_product(test_data1, test_data2, test_size); + return fabsf(result - dot_ref) / test_size; + } + + // nrc == 2: kernel computes a 2x2 dot product matrix + // Output layout: s[0]=dot(vx0,vy0), s[1]=dot(vx1,vy0), s[bs]=dot(vx0,vy1), s[bs+1]=dot(vx1,vy1) + // row and output strides are padded, same as in the mul_mat path + qfns_cpu->from_float(test_data3, tmp_q1.data() + bx, test_size); + vdot->from_float(test_data4, tmp_q2.data() + by, test_size); + + const size_t bs = 16; + std::vector result(bs + 2, INFINITY); + qfns_cpu->vec_dot(test_size, result.data(), bs, tmp_q1.data(), bx, tmp_q2.data(), by, 2); + + const float ref00 = dot_product(test_data1, test_data2, test_size); + const float ref10 = dot_product(test_data3, test_data2, test_size); + const float ref01 = dot_product(test_data1, test_data4, test_size); + const float ref11 = dot_product(test_data3, test_data4, test_size); - const float dot_ref = dot_product(test_data1, test_data2, test_size); + const auto err = [test_size](float val, float ref) { + const float e = fabsf(val - ref) / test_size; + return std::isfinite(e) ? e : INFINITY; + }; - return fabsf(result - dot_ref) / test_size; + return std::max({err(result[0], ref00), err(result[1], ref10), err(result[bs], ref01), err(result[bs + 1], ref11)}); } static int test_vec_dot_f32(bool verbose) { @@ -133,9 +162,13 @@ static int test_vec_dot_q(bool verbose) { std::vector test_data(test_size); std::vector test_data2(test_size); + std::vector test_data3(test_size); + std::vector test_data4(test_size); generate_data(0.0, test_data.size(), test_data.data()); generate_data(1.0, test_data2.size(), test_data2.data()); + generate_data(3.0, test_data3.size(), test_data3.data(), 1.0f); + generate_data(4.0, test_data4.size(), test_data4.data(), 1.5f); for (int i = 0; i < GGML_TYPE_COUNT; i++) { ggml_type type = (ggml_type) i; @@ -178,7 +211,7 @@ static int test_vec_dot_q(bool verbose) { printf("%5s reference implementation error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], reference_error); } - const float vec_dot_error = dot_product_error(qfns, qfns_cpu, test_size, test_data.data(), test_data2.data()); + const float vec_dot_error = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), nullptr, nullptr, 1); const float max_allowed_error = type == GGML_TYPE_Q2_K || type == GGML_TYPE_IQ2_XS || type == GGML_TYPE_IQ2_XXS || type == GGML_TYPE_IQ3_XXS || type == GGML_TYPE_IQ3_S || type == GGML_TYPE_IQ2_S ? MAX_DOT_PRODUCT_ERROR_LOWBIT @@ -194,6 +227,16 @@ static int test_vec_dot_q(bool verbose) { if (failed || verbose) { printf("%5s dot product error: %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error); } + + // Test nrc=2 path for types that support it + if (qfns_cpu->nrows == 2) { + const float vec_dot_error_nrc2 = dot_product_error(qfns_cpu, type, test_size, test_data.data(), test_data2.data(), test_data3.data(), test_data4.data(), 2); + failed = !(vec_dot_error_nrc2 < max_allowed_error); + num_failed += failed; + if (failed || verbose) { + printf("%5s dot product error (nrc=2): %s (%f)\n", ggml_type_name(type), RESULT_STR[failed], vec_dot_error_nrc2); + } + } } } diff --git a/src/arch/funasr_nano/adaptor.cpp b/src/arch/funasr_nano/adaptor.cpp index 4d480ee5..4537c8cb 100644 --- a/src/arch/funasr_nano/adaptor.cpp +++ b/src/arch/funasr_nano/adaptor.cpp @@ -29,7 +29,7 @@ ggml_tensor * named(ggml_tensor * t, const char * name) { ggml_tensor * mul_mat_f32acc(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * y = ggml_mul_mat(ctx, w, x); if (w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(y, GGML_PREC_F32); + ggml_prec_set_acc(y, GGML_PREC_F32); } return y; } diff --git a/src/arch/medasr/encoder.cpp b/src/arch/medasr/encoder.cpp index e096fee8..804ade96 100644 --- a/src/arch/medasr/encoder.cpp +++ b/src/arch/medasr/encoder.cpp @@ -39,7 +39,7 @@ namespace conf = transcribe::conformer; ggml_tensor * mul_mat_f32acc(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * y = ggml_mul_mat(ctx, w, x); if (w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(y, GGML_PREC_F32); + ggml_prec_set_acc(y, GGML_PREC_F32); } return y; } diff --git a/src/causal_lm/causal_lm.cpp b/src/causal_lm/causal_lm.cpp index f8ecb2da..1da80bdf 100644 --- a/src/causal_lm/causal_lm.cpp +++ b/src/causal_lm/causal_lm.cpp @@ -36,7 +36,7 @@ ggml_tensor * rms_norm(ggml_context * ctx, ggml_tensor * x, ggml_tensor * weight ggml_tensor * mul_mat_f32acc(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * y = ggml_mul_mat(ctx, w, x); if (w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(y, GGML_PREC_F32); + ggml_prec_set_acc(y, GGML_PREC_F32); } return y; } diff --git a/src/conformer/conformer.cpp b/src/conformer/conformer.cpp index 7643a938..f811d36d 100644 --- a/src/conformer/conformer.cpp +++ b/src/conformer/conformer.cpp @@ -143,7 +143,7 @@ ggml_tensor * conv_1d_f32(ggml_context * ctx, ggml_tensor * result = ggml_mul_mat(ctx, ggml_reshape_2d(ctx, im2col, im2col->ne[0], im2col->ne[1]), kernel_2d); // [OW, OC] if (kernel_needs_f32_acc) { - ggml_mul_mat_set_prec(result, GGML_PREC_F32); + ggml_prec_set_acc(result, GGML_PREC_F32); } result = ggml_reshape_3d(ctx, result, im2col->ne[1], kernel->ne[2], 1); return result; @@ -156,7 +156,7 @@ ggml_tensor * conv_1d_f32(ggml_context * ctx, // permute to the [OW, OC, N] = [time, channels, batch] convention. ggml_tensor * result = ggml_mul_mat(ctx, kernel_2d, im2col); // [OC, OW, N] if (kernel_needs_f32_acc) { - ggml_mul_mat_set_prec(result, GGML_PREC_F32); + ggml_prec_set_acc(result, GGML_PREC_F32); } result = ggml_cont(ctx, ggml_permute(ctx, result, 1, 0, 2, 3)); return result; // [OW, OC, N] @@ -357,7 +357,7 @@ ggml_tensor * conv_module(ggml_context * ctx, ggml_tensor * x, const BlockView & ggml_tensor * pw1 = ggml_reshape_2d(ctx, b.conv_pw1_w, d_model, 2 * d_model); x = ggml_mul_mat(ctx, pw1, x); // [2*d_model, T, B] if (b.conv_pw1_w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(x, GGML_PREC_F32); + ggml_prec_set_acc(x, GGML_PREC_F32); } if (b.conv_pw1_b != nullptr) { x = ggml_add(ctx, x, b.conv_pw1_b); @@ -509,7 +509,7 @@ ggml_tensor * conv_module(ggml_context * ctx, ggml_tensor * x, const BlockView & ggml_tensor * pw2 = ggml_reshape_2d(ctx, b.conv_pw2_w, d_model, d_model); x = ggml_mul_mat(ctx, pw2, x); if (b.conv_pw2_w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(x, GGML_PREC_F32); + ggml_prec_set_acc(x, GGML_PREC_F32); } if (b.conv_pw2_b != nullptr) { x = ggml_add(ctx, x, b.conv_pw2_b); diff --git a/src/sanm/sanm.cpp b/src/sanm/sanm.cpp index fb231f6c..f3fcc253 100644 --- a/src/sanm/sanm.cpp +++ b/src/sanm/sanm.cpp @@ -18,7 +18,7 @@ namespace transcribe::sanm { static ggml_tensor * mul_mat_f32acc(ggml_context * ctx, ggml_tensor * w, ggml_tensor * x) { ggml_tensor * y = ggml_mul_mat(ctx, w, x); if (w->type == GGML_TYPE_F16) { - ggml_mul_mat_set_prec(y, GGML_PREC_F32); + ggml_prec_set_acc(y, GGML_PREC_F32); } return y; }