diff --git a/.devops/openvino.Dockerfile b/.devops/openvino.Dockerfile index 13301ba287..e301aa8f5c 100644 --- a/.devops/openvino.Dockerfile +++ b/.devops/openvino.Dockerfile @@ -1,5 +1,5 @@ -ARG OPENVINO_VERSION_MAJOR=2026.3.1 -ARG OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d +ARG OPENVINO_VERSION_MAJOR=2026.4 +ARG OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3 ARG UBUNTU_VERSION=24.04 # Intel GPU driver versions. https://github.com/intel/compute-runtime/releases @@ -10,9 +10,9 @@ ARG COMPUTE_RUNTIME_VERSION_FULL=26.31.39395.13-0 ARG IGDGMM_VERSION=22.10.0 # Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases -ARG NPU_DRIVER_VERSION=v1.35.0 -ARG NPU_DRIVER_FULL=v1.35.0.20260722-29947505341 -ARG LIBZE1_VERSION=1.28.2-1~24.04~ppa1 +ARG NPU_DRIVER_VERSION=v1.38.0 +ARG NPU_DRIVER_FULL=v1.38.0.20260910-34487311128 +ARG LIBZE1_VERSION=1.32.0-1~24.04~ppa1 # Optional proxy build arguments ARG http_proxy= @@ -173,7 +173,7 @@ RUN --mount=type=cache,target=/var/cache/intel-npu,sharing=locked \ fi; \ DEB=/var/cache/intel-npu/libze1_${LIBZE1_VERSION}_amd64.deb; \ if [ ! -f "$DEB" ]; then \ - wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260606T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \ + wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260830T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \ fi; \ mkdir /tmp/npu/ && cd /tmp/npu/ && tar -xf "$TGZ" && cp "$DEB" .; \ apt-get update; \ diff --git a/.github/workflows/build-cache.yml b/.github/workflows/build-cache.yml index 4a23ec2d4d..27512a142e 100644 --- a/.github/workflows/build-cache.yml +++ b/.github/workflows/build-cache.yml @@ -41,8 +41,8 @@ jobs: env: # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone @@ -69,8 +69,8 @@ jobs: env: # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone diff --git a/.github/workflows/build-openvino.yml b/.github/workflows/build-openvino.yml index 86aba456ce..daa08b1bf9 100644 --- a/.github/workflows/build-openvino.yml +++ b/.github/workflows/build-openvino.yml @@ -41,8 +41,8 @@ jobs: env: # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone @@ -96,8 +96,8 @@ jobs: env: # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone diff --git a/.github/workflows/build-self-hosted.yml b/.github/workflows/build-self-hosted.yml index fd3722bcf5..d54f71ac51 100644 --- a/.github/workflows/build-self-hosted.yml +++ b/.github/workflows/build-self-hosted.yml @@ -412,8 +412,8 @@ jobs: env: # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Clone diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 8389f017b9..cace91037b 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -555,8 +555,8 @@ jobs: env: # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Set OpenVINO version output @@ -669,8 +669,8 @@ jobs: env: # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile - OPENVINO_VERSION_MAJOR: "2026.3.1" - OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d" + OPENVINO_VERSION_MAJOR: "2026.4" + OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3" steps: - name: Set OpenVINO version output diff --git a/docs/backend/OPENVINO.md b/docs/backend/OPENVINO.md index c1e39c5bf1..3d79197755 100644 --- a/docs/backend/OPENVINO.md +++ b/docs/backend/OPENVINO.md @@ -12,6 +12,8 @@ The OpenVINO backend is implemented in `ggml/src/ggml-openvino` and provides a t - Compiles and caches the model for the target device. - Binds GGML tensor memory to OpenVINO inference tensors and runs inference. +For guidance on contributing to the OpenVINO backend, see the [OpenVINO Backend Contributing Guide](https://github.com/ravi9/llamacpp-ov-dev-guide/blob/main/contributing-llamacpp-ov.md). + ## Contents - [Supported Devices](#supported-devices) @@ -96,7 +98,7 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_ - **SL** = Stateless (`GGML_OPENVINO_STATEFUL_EXECUTION=0`) - **SF** = Stateful (`GGML_OPENVINO_STATEFUL_EXECUTION=1`) - Note: The NPU operates in stateless mode only. -- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.35.0. +- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel Graphics Compiler 2.41.5 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.38.0. - See [Known Limitations](#known-limitations) for context on observed failures. | Model | CPU (SL / SF) | GPU (SL / SF) | NPU (SL) | @@ -117,9 +119,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_ | [lmstudio-community/Qwen3.5-9B-Q4_K_M](https://huggingface.co/lmstudio-community/Qwen3.5-9B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ | | | | | | | [unsloth/gemma-3-4b-it-Q4_K_M](https://huggingface.co/unsloth/gemma-3-4b-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | -| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ | -| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ | -| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ | +| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ | +| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | +| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ | | | | | | | [bartowski/Phi-3-mini-4k-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3-mini-4k-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | | [bartowski/Phi-3.5-mini-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3.5-mini-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | @@ -132,9 +134,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_ | [bartowski/DeepSeek-R1-Distill-Llama-8B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Llama-8B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | | [bartowski/DeepSeek-R1-Distill-Qwen-7B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | | | | | | -| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✗ / ✗ | ✓ | +| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | | [ibm-granite/granite-4.0-micro-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-micro-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | -| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✗ / ✗ | ✗ | +| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ | | [ibm-research/granite-3.2-8b-instruct-Q4_K_M](https://huggingface.co/ibm-research/granite-3.2-8b-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | | | | | | | [HuggingFaceTB/smollm2-1.7b-instruct-q4_k_m](https://huggingface.co/HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ | @@ -242,8 +244,8 @@ chmod +x build-llamacpp-ov.sh # ============================================ set -euo pipefail -OPENVINO_VERSION_MAJOR="2026.3.1" -OPENVINO_VERSION_FULL="2026.3.1.22476.56d9685302d" +OPENVINO_VERSION_MAJOR="2026.4" +OPENVINO_VERSION_FULL="2026.4.0.22959.99c81491cc3" SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" OPENVINO_INSTALL_DIR="/opt/intel/openvino_${OPENVINO_VERSION_MAJOR}" @@ -340,7 +342,7 @@ echo " ./build/ReleaseOV/bin/llama-cli -m model.gguf" ``` > [!NOTE] -> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. +> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. @@ -370,8 +372,8 @@ REM ============================================ REM llama.cpp OpenVINO Build Script (Ninja) REM ============================================ -set "OPENVINO_VERSION_MAJOR=2026.3.1" -set "OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d" +set "OPENVINO_VERSION_MAJOR=2026.4" +set "OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3" set "SCRIPT_DIR=%~dp0" set "VCPKG_DIR=C:\vcpkg" @@ -550,7 +552,7 @@ endlocal ``` > [!NOTE] -> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**. +> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**. diff --git a/ggml/src/ggml-openvino/ggml-decoder.cpp b/ggml/src/ggml-openvino/ggml-decoder.cpp index 0b99834aa8..cd06b22e8a 100644 --- a/ggml/src/ggml-openvino/ggml-decoder.cpp +++ b/ggml/src/ggml-openvino/ggml-decoder.cpp @@ -245,7 +245,7 @@ void GgmlOvDecoder::set_input_output() { if (src->op == GGML_OP_VIEW) { // Traverse upward through nested VIEW operations std::remove_reference_t view_chain; - auto current = src; + auto * current = src; while (current != nullptr) { auto current_name = get_tensor_ov_name(m_cgraph, current); @@ -612,9 +612,8 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr if (node->src[1]->view_src != nullptr) { if (node->src[3] != nullptr) { return 4; // decoder self-attention - } else { - return 5; // cross-attention or encoder self-attention - }; + } + return 5; // cross-attention or encoder self-attention } break; default: @@ -736,8 +735,7 @@ std::pair GgmlOvDecoder::compute_llm_params(ggml_cgr bool rope_seen = false; for (int i = 0; i < cgraph->n_nodes; i++) { - auto * node = cgraph->nodes[i]; - std::string name = std::string(node->name); + ggml_tensor * node = cgraph->nodes[i]; const int attention_pattern_case = get_attention_pattern_case(node); if (attention_pattern_case != -1) { ggml_tensor * cache_k_permute = nullptr; @@ -948,7 +946,6 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op, if (m_naive) { return input != nullptr ? ov::PartialShape{get_shape(input)} : ov::PartialShape{get_shape(op)}; } - auto name = std::string(input->name); ov::PartialShape input_shape; if (is_inp_tok(input, op) || is_inp_pos(input, op)) { @@ -1474,7 +1471,7 @@ std::shared_ptr GgmlOvDecoder::create_weight_node(ggml_tensor * tensor void GgmlOvDecoder::dump_cgraph(const ggml_cgraph * cgraph, std::string & filename) { std::ofstream file(filename); if (!file.is_open()) { - std::cerr << "Failed to open file" << std::endl; + std::cerr << "Failed to open file" << '\n'; return; } @@ -1580,11 +1577,11 @@ void print_tensor_address_map(const ggml_cgraph * cgraph) { } } for (const auto & pair : address_map) { - std::cout << "Address: " << pair.first << std::endl; + std::cout << "Address: " << pair.first << '\n'; for (const auto & name : pair.second) { std::cout << name << " ; "; } - std::cout << std::endl << std::endl; + std::cout << "\n\n"; } } @@ -2226,7 +2223,7 @@ void GgmlOvDecoder::compute_node_dynamic_dims() { std::cout << ", "; } } - std::cout << "]" << std::endl; + std::cout << "]" << '\n'; // print the src name & shape with the dynamic dim for debugging for (int j = 0; j < GGML_MAX_SRC; j++) { ggml_tensor * src = node->src[j]; @@ -2245,9 +2242,9 @@ void GgmlOvDecoder::compute_node_dynamic_dims() { std::cout << ", "; } } - std::cout << "]" << std::endl; + std::cout << "]" << '\n'; } - std::cout << std::endl; + std::cout << '\n'; } } } diff --git a/ggml/src/ggml-openvino/ggml-decoder.h b/ggml/src/ggml-openvino/ggml-decoder.h index 7f9d45a48a..056e39e871 100644 --- a/ggml/src/ggml-openvino/ggml-decoder.h +++ b/ggml/src/ggml-openvino/ggml-decoder.h @@ -354,41 +354,41 @@ public: void update_io(ggml_cgraph * cgraph); - inline static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op == GGML_OP_NONE; } - inline static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_ROPE && tensor == op->src[1]; } // IMROPE packs 4 stacked position planes (t/h/w/e) into inp_pos, each of length // n_tokens; other modes carry a single position per token. - inline static int get_inp_pos_n_planes(const ggml_tensor * op) { + static int get_inp_pos_n_planes(const ggml_tensor * op) { return op->op_params[2] == GGML_ROPE_TYPE_IMROPE ? 4 : 1; } - inline static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) { return tensor->op == GGML_OP_GET_ROWS && op->op == GGML_OP_RMS_NORM; } - inline static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_CPY || (op->op == GGML_OP_FLASH_ATTN_EXT && tensor == op->src[3]) || (op->op == GGML_OP_SOFT_MAX && tensor == op->src[1]); } - inline static bool is_inp_mean(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_mean(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_MUL_MAT && tensor == op->src[1] && tensor->op == GGML_OP_NONE && (tensor->flags & GGML_TENSOR_FLAG_INPUT) && tensor->type == GGML_TYPE_F32 && op->src[0] != nullptr && op->src[0]->op != GGML_OP_NONE; } - inline static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_ROPE && tensor == op->src[2]; } // also returns true for cache_s and cache_r in SSM/DeltaNet models - inline static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) { if (tensor == nullptr) { return false; } @@ -396,14 +396,14 @@ public: (op != nullptr && op->op == GGML_OP_SET_ROWS && op->src[2] == tensor); } - inline static bool is_conv_state_writeback(const ggml_tensor * node) { + static bool is_conv_state_writeback(const ggml_tensor * node) { return node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) && node->src[0] != nullptr && node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr && node->src[0]->src[0]->op == GGML_OP_CONCAT && node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src; } - inline static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_SET_ROWS && op->src[1] == tensor; } @@ -411,13 +411,13 @@ public: return m_model_params.swa_mask != nullptr && tensor == m_model_params.swa_mask; } - inline static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op != GGML_OP_NONE && op->src[1]->op == GGML_OP_NONE; } // the state permutation index input used in SSM/DeltaNet models (inp->s_copy in llama-graph.cpp) - inline static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) { + static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) { return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY; } @@ -481,5 +481,3 @@ private: }; void print_tensor_address_map(const ggml_cgraph * cgraph); - -std::optional extract_layer_from_name(const std::string & name); diff --git a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp index 52e1a297c2..216e3b8a69 100644 --- a/ggml/src/ggml-openvino/ggml-openvino-extra.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino-extra.cpp @@ -472,10 +472,6 @@ ggml_openvino_extracted_layout ggml_openvino_get_extracted_layout(const ggml_ten switch (tensor->type) { case GGML_TYPE_MXFP4: - layout.is_u4 = true; - layout.is_symmetric = true; - break; - case GGML_TYPE_Q4_0: layout.is_u4 = true; layout.is_symmetric = true; diff --git a/ggml/src/ggml-openvino/ggml-openvino.cpp b/ggml/src/ggml-openvino/ggml-openvino.cpp index 044b4da1c9..02c5962238 100644 --- a/ggml/src/ggml-openvino/ggml-openvino.cpp +++ b/ggml/src/ggml-openvino/ggml-openvino.cpp @@ -28,12 +28,7 @@ #include #include -#ifndef _WIN32 -# include -# include -#endif - -#if defined(_WIN32) +#ifdef _WIN32 # define WIN32_LEAN_AND_MEAN # ifndef NOMINMAX # define NOMINMAX @@ -61,6 +56,7 @@ // - CPU repack buffer: tensor->extra stores tensor_traits with repacked data // ===================================================== +namespace { // Buffer context that manages per-tensor allocations (no contiguous buffer for weights) struct ggml_backend_openvino_buffer_context { int device; @@ -199,6 +195,7 @@ struct ggml_backend_openvino_buffer_type_context { int device; std::string name; }; +} // namespace // ===================================================== // Host weight-buffer release (GGML_OPENVINO_RELEASE_WEIGHTS) @@ -258,14 +255,16 @@ void ggml_openvino_release_weight_buffers() { for (const auto & b : reg.buffers) { // Align down/up to page boundaries so madvise only drops whole pages // fully owned by this buffer. - const long page = sysconf(_SC_PAGESIZE); - uintptr_t start = reinterpret_cast(b.first); - uintptr_t end = start + b.second; - uintptr_t astart = (start + page - 1) & ~(uintptr_t) (page - 1); - uintptr_t aend = end & ~(uintptr_t) (page - 1); - if (aend > astart) { - if (madvise(reinterpret_cast(astart), aend - astart, MADV_DONTNEED) == 0) { - total += aend - astart; + const size_t page = (size_t) sysconf(_SC_PAGESIZE); + const uintptr_t ustart = reinterpret_cast(b.first); + const size_t offset_to_page = (page - (ustart & (page - 1))) & (page - 1); + if (b.second > offset_to_page) { + const size_t aligned_len = (b.second - offset_to_page) & ~(page - 1); + if (aligned_len > 0) { + char * astart = static_cast(b.first) + offset_to_page; + if (madvise(astart, aligned_len, MADV_DONTNEED) == 0) { + total += aligned_len; + } } } } @@ -876,11 +875,13 @@ GGML_BACKEND_API bool ggml_backend_is_openvino(ggml_backend_t backend) { return backend != NULL && ggml_guid_matches(backend->guid, ggml_backend_openvino_guid()); } +namespace { struct ggml_backend_openvino_device_context { int device; std::string name; std::string description; }; +} static const char * ggml_backend_openvino_device_get_name(ggml_backend_dev_t dev) { ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context; @@ -1588,9 +1589,11 @@ static const struct ggml_backend_device_i ggml_backend_openvino_device_interface /* .event_synchronize = */ NULL, }; +namespace { struct ggml_backend_openvino_reg_context { std::vector devices; }; +} static const char * ggml_backend_openvino_reg_get_name(ggml_backend_reg_t reg) { return GGML_OPENVINO_NAME; diff --git a/ggml/src/ggml-openvino/ggml-quants.cpp b/ggml/src/ggml-openvino/ggml-quants.cpp index 93f9e8254a..824d244782 100644 --- a/ggml/src/ggml-openvino/ggml-quants.cpp +++ b/ggml/src/ggml-openvino/ggml-quants.cpp @@ -34,6 +34,15 @@ #include #include +// From /src/common/transformations/include/transformations/utils/utils.hpp +namespace ov::op::util { +// From /src/common/transformations/include/transformations/utils/utils.hpp +bool get_single_value(const std::shared_ptr & const_node, + float & value, + bool check_value_range = true); +} // namespace ov::op::util + +namespace { void unpack_32_4(const uint8_t * data, uint8_t * dst) { std::fill_n(dst, 16, 0); for (int j = 0; j < 16; ++j) { @@ -48,11 +57,11 @@ void unpack_32_4(const uint8_t * data, uint8_t * dst) { } } -static constexpr size_t MXFP4_BLOCK_SIZE = 32; -static constexpr size_t MXFP4_BLOCK_QS_SIZE = MXFP4_BLOCK_SIZE / 2; -static constexpr size_t MXFP4_BLOCK_BYTES = sizeof(uint8_t) + MXFP4_BLOCK_QS_SIZE; +constexpr size_t MXFP4_BLOCK_SIZE = 32; +constexpr size_t MXFP4_BLOCK_QS_SIZE = MXFP4_BLOCK_SIZE / 2; +constexpr size_t MXFP4_BLOCK_BYTES = sizeof(uint8_t) + MXFP4_BLOCK_QS_SIZE; -static void pack_32_mxfp4_for_openvino(const uint8_t * data, uint8_t * dst) { +void pack_32_mxfp4_for_openvino(const uint8_t * data, uint8_t * dst) { for (int j = 0; j < static_cast(MXFP4_BLOCK_QS_SIZE); j += 2) { const uint8_t v0 = data[j] & 0x0F; const uint8_t v1 = (data[j + 1] & 0x0F) << 4; @@ -419,7 +428,7 @@ void extract_q6_k_data(const ggml_tensor * tensor, } } -static inline void get_scale_min_k4(int j, const uint8_t * q, uint8_t * d, uint8_t * m) { +inline void get_scale_min_k4(int j, const uint8_t * q, uint8_t * d, uint8_t * m) { if (j < 4) { *d = q[j] & 63; *m = q[j + 4] & 63; @@ -514,9 +523,9 @@ void extract_q5_k_data(const ggml_tensor * tensor, ov::Output make_int8_weights(ov::Tensor & weight, ov::Tensor & scales, ov::Tensor & zp, - size_t group_size, - bool use_bias, - bool for_gather_matmul) { + size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, + bool use_bias = false, + bool for_gather_matmul = false) { ov::Shape orig_shape = weight.get_shape(); bool is_signed = (weight.get_element_type() == ov::element::i8); // Symmetric: signed weights, no ZP @@ -611,13 +620,24 @@ ov::Output make_int8_weights(ov::Tensor & weight, return std::make_shared(result, ov::element::f32); } +// If for_gather_matmul is true, the weight tensor may be N-D (e.g. 3D MoE expert weights +// [n_expert, rows, cols]). The dequantization chain (Convert->[Subtract]->Multiply) is built as +// usual but left in f16 (no final Convert to f32) -- ov::pass::MarkDequantization (registered in +// translate_session.cpp) marks the chain so it survives model-build-time ConstantFolding -- see +// make_int8_weights.cpp/make_int4_weights.cpp. mul_mat_id.cpp constructs ov::op::internal::GatherMatmul +// directly from the resulting f16 dequant chain. +// +// When use_bias is true (explicitly, or implicitly because for_gather_matmul is true), the zp +// tensor is expected to hold an exact f16 bias value (rather than a rounded integer zero point); +// it is converted in place into an exact zero_point = -bias/scale and consumed via Subtract, not +// Add, so the chain still matches OpenVINO's Convert->Subtract->Multiply decompression pattern. // See make_int8_weights for the meaning of for_gather_matmul. ov::Output make_int4_weights(ov::Tensor & weight, ov::Tensor & scales, ov::Tensor & zp, - size_t group_size, - bool use_bias, - bool for_gather_matmul) { + size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, + bool use_bias = false, + bool for_gather_matmul = false) { ov::Shape orig_weight_shape = weight.get_shape(); bool is_signed = (weight.get_element_type() == ov::element::i4); // Symmetric: signed weights, no ZP @@ -746,357 +766,6 @@ ov::Output make_mxfp4_moe_packed_weights(ov::Tensor & weight) { return weights_node; } -// Extract quantized weights from tensor and create weight subgraph -std::shared_ptr extract_quantized_weights(const ggml_tensor * tensor, - const void * data, - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp, - bool use_bias) { - // Create a temporary tensor for extraction functions that read from tensor->data - ggml_tensor temp_tensor = *tensor; - temp_tensor.data = const_cast(data); - - if (tensor->type == GGML_TYPE_MXFP4) { - extract_mxfp4_data(&temp_tensor, weights, scales); - auto result = make_mxfp4_weights(weights, scales).get_node_shared_ptr(); - result->set_friendly_name(tensor->name); - return result; - } - - // Determine block size based on tensor type - int64_t weights_per_block; - bool is_u4; - switch (tensor->type) { - case GGML_TYPE_Q4_0: - case GGML_TYPE_Q4_1: - case GGML_TYPE_Q4_K: - is_u4 = true; - weights_per_block = 32; - break; - case GGML_TYPE_Q8_0: - case GGML_TYPE_Q5_1: - case GGML_TYPE_Q5_K: - is_u4 = false; - weights_per_block = 32; - break; - case GGML_TYPE_Q6_K: - is_u4 = false; - weights_per_block = 16; - break; - default: - throw std::runtime_error("Unsupported quantized type for extraction: " + - std::string(ggml_type_name(tensor->type))); - } - - // 3D MoE expert weights (for_gather_matmul) always use the exact f16 zero-point extraction - // (see make_int8_weights/make_int4_weights) rather than the rounded integer zero point -- - // round(min/scale) error is what corrupts Q4_K/Q5_1 experts, and the f16-zp form still fuses - // into GatherMatmulCompressed since it stays a Subtract, not an Add. - const bool for_gather_matmul = tensor->ne[2] > 1; - use_bias = use_bias || for_gather_matmul; - - // Extract quantized data - switch (tensor->type) { - case GGML_TYPE_Q4_0: - extract_q4_0_data(&temp_tensor, weights, scales, zp); - break; - case GGML_TYPE_Q4_1: - extract_q4_1_data(&temp_tensor, weights, scales, zp, use_bias); - break; - case GGML_TYPE_Q4_K: - extract_q4_k_data(&temp_tensor, weights, scales, zp, use_bias); - break; - case GGML_TYPE_Q5_1: - extract_q5_1_data(&temp_tensor, weights, scales, zp, use_bias); - break; - case GGML_TYPE_Q8_0: - extract_q8_0_data(&temp_tensor, weights, scales, zp); - break; - case GGML_TYPE_Q6_K: - extract_q6_k_data(&temp_tensor, weights, scales, zp); - break; - case GGML_TYPE_Q5_K: - extract_q5_k_data(&temp_tensor, weights, scales, zp, use_bias); - break; - default: - throw std::runtime_error("Unsupported quantized type: " + std::string(ggml_type_name(tensor->type))); - } - - // Create the OpenVINO weight subgraph. 3D expert weights (MoE) are routed through the - // GatherMatmul-oriented path: dequantized in f16, with constant folding disabled on the chain. - ov::Output weight_node; - if (is_u4) { - weight_node = make_int4_weights(weights, scales, zp, weights_per_block, use_bias, for_gather_matmul); - } else { - weight_node = make_int8_weights(weights, scales, zp, weights_per_block, use_bias, for_gather_matmul); - } - - auto result = weight_node.get_node_shared_ptr(); - result->set_friendly_name(tensor->name); - return result; -} - -// Requantize weights to target format, writing to provided buffers -std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, - const void * data, - ExtraQuantType requant_type, - int64_t block_size, - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp) { - int64_t n_elements = ggml_nelements(tensor); - const int64_t ne0 = tensor->ne[0]; // elements per row - const int64_t n_rows = n_elements / ne0; - const auto * type_traits = ggml_get_type_traits(tensor->type); - const size_t src_row_bytes = ggml_row_size(tensor->type, ne0); - - bool is_u4 = (requant_type == ExtraQuantType::Q4_0_C || requant_type == ExtraQuantType::Q4_0_128 || - requant_type == ExtraQuantType::Q4_0_64 || requant_type == ExtraQuantType::Q4_1_64); - - // Streaming dequant (opt-in via GGML_OPENVINO_REDUCE_COMPILE_MEM or - // GGML_OPENVINO_MEMORY_OPTIMIZE): instead of - // materializing the full n_elements F32 array (e.g. ~1 GB for token_embd), dequantize - // a chunk of complete rows into a small scratch and quantize/convert it straight into - // the output buffers, capping the transient F32 footprint at CHUNK_ROWS*ne0 floats. - // - // Only valid (and only used) for the Q8_0_C / Q8_1_C / F16 targets whose block size - // divides a row (channel-wise _C uses block_size == ne0) so no target block straddles - // a row boundary, and Q8/F16 have no cross-block packing. The u4 (Q4_0) path packs two - // weights per byte with running zp ORs that assume a single whole-array call, so it is - // never streamed. When the flag is off, behavior is identical to the original - // full-materialization path. - const bool stream_requant = ggml_openvino_reduce_compile_mem_enabled() && !is_u4 && - !(block_size > 0 && ne0 % block_size != 0); - - if (!stream_requant) { - // Full materialization (original behavior): dequantize the whole tensor to F32, - // then convert/quantize in one call. - std::vector weights_f32(n_elements); - type_traits->to_float(data, weights_f32.data(), n_elements); - if (requant_type == ExtraQuantType::F16) { - ggml_get_type_traits(GGML_TYPE_F16)->from_float_ref(weights_f32.data(), weights.data(), n_elements); - auto result = std::make_shared(weights); - result->set_friendly_name(tensor->name); - return result; - } - if (requant_type == ExtraQuantType::Q4_1_64) { - quantize_q4_1_asym(weights_f32.data(), weights, scales, zp, n_elements, block_size); - } else if (is_u4) { - quantize_q4_0(weights_f32.data(), weights, scales, zp, n_elements, block_size); - } else if (requant_type == ExtraQuantType::Q8_1_C) { - quantize_q8_1(weights_f32.data(), weights, scales, zp, n_elements, block_size); - } else { - quantize_q8_0(weights_f32.data(), weights, scales, zp, n_elements, block_size); - } - } else { - // Streaming path for Q8_0_C / Q8_1_C / F16 (covers token_embd, output.weight, - // and per-layer Q6_K/Q5_K requant — the large transient cases). - const int64_t CHUNK_ROWS = std::min(n_rows, 256); - std::vector scratch(CHUNK_ROWS * ne0); - // F16 destination: 2 bytes/element, advanced per chunk by r0*ne0 elements. - auto * f16_base = static_cast(weights.data()); - for (int64_t r0 = 0; r0 < n_rows; r0 += CHUNK_ROWS) { - const int64_t rows = std::min(CHUNK_ROWS, n_rows - r0); - const int64_t elems = rows * ne0; - const auto * src = static_cast(data) + r0 * src_row_bytes; - type_traits->to_float(src, scratch.data(), elems); - - if (requant_type == ExtraQuantType::F16) { - ggml_get_type_traits(GGML_TYPE_F16) - ->from_float_ref(scratch.data(), f16_base + (r0 * ne0) * sizeof(uint16_t), elems); - } else { - const int64_t block_offset = (r0 * ne0) / block_size; - if (requant_type == ExtraQuantType::Q8_1_C) { - quantize_q8_1(scratch.data(), weights, scales, zp, elems, block_size, block_offset); - } else { - quantize_q8_0(scratch.data(), weights, scales, zp, elems, block_size, block_offset); - } - } - } - if (requant_type == ExtraQuantType::F16) { - auto result = std::make_shared(weights); - result->set_friendly_name(tensor->name); - return result; - } - } - - // Create the OpenVINO weight subgraph - ov::Output weight_node; - if (is_u4) { - weight_node = make_int4_weights(weights, scales, zp, block_size); - } else { - weight_node = make_int8_weights(weights, scales, zp, block_size); - } - - auto result = weight_node.get_node_shared_ptr(); - result->set_friendly_name(tensor->name); - return result; -} - -OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, void * output_base_ptr, bool use_bias) { - GGML_ASSERT(tensor != nullptr); - GGML_ASSERT(data != nullptr); - - OvWeight result; - - // Get shape for weights: [rows, cols], or [n_expert, rows, cols] for 3D MoE expert weights. - ov::Shape node_shape = (tensor->ne[2] > 1) ? - ov::Shape{static_cast(tensor->ne[2]), static_cast(tensor->ne[1]), - static_cast(tensor->ne[0])} : - ov::Shape{static_cast(tensor->ne[1]), static_cast(tensor->ne[0])}; - - // Handle F16/F32/BF16 weights - if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) { - ov::element::Type element_type; - switch (tensor->type) { - case GGML_TYPE_F32: - element_type = ov::element::f32; - break; - case GGML_TYPE_F16: - element_type = ov::element::f16; - break; - case GGML_TYPE_BF16: - element_type = ov::element::bf16; - break; - default: - OPENVINO_THROW("Unexpected tensor type in F16/F32/BF16 path"); - } - - if (output_base_ptr && output_base_ptr != data) { - // Using external buffer - copy data and create shared-memory constant - size_t tensor_bytes = ggml_nbytes(tensor); - memcpy(output_base_ptr, data, tensor_bytes); - result.weights = ov::Tensor(element_type, node_shape, output_base_ptr); - } else { - result.weights = ov::Tensor(element_type, node_shape, data); - } - result.weight_node = std::make_shared(result.weights); - return result; - } - - // Handle quantized weights - if (!ggml_is_quantized(tensor->type)) { - OPENVINO_THROW("Unsupported weight tensor type: ", ggml_type_name(tensor->type)); - } - - result.layout = ggml_openvino_get_extracted_layout(tensor, use_bias); - const auto & layout = result.layout; - if (layout.total_size == 0) { - OPENVINO_THROW("Unsupported quantized type: ", ggml_type_name(tensor->type)); - } - - // 3D MoE expert weights (for_gather_matmul) always use the exact f16 zero-point path (see - // extract_quantized_weights) -- must be kept in sync with the "use_bias || for_gather_matmul" - // check in ggml_openvino_get_extracted_layout, which sizes/offsets the zp slot accordingly. - // Requantized tensors (layout.is_requant) are handled by requantize_to_buffers instead, whose - // zp sizing/type is unaffected by for_gather_matmul, so they are excluded here. - const bool for_gather_matmul = tensor->ne[2] > 1; - const bool zp_is_f16 = !layout.is_requant && (use_bias || for_gather_matmul); - - const bool is_3d_mxfp4_moe = tensor->type == GGML_TYPE_MXFP4 && (tensor->ne[2] > 1 || tensor->ne[3] > 1); - if (is_3d_mxfp4_moe) { - ov::Shape packed_shape = {static_cast(tensor->ne[3]), - static_cast(tensor->ne[2]), - static_cast(tensor->ne[1]), - static_cast(tensor->ne[0] / MXFP4_BLOCK_SIZE), - MXFP4_BLOCK_BYTES}; - const size_t tensor_bytes = ggml_nbytes(tensor); - if (output_base_ptr) { - auto * buf_base = static_cast(output_base_ptr); - memcpy(buf_base + layout.weights_offset, data, tensor_bytes); - result.weights = ov::Tensor(ov::element::u8, packed_shape, buf_base + layout.weights_offset); - } else { - result.weights = ov::Tensor(ov::element::u8, packed_shape); - memcpy(result.weights.data(), data, tensor_bytes); - } - result.weight_node = make_mxfp4_moe_packed_weights(result.weights).get_node_shared_ptr(); - result.weight_node->set_friendly_name(tensor->name); - return result; - } - - if (use_bias) { - OPENVINO_ASSERT(!layout.is_requant, - "use_bias is only used for test-backend-ops, which should not have requantization"); - // bias node will be created on the fly and not use backend buffer - output_base_ptr = nullptr; - } - - // F16 requant path - no separate scales/zp needed in result - if (layout.is_requant && layout.requant_type.has_value() && layout.requant_type.value() == ExtraQuantType::F16) { - if (output_base_ptr) { - result.weights = ov::Tensor(ov::element::f16, node_shape, - static_cast(output_base_ptr) + layout.weights_offset); - } else { - result.weights = ov::Tensor(ov::element::f16, node_shape); - } - ov::Tensor dummy_scales, dummy_zp; // Not used for F16 - result.weight_node = - requantize_to_buffers(tensor, data, ExtraQuantType::F16, 0, result.weights, dummy_scales, dummy_zp); - return result; - } - - // Quantized path (normal extraction or quantized requant) - // Create weight/scale/zp tensors - shared between both paths - // For symmetric quantization, use signed types (i4/i8) and no ZP tensor - ov::element::Type weight_type = tensor->type == GGML_TYPE_MXFP4 ? - ov::element::f4e2m1 : - (layout.is_symmetric ? (layout.is_u4 ? ov::element::i4 : ov::element::i8) : - (layout.is_u4 ? ov::element::u4 : ov::element::u8)); - ov::Shape scale_shape = node_shape; - scale_shape.back() /= layout.weights_per_block; - - if (tensor->type == GGML_TYPE_MXFP4) { - if (tensor->ne[2] == 1 && tensor->ne[3] == 1) { - node_shape = {static_cast(tensor->ne[1]), static_cast(tensor->ne[0])}; - } else { - node_shape.clear(); - for (int i = GGML_MAX_DIMS - 1; i >= 0; --i) { - node_shape.push_back(static_cast(tensor->ne[i])); - } - } - - scale_shape = node_shape; - scale_shape.back() /= layout.weights_per_block; - } - - if (output_base_ptr) { - uint8_t * buf_base = static_cast(output_base_ptr); - result.weights = ov::Tensor(weight_type, node_shape, buf_base + layout.weights_offset); - const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; - result.scales = ov::Tensor(scale_type, scale_shape, buf_base + layout.scales_offset); - if (!layout.is_symmetric) { - ov::element::Type zp_type = - zp_is_f16 ? ov::element::f16 : (layout.is_u4 ? ov::element::u4 : ov::element::u8); - result.zp = ov::Tensor(zp_type, scale_shape, buf_base + layout.zp_offset); - } - // else: result.zp remains default-constructed (empty) for symmetric - } else { - result.weights = ov::Tensor(weight_type, node_shape); - const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; - result.scales = ov::Tensor(scale_type, scale_shape); - if (!layout.is_symmetric) { - if (zp_is_f16) { - result.zp = ov::Tensor(ov::element::f16, scale_shape); - } else { - ov::element::Type zp_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; - result.zp = ov::Tensor(zp_type, scale_shape); - } - } - // else: result.zp remains default-constructed (empty) for symmetric - } - - if (layout.is_requant && layout.requant_type.has_value()) { - result.weight_node = requantize_to_buffers(tensor, data, layout.requant_type.value(), layout.weights_per_block, - result.weights, result.scales, result.zp); - } else { - result.weight_node = - extract_quantized_weights(tensor, data, result.weights, result.scales, result.zp, use_bias); - } - - return result; -} - void quantize_q4_0(const float * x, ov::Tensor & weights_arr, ov::Tensor & scales_arr, @@ -1252,7 +921,7 @@ void quantize_q8_0(const float * x, ov::Tensor & zp_arr, int64_t k, int64_t qk, - int64_t block_offset) { + int64_t block_offset = 0) { assert(k % qk == 0); const int nb = k / qk; @@ -1308,7 +977,7 @@ void quantize_q8_1(const float * x, ov::Tensor & zp_arr, int64_t k, int64_t qk, - int64_t block_offset) { + int64_t block_offset = 0) { assert(k % qk == 0); const int nb = k / qk; @@ -1339,3 +1008,366 @@ void quantize_q8_1(const float * x, } } } + +// Extract quantized weights from tensor and create weight subgraph +// If weights/scales/zp are provided (non-empty), uses them as output buffers +// Otherwise allocates new ov::Tensors internally +// Returns the weight node (make_int4_weights or make_int8_weights result) +std::shared_ptr extract_quantized_weights(const ggml_tensor * tensor, + const void * data, // Source data pointer (may differ from tensor->data) + ov::Tensor & weights, + ov::Tensor & scales, + ov::Tensor & zp, + // Use an exact f16 zero point (vs. a rounded integer one); always + // used for for_gather_matmul (3D MoE expert) weights regardless of + // this flag, and also settable explicitly for test-backend-ops. + bool use_bias = false) { + // Create a temporary tensor for extraction functions that read from tensor->data + ggml_tensor temp_tensor = *tensor; + temp_tensor.data = const_cast(data); + + if (tensor->type == GGML_TYPE_MXFP4) { + extract_mxfp4_data(&temp_tensor, weights, scales); + auto result = make_mxfp4_weights(weights, scales).get_node_shared_ptr(); + result->set_friendly_name(tensor->name); + return result; + } + + // Determine block size based on tensor type + int64_t weights_per_block; + bool is_u4; + switch (tensor->type) { + case GGML_TYPE_Q4_0: + case GGML_TYPE_Q4_1: + case GGML_TYPE_Q4_K: + is_u4 = true; + weights_per_block = 32; + break; + case GGML_TYPE_Q8_0: + case GGML_TYPE_Q5_1: + case GGML_TYPE_Q5_K: + is_u4 = false; + weights_per_block = 32; + break; + case GGML_TYPE_Q6_K: + is_u4 = false; + weights_per_block = 16; + break; + default: + throw std::runtime_error("Unsupported quantized type for extraction: " + + std::string(ggml_type_name(tensor->type))); + } + + // 3D MoE expert weights (for_gather_matmul) always use the exact f16 zero-point extraction + // (see make_int8_weights/make_int4_weights) rather than the rounded integer zero point -- + // round(min/scale) error is what corrupts Q4_K/Q5_1 experts, and the f16-zp form still fuses + // into GatherMatmulCompressed since it stays a Subtract, not an Add. + const bool for_gather_matmul = tensor->ne[2] > 1; + use_bias = use_bias || for_gather_matmul; + + // Extract quantized data + switch (tensor->type) { + case GGML_TYPE_Q4_0: + extract_q4_0_data(&temp_tensor, weights, scales, zp); + break; + case GGML_TYPE_Q4_1: + extract_q4_1_data(&temp_tensor, weights, scales, zp, use_bias); + break; + case GGML_TYPE_Q4_K: + extract_q4_k_data(&temp_tensor, weights, scales, zp, use_bias); + break; + case GGML_TYPE_Q5_1: + extract_q5_1_data(&temp_tensor, weights, scales, zp, use_bias); + break; + case GGML_TYPE_Q8_0: + extract_q8_0_data(&temp_tensor, weights, scales, zp); + break; + case GGML_TYPE_Q6_K: + extract_q6_k_data(&temp_tensor, weights, scales, zp); + break; + case GGML_TYPE_Q5_K: + extract_q5_k_data(&temp_tensor, weights, scales, zp, use_bias); + break; + default: + throw std::runtime_error("Unsupported quantized type: " + std::string(ggml_type_name(tensor->type))); + } + + // Create the OpenVINO weight subgraph. 3D expert weights (MoE) are routed through the + // GatherMatmul-oriented path: dequantized in f16, with constant folding disabled on the chain. + ov::Output weight_node; + if (is_u4) { + weight_node = make_int4_weights(weights, scales, zp, weights_per_block, use_bias, for_gather_matmul); + } else { + weight_node = make_int8_weights(weights, scales, zp, weights_per_block, use_bias, for_gather_matmul); + } + + auto result = weight_node.get_node_shared_ptr(); + result->set_friendly_name(tensor->name); + return result; +} + +// Requantize weights from tensor to target format, writing to provided buffers +// For F16 target, only weights buffer is used (scales/zp ignored) +// Returns the weight node +std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, + const void * data, // Source data pointer + ExtraQuantType requant_type, + int64_t block_size, + ov::Tensor & weights, + ov::Tensor & scales, + ov::Tensor & zp) { + int64_t n_elements = ggml_nelements(tensor); + const int64_t ne0 = tensor->ne[0]; // elements per row + const int64_t n_rows = n_elements / ne0; + const auto * type_traits = ggml_get_type_traits(tensor->type); + const size_t src_row_bytes = ggml_row_size(tensor->type, ne0); + + bool is_u4 = (requant_type == ExtraQuantType::Q4_0_C || requant_type == ExtraQuantType::Q4_0_128 || + requant_type == ExtraQuantType::Q4_0_64 || requant_type == ExtraQuantType::Q4_1_64); + + // Streaming dequant (opt-in via GGML_OPENVINO_REDUCE_COMPILE_MEM or + // GGML_OPENVINO_MEMORY_OPTIMIZE): instead of + // materializing the full n_elements F32 array (e.g. ~1 GB for token_embd), dequantize + // a chunk of complete rows into a small scratch and quantize/convert it straight into + // the output buffers, capping the transient F32 footprint at CHUNK_ROWS*ne0 floats. + // + // Only valid (and only used) for the Q8_0_C / Q8_1_C / F16 targets whose block size + // divides a row (channel-wise _C uses block_size == ne0) so no target block straddles + // a row boundary, and Q8/F16 have no cross-block packing. The u4 (Q4_0) path packs two + // weights per byte with running zp ORs that assume a single whole-array call, so it is + // never streamed. When the flag is off, behavior is identical to the original + // full-materialization path. + const bool stream_requant = ggml_openvino_reduce_compile_mem_enabled() && !is_u4 && + !(block_size > 0 && ne0 % block_size != 0); + + if (!stream_requant) { + // Full materialization (original behavior): dequantize the whole tensor to F32, + // then convert/quantize in one call. + std::vector weights_f32(n_elements); + type_traits->to_float(data, weights_f32.data(), n_elements); + if (requant_type == ExtraQuantType::F16) { + ggml_get_type_traits(GGML_TYPE_F16)->from_float_ref(weights_f32.data(), weights.data(), n_elements); + auto result = std::make_shared(weights); + result->set_friendly_name(tensor->name); + return result; + } + if (requant_type == ExtraQuantType::Q4_1_64) { + quantize_q4_1_asym(weights_f32.data(), weights, scales, zp, n_elements, block_size); + } else if (is_u4) { + quantize_q4_0(weights_f32.data(), weights, scales, zp, n_elements, block_size); + } else if (requant_type == ExtraQuantType::Q8_1_C) { + quantize_q8_1(weights_f32.data(), weights, scales, zp, n_elements, block_size); + } else { + quantize_q8_0(weights_f32.data(), weights, scales, zp, n_elements, block_size); + } + } else { + // Streaming path for Q8_0_C / Q8_1_C / F16 (covers token_embd, output.weight, + // and per-layer Q6_K/Q5_K requant — the large transient cases). + const int64_t CHUNK_ROWS = std::min(n_rows, 256); + std::vector scratch(CHUNK_ROWS * ne0); + // F16 destination: 2 bytes/element, advanced per chunk by r0*ne0 elements. + auto * f16_base = static_cast(weights.data()); + for (int64_t r0 = 0; r0 < n_rows; r0 += CHUNK_ROWS) { + const int64_t rows = std::min(CHUNK_ROWS, n_rows - r0); + const int64_t elems = rows * ne0; + const auto * src = static_cast(data) + r0 * src_row_bytes; + type_traits->to_float(src, scratch.data(), elems); + + if (requant_type == ExtraQuantType::F16) { + ggml_get_type_traits(GGML_TYPE_F16) + ->from_float_ref(scratch.data(), f16_base + (r0 * ne0) * sizeof(uint16_t), elems); + } else { + const int64_t block_offset = (r0 * ne0) / block_size; + if (requant_type == ExtraQuantType::Q8_1_C) { + quantize_q8_1(scratch.data(), weights, scales, zp, elems, block_size, block_offset); + } else { + quantize_q8_0(scratch.data(), weights, scales, zp, elems, block_size, block_offset); + } + } + } + if (requant_type == ExtraQuantType::F16) { + auto result = std::make_shared(weights); + result->set_friendly_name(tensor->name); + return result; + } + } + + // Create the OpenVINO weight subgraph + ov::Output weight_node; + if (is_u4) { + weight_node = make_int4_weights(weights, scales, zp, block_size); + } else { + weight_node = make_int8_weights(weights, scales, zp, block_size); + } + + auto result = weight_node.get_node_shared_ptr(); + result->set_friendly_name(tensor->name); + return result; +} +} // namespace + +OvWeight process_weight_tensor(const ggml_tensor * tensor, const void * data, void * output_base_ptr, bool use_bias) { + GGML_ASSERT(tensor != nullptr); + GGML_ASSERT(data != nullptr); + + OvWeight result; + + // Get shape for weights: [rows, cols], or [n_expert, rows, cols] for 3D MoE expert weights. + ov::Shape node_shape = (tensor->ne[2] > 1) ? + ov::Shape{static_cast(tensor->ne[2]), static_cast(tensor->ne[1]), + static_cast(tensor->ne[0])} : + ov::Shape{static_cast(tensor->ne[1]), static_cast(tensor->ne[0])}; + + // Handle F16/F32/BF16 weights + if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) { + ov::element::Type element_type; + switch (tensor->type) { + case GGML_TYPE_F32: + element_type = ov::element::f32; + break; + case GGML_TYPE_F16: + element_type = ov::element::f16; + break; + case GGML_TYPE_BF16: + element_type = ov::element::bf16; + break; + default: + OPENVINO_THROW("Unexpected tensor type in F16/F32/BF16 path"); + } + + if (output_base_ptr && output_base_ptr != data) { + // Using external buffer - copy data and create shared-memory constant + size_t tensor_bytes = ggml_nbytes(tensor); + memcpy(output_base_ptr, data, tensor_bytes); + result.weights = ov::Tensor(element_type, node_shape, output_base_ptr); + } else { + result.weights = ov::Tensor(element_type, node_shape, data); + } + result.weight_node = std::make_shared(result.weights); + return result; + } + + // Handle quantized weights + if (!ggml_is_quantized(tensor->type)) { + OPENVINO_THROW("Unsupported weight tensor type: ", ggml_type_name(tensor->type)); + } + + result.layout = ggml_openvino_get_extracted_layout(tensor, use_bias); + const auto & layout = result.layout; + if (layout.total_size == 0) { + OPENVINO_THROW("Unsupported quantized type: ", ggml_type_name(tensor->type)); + } + + // 3D MoE expert weights (for_gather_matmul) always use the exact f16 zero-point path (see + // extract_quantized_weights) -- must be kept in sync with the "use_bias || for_gather_matmul" + // check in ggml_openvino_get_extracted_layout, which sizes/offsets the zp slot accordingly. + // Requantized tensors (layout.is_requant) are handled by requantize_to_buffers instead, whose + // zp sizing/type is unaffected by for_gather_matmul, so they are excluded here. + const bool for_gather_matmul = tensor->ne[2] > 1; + const bool zp_is_f16 = !layout.is_requant && (use_bias || for_gather_matmul); + + const bool is_3d_mxfp4_moe = tensor->type == GGML_TYPE_MXFP4 && (tensor->ne[2] > 1 || tensor->ne[3] > 1); + if (is_3d_mxfp4_moe) { + ov::Shape packed_shape = {static_cast(tensor->ne[3]), + static_cast(tensor->ne[2]), + static_cast(tensor->ne[1]), + static_cast(tensor->ne[0] / MXFP4_BLOCK_SIZE), + MXFP4_BLOCK_BYTES}; + const size_t tensor_bytes = ggml_nbytes(tensor); + if (output_base_ptr) { + auto * buf_base = static_cast(output_base_ptr); + memcpy(buf_base + layout.weights_offset, data, tensor_bytes); + result.weights = ov::Tensor(ov::element::u8, packed_shape, buf_base + layout.weights_offset); + } else { + result.weights = ov::Tensor(ov::element::u8, packed_shape); + memcpy(result.weights.data(), data, tensor_bytes); + } + result.weight_node = make_mxfp4_moe_packed_weights(result.weights).get_node_shared_ptr(); + result.weight_node->set_friendly_name(tensor->name); + return result; + } + + if (use_bias) { + OPENVINO_ASSERT(!layout.is_requant, + "use_bias is only used for test-backend-ops, which should not have requantization"); + // bias node will be created on the fly and not use backend buffer + output_base_ptr = nullptr; + } + + // F16 requant path - no separate scales/zp needed in result + if (layout.is_requant && layout.requant_type.has_value() && layout.requant_type.value() == ExtraQuantType::F16) { + if (output_base_ptr) { + result.weights = ov::Tensor(ov::element::f16, node_shape, + static_cast(output_base_ptr) + layout.weights_offset); + } else { + result.weights = ov::Tensor(ov::element::f16, node_shape); + } + // Not used for F16: + ov::Tensor dummy_scales; + ov::Tensor dummy_zp; + result.weight_node = + requantize_to_buffers(tensor, data, ExtraQuantType::F16, 0, result.weights, dummy_scales, dummy_zp); + return result; + } + + // Quantized path (normal extraction or quantized requant) + // Create weight/scale/zp tensors - shared between both paths + // For symmetric quantization, use signed types (i4/i8) and no ZP tensor + ov::element::Type weight_type; + if (tensor->type == GGML_TYPE_MXFP4) { + weight_type = ov::element::f4e2m1; + } else if (layout.is_symmetric) { + weight_type = layout.is_u4 ? ov::element::i4 : ov::element::i8; + } else { + weight_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; + } + ov::Shape scale_shape = node_shape; + scale_shape.back() /= layout.weights_per_block; + + if (tensor->type == GGML_TYPE_MXFP4) { + if (tensor->ne[2] == 1 && tensor->ne[3] == 1) { + node_shape = {static_cast(tensor->ne[1]), static_cast(tensor->ne[0])}; + } else { + node_shape.clear(); + for (int i = GGML_MAX_DIMS - 1; i >= 0; --i) { + node_shape.push_back(static_cast(tensor->ne[i])); + } + } + + scale_shape = node_shape; + scale_shape.back() /= layout.weights_per_block; + } + + const ov::element::Type scale_type = tensor->type == GGML_TYPE_MXFP4 ? ov::element::f8e8m0 : ov::element::f16; + ov::element::Type zp_type = layout.is_u4 ? ov::element::u4 : ov::element::u8; + if (zp_is_f16) { + zp_type = ov::element::f16; + } + + if (output_base_ptr) { + uint8_t * buf_base = static_cast(output_base_ptr); + result.weights = ov::Tensor(weight_type, node_shape, buf_base + layout.weights_offset); + result.scales = ov::Tensor(scale_type, scale_shape, buf_base + layout.scales_offset); + if (!layout.is_symmetric) { + result.zp = ov::Tensor(zp_type, scale_shape, buf_base + layout.zp_offset); + } + // else: result.zp remains default-constructed (empty) for symmetric + } else { + result.weights = ov::Tensor(weight_type, node_shape); + result.scales = ov::Tensor(scale_type, scale_shape); + if (!layout.is_symmetric) { + result.zp = ov::Tensor(zp_type, scale_shape); + } + // else: result.zp remains default-constructed (empty) for symmetric + } + + if (layout.is_requant && layout.requant_type.has_value()) { + result.weight_node = requantize_to_buffers(tensor, data, layout.requant_type.value(), layout.weights_per_block, + result.weights, result.scales, result.zp); + } else { + result.weight_node = + extract_quantized_weights(tensor, data, result.weights, result.scales, result.zp, use_bias); + } + + return result; +} diff --git a/ggml/src/ggml-openvino/ggml-quants.h b/ggml/src/ggml-openvino/ggml-quants.h index d5273727e8..04fe0218a6 100644 --- a/ggml/src/ggml-openvino/ggml-quants.h +++ b/ggml/src/ggml-openvino/ggml-quants.h @@ -2,112 +2,12 @@ #include "ggml-openvino-extra.h" // For ExtraQuantType #include "ggml.h" -#include -#include #include +#include #include -void unpack_32_4(const uint8_t * data, uint8_t * dst); - -void extract_q4_0_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void extract_q4_1_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q5_1_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q8_0_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void unpack_256_4(const uint8_t * data, uint8_t * dst); - -void extract_q4_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q5_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - bool use_bias = false); - -void extract_q6_k_data(const ggml_tensor * tensor, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr); - -void extract_mxfp4_data(const ggml_tensor * tensor, ov::Tensor & weights_arr, ov::Tensor & scales_arr); - static constexpr size_t GGML_QUANTIZATION_GROUP_SIZE = 32; -// If for_gather_matmul is true, the weight tensor may be N-D (e.g. 3D MoE expert weights -// [n_expert, rows, cols]). The dequantization chain (Convert->[Subtract]->Multiply) is built as -// usual but left in f16 (no final Convert to f32) -- ov::pass::MarkDequantization (registered in -// translate_session.cpp) marks the chain so it survives model-build-time ConstantFolding -- see -// make_int8_weights.cpp/make_int4_weights.cpp. mul_mat_id.cpp constructs ov::op::internal::GatherMatmul -// directly from the resulting f16 dequant chain. -// -// When use_bias is true (explicitly, or implicitly because for_gather_matmul is true), the zp -// tensor is expected to hold an exact f16 bias value (rather than a rounded integer zero point); -// it is converted in place into an exact zero_point = -bias/scale and consumed via Subtract, not -// Add, so the chain still matches OpenVINO's Convert->Subtract->Multiply decompression pattern. -ov::Output make_int8_weights(ov::Tensor & weight, - ov::Tensor & scales, - ov::Tensor & zp, - size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, - bool use_bias = false, - bool for_gather_matmul = false); - -ov::Output make_int4_weights(ov::Tensor & weight, - ov::Tensor & scales, - ov::Tensor & zp, - size_t group_size = GGML_QUANTIZATION_GROUP_SIZE, - bool use_bias = false, - bool for_gather_matmul = false); - -ov::Output make_mxfp4_weights(ov::Tensor & weight, ov::Tensor & scales); - -ov::Output make_mxfp4_moe_packed_weights(ov::Tensor & weight); - -// Extract quantized weights from tensor and create weight subgraph -// If weights/scales/zp are provided (non-empty), uses them as output buffers -// Otherwise allocates new ov::Tensors internally -// Returns the weight node (make_int4_weights or make_int8_weights result) -std::shared_ptr extract_quantized_weights( - const ggml_tensor * tensor, - const void * data, // Source data pointer (may differ from tensor->data) - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp, - bool use_bias = false); // Use an exact f16 zero point (vs. a rounded integer one); always - // used for for_gather_matmul (3D MoE expert) weights regardless of - // this flag, and also settable explicitly for test-backend-ops. - -// Requantize weights from tensor to target format, writing to provided buffers -// For F16 target, only weights buffer is used (scales/zp ignored) -// Returns the weight node -std::shared_ptr requantize_to_buffers(const ggml_tensor * tensor, - const void * data, // Source data pointer - ExtraQuantType requant_type, - int64_t block_size, - ov::Tensor & weights, - ov::Tensor & scales, - ov::Tensor & zp); - inline const char * extra_quant_type_name(ExtraQuantType t) { switch (t) { case ExtraQuantType::F16: @@ -156,41 +56,3 @@ OvWeight process_weight_tensor( // always used for for_gather_matmul (3D MoE expert) weights // regardless of this flag, and also settable explicitly for // test-backend-ops. - -void quantize_q4_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk); -void quantize_q8_1(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset = 0); -void quantize_q4_1_asym(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk); -void quantize_q8_0(const float * x, - ov::Tensor & weights_arr, - ov::Tensor & scales_arr, - ov::Tensor & zp_arr, - int64_t k, - int64_t qk, - int64_t block_offset = 0); - -namespace ov { -namespace op { -namespace util { -// From /src/common/transformations/include/transformations/utils/utils.hpp -bool get_single_value(const std::shared_ptr & const_node, - float & value, - bool check_value_range = true); -} // namespace util -} // namespace op -} // namespace ov diff --git a/ggml/src/ggml-openvino/model-cache.cpp b/ggml/src/ggml-openvino/model-cache.cpp index 3fc7028d88..3725fbd225 100644 --- a/ggml/src/ggml-openvino/model-cache.cpp +++ b/ggml/src/ggml-openvino/model-cache.cpp @@ -237,7 +237,8 @@ bool ggml_openvino_model_cache_verify_manifest(const std::string & path, if (!f.is_open()) { return false; } - std::string tag, val; + std::string tag; + std::string val; // header: fingerprint if (!(f >> tag >> val) || tag != "fingerprint" || val != hex64(fingerprint)) { return false; diff --git a/ggml/src/ggml-openvino/openvino/frontend.h b/ggml/src/ggml-openvino/openvino/frontend.h index 72134a3e8c..4e301d32e0 100644 --- a/ggml/src/ggml-openvino/openvino/frontend.h +++ b/ggml/src/ggml-openvino/openvino/frontend.h @@ -12,7 +12,6 @@ namespace ggml { class FrontEnd { public: - using Ptr = std::shared_ptr; FrontEnd(); static std::shared_ptr convert(const InputModel::Ptr & model, bool naive = false); diff --git a/ggml/src/ggml-openvino/openvino/op/add_id.cpp b/ggml/src/ggml-openvino/openvino/op/add_id.cpp index e54d700d42..79bdbe8773 100644 --- a/ggml/src/ggml-openvino/openvino/op/add_id.cpp +++ b/ggml/src/ggml-openvino/openvino/op/add_id.cpp @@ -20,7 +20,7 @@ namespace op { static ov::Output reshape_add_id_input_to_2d(const ov::Output & input, const ov::PartialShape & input_shape, const std::vector & dims) { - const auto actual_shape = input.get_partial_shape(); + const auto & actual_shape = input.get_partial_shape(); if (actual_shape.rank().is_static() && actual_shape.rank().get_length() == 2) { return input; } diff --git a/ggml/src/ggml-openvino/openvino/op/cont.cpp b/ggml/src/ggml-openvino/openvino/op/cont.cpp index 1d6cc67212..9888f6b93f 100644 --- a/ggml/src/ggml-openvino/openvino/op/cont.cpp +++ b/ggml/src/ggml-openvino/openvino/op/cont.cpp @@ -3,12 +3,9 @@ #include "../op_table.h" #include "../utils.h" -#include -#include #include #include #include -#include namespace ov { namespace frontend { diff --git a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp index 06547f3d29..b06d01dcac 100644 --- a/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp +++ b/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp @@ -195,7 +195,9 @@ OutputVector translate_flash_attn_ext(const NodeContext & context) { auto tile_kv = [&](int64_t n_heads, int64_t n_heads_kv, int64_t hs, ov::Output kv) { int64_t f = n_heads / n_heads_kv; if (f > 1 && n_heads_kv > 1) { - ov::Output kv_broadcast_shape, kv_unsqueezed, new_kv_shape; + ov::Output kv_broadcast_shape; + ov::Output kv_unsqueezed; + ov::Output new_kv_shape; auto unsqueeze_axes = ov::op::v0::Constant::create(ov::element::i64, Shape{}, {2}); kv_unsqueezed = std::make_shared(kv, unsqueeze_axes); diff --git a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp index 07eeb3c8fd..8d07c90bfe 100644 --- a/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp +++ b/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp @@ -196,7 +196,7 @@ static OutputVector translate_gated_delta_net_ref(const NodeContext & context) { } // Merge batch and head dims: [B*H_v, T, S_v] - auto merge_bh = [&](ov::Output x, int64_t last_dim) { + auto merge_bh = [&](const ov::Output & x, int64_t last_dim) { auto shape = ov::op::v0::Constant::create(ov::element::i64, {3}, std::vector{B * H_v, T, last_dim}); return std::make_shared(x, shape, false); }; diff --git a/ggml/src/ggml-openvino/openvino/op/im2col.cpp b/ggml/src/ggml-openvino/openvino/op/im2col.cpp index 856e97f79d..08b53f260d 100644 --- a/ggml/src/ggml-openvino/openvino/op/im2col.cpp +++ b/ggml/src/ggml-openvino/openvino/op/im2col.cpp @@ -1,7 +1,6 @@ #include "../node_context.h" #include "../op_table.h" #include "../utils.h" -#include "ggml-impl.h" #include #include diff --git a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp index 0de6161bed..a336924e14 100644 --- a/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp +++ b/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp @@ -42,7 +42,7 @@ ov::Output slice_axis(const ov::Output & input, int64_t axis ov::Output static_shape_dims_or_shapeof(const ov::Output & input, const std::vector & dims) { - const auto partial_shape = input.get_partial_shape(); + const auto & partial_shape = input.get_partial_shape(); if (partial_shape.is_static()) { std::vector values; values.reserve(dims.size()); diff --git a/ggml/src/ggml-openvino/openvino/op/pad.cpp b/ggml/src/ggml-openvino/openvino/op/pad.cpp index d2b8611423..ae3d7be18e 100644 --- a/ggml/src/ggml-openvino/openvino/op/pad.cpp +++ b/ggml/src/ggml-openvino/openvino/op/pad.cpp @@ -8,6 +8,7 @@ #include #include #include +#include #include namespace ov { @@ -20,7 +21,7 @@ namespace { ov::Output translate_circular_pad(ov::Output input, const std::array & pads, const ov::Shape & input_shape) { - ov::Output result = input; + ov::Output result = std::move(input); const std::array pads_begin = {pads[6], pads[4], pads[2], pads[0]}; const std::array pads_end = {pads[7], pads[5], pads[3], pads[1]}; diff --git a/ggml/src/ggml-openvino/openvino/op/repeat.cpp b/ggml/src/ggml-openvino/openvino/op/repeat.cpp index d58b59e4e3..b7aeaa24fa 100644 --- a/ggml/src/ggml-openvino/openvino/op/repeat.cpp +++ b/ggml/src/ggml-openvino/openvino/op/repeat.cpp @@ -1,7 +1,6 @@ #include "../node_context.h" #include "../op_table.h" #include "../utils.h" -#include "ggml.h" #include #include diff --git a/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp b/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp index 9cbce7db0d..25c9535454 100644 --- a/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp +++ b/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp @@ -25,9 +25,7 @@ OutputVector translate_rms_norm(const NodeContext & context) { auto op_case = context.get_op_case(); ov::Output input_node; - if (op_case == 1) { - input_node = process_view_input_new(context, 0); - } else if (op_case == 2) { + if (op_case == 2) { auto ssm_state_size = context.get_ssm_state_size(); // The GDN op packs [attn | new_state] along the row axis; the state occupies the last // ssm_state_size * n_seqs rows. Slice it off (scaling by the active sequence count) to keep diff --git a/ggml/src/ggml-openvino/openvino/op/view.cpp b/ggml/src/ggml-openvino/openvino/op/view.cpp index 56f5ceec9b..ca2d2dc087 100644 --- a/ggml/src/ggml-openvino/openvino/op/view.cpp +++ b/ggml/src/ggml-openvino/openvino/op/view.cpp @@ -7,7 +7,6 @@ #include #include #include -#include namespace ov { namespace frontend { @@ -153,7 +152,8 @@ OutputVector translate_view(const NodeContext & context) { return {input}; } - int64_t src_elems = 1, dst_elems = 1; + int64_t src_elems = 1; + int64_t dst_elems = 1; for (int64_t i = 0; i < src_shape.rank().get_length(); ++i) { if (src_shape[i].is_dynamic()) { return {input}; diff --git a/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp index c9952b1d52..04de2d008c 100644 --- a/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp +++ b/ggml/src/ggml-openvino/openvino/pass/kv_state_seq_axis.cpp @@ -84,7 +84,7 @@ bool KVStateSeqAxis::run_on_model(const std::shared_ptr & model) { // Readers still expect seq at dim 1. A reader that is itself the inverse // Transpose wanted seq at dim 2 all along, so drop it; give anything else the // inverse Transpose so its input is unchanged. - for (auto & reader : readers) { + for (const auto & reader : readers) { auto * node = reader.get_node(); if (ov::is_type(node)) { continue; diff --git a/ggml/src/ggml-openvino/openvino/translate_session.cpp b/ggml/src/ggml-openvino/openvino/translate_session.cpp index 3170c2e4ce..e56a4e41d0 100644 --- a/ggml/src/ggml-openvino/openvino/translate_session.cpp +++ b/ggml/src/ggml-openvino/openvino/translate_session.cpp @@ -344,7 +344,7 @@ std::shared_ptr TranslateSession::translate_graph(const frontend::InputMo } }; - auto node_visitor = [&](std::shared_ptr decoder, int node_idx) { + auto node_visitor = [&](const std::shared_ptr & decoder, int node_idx) { auto converted_outputs = translate_node(decoder, node_idx); if (converted_outputs.empty()) { return; diff --git a/ggml/src/ggml-openvino/openvino/utils.cpp b/ggml/src/ggml-openvino/openvino/utils.cpp index 8bb7678ee3..98a85e632a 100644 --- a/ggml/src/ggml-openvino/openvino/utils.cpp +++ b/ggml/src/ggml-openvino/openvino/utils.cpp @@ -1,7 +1,5 @@ #include "utils.h" -#include "ggml-impl.h" - #include #include #include @@ -28,13 +26,6 @@ namespace ov { namespace frontend { namespace ggml { -std::string getCurrentTime() { - std::time_t now = std::time(nullptr); - char buf[100]; - std::strftime(buf, sizeof(buf), "%Y-%m-%d %H:%M:%S", std::localtime(&now)); - return buf; -} - void num_inputs_check(const NodeContext & context, size_t min_inputs, size_t max_inputs) { auto input_size = context.get_input_size(); FRONT_END_OP_CONVERSION_CHECK(input_size >= min_inputs, "Got less inputs than expected"); @@ -82,7 +73,7 @@ namespace { ov::Output rope_yarn_ramp_mix(int n_dims, const float corr_dims[2], float ext_factor) { int half_n_dims = n_dims / 2; std::vector dim_ids_vec(half_n_dims); - std::iota(dim_ids_vec.begin(), dim_ids_vec.end(), 0); + std::iota(dim_ids_vec.begin(), dim_ids_vec.end(), 0.0f); auto dim_ids = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, (size_t) half_n_dims}, dim_ids_vec); auto corr_low = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, 1}, {corr_dims[0]}); auto corr_high = ov::op::v0::Constant::create(ov::element::f32, Shape{1, 1, 1, 1}, {corr_dims[1]}); @@ -551,6 +542,7 @@ ov::Output process_view_input_new(const NodeContext & context, int inp if (tail_begin >= 0 && tail_end <= tail_src_elems) { std::vector flat_shape; + flat_shape.reserve(slice_dim); for (int i = 0; i < slice_dim; ++i) { flat_shape.push_back(static_cast(view_src_ggml_shape[i])); } diff --git a/ggml/src/ggml-openvino/openvino/utils.h b/ggml/src/ggml-openvino/openvino/utils.h index 5d4c353866..d9858f9236 100644 --- a/ggml/src/ggml-openvino/openvino/utils.h +++ b/ggml/src/ggml-openvino/openvino/utils.h @@ -14,8 +14,6 @@ namespace ggml { std::string getCurrentTime(); -void dump_ov_model(std::shared_ptr model); - void num_inputs_check(const NodeContext & context, size_t min_inputs, size_t max_inputs); int non_cont_dim(std::vector ne, std::vector nb); diff --git a/ggml/src/ggml-openvino/utils.cpp b/ggml/src/ggml-openvino/utils.cpp index 44a9b2c788..b1ee792fdb 100644 --- a/ggml/src/ggml-openvino/utils.cpp +++ b/ggml/src/ggml-openvino/utils.cpp @@ -1,8 +1,8 @@ #include "utils.h" #include "ggml-impl.h" -#include "ggml-openvino.h" #include "ggml-openvino-extra.h" +#include "ggml-openvino.h" #include "ggml-openvino/ggml-decoder.h" #include "ggml.h" #include "model-cache.h" @@ -18,7 +18,6 @@ #include #include #include -#include #include #include #include @@ -39,42 +38,7 @@ #include #include -// Suppress deprecation warning for ov::Tensor::data() -#pragma GCC diagnostic push -#pragma GCC diagnostic ignored "-Wdeprecated-declarations" - -// Both execution paths use two cache levels: -// 1. Reuse this backend's decoder/request via graph_key and compatibility checks. -// 2. On a local miss, look up compiled_graph_key in the shared compilation cache, -// compile if needed, then create a private request from the compiled model. -// The shared lock covers compilation and frontend cleanup, never inference. -enum ggml_status ov_graph_compute(ggml_cgraph * cgraph, ggml_backend_t backend) { - ggml_backend_openvino_context * ctx = (ggml_backend_openvino_context *) backend->context; - try { - if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_CGRAPH")) { - std::string filename = "cgraph_ov.txt"; - GgmlOvDecoder::dump_cgraph(cgraph, filename); - } - - const auto is_static = ggml_openvino_is_npu() || ggml_openvino_getenv_int("GGML_OPENVINO_FORCE_STATIC"); - - GGML_ASSERT(ctx->runtime_context != nullptr); - std::shared_ptr r_ctx = std::static_pointer_cast(ctx->runtime_context); - std::lock_guard execution_lock(r_ctx->execution_mutex); - - return is_static ? ov_graph_compute_static(cgraph, r_ctx) : ov_graph_compute_dynamic(cgraph, r_ctx); - } catch (const ov::Exception & e) { - GGML_LOG_ERROR("GGML OpenVINO backend ov::Exception: %s\n", e.what()); - return GGML_STATUS_FAILED; - } catch (const std::exception & e) { - GGML_LOG_ERROR("GGML OpenVINO backend std::exception: %s\n", e.what()); - return GGML_STATUS_FAILED; - } catch (...) { - GGML_LOG_ERROR("GGML OpenVINO backend unknown exception\n"); - return GGML_STATUS_FAILED; - } -} - +namespace { // For a KV cache input, return an ov::Tensor sized to n_kv (== attention_size // for that layer) instead of the fully-allocated ctx_per_seq. Pre-conditions: // * non-static (CPU/GPU) backend, single sequence, seq_active_start == 0 @@ -85,9 +49,9 @@ enum ggml_status ov_graph_compute(ggml_cgraph * cgraph, ggml_backend_t backend) // n_kv rows no longer contain the live prefix // On any unmet pre-condition returns std::nullopt; the caller falls back to // the full-size tensor. -static std::optional try_make_kv_sliced_tensor(std::shared_ptr ggml_decoder, - const std::string & name, - const ggml_tensor * ggml_tensor) { +std::optional try_make_kv_sliced_tensor(const std::shared_ptr & ggml_decoder, + const std::string & name, + const ggml_tensor * ggml_tensor) { static const bool kv_slice_disabled = ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_SLICE"); if (kv_slice_disabled) { return std::nullopt; @@ -125,7 +89,7 @@ static std::optional try_make_kv_sliced_tensor(std::shared_ptrget_shape(ggml_tensor); + ov::Shape full_shape = GgmlOvDecoder::get_shape(ggml_tensor); if (full_shape.size() != 4 || full_shape[0] != 1 || full_shape[1] != 1 || static_cast(full_shape[2]) != ctx_per_seq) { return std::nullopt; @@ -141,10 +105,10 @@ static std::optional try_make_kv_sliced_tensor(std::shared_ptrget_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); // } - return ov::Tensor(ggml_decoder->get_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); + return ov::Tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), sliced_shape, ggml_tensor->data); } -static uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, bool stateful) { +uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, bool stateful) { const char * manual_gqa_env = ggml_openvino_getenv_str("GGML_OPENVINO_MANUAL_GQA_ATTN"); const bool manual_gqa_enabled = manual_gqa_env != nullptr ? ggml_openvino_getenv_int("GGML_OPENVINO_MANUAL_GQA_ATTN") > 0 : @@ -158,7 +122,7 @@ static uint64_t ggml_openvino_model_cache_extra_cfg(const std::string & device, return extra_cfg; } -static std::map> get_weight_names(ggml_cgraph * cgraph) { +std::map> get_weight_names(ggml_cgraph * cgraph) { std::map> names; for (const auto & name : GgmlOvDecoder::collect_weight_names(cgraph)) { names[name] = nullptr; @@ -170,8 +134,10 @@ static std::map> get_weight_names(ggml_cg // miss. Include topology, layouts, op parameters, constant extra inputs and weight // allocation identities. Never use a sampled weight hash or a graph name alone: // different models can have identical topology. OV buffer IDs survive address reuse. -static std::string compiled_graph_key(const ggml_cgraph * graph, const GgmlOvDecoder & decoder, - const std::string & device, int prefill_chunk_size = 0) { +std::string compiled_graph_key(const ggml_cgraph * graph, + const GgmlOvDecoder & decoder, + const std::string & device, + int prefill_chunk_size = 0) { std::string key; auto append = [&key](const auto & value) { key.append(reinterpret_cast(&value), sizeof(value)); @@ -243,8 +209,8 @@ static std::string compiled_graph_key(const ggml_cgraph * graph, const GgmlOvDec return has_weight_buffer_id ? key : std::string{}; } -ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, - std::shared_ptr infer_request, +ov::Tensor create_ov_output_tensor(const std::shared_ptr & ggml_decoder, + const std::shared_ptr & infer_request, int output_index, const ggml_tensor * ggml_tensor) { if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, std::string(ggml_tensor->name), ggml_tensor)) { @@ -260,7 +226,7 @@ ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, // } // } - auto output_type = ggml_decoder->get_ov_type(ggml_tensor); + auto output_type = GgmlOvDecoder::get_ov_type(ggml_tensor); ov::Shape output_shape; void * output_data = ggml_tensor->data; if (ggml_decoder->is_static()) { @@ -273,10 +239,10 @@ ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, // pointer instead so the OV tensor matches the model output exactly. if (ggml_tensor->op == GGML_OP_CPY && ggml_tensor->view_src != nullptr && ggml_nbytes(ggml_tensor) != ggml_nbytes(ggml_tensor->view_src)) { - output_shape = ggml_decoder->get_shape(ggml_tensor->view_src); + output_shape = GgmlOvDecoder::get_shape(ggml_tensor->view_src); output_data = ggml_tensor->view_src->data; } else { - output_shape = ggml_decoder->get_shape(ggml_tensor); + output_shape = GgmlOvDecoder::get_shape(ggml_tensor); } } ov::Tensor output_tensor(output_type, output_shape, output_data); @@ -286,7 +252,7 @@ ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, // Rewrite ggml's KV rows into a relayout state that keeps the sequence on dim 2. // ggml stores [seq][n_heads_kv * head_size]; the state wants [1, n_heads_kv, seq, head_size], // a different element order, so the rows are copied instead of reinterpreted. -static ov::Tensor kv_rows_to_seq_axis_2(const ov::Tensor & kv_tensor, size_t n_heads_kv) { +ov::Tensor kv_rows_to_seq_axis_2(const ov::Tensor & kv_tensor, size_t n_heads_kv) { const size_t rows = kv_tensor.get_shape()[2]; const size_t head_size = kv_tensor.get_shape()[3] / n_heads_kv; const size_t elem = kv_tensor.get_element_type().size(); @@ -303,7 +269,366 @@ static ov::Tensor kv_rows_to_seq_axis_2(const ov::Tensor & kv_tensor, size_t n_h return out; } -enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr r_ctx) { +template void set_zero_diagonal(std::vector & matrix, size_t rows, size_t cols, T zero_value = T{}) { + for (size_t i = 0; i < rows; ++i) { + size_t diag_col = std::min(i, cols - 1); + matrix[i * cols + diag_col] = zero_value; + } +} + +ov::Tensor make_contiguous_split_input_tensor(const struct ggml_tensor * ggml_tensor, const ov::Shape & input_shape) { + const size_t element_size = ggml_type_size(ggml_tensor->type); + const size_t block_size = ggml_blck_size(ggml_tensor->type); + + GGML_ASSERT(block_size == 1 && "non-contiguous split inputs must be plain element types"); + + const struct ggml_tensor * source_tensor = ggml_tensor->view_src != nullptr ? ggml_tensor->view_src : ggml_tensor; + const size_t source_offset = ggml_tensor->view_src != nullptr ? ggml_tensor->view_offs : 0; + + std::vector source_data(ggml_nbytes(source_tensor)); + ggml_backend_tensor_get(source_tensor, source_data.data(), 0, source_data.size()); + + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + auto * dst = static_cast(input_tensor.data()); + size_t dst_offset = 0; + + for (size_t i3 = 0; i3 < static_cast(ggml_tensor->ne[3]); ++i3) { + for (size_t i2 = 0; i2 < static_cast(ggml_tensor->ne[2]); ++i2) { + for (size_t i1 = 0; i1 < static_cast(ggml_tensor->ne[1]); ++i1) { + for (size_t i0 = 0; i0 < static_cast(ggml_tensor->ne[0]); ++i0) { + const size_t src_offset = source_offset + i3 * ggml_tensor->nb[3] + i2 * ggml_tensor->nb[2] + + i1 * ggml_tensor->nb[1] + i0 * ggml_tensor->nb[0]; + std::memcpy(dst + dst_offset, source_data.data() + src_offset, element_size); + dst_offset += element_size; + } + } + } + } + + return input_tensor; +} + +ov::Tensor convert_ggml_input_to_ov(const std::shared_ptr & ggml_decoder, const std::string & name) { + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(name); + + if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, name, ggml_tensor)) { + return *sliced; + } + + if (ggml_tensor->extra != nullptr && !ggml_decoder->is_splited_model()) { + auto * extra_base = static_cast(ggml_tensor->extra); + if (extra_base->type == ggml_openvino_extra_base::Type::TENSOR) { + // GGML_LOG_DEBUG("Using ggml_tensor->extra as ov::Tensor for input: %s\n", name.c_str()); + auto * tensor_extra = static_cast(extra_base); + return *tensor_extra->tensor; + } + } + + // GGML_LOG_DEBUG("Converting ggml tensor to ov::Tensor for input: %s\n", name.c_str()); + auto * input_data = ggml_tensor->data; + ov::Shape input_shape; + if (ggml_tensor->op == GGML_OP_VIEW && !ggml_decoder->is_splited_model()) { + // This case is added to make test-backend-ops work + input_shape = GgmlOvDecoder::get_shape(ggml_tensor->view_src); + } else { + input_shape = GgmlOvDecoder::get_shape(ggml_tensor); + } + + if (ggml_decoder->is_splited_model() && !ggml_is_contiguous(ggml_tensor)) { + return make_contiguous_split_input_tensor(ggml_tensor, input_shape); + } + + auto input_tensor = ov::Tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape, input_data); + return input_tensor; +} + +ov::Tensor get_ov_input_tensor(const std::shared_ptr & ggml_decoder, const std::string & param_name) { + ov::Tensor input_tensor; + auto extra_input = ggml_decoder->get_model_extra_inputs().find(param_name); + if (extra_input != ggml_decoder->get_model_extra_inputs().end()) { + input_tensor = ov::Tensor(extra_input->second.type, extra_input->second.shape); + *input_tensor.data() = extra_input->second.value; + } else { + input_tensor = convert_ggml_input_to_ov(ggml_decoder, param_name); + } + return input_tensor; +} + +ov::Tensor get_ov_input_tensor_static_decode(const std::shared_ptr & ggml_decoder, + const std::string & param_name) { + // NPU decoding stage + if (ggml_decoder->get_model_extra_inputs().count(param_name)) { + return get_ov_input_tensor(ggml_decoder, param_name); + } + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); + const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); + + if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || + GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { + // IMROPE's inp_pos holds one value per t/h/w/e plane instead of a single position; + // with a single decode token the planes are still contiguous, so a flat copy works. + const int n_planes = GgmlOvDecoder::is_inp_pos(ggml_tensor, op) ? GgmlOvDecoder::get_inp_pos_n_planes(op) : 1; + assert(ggml_tensor->ne[0] == n_planes); + ov::Shape input_shape = {1, 1, 1, (size_t) n_planes}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + std::memcpy(input_tensor.data(), ggml_tensor->data, n_planes * ggml_type_size(ggml_tensor->type)); + return input_tensor; + } + + if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { + ov::Shape input_shape = {1, 1, 1, 1}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + int32_t inp_out_id = *((int32_t *) ggml_tensor->data); + assert(ggml_tensor->ne[0] == 1); + assert(inp_out_id == 0); + *input_tensor.data() = inp_out_id; + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { + size_t context_size = ggml_decoder->get_ctx_size(); + if (ggml_tensor->type == GGML_TYPE_F16) { + std::vector padded_data = + pad_input(ggml_tensor, 1, context_size, GGML_FP32_TO_FP16(-INFINITY)); + ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, 1, context_size}); + std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); + return input_tensor; + } + + std::vector padded_data = pad_input(ggml_tensor, 1, context_size, -INFINITY); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, 1, context_size}); + auto * data_ptr = input_tensor.data(); + std::copy(padded_data.begin(), padded_data.begin() + context_size, data_ptr); + return input_tensor; + } + + return get_ov_input_tensor(ggml_decoder, param_name); +} + +ov::Tensor get_ov_input_tensor_static_prefill(const std::shared_ptr & ggml_decoder, + const std::string & param_name, + int chunk_index) { + // NPU prompt processing stage + const size_t input_len = ggml_decoder->get_input_len(); + const size_t chunk_size = ggml_decoder->m_prefill_chunk_size; + const size_t chunk_valid_size = std::min(chunk_size, input_len - chunk_index * chunk_size); + const size_t chunk_pad_size = chunk_size - chunk_valid_size; + + if (param_name == "chunk_valid_len") { + ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); + *input_tensor.data() = (int64_t) chunk_valid_size; + return input_tensor; + } + if (chunk_index > 0 && param_name == "cache_rs_reset_len") { + // The recurrent-state clear belongs to the start of the sequence. Re-applying it on every + // chunk would wipe the state accumulated by the preceding chunks, so disable it (a zero + // length makes scale.cpp's keep-mask select every slot) after the first chunk. + ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); + *input_tensor.data() = 0; + return input_tensor; + } + if (ggml_decoder->get_model_extra_inputs().count(param_name)) { + return get_ov_input_tensor(ggml_decoder, param_name); + } + const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); + const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); + + if (GgmlOvDecoder::is_inp_pos(ggml_tensor, op) && GgmlOvDecoder::get_inp_pos_n_planes(op) > 1) { + // IMROPE: inp_pos stacks n_planes (t/h/w/e) position planes, each of length + // input_len; pad every plane independently so they stay aligned to chunk_size. + const int n_planes = GgmlOvDecoder::get_inp_pos_n_planes(op); + const size_t element_size = ggml_type_size(ggml_tensor->type); + ov::Shape input_shape = {1, 1, 1, (size_t) n_planes * chunk_size}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + for (int p = 0; p < n_planes; p++) { + const char * src = + (const char *) ggml_tensor->data + (p * input_len + chunk_index * chunk_size) * element_size; + char * dst = (char *) input_tensor.data() + p * chunk_size * element_size; + std::memcpy(dst, src, chunk_valid_size * element_size); + if (chunk_pad_size > 0) { + if (ggml_tensor->type == GGML_TYPE_I32) { + int32_t last_value = *((const int32_t *) src + chunk_valid_size - 1); + int32_t * out = (int32_t *) dst; + std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); + } else if (ggml_tensor->type == GGML_TYPE_I64) { + int64_t last_value = *((const int64_t *) src + chunk_valid_size - 1); + int64_t * out = (int64_t *) dst; + std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); + } else { + throw std::runtime_error("Unexpected tensor type for " + param_name); + } + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || + GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { + ov::Shape input_shape = {1, 1, 1, chunk_size}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + // copy the chunk_index-th chunk from ggml_tensor + size_t element_size = ggml_type_size(ggml_tensor->type); + void * input_data = (char *) ggml_tensor->data + chunk_index * chunk_size * element_size; + std::memcpy(input_tensor.data(), input_data, chunk_valid_size * element_size); + // pad the rest with last_value + 1, so that kv's of padded positions are inserted + // to the next row after the valids row in the kvcache + if (chunk_pad_size > 0) { + if (ggml_tensor->type == GGML_TYPE_I32) { + int32_t last_value = + *((int32_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); + int32_t * output_data = input_tensor.data(); + std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); + } else if (ggml_tensor->type == GGML_TYPE_I64) { + int64_t last_value = + *((int64_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); + int64_t * output_data = input_tensor.data(); + std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); + } else { + throw std::runtime_error("Unexpected tensor type for " + param_name); + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { + size_t output_len = ggml_decoder->get_compute_params().output_len; + ov::Shape input_shape = {1, 1, 1, output_len}; + ov::Tensor input_tensor(GgmlOvDecoder::get_ov_type(ggml_tensor), input_shape); + if (ggml_tensor->ne[0] == 0) { + *input_tensor.data() = 0; + } else { + auto * data_addr = input_tensor.data(); + for (size_t i = 0; i < output_len; i++) { + data_addr[i] = ((int32_t *) ggml_tensor->data)[i] % chunk_size; + } + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mean(ggml_tensor, op)) { + const size_t n_seqs = ggml_tensor->ne[1]; + const size_t src_stride = ggml_tensor->ne[0]; + const size_t copy_len = std::min(chunk_valid_size, src_stride - chunk_index * chunk_size); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, n_seqs, chunk_size}); + auto * dst = input_tensor.data(); + std::fill(dst, dst + n_seqs * chunk_size, 0.0f); + const auto * src = static_cast(ggml_tensor->data) + chunk_index * chunk_size; + for (size_t s = 0; s < n_seqs; s++) { + std::memcpy(dst + s * chunk_size, src + s * src_stride, copy_len * sizeof(float)); + } + return input_tensor; + } + + if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { + size_t cols = ggml_tensor->ne[0]; + size_t rows = ggml_tensor->ne[1]; + size_t chunk_valid_rows = std::min(chunk_size, rows - chunk_index * chunk_size); + size_t context_size = ggml_decoder->get_ctx_size(); + if (ggml_tensor->type == GGML_TYPE_F16) { + const auto * ggml_data = + static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; + std::vector padded_data = pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, + context_size, GGML_FP32_TO_FP16(-INFINITY)); + set_zero_diagonal(padded_data, chunk_size, context_size, GGML_FP32_TO_FP16(0.0f)); + ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, chunk_size, context_size}); + std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); + return input_tensor; + } + + const auto * ggml_data = static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; + std::vector padded_data = + pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, context_size, -INFINITY); + set_zero_diagonal(padded_data, chunk_size, context_size); + ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, chunk_size, context_size}); + auto * data_ptr = input_tensor.data(); + std::copy(padded_data.begin(), padded_data.begin() + chunk_size * context_size, data_ptr); + return input_tensor; + } + + return get_ov_input_tensor(ggml_decoder, param_name); +} + +enum ggml_status naive_compute(ggml_cgraph * cgraph, + ov::Core & core, + const std::string & device, + const ov::AnyMap & config, + ov_compiled_model_cache & cache) { + if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { + return GGML_STATUS_SUCCESS; + } + + std::unique_lock compile_lock(cache.mutex); + bool naive = true; + auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); + auto decoder = std::make_shared(cgraph, model_weights); + auto input_model = std::make_shared(decoder); + auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); + if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { + ov::serialize(model, "IR_naive.xml"); + } + + std::shared_ptr infer_request; + auto remote_context = ggml_openvino_get_remote_context(); + ov::AnyMap compile_config = config; + if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { + // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; + } else { + compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; + } + if (remote_context.has_value()) { + infer_request = std::make_shared( + core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); + } else { + infer_request = std::make_shared( + core.compile_model(model, device, compile_config).create_infer_request()); + } + std::vector input_names; + std::vector output_names; + for (const auto & param : model->get_parameters()) { + input_names.push_back(param->get_friendly_name()); + } + for (const auto & result : model->get_results()) { + output_names.push_back(result->get_friendly_name()); + } + // Destroy the frontend graph under the compilation lock as well: it can + // still own edges into the shared weight nodes. + model.reset(); + input_model.reset(); + decoder->clear_model_weights(); + model_weights.clear(); + compile_lock.unlock(); + + for (size_t i = 0; i < input_names.size(); i++) { + const auto & param_name = input_names[i]; + auto input_tensor = get_ov_input_tensor(decoder, param_name); + infer_request->set_input_tensor(i, input_tensor); + } + + // Use get_output_tensor + memcpy instead of set_output_tensor to avoid memory overwritten + // when i/o buffer overlaps, e.g. the cgraph is a single PERMUTE + + infer_request->infer(); + + for (size_t i = 0; i < output_names.size(); i++) { + auto output_tensor = infer_request->get_output_tensor(i); + const auto & model_outputs = decoder->get_model_outputs(); + auto model_output_it = model_outputs.find(output_names[i]); + if (model_output_it == model_outputs.end()) { + // Debug-only output added via GGML_OPENVINO_DEBUG_NODE; nothing to copy into. + if (ggml_openvino_getenv_int("GGML_OPENVINO_DEBUG_OUTPUT") || + ggml_openvino_getenv_str("GGML_OPENVINO_DEBUG_NODE")) { + print_output_tensor_info(output_names[i], output_tensor, output_tensor.data()); + } + continue; + } + auto * ggml_tensor = model_output_it->second; + std::memcpy(ggml_tensor->data, output_tensor.data(), output_tensor.get_byte_size()); + } + return GGML_STATUS_SUCCESS; +} + +enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, const std::shared_ptr & r_ctx) { auto & core = ov_singleton_core(); const auto & config = ggml_openvino_get_compile_config(); const auto & device = r_ctx->device; @@ -403,7 +728,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< if (stateful) { const auto * inp_pos = get_inp_pos_tensor(cgraph); int32_t * pos_data = (int32_t *) inp_pos->data; - auto pos_shape = ggml_decoder->get_shape(inp_pos); + auto pos_shape = GgmlOvDecoder::get_shape(inp_pos); if (pos_data[0] == 0) { infer_request->reset_state(); r_ctx->stateful_kv_size = pos_shape[3]; @@ -430,8 +755,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } } - const bool relayout_enabled = - !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT"); + const bool relayout_enabled = !ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT"); auto states = infer_request->query_state(); for (auto state : states) { @@ -506,8 +830,8 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< auto shared_cache = r_ctx->compiled_cache; std::unique_lock compile_lock(shared_cache->mutex); auto weight_names = get_weight_names(cgraph); - ggml_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, - is_static, stateful, model_is_splitted); + ggml_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, is_static, + stateful, model_is_splitted); const std::string shared_key = cache_enabled ? compiled_graph_key(cgraph, *ggml_decoder, device) : ""; ov::CompiledModel shared_model; bool imported = false; @@ -543,7 +867,8 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< // the weights are baked into the imported CompiledModel. const std::string model_cache_dir = ggml_openvino_model_cache_dir(); uint64_t model_fp = 0; - std::string blob_path, manifest_path; + std::string blob_path; + std::string manifest_path; // When the frontend model cache is active it supersedes the plugin-level // ov::cache_dir: a blob exported from a model compiled WITH cache_dir cannot // be re-imported (import returns an uninitialized model). Strip cache_dir / @@ -555,14 +880,15 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } if (!imported && !model_cache_dir.empty() && !model_is_splitted) { const uint64_t extra_cfg = ggml_openvino_model_cache_extra_cfg(device, stateful); - model_fp = ggml_openvino_model_fingerprint(cgraph, device, /*fa=*/true, m_params.rope_params, - 16, extra_cfg); + model_fp = + ggml_openvino_model_fingerprint(cgraph, device, /*fa=*/true, m_params.rope_params, 16, extra_cfg); blob_path = ggml_openvino_model_cache_blob_path(model_cache_dir, model_fp); manifest_path = ggml_openvino_model_cache_manifest_path(model_cache_dir, model_fp); std::ifstream blob_in(blob_path, std::ios::binary); bool blob_ok = blob_in.is_open(); - bool manifest_ok = blob_ok && ggml_openvino_model_cache_verify_manifest(manifest_path, cgraph, model_fp); + bool manifest_ok = + blob_ok && ggml_openvino_model_cache_verify_manifest(manifest_path, cgraph, model_fp); if (blob_ok && manifest_ok) { int64_t import_start = ggml_time_us(); try { @@ -689,8 +1015,8 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< entry->ptr = ggml_decoder; if (!shared_key.empty() && shared_it == shared_cache->graphs.end()) { - shared_cache->graphs.emplace(shared_key, ov_compiled_graph{shared_model, {}, ov_input_names, - ov_output_names}); + shared_cache->graphs.emplace(shared_key, + ov_compiled_graph{shared_model, {}, ov_input_names, ov_output_names}); } if (cache_enabled) { std::lock_guard map_lock(r_ctx->ctx_mutex); @@ -701,7 +1027,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< if (stateful && cache_enabled) { const auto * inp_pos = get_inp_pos_tensor(cgraph); - auto pos_shape = ggml_decoder->get_shape(inp_pos); + auto pos_shape = GgmlOvDecoder::get_shape(inp_pos); // A freshly compiled model starts with an empty state, so it can only serve a // sequence from its beginning. A non-zero start position means the KV history was // built elsewhere (a restored ggml cache), which the state cannot adopt. @@ -723,7 +1049,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< } for (size_t i = 0; i < ov_input_names.size(); i++) { - auto param_name = ov_input_names[i]; + const auto & param_name = ov_input_names[i]; auto input_tensor = get_ov_input_tensor(ggml_decoder, param_name); infer_request->set_input_tensor(i, input_tensor); @@ -792,7 +1118,7 @@ enum ggml_status ov_graph_compute_dynamic(ggml_cgraph * cgraph, std::shared_ptr< return GGML_STATUS_SUCCESS; } -static ov::AnyMap without_npuw(const ov::AnyMap & config) { +ov::AnyMap without_npuw(const ov::AnyMap & config) { ov::AnyMap out; for (const auto & kv : config) { if (kv.first.rfind("NPUW", 0) == 0 || kv.first == "NPU_USE_NPUW") { @@ -803,7 +1129,7 @@ static ov::AnyMap without_npuw(const ov::AnyMap & config) { return out; } -enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptr r_ctx) { +enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, const std::shared_ptr & r_ctx) { auto & core = ov_singleton_core(); auto get_prefill_chunk_size = [] { @@ -919,16 +1245,17 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrcompiled_cache; std::unique_lock compile_lock(shared_cache->mutex); auto weight_names = get_weight_names(cgraph); - auto local_decoder = std::make_shared( - cgraph, m_params, c_params, weight_names, is_static, stateful, false, is_prefill, prefill_chunk_size); - const std::string shared_key = cache_enabled ? - compiled_graph_key(cgraph, *local_decoder, device, prefill_chunk_size) : ""; + auto local_decoder = std::make_shared(cgraph, m_params, c_params, weight_names, is_static, + stateful, false, is_prefill, prefill_chunk_size); + const std::string shared_key = + cache_enabled ? compiled_graph_key(cgraph, *local_decoder, device, prefill_chunk_size) : ""; auto shared_it = shared_cache->graphs.find(shared_key); if (!shared_key.empty() && shared_it != shared_cache->graphs.end()) { auto & compiled = shared_it->second; auto prefill_request = std::make_shared(compiled.prefill.create_infer_request()); - auto decode_request = no_kv_cache ? prefill_request : - std::make_shared(compiled.decode.create_infer_request()); + auto decode_request = no_kv_cache ? + prefill_request : + std::make_shared(compiled.decode.create_infer_request()); ggml_decoder = local_decoder; entry->ptr = ggml_decoder; infer_request = is_prefill ? prefill_request : decode_request; @@ -956,13 +1283,10 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptr(ggml_time_us()); auto build_static_model = [&core, &compile_config, dump_ir, dump_ir_timestamp]( - std::shared_ptr decoder, - const char * tag, - std::shared_ptr & model, - ov::CompiledModel & compiled_model, - std::shared_ptr & infer_request, - int64_t & local_conversion_end_time, - int64_t & local_compile_end_time) { + const std::shared_ptr & decoder, const char * tag, + std::shared_ptr & model, ov::CompiledModel & compiled_model, + std::shared_ptr & infer_request, + int64_t & local_conversion_end_time, int64_t & local_compile_end_time) { auto input_model = std::make_shared(decoder); model = ov::frontend::ggml::FrontEnd::convert(input_model); decoder->clear_model_weights(); @@ -990,7 +1314,7 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrgraphs.emplace(shared_key, ov_compiled_graph{compiled_model_decode, compiled_model_prefill, - ov_input_names_local, ov_output_names_local}); + shared_cache->graphs.emplace( + shared_key, ov_compiled_graph{compiled_model_decode, compiled_model_prefill, ov_input_names_local, + ov_output_names_local}); } if (cache_enabled) { @@ -1029,14 +1354,13 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrov_output_names_cache[key] = ov_output_names_local; } } - } if (is_prefill) { auto inp_len = get_inp_pos_n_tokens(cgraph, inp_pos); for (int chunk_index = 0; chunk_index * prefill_chunk_size < inp_len; chunk_index++) { for (size_t i = 0; i < ov_input_names_local.size(); i++) { - auto param_name = ov_input_names_local[i]; + const auto & param_name = ov_input_names_local[i]; auto input_tensor = get_ov_input_tensor_static_prefill(ggml_decoder, param_name, chunk_index); infer_request->set_input_tensor(i, input_tensor); @@ -1077,7 +1401,7 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrset_input_tensor(i, input_tensor); @@ -1128,6 +1452,39 @@ enum ggml_status ov_graph_compute_static(ggml_cgraph * cgraph, std::shared_ptrcontext; + try { + if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_CGRAPH")) { + std::string filename = "cgraph_ov.txt"; + GgmlOvDecoder::dump_cgraph(cgraph, filename); + } + + const auto is_static = ggml_openvino_is_npu() || ggml_openvino_getenv_int("GGML_OPENVINO_FORCE_STATIC"); + + GGML_ASSERT(ctx->runtime_context != nullptr); + std::shared_ptr r_ctx = std::static_pointer_cast(ctx->runtime_context); + std::lock_guard execution_lock(r_ctx->execution_mutex); + + return is_static ? ov_graph_compute_static(cgraph, r_ctx) : ov_graph_compute_dynamic(cgraph, r_ctx); + } catch (const ov::Exception & e) { + GGML_LOG_ERROR("GGML OpenVINO backend ov::Exception: %s\n", e.what()); + return GGML_STATUS_FAILED; + } catch (const std::exception & e) { + GGML_LOG_ERROR("GGML OpenVINO backend std::exception: %s\n", e.what()); + return GGML_STATUS_FAILED; + } catch (...) { + GGML_LOG_ERROR("GGML OpenVINO backend unknown exception\n"); + return GGML_STATUS_FAILED; + } +} // Detect whether a cgraph is a split subgraph or not. // Step 1 compares each node's recorded use_count with actual fan-out references in node->src. @@ -1217,369 +1574,6 @@ bool is_naive(ggml_cgraph * cgraph) { return count < naive_graph_size_threshold; } -enum ggml_status naive_compute(ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config, - ov_compiled_model_cache & cache) { - if (cgraph->n_nodes == 1 && (cgraph->nodes[0]->op == GGML_OP_NONE || cgraph->nodes[0]->op == GGML_OP_VIEW)) { - return GGML_STATUS_SUCCESS; - } - - std::unique_lock compile_lock(cache.mutex); - bool naive = true; - auto model_weights = GgmlOvDecoder::create_weight_nodes(cgraph, naive); - auto decoder = std::make_shared(cgraph, model_weights); - auto input_model = std::make_shared(decoder); - auto model = ov::frontend::ggml::FrontEnd::convert(input_model, naive); - if (ggml_openvino_getenv_int("GGML_OPENVINO_DUMP_IR")) { - ov::serialize(model, "IR_naive.xml"); - } - - std::shared_ptr infer_request; - auto remote_context = ggml_openvino_get_remote_context(); - ov::AnyMap compile_config = config; - if (cgraph->nodes[0]->op == GGML_OP_MUL_MAT) { - // TODO ACCURACY hint triggers a bug in GPU plugin/driver on Lunar Lake. Remove once CVS-182166 is resolved - compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::PERFORMANCE; - } else { - compile_config[ov::hint::execution_mode.name()] = ov::hint::ExecutionMode::ACCURACY; - } - if (remote_context.has_value()) { - infer_request = std::make_shared( - core.compile_model(model, remote_context.value(), compile_config).create_infer_request()); - } else { - infer_request = - std::make_shared(core.compile_model(model, device, compile_config).create_infer_request()); - } - std::vector input_names; - std::vector output_names; - for (const auto & param : model->get_parameters()) { - input_names.push_back(param->get_friendly_name()); - } - for (const auto & result : model->get_results()) { - output_names.push_back(result->get_friendly_name()); - } - // Destroy the frontend graph under the compilation lock as well: it can - // still own edges into the shared weight nodes. - model.reset(); - input_model.reset(); - decoder->clear_model_weights(); - model_weights.clear(); - compile_lock.unlock(); - - for (size_t i = 0; i < input_names.size(); i++) { - const auto & param_name = input_names[i]; - auto input_tensor = get_ov_input_tensor(decoder, param_name); - infer_request->set_input_tensor(i, input_tensor); - } - - // Use get_output_tensor + memcpy instead of set_output_tensor to avoid memory overwritten - // when i/o buffer overlaps, e.g. the cgraph is a single PERMUTE - - infer_request->infer(); - - for (size_t i = 0; i < output_names.size(); i++) { - auto output_tensor = infer_request->get_output_tensor(i); - const auto & model_outputs = decoder->get_model_outputs(); - auto model_output_it = model_outputs.find(output_names[i]); - if (model_output_it == model_outputs.end()) { - // Debug-only output added via GGML_OPENVINO_DEBUG_NODE; nothing to copy into. - if (ggml_openvino_getenv_int("GGML_OPENVINO_DEBUG_OUTPUT") || - ggml_openvino_getenv_str("GGML_OPENVINO_DEBUG_NODE")) { - print_output_tensor_info(output_names[i], output_tensor, output_tensor.data()); - } - continue; - } - auto * ggml_tensor = model_output_it->second; - std::memcpy(ggml_tensor->data, output_tensor.data(), output_tensor.get_byte_size()); - } - return GGML_STATUS_SUCCESS; -} - -namespace { -template void set_zero_diagonal(std::vector & matrix, size_t rows, size_t cols, T zero_value = T{}) { - for (size_t i = 0; i < rows; ++i) { - size_t diag_col = std::min(i, cols - 1); - matrix[i * cols + diag_col] = zero_value; - } -} - -ov::Tensor make_contiguous_split_input_tensor(std::shared_ptr ggml_decoder, - const struct ggml_tensor * ggml_tensor, - const ov::Shape & input_shape) { - const size_t element_size = ggml_type_size(ggml_tensor->type); - const size_t block_size = ggml_blck_size(ggml_tensor->type); - - GGML_ASSERT(block_size == 1 && "non-contiguous split inputs must be plain element types"); - - const struct ggml_tensor * source_tensor = ggml_tensor->view_src != nullptr ? ggml_tensor->view_src : ggml_tensor; - const size_t source_offset = ggml_tensor->view_src != nullptr ? ggml_tensor->view_offs : 0; - - std::vector source_data(ggml_nbytes(source_tensor)); - ggml_backend_tensor_get(source_tensor, source_data.data(), 0, source_data.size()); - - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - auto * dst = static_cast(input_tensor.data()); - size_t dst_offset = 0; - - for (size_t i3 = 0; i3 < static_cast(ggml_tensor->ne[3]); ++i3) { - for (size_t i2 = 0; i2 < static_cast(ggml_tensor->ne[2]); ++i2) { - for (size_t i1 = 0; i1 < static_cast(ggml_tensor->ne[1]); ++i1) { - for (size_t i0 = 0; i0 < static_cast(ggml_tensor->ne[0]); ++i0) { - const size_t src_offset = source_offset + i3 * ggml_tensor->nb[3] + i2 * ggml_tensor->nb[2] + - i1 * ggml_tensor->nb[1] + i0 * ggml_tensor->nb[0]; - std::memcpy(dst + dst_offset, source_data.data() + src_offset, element_size); - dst_offset += element_size; - } - } - } - } - - return input_tensor; -} - -ov::Tensor convert_ggml_input_to_ov(std::shared_ptr ggml_decoder, const std::string & name) { - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(name); - - if (auto sliced = try_make_kv_sliced_tensor(ggml_decoder, name, ggml_tensor)) { - return *sliced; - } - - if (ggml_tensor->extra != nullptr && !ggml_decoder->is_splited_model()) { - auto * extra_base = static_cast(ggml_tensor->extra); - if (extra_base->type == ggml_openvino_extra_base::Type::TENSOR) { - // GGML_LOG_DEBUG("Using ggml_tensor->extra as ov::Tensor for input: %s\n", name.c_str()); - auto * tensor_extra = static_cast(extra_base); - return *tensor_extra->tensor; - } - } - - // GGML_LOG_DEBUG("Converting ggml tensor to ov::Tensor for input: %s\n", name.c_str()); - auto * input_data = ggml_tensor->data; - ov::Shape input_shape; - if (ggml_tensor->op == GGML_OP_VIEW && !ggml_decoder->is_splited_model()) { - // This case is added to make test-backend-ops work - input_shape = ggml_decoder->get_shape(ggml_tensor->view_src); - } else { - input_shape = ggml_decoder->get_shape(ggml_tensor); - } - - if (ggml_decoder->is_splited_model() && !ggml_is_contiguous(ggml_tensor)) { - return make_contiguous_split_input_tensor(ggml_decoder, ggml_tensor, input_shape); - } - - auto input_tensor = ov::Tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape, input_data); - return input_tensor; -} -} // namespace - -ov::Tensor get_ov_input_tensor(std::shared_ptr ggml_decoder, const std::string & param_name) { - ov::Tensor input_tensor; - auto extra_input = ggml_decoder->get_model_extra_inputs().find(param_name); - if (extra_input != ggml_decoder->get_model_extra_inputs().end()) { - input_tensor = ov::Tensor(extra_input->second.type, extra_input->second.shape); - *input_tensor.data() = extra_input->second.value; - } else { - input_tensor = convert_ggml_input_to_ov(ggml_decoder, param_name); - } - return input_tensor; -} - -ov::Tensor get_ov_input_tensor_static_decode(std::shared_ptr ggml_decoder, - const std::string & param_name) { - // NPU decoding stage - if (ggml_decoder->get_model_extra_inputs().count(param_name)) { - return get_ov_input_tensor(ggml_decoder, param_name); - } - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); - const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); - - if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || - GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { - // IMROPE's inp_pos holds one value per t/h/w/e plane instead of a single position; - // with a single decode token the planes are still contiguous, so a flat copy works. - const int n_planes = GgmlOvDecoder::is_inp_pos(ggml_tensor, op) ? GgmlOvDecoder::get_inp_pos_n_planes(op) : 1; - assert(ggml_tensor->ne[0] == n_planes); - ov::Shape input_shape = {1, 1, 1, (size_t) n_planes}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - std::memcpy(input_tensor.data(), ggml_tensor->data, n_planes * ggml_type_size(ggml_tensor->type)); - return input_tensor; - } - - if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { - ov::Shape input_shape = {1, 1, 1, 1}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - int32_t inp_out_id = *((int32_t *) ggml_tensor->data); - assert(ggml_tensor->ne[0] == 1); - assert(inp_out_id == 0); - *input_tensor.data() = inp_out_id; - return input_tensor; - } - - if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { - size_t context_size = ggml_decoder->get_ctx_size(); - if (ggml_tensor->type == GGML_TYPE_F16) { - std::vector padded_data = - pad_input(ggml_tensor, 1, context_size, GGML_FP32_TO_FP16(-INFINITY)); - ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, 1, context_size}); - std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); - return input_tensor; - } - - std::vector padded_data = pad_input(ggml_tensor, 1, context_size, -INFINITY); - ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, 1, context_size}); - auto * data_ptr = input_tensor.data(); - std::copy(padded_data.begin(), padded_data.begin() + context_size, data_ptr); - return input_tensor; - } - - return get_ov_input_tensor(ggml_decoder, param_name); -} - -ov::Tensor get_ov_input_tensor_static_prefill(std::shared_ptr ggml_decoder, - const std::string & param_name, - int chunk_index) { - // NPU prompt processing stage - const size_t input_len = ggml_decoder->get_input_len(); - const size_t chunk_size = ggml_decoder->m_prefill_chunk_size; - const size_t chunk_valid_size = std::min(chunk_size, input_len - chunk_index * chunk_size); - const size_t chunk_pad_size = chunk_size - chunk_valid_size; - - if (param_name == "chunk_valid_len") { - ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); - *input_tensor.data() = (int64_t) chunk_valid_size; - return input_tensor; - } - if (chunk_index > 0 && param_name == "cache_rs_reset_len") { - // The recurrent-state clear belongs to the start of the sequence. Re-applying it on every - // chunk would wipe the state accumulated by the preceding chunks, so disable it (a zero - // length makes scale.cpp's keep-mask select every slot) after the first chunk. - ov::Tensor input_tensor(ov::element::i64, ov::Shape{1}); - *input_tensor.data() = 0; - return input_tensor; - } - if (ggml_decoder->get_model_extra_inputs().count(param_name)) { - return get_ov_input_tensor(ggml_decoder, param_name); - } - const auto * ggml_tensor = ggml_decoder->get_input_ggml_tensor(param_name); - const auto * op = ggml_decoder->get_tensor_used_op(ggml_tensor); - - if (GgmlOvDecoder::is_inp_pos(ggml_tensor, op) && GgmlOvDecoder::get_inp_pos_n_planes(op) > 1) { - // IMROPE: inp_pos stacks n_planes (t/h/w/e) position planes, each of length - // input_len; pad every plane independently so they stay aligned to chunk_size. - const int n_planes = GgmlOvDecoder::get_inp_pos_n_planes(op); - const size_t element_size = ggml_type_size(ggml_tensor->type); - ov::Shape input_shape = {1, 1, 1, (size_t) n_planes * chunk_size}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - for (int p = 0; p < n_planes; p++) { - const char * src = - (const char *) ggml_tensor->data + (p * input_len + chunk_index * chunk_size) * element_size; - char * dst = (char *) input_tensor.data() + p * chunk_size * element_size; - std::memcpy(dst, src, chunk_valid_size * element_size); - if (chunk_pad_size > 0) { - if (ggml_tensor->type == GGML_TYPE_I32) { - int32_t last_value = *((const int32_t *) src + chunk_valid_size - 1); - int32_t * out = (int32_t *) dst; - std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); - } else if (ggml_tensor->type == GGML_TYPE_I64) { - int64_t last_value = *((const int64_t *) src + chunk_valid_size - 1); - int64_t * out = (int64_t *) dst; - std::fill(out + chunk_valid_size, out + chunk_size, last_value + 1); - } else { - throw std::runtime_error("Unexpected tensor type for " + param_name); - } - } - } - return input_tensor; - } - - if (GgmlOvDecoder::is_inp_tok(ggml_tensor, op) || GgmlOvDecoder::is_inp_pos(ggml_tensor, op) || - GgmlOvDecoder::is_kv_idx(ggml_tensor, op)) { - ov::Shape input_shape = {1, 1, 1, chunk_size}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - // copy the chunk_index-th chunk from ggml_tensor - size_t element_size = ggml_type_size(ggml_tensor->type); - void * input_data = (char *) ggml_tensor->data + chunk_index * chunk_size * element_size; - std::memcpy(input_tensor.data(), input_data, chunk_valid_size * element_size); - // pad the rest with last_value + 1, so that kv's of padded positions are inserted - // to the next row after the valids row in the kvcache - if (chunk_pad_size > 0) { - if (ggml_tensor->type == GGML_TYPE_I32) { - int32_t last_value = - *((int32_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); - int32_t * output_data = input_tensor.data(); - std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); - } else if (ggml_tensor->type == GGML_TYPE_I64) { - int64_t last_value = - *((int64_t *) ggml_tensor->data + (chunk_index * chunk_size + chunk_valid_size - 1)); - int64_t * output_data = input_tensor.data(); - std::fill(output_data + chunk_valid_size, output_data + chunk_size, last_value + 1); - } else { - throw std::runtime_error("Unexpected tensor type for " + param_name); - } - } - return input_tensor; - } - - if (GgmlOvDecoder::is_output_idx(ggml_tensor, op)) { - size_t output_len = ggml_decoder->get_compute_params().output_len; - ov::Shape input_shape = {1, 1, 1, output_len}; - ov::Tensor input_tensor(ggml_decoder->get_ov_type(ggml_tensor), input_shape); - if (ggml_tensor->ne[0] == 0) { - *input_tensor.data() = 0; - } else { - auto * data_addr = input_tensor.data(); - for (size_t i = 0; i < output_len; i++) { - data_addr[i] = ((int32_t *) ggml_tensor->data)[i] % chunk_size; - } - } - return input_tensor; - } - - if (GgmlOvDecoder::is_inp_mean(ggml_tensor, op)) { - const size_t n_seqs = ggml_tensor->ne[1]; - const size_t src_stride = ggml_tensor->ne[0]; - const size_t copy_len = std::min(chunk_valid_size, src_stride - chunk_index * chunk_size); - ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, n_seqs, chunk_size}); - auto * dst = input_tensor.data(); - std::fill(dst, dst + n_seqs * chunk_size, 0.0f); - const auto * src = static_cast(ggml_tensor->data) + chunk_index * chunk_size; - for (size_t s = 0; s < n_seqs; s++) { - std::memcpy(dst + s * chunk_size, src + s * src_stride, copy_len * sizeof(float)); - } - return input_tensor; - } - - if (GgmlOvDecoder::is_inp_mask(ggml_tensor, op)) { - size_t cols = ggml_tensor->ne[0]; - size_t rows = ggml_tensor->ne[1]; - size_t chunk_valid_rows = std::min(chunk_size, rows - chunk_index * chunk_size); - size_t context_size = ggml_decoder->get_ctx_size(); - if (ggml_tensor->type == GGML_TYPE_F16) { - const auto * ggml_data = - static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; - std::vector padded_data = pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, - context_size, GGML_FP32_TO_FP16(-INFINITY)); - set_zero_diagonal(padded_data, chunk_size, context_size, GGML_FP32_TO_FP16(0.0f)); - ov::Tensor input_tensor(ov::element::f16, ov::Shape{1, 1, chunk_size, context_size}); - std::memcpy(input_tensor.data(), padded_data.data(), padded_data.size() * sizeof(ggml_fp16_t)); - return input_tensor; - } - - const auto * ggml_data = static_cast(ggml_tensor->data) + chunk_index * chunk_size * cols; - std::vector padded_data = - pad_input(ggml_data, chunk_valid_rows, cols, chunk_size, context_size, -INFINITY); - set_zero_diagonal(padded_data, chunk_size, context_size); - ov::Tensor input_tensor(ov::element::f32, ov::Shape{1, 1, chunk_size, context_size}); - auto * data_ptr = input_tensor.data(); - std::copy(padded_data.begin(), padded_data.begin() + chunk_size * context_size, data_ptr); - return input_tensor; - } - - return get_ov_input_tensor(ggml_decoder, param_name); -} - size_t checksum(const void * data, size_t size) { const uint8_t * bytes = static_cast(data); size_t sum = 0; @@ -1651,15 +1645,15 @@ bool save_ggml_tensor_data_to_txt(const ggml_tensor * tensor, const std::string void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor) { std::cout << "Input name: " << name << ", Input shape: " << tensor.get_shape() << ", Address: " << tensor.data() - << std::endl; + << '\n'; switch (tensor.get_element_type()) { case ov::element::f32: { if (name.find("self_kq_mask") == std::string::npos && name.find("KQ_mask") == std::string::npos) { - std::cout << *(tensor.data()) << std::endl; + std::cout << *(tensor.data()) << '\n'; } else { size_t rows = tensor.get_shape()[2]; size_t cols = tensor.get_shape()[3]; - auto * data = tensor.data(); + const float * data = tensor.data(); for (size_t i = 0; i < rows; ++i) { for (size_t j = 0; j < cols; ++j) { float val = data[i * cols + j]; @@ -1669,26 +1663,26 @@ void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor std::cout << std::setw(5) << val; } } - std::cout << std::endl; + std::cout << '\n'; } } break; } case ov::element::f16: - std::cout << *(tensor.data()) << std::endl; + std::cout << *(tensor.data()) << '\n'; break; case ov::element::i32: for (size_t i = 0; i < tensor.get_size(); ++i) { - std::cout << tensor.data()[i] << " "; + std::cout << tensor.data()[i] << ' '; } - std::cout << std::endl; + std::cout << '\n'; break; case ov::element::i64: for (size_t i = 0; i < tensor.get_size(); ++i) { - std::cout << tensor.data()[i] << " "; + std::cout << tensor.data()[i] << ' '; } - std::cout << std::endl; + std::cout << '\n'; break; default: break; @@ -1697,7 +1691,7 @@ void print_input_tensor_info(const std::string & name, const ov::Tensor & tensor void print_output_tensor_info(const std::string & name, const ov::Tensor & tensor, const void * output_dst) { std::cout << "Output name: " << name << ", Output shape: " << tensor.get_shape() << ", Address: " << output_dst - << std::endl; + << '\n'; auto print_float_stats = [](const std::string & type_name, size_t size, auto get_value) { if (size == 0) { @@ -1711,20 +1705,16 @@ void print_output_tensor_info(const std::string & name, const ov::Tensor & tenso for (size_t i = 1; i < size; ++i) { float v = get_value(i); - if (v < min) { - min = v; - } - if (v > max) { - max = v; - } + min = std::min(v, min); + max = std::max(v, max); sum += v; } double mean = sum / size; std::cout << std::right << std::setw(6) << type_name << std::right << std::setw(12) << "First" << std::setw(12) - << "Min" << std::setw(12) << "Max" << std::setw(12) << "Mean" << std::endl; + << "Min" << std::setw(12) << "Max" << std::setw(12) << "Mean" << '\n'; std::cout << std::right << std::setw(6) << "" << std::right << std::setw(12) << first << std::setw(12) << min - << std::setw(12) << max << std::setw(12) << mean << std::endl; + << std::setw(12) << max << std::setw(12) << mean << '\n'; }; switch (tensor.get_element_type()) { @@ -1781,5 +1771,3 @@ int64_t get_inp_pos_n_tokens(ggml_cgraph * cgraph, const ggml_tensor * inp_pos) bool get_is_prefill(ggml_cgraph * cgraph, const ggml_tensor * inp_pos) { return get_inp_pos_n_tokens(cgraph, inp_pos) > 1; } - -#pragma GCC diagnostic pop diff --git a/ggml/src/ggml-openvino/utils.h b/ggml/src/ggml-openvino/utils.h index 235b15d7e9..74c25f0ace 100644 --- a/ggml/src/ggml-openvino/utils.h +++ b/ggml/src/ggml-openvino/utils.h @@ -142,9 +142,6 @@ struct ov_runtime_context { enum ggml_status ov_graph_compute(struct ggml_cgraph * cgraph, ggml_backend_t backend); -enum ggml_status ov_graph_compute_dynamic(struct ggml_cgraph * cgraph, std::shared_ptr r_ctx); -enum ggml_status ov_graph_compute_static(struct ggml_cgraph * cgraph, std::shared_ptr r_ctx); - size_t checksum(const void * data, size_t size); bool save_ggml_tensor_data_to_txt(const ggml_tensor * tensor, const std::string & file_path); @@ -185,18 +182,6 @@ int64_t get_inp_pos_n_tokens(struct ggml_cgraph * cgraph, const ggml_tensor * in bool get_is_prefill(struct ggml_cgraph * cgraph, const ggml_tensor * inp_pos); -ov::Tensor get_ov_input_tensor(std::shared_ptr ggml_decoder, const std::string & param_name); -ov::Tensor get_ov_input_tensor_static_decode(std::shared_ptr ggml_decoder, - const std::string & param_name); -ov::Tensor get_ov_input_tensor_static_prefill(std::shared_ptr ggml_decoder, - const std::string & param_name, - int chunk_index); - -ov::Tensor create_ov_output_tensor(std::shared_ptr ggml_decoder, - std::shared_ptr infer_request, - int output_index, - const ggml_tensor * ggml_tensor); - bool is_naive(struct ggml_cgraph * cgraph); /** @@ -205,9 +190,3 @@ bool is_naive(struct ggml_cgraph * cgraph); * @return true if the graph is identified as split; otherwise false. */ bool is_model_splitted(struct ggml_cgraph * cgraph); - -enum ggml_status naive_compute(struct ggml_cgraph * cgraph, - ov::Core & core, - const std::string & device, - const ov::AnyMap & config, - ov_compiled_model_cache & cache);