Skip to content

Commit 13e3155

Browse files
authored
Merge branch 'ggml-org:master' into master
2 parents d229b65 + c77ae69 commit 13e3155

43 files changed

Lines changed: 7216 additions & 7150 deletions

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎.devops/openvino.Dockerfile‎

Lines changed: 6 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,5 @@
1-
ARG OPENVINO_VERSION_MAJOR=2026.3.1
2-
ARG OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d
1+
ARG OPENVINO_VERSION_MAJOR=2026.4
2+
ARG OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3
33
ARG UBUNTU_VERSION=24.04
44

55
# Intel GPU driver versions. https://github.com/intel/compute-runtime/releases
@@ -10,9 +10,9 @@ ARG COMPUTE_RUNTIME_VERSION_FULL=26.31.39395.13-0
1010
ARG IGDGMM_VERSION=22.10.0
1111

1212
# Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases
13-
ARG NPU_DRIVER_VERSION=v1.35.0
14-
ARG NPU_DRIVER_FULL=v1.35.0.20260722-29947505341
15-
ARG LIBZE1_VERSION=1.28.2-1~24.04~ppa1
13+
ARG NPU_DRIVER_VERSION=v1.38.0
14+
ARG NPU_DRIVER_FULL=v1.38.0.20260910-34487311128
15+
ARG LIBZE1_VERSION=1.32.0-1~24.04~ppa1
1616

1717
# Optional proxy build arguments
1818
ARG http_proxy=
@@ -173,7 +173,7 @@ RUN --mount=type=cache,target=/var/cache/intel-npu,sharing=locked \
173173
fi; \
174174
DEB=/var/cache/intel-npu/libze1_${LIBZE1_VERSION}_amd64.deb; \
175175
if [ ! -f "$DEB" ]; then \
176-
wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260606T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \
176+
wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260830T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \
177177
fi; \
178178
mkdir /tmp/npu/ && cd /tmp/npu/ && tar -xf "$TGZ" && cp "$DEB" .; \
179179
apt-get update; \

‎.github/workflows/build-cache.yml‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -41,8 +41,8 @@ jobs:
4141

4242
env:
4343
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
44-
OPENVINO_VERSION_MAJOR: "2026.3.1"
45-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
44+
OPENVINO_VERSION_MAJOR: "2026.4"
45+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
4646

4747
steps:
4848
- name: Clone
@@ -69,8 +69,8 @@ jobs:
6969

7070
env:
7171
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
72-
OPENVINO_VERSION_MAJOR: "2026.3.1"
73-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
72+
OPENVINO_VERSION_MAJOR: "2026.4"
73+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
7474

7575
steps:
7676
- name: Clone

‎.github/workflows/build-openvino.yml‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -41,8 +41,8 @@ jobs:
4141

4242
env:
4343
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
44-
OPENVINO_VERSION_MAJOR: "2026.3.1"
45-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
44+
OPENVINO_VERSION_MAJOR: "2026.4"
45+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
4646

4747
steps:
4848
- name: Clone
@@ -96,8 +96,8 @@ jobs:
9696

9797
env:
9898
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
99-
OPENVINO_VERSION_MAJOR: "2026.3.1"
100-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
99+
OPENVINO_VERSION_MAJOR: "2026.4"
100+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
101101

102102
steps:
103103
- name: Clone

‎.github/workflows/build-self-hosted.yml‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -412,8 +412,8 @@ jobs:
412412

413413
env:
414414
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
415-
OPENVINO_VERSION_MAJOR: "2026.3.1"
416-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
415+
OPENVINO_VERSION_MAJOR: "2026.4"
416+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
417417

418418
steps:
419419
- name: Clone

‎.github/workflows/release.yml‎

Lines changed: 4 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -555,8 +555,8 @@ jobs:
555555

556556
env:
557557
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
558-
OPENVINO_VERSION_MAJOR: "2026.3.1"
559-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
558+
OPENVINO_VERSION_MAJOR: "2026.4"
559+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
560560

561561
steps:
562562
- name: Set OpenVINO version output
@@ -669,8 +669,8 @@ jobs:
669669

670670
env:
671671
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
672-
OPENVINO_VERSION_MAJOR: "2026.3.1"
673-
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
672+
OPENVINO_VERSION_MAJOR: "2026.4"
673+
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
674674

675675
steps:
676676
- name: Set OpenVINO version output

‎docs/backend/OPENVINO.md‎

Lines changed: 14 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -12,6 +12,8 @@ The OpenVINO backend is implemented in `ggml/src/ggml-openvino` and provides a t
1212
- Compiles and caches the model for the target device.
1313
- Binds GGML tensor memory to OpenVINO inference tensors and runs inference.
1414

15+
For guidance on contributing to the OpenVINO backend, see the [OpenVINO Backend Contributing Guide](https://github.com/ravi9/llamacpp-ov-dev-guide/blob/main/contributing-llamacpp-ov.md).
16+
1517
## Contents
1618

1719
- [Supported Devices](#supported-devices)
@@ -96,7 +98,7 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
9698
- **SL** = Stateless (`GGML_OPENVINO_STATEFUL_EXECUTION=0`)
9799
- **SF** = Stateful (`GGML_OPENVINO_STATEFUL_EXECUTION=1`)
98100
- Note: The NPU operates in stateless mode only.
99-
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.35.0.
101+
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel Graphics Compiler 2.41.5 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.38.0.
100102
- See [Known Limitations](#known-limitations) for context on observed failures.
101103

102104
| Model | CPU (SL / SF) | GPU (SL / SF) | NPU (SL) |
@@ -117,9 +119,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
117119
| [lmstudio-community/Qwen3.5-9B-Q4_K_M](https://huggingface.co/lmstudio-community/Qwen3.5-9B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
118120
| | | | |
119121
| [unsloth/gemma-3-4b-it-Q4_K_M](https://huggingface.co/unsloth/gemma-3-4b-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
120-
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
121-
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ |
122-
| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ |
122+
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
123+
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
124+
| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
123125
| | | | |
124126
| [bartowski/Phi-3-mini-4k-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3-mini-4k-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
125127
| [bartowski/Phi-3.5-mini-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3.5-mini-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
@@ -132,9 +134,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
132134
| [bartowski/DeepSeek-R1-Distill-Llama-8B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Llama-8B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
133135
| [bartowski/DeepSeek-R1-Distill-Qwen-7B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
134136
| | | | |
135-
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✗ / ✗ | ✓ |
137+
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
136138
| [ibm-granite/granite-4.0-micro-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-micro-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
137-
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✗ / ✗ | ✗ |
139+
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
138140
| [ibm-research/granite-3.2-8b-instruct-Q4_K_M](https://huggingface.co/ibm-research/granite-3.2-8b-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
139141
| | | | |
140142
| [HuggingFaceTB/smollm2-1.7b-instruct-q4_k_m](https://huggingface.co/HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
@@ -242,8 +244,8 @@ chmod +x build-llamacpp-ov.sh
242244
# ============================================
243245
set -euo pipefail
244246
245-
OPENVINO_VERSION_MAJOR="2026.3.1"
246-
OPENVINO_VERSION_FULL="2026.3.1.22476.56d9685302d"
247+
OPENVINO_VERSION_MAJOR="2026.4"
248+
OPENVINO_VERSION_FULL="2026.4.0.22959.99c81491cc3"
247249
248250
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
249251
OPENVINO_INSTALL_DIR="/opt/intel/openvino_${OPENVINO_VERSION_MAJOR}"
@@ -340,7 +342,7 @@ echo " ./build/ReleaseOV/bin/llama-cli -m model.gguf"
340342
```
341343

342344
> [!NOTE]
343-
> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
345+
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
344346

345347
</details>
346348

@@ -370,8 +372,8 @@ REM ============================================
370372
REM llama.cpp OpenVINO Build Script (Ninja)
371373
REM ============================================
372374
373-
set "OPENVINO_VERSION_MAJOR=2026.3.1"
374-
set "OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d"
375+
set "OPENVINO_VERSION_MAJOR=2026.4"
376+
set "OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3"
375377
376378
set "SCRIPT_DIR=%~dp0"
377379
set "VCPKG_DIR=C:\vcpkg"
@@ -550,7 +552,7 @@ endlocal
550552
```
551553
552554
> [!NOTE]
553-
> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
555+
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
554556
555557
</details>
556558

‎ggml/src/ggml-openvino/ggml-decoder.cpp‎

Lines changed: 10 additions & 13 deletions
Original file line numberDiff line numberDiff line change
@@ -245,7 +245,7 @@ void GgmlOvDecoder::set_input_output() {
245245
if (src->op == GGML_OP_VIEW) {
246246
// Traverse upward through nested VIEW operations
247247
std::remove_reference_t<decltype(current_node_info.node_inputs_views[src_name])> view_chain;
248-
auto current = src;
248+
auto * current = src;
249249

250250
while (current != nullptr) {
251251
auto current_name = get_tensor_ov_name(m_cgraph, current);
@@ -612,9 +612,8 @@ std::pair<ModelParams, ComputeParams> GgmlOvDecoder::compute_llm_params(ggml_cgr
612612
if (node->src[1]->view_src != nullptr) {
613613
if (node->src[3] != nullptr) {
614614
return 4; // decoder self-attention
615-
} else {
616-
return 5; // cross-attention or encoder self-attention
617-
};
615+
}
616+
return 5; // cross-attention or encoder self-attention
618617
}
619618
break;
620619
default:
@@ -736,8 +735,7 @@ std::pair<ModelParams, ComputeParams> GgmlOvDecoder::compute_llm_params(ggml_cgr
736735

737736
bool rope_seen = false;
738737
for (int i = 0; i < cgraph->n_nodes; i++) {
739-
auto * node = cgraph->nodes[i];
740-
std::string name = std::string(node->name);
738+
ggml_tensor * node = cgraph->nodes[i];
741739
const int attention_pattern_case = get_attention_pattern_case(node);
742740
if (attention_pattern_case != -1) {
743741
ggml_tensor * cache_k_permute = nullptr;
@@ -948,7 +946,6 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op,
948946
if (m_naive) {
949947
return input != nullptr ? ov::PartialShape{get_shape(input)} : ov::PartialShape{get_shape(op)};
950948
}
951-
auto name = std::string(input->name);
952949
ov::PartialShape input_shape;
953950

954951
if (is_inp_tok(input, op) || is_inp_pos(input, op)) {
@@ -1474,7 +1471,7 @@ std::shared_ptr<ov::Node> GgmlOvDecoder::create_weight_node(ggml_tensor * tensor
14741471
void GgmlOvDecoder::dump_cgraph(const ggml_cgraph * cgraph, std::string & filename) {
14751472
std::ofstream file(filename);
14761473
if (!file.is_open()) {
1477-
std::cerr << "Failed to open file" << std::endl;
1474+
std::cerr << "Failed to open file" << '\n';
14781475
return;
14791476
}
14801477

@@ -1580,11 +1577,11 @@ void print_tensor_address_map(const ggml_cgraph * cgraph) {
15801577
}
15811578
}
15821579
for (const auto & pair : address_map) {
1583-
std::cout << "Address: " << pair.first << std::endl;
1580+
std::cout << "Address: " << pair.first << '\n';
15841581
for (const auto & name : pair.second) {
15851582
std::cout << name << " ; ";
15861583
}
1587-
std::cout << std::endl << std::endl;
1584+
std::cout << "\n\n";
15881585
}
15891586
}
15901587

@@ -2226,7 +2223,7 @@ void GgmlOvDecoder::compute_node_dynamic_dims() {
22262223
std::cout << ", ";
22272224
}
22282225
}
2229-
std::cout << "]" << std::endl;
2226+
std::cout << "]" << '\n';
22302227
// print the src name & shape with the dynamic dim for debugging
22312228
for (int j = 0; j < GGML_MAX_SRC; j++) {
22322229
ggml_tensor * src = node->src[j];
@@ -2245,9 +2242,9 @@ void GgmlOvDecoder::compute_node_dynamic_dims() {
22452242
std::cout << ", ";
22462243
}
22472244
}
2248-
std::cout << "]" << std::endl;
2245+
std::cout << "]" << '\n';
22492246
}
2250-
std::cout << std::endl;
2247+
std::cout << '\n';
22512248
}
22522249
}
22532250
}

‎ggml/src/ggml-openvino/ggml-decoder.h‎

Lines changed: 12 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -354,70 +354,70 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder {
354354

355355
void update_io(ggml_cgraph * cgraph);
356356

357-
inline static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) {
357+
static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) {
358358
return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op == GGML_OP_NONE;
359359
}
360360

361-
inline static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) {
361+
static bool is_inp_pos(const ggml_tensor * tensor, const ggml_tensor * op) {
362362
return op->op == GGML_OP_ROPE && tensor == op->src[1];
363363
}
364364

365365
// IMROPE packs 4 stacked position planes (t/h/w/e) into inp_pos, each of length
366366
// n_tokens; other modes carry a single position per token.
367-
inline static int get_inp_pos_n_planes(const ggml_tensor * op) {
367+
static int get_inp_pos_n_planes(const ggml_tensor * op) {
368368
return op->op_params[2] == GGML_ROPE_TYPE_IMROPE ? 4 : 1;
369369
}
370370

371-
inline static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) {
371+
static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) {
372372
return tensor->op == GGML_OP_GET_ROWS && op->op == GGML_OP_RMS_NORM;
373373
}
374374

375-
inline static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) {
375+
static bool is_inp_mask(const ggml_tensor * tensor, const ggml_tensor * op) {
376376
return op->op == GGML_OP_CPY || (op->op == GGML_OP_FLASH_ATTN_EXT && tensor == op->src[3]) ||
377377
(op->op == GGML_OP_SOFT_MAX && tensor == op->src[1]);
378378
}
379379

380-
inline static bool is_inp_mean(const ggml_tensor * tensor, const ggml_tensor * op) {
380+
static bool is_inp_mean(const ggml_tensor * tensor, const ggml_tensor * op) {
381381
return op->op == GGML_OP_MUL_MAT && tensor == op->src[1] && tensor->op == GGML_OP_NONE &&
382382
(tensor->flags & GGML_TENSOR_FLAG_INPUT) && tensor->type == GGML_TYPE_F32 &&
383383
op->src[0] != nullptr && op->src[0]->op != GGML_OP_NONE;
384384
}
385385

386-
inline static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) {
386+
static bool is_rope_freqs_weight(const ggml_tensor * tensor, const ggml_tensor * op) {
387387
return op->op == GGML_OP_ROPE && tensor == op->src[2];
388388
}
389389

390390
// also returns true for cache_s and cache_r in SSM/DeltaNet models
391-
inline static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) {
391+
static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) {
392392
if (tensor == nullptr) {
393393
return false;
394394
}
395395
return (tensor->buffer != nullptr && tensor->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY) ||
396396
(op != nullptr && op->op == GGML_OP_SET_ROWS && op->src[2] == tensor);
397397
}
398398

399-
inline static bool is_conv_state_writeback(const ggml_tensor * node) {
399+
static bool is_conv_state_writeback(const ggml_tensor * node) {
400400
return node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) &&
401401
node->src[0] != nullptr && node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr &&
402402
node->src[0]->src[0]->op == GGML_OP_CONCAT && node->src[1] != nullptr &&
403403
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src;
404404
}
405405

406-
inline static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) {
406+
static bool is_kv_idx(const ggml_tensor * tensor, const ggml_tensor * op) {
407407
return op->op == GGML_OP_SET_ROWS && op->src[1] == tensor;
408408
}
409409

410410
bool is_swa_mask(const ggml_tensor * tensor) const {
411411
return m_model_params.swa_mask != nullptr && tensor == m_model_params.swa_mask;
412412
}
413413

414-
inline static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) {
414+
static bool is_output_idx(const ggml_tensor * tensor, const ggml_tensor * op) {
415415
return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op != GGML_OP_NONE &&
416416
op->src[1]->op == GGML_OP_NONE;
417417
}
418418

419419
// the state permutation index input used in SSM/DeltaNet models (inp->s_copy in llama-graph.cpp)
420-
inline static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) {
420+
static bool is_inp_s_copy(const ggml_tensor * tensor, const ggml_tensor * op) {
421421
return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] &&
422422
op->src[0]->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY;
423423
}
@@ -481,5 +481,3 @@ class GgmlOvDecoder : public ov::frontend::ggml::GgmlDecoder {
481481
};
482482

483483
void print_tensor_address_map(const ggml_cgraph * cgraph);
484-
485-
std::optional<int> extract_layer_from_name(const std::string & name);

‎ggml/src/ggml-openvino/ggml-openvino-extra.cpp‎

Lines changed: 0 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -472,10 +472,6 @@ ggml_openvino_extracted_layout ggml_openvino_get_extracted_layout(const ggml_ten
472472

473473
switch (tensor->type) {
474474
case GGML_TYPE_MXFP4:
475-
layout.is_u4 = true;
476-
layout.is_symmetric = true;
477-
break;
478-
479475
case GGML_TYPE_Q4_0:
480476
layout.is_u4 = true;
481477
layout.is_symmetric = true;

0 commit comments

Comments
 (0)