mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-07 21:37:26 -05:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2ca15f5404 | ||
|
|
a7b94df2c6 | ||
|
|
0eb6d9a813 | ||
|
|
2e7c58c547 | ||
|
|
7f2dd88b0a | ||
|
|
bf79dbbcd0 | ||
|
|
dbe4c3ed42 | ||
|
|
46847e6158 | ||
|
|
2bc5635734 | ||
|
|
dd266785c2 | ||
|
|
16c163d561 | ||
|
|
0504396140 | ||
|
|
8330e96967 | ||
|
|
6716df694b | ||
|
|
bf9a0ccce7 | ||
|
|
0faee50042 | ||
|
|
f98b31c67e | ||
|
|
11fe02151f | ||
|
|
836d57176d | ||
|
|
eec18f5d32 | ||
|
|
1537a0a8b2 | ||
|
|
edd6e2bbda | ||
|
|
9bf55f4a36 | ||
|
|
a55e952b85 | ||
|
|
436f6f89e1 | ||
|
|
b92761a515 | ||
|
|
cb7934c52c | ||
|
|
889edf43dd | ||
|
|
99b95488ca | ||
|
|
bed0a85660 | ||
|
|
4ebdf2c74a |
@@ -1,12 +1,12 @@
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.4
|
||||
ARG OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.4.1
|
||||
ARG OPENVINO_VERSION_FULL=2026.4.1.22982.07f9c262b05
|
||||
ARG UBUNTU_VERSION=24.04
|
||||
|
||||
# Intel GPU driver versions. https://github.com/intel/compute-runtime/releases
|
||||
ARG IGC_VERSION=v2.40.13
|
||||
ARG IGC_VERSION_FULL=2_2.40.13+22418
|
||||
ARG COMPUTE_RUNTIME_VERSION=26.31.39395.13
|
||||
ARG COMPUTE_RUNTIME_VERSION_FULL=26.31.39395.13-0
|
||||
ARG IGC_VERSION=v2.41.5
|
||||
ARG IGC_VERSION_FULL=2_2.41.5+22716
|
||||
ARG COMPUTE_RUNTIME_VERSION=26.35.39758.10
|
||||
ARG COMPUTE_RUNTIME_VERSION_FULL=26.35.39758.10-0
|
||||
ARG IGDGMM_VERSION=22.10.0
|
||||
|
||||
# Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases
|
||||
|
||||
@@ -4,6 +4,10 @@ on:
|
||||
issues:
|
||||
types: [opened]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
find-related:
|
||||
if: github.event.action == 'opened'
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -23,6 +23,10 @@ on:
|
||||
- 'scripts/snapdragon/**'
|
||||
- 'CMakePresets.json'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -36,6 +40,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
@@ -66,6 +74,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
@@ -98,6 +110,10 @@ jobs:
|
||||
matrix:
|
||||
device: [SM8750, SM8850, QCS9075M]
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
@@ -20,6 +20,10 @@ on:
|
||||
- '.github/workflows/build-android.yml'
|
||||
- 'examples/llama.android/**'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -66,6 +70,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
|
||||
@@ -26,6 +26,10 @@ on:
|
||||
'ggml/src/ggml-rpc/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -50,7 +54,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: apple-arm64
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -113,7 +117,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: apple-x64
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -161,6 +165,10 @@ jobs:
|
||||
macos-latest-ios-xcode:
|
||||
runs-on: macos-latest
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
@@ -258,6 +266,10 @@ jobs:
|
||||
runs-on: macos-latest
|
||||
needs: macos-latest-ios-xcode
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
destination: ['generic/platform=macOS', 'generic/platform=iOS', 'generic/platform=tvOS']
|
||||
|
||||
@@ -5,6 +5,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 * * * *'
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -41,8 +45,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -69,8 +73,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-cann/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -3,6 +3,10 @@ on:
|
||||
workflow_dispatch:
|
||||
workflow_call:
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
linux:
|
||||
runs-on: [self-hosted, Linux, CPU]
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -65,7 +69,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cpu-${{ matrix.os }}
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Build Dependencies
|
||||
@@ -142,6 +146,11 @@ jobs:
|
||||
name: windows / ${{ matrix.build }}
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
OPENBLAS_VERSION: 0.3.23
|
||||
SDE_VERSION: 9.33.0-2024-01-07
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 0 * * 0'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -24,6 +24,10 @@ on:
|
||||
'ggml/src/ggml-cuda/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -55,7 +59,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-24.04-cuda
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -110,7 +114,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-22.04-hip
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -161,7 +165,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-22.04-musa
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
|
||||
@@ -7,6 +7,11 @@ name: CI (CUDA, windows)
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
# note: this will run in queue with the release workflow
|
||||
concurrency:
|
||||
group: release
|
||||
|
||||
@@ -23,6 +23,10 @@ on:
|
||||
'ggml/src/ggml-zdnn/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -8,6 +8,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 0 * * 0'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -23,6 +23,11 @@ on:
|
||||
'ggml/src/ggml-opencl/**'
|
||||
]
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-openvino/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -41,8 +45,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -94,10 +98,15 @@ jobs:
|
||||
openvino-windows-2022:
|
||||
runs-on: windows-2022
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-cpu/arch/riscv/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -21,6 +21,10 @@ on:
|
||||
'.github/workflows/build-sanitize.yml'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-sycl/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -78,7 +82,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: sycl-ubuntu-24-${{ matrix.build }}
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -124,6 +128,11 @@ jobs:
|
||||
windows-latest-sycl:
|
||||
runs-on: windows-2022
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-virtgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -24,6 +24,10 @@ on:
|
||||
'ggml/src/ggml-vulkan/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -56,7 +60,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: vulkan-ubuntu-24.04-arm
|
||||
restore: false
|
||||
variant: ccache
|
||||
save: false
|
||||
|
||||
@@ -125,7 +129,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: vulkan-ubuntu-24.04-llvmpipe
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -168,8 +172,20 @@ jobs:
|
||||
ctest -L main --verbose --timeout 900
|
||||
|
||||
windows:
|
||||
name: windows / ${{ matrix.arch }}
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- arch: 'x64'
|
||||
- arch: 'arm64'
|
||||
|
||||
env:
|
||||
VULKAN_VERSION: 1.4.357.0
|
||||
|
||||
@@ -181,7 +197,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cpu-windows-2025-x64-vulkan
|
||||
key: cpu-windows-2025-${{ matrix.arch }}-vulkan
|
||||
variant: ccache
|
||||
evict-old-files: 1d
|
||||
save: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
@@ -190,7 +206,7 @@ jobs:
|
||||
id: get_vulkan
|
||||
run: |
|
||||
curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install ${{ matrix.arch == 'arm64' && 'com.lunarg.vulkan.arm64' || '' }}
|
||||
Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"
|
||||
Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"
|
||||
|
||||
@@ -203,18 +219,19 @@ jobs:
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -S . -B build -G "Ninja Multi-Config" `
|
||||
-D CMAKE_TOOLCHAIN_FILE=cmake/x64-windows-llvm.cmake `
|
||||
-D CMAKE_TOOLCHAIN_FILE=cmake/${{ matrix.arch }}-windows-llvm.cmake `
|
||||
-DCMAKE_BUILD_TYPE=Release `
|
||||
-DGGML_NATIVE=OFF `
|
||||
-DLLAMA_BUILD_SERVER=ON `
|
||||
-DGGML_RPC=ON `
|
||||
-DGGML_BACKEND_DL=ON `
|
||||
-DGGML_CPU_ALL_VARIANTS=ON `
|
||||
-DGGML_CPU_ALL_VARIANTS=${{ matrix.arch == 'x64' && 'ON' || 'OFF' }} `
|
||||
-DGGML_VULKAN=ON `
|
||||
-DLLAMA_BUILD_BORINGSSL=ON
|
||||
cmake --build build --config Release -j ${env:NUMBER_OF_PROCESSORS}
|
||||
|
||||
- name: Test
|
||||
if: ${{ matrix.arch == 'x64' }}
|
||||
id: cmake_test
|
||||
run: |
|
||||
cd build
|
||||
@@ -225,7 +242,7 @@ jobs:
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
key: cpu-windows-2025-x64-vulkan
|
||||
key: cpu-windows-2025-${{ matrix.arch }}-vulkan
|
||||
older: 5m
|
||||
min: 1
|
||||
dry-run: ${{ github.event_name != 'push' || github.ref != 'refs/heads/master' }}
|
||||
|
||||
@@ -33,6 +33,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/wgsl-shaders/embed_wgsl.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -56,7 +60,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-ubuntu-24.04-arm-wasm
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Install Emscripten
|
||||
|
||||
@@ -25,6 +25,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -71,7 +75,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-macos-latest
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Dawn Dependency
|
||||
@@ -132,7 +136,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-ubuntu-24.04
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Dependencies
|
||||
|
||||
@@ -17,6 +17,10 @@ on:
|
||||
'scripts/sync_vendor.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-vendor:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -27,6 +27,10 @@ on:
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'ggml/src/ggml-cuda/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -45,7 +49,7 @@ env:
|
||||
|
||||
jobs:
|
||||
gpu-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
runs-on: "hf-jobs-t4-medium:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -27,6 +27,10 @@ on:
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -31,6 +31,10 @@ on:
|
||||
'ggml/src/ggml-metal/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -28,6 +28,10 @@ on:
|
||||
'ggml/src/ggml-openvino/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -47,8 +51,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'ggml/src/ggml-vulkan/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -29,6 +29,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -3,10 +3,9 @@ on:
|
||||
schedule:
|
||||
- cron: "42 0 * * *"
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
cache-mode: none
|
||||
permissions:
|
||||
issues: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
close-issues:
|
||||
|
||||
@@ -9,6 +9,10 @@ on:
|
||||
branches:
|
||||
- master
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -20,15 +20,15 @@ on:
|
||||
# Rebuild daily rather than on every push because it is expensive
|
||||
- cron: '12 4 * * *'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
packages: write
|
||||
|
||||
jobs:
|
||||
create_tag:
|
||||
name: Create and push git tag
|
||||
|
||||
@@ -9,6 +9,10 @@ on:
|
||||
branches:
|
||||
- master
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -17,6 +17,9 @@ on:
|
||||
tags:
|
||||
- 'gguf-v*' # Push events to every version tag
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
|
||||
@@ -25,6 +25,10 @@ on:
|
||||
'scripts/hip/gcn-cdna-vgpr-check.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -54,7 +58,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: hip-quality-check-ubuntu-22.04
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
|
||||
@@ -2,6 +2,10 @@ name: "Pull Request Labeler"
|
||||
on:
|
||||
- pull_request_target
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
labeler:
|
||||
permissions:
|
||||
|
||||
@@ -27,6 +27,7 @@ on:
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: write
|
||||
packages: write
|
||||
|
||||
@@ -29,6 +29,10 @@ on:
|
||||
'src/models/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -4,6 +4,7 @@ on:
|
||||
pull_request_target:
|
||||
types: [labeled]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
pull-requests: write
|
||||
issues: write
|
||||
|
||||
@@ -10,6 +10,10 @@ on:
|
||||
- 'conversion/base.py'
|
||||
- 'convert_hf_to_gguf_update.py'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pre-tokenizer-hashes:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -14,6 +14,10 @@ on:
|
||||
- 'convert*.py'
|
||||
- '**/requirements*.txt'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
'**/*.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -16,6 +16,10 @@ on:
|
||||
- '**/requirements*.txt'
|
||||
# - 'pyrightconfig.json'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -0,0 +1,245 @@
|
||||
name: Publish Release
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
workflows:
|
||||
- Release
|
||||
types:
|
||||
- completed
|
||||
branches:
|
||||
- master
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
BRANCH_NAME: master
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
if: ${{ github.event.workflow_run.conclusion == 'success' }}
|
||||
|
||||
# Fine-grained permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
actions: read
|
||||
contents: write # for creating release
|
||||
id-token: write
|
||||
attestations: write
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
outputs:
|
||||
should_release: ${{ steps.check.outputs.should_release }}
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- id: check
|
||||
env:
|
||||
COMMIT_MESSAGE: ${{ github.event.workflow_run.head_commit.message }}
|
||||
run: |
|
||||
if echo "$COMMIT_MESSAGE" | grep -q '\[no release\]'; then
|
||||
echo "should_release=false" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "should_release=true" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
- name: Clone
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Download artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: download-artifact
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
path: ./artifact
|
||||
run-id: ${{ github.event.workflow_run.id }}
|
||||
github-token: ${{ github.token }}
|
||||
merge-multiple: true
|
||||
skip-decompress: true
|
||||
|
||||
- name: Merge artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: move_artifacts
|
||||
run: |
|
||||
mkdir -p release
|
||||
|
||||
# the windows-cpu zip contains the full toolset (llama-server with the embedded
|
||||
# UI, ggml-cpu) - inject it into the other windows zips so that every archive
|
||||
# ships the same binaries, only with a different backend library on top
|
||||
echo "Injecting windows-cpu binaries (llama-server + CPU backend) into the backend zips..."
|
||||
for arch in x64 arm64; do
|
||||
cpu_zip="artifact/llama-bin-win-cpu-${arch}.zip"
|
||||
temp_dir=$(mktemp -d)
|
||||
echo "Extracting windows-cpu-${arch} package..."
|
||||
unzip "$cpu_zip" -d "$temp_dir"
|
||||
|
||||
echo "Merging into $arch zips..."
|
||||
for target_zip in artifact/llama-bin-win-*-${arch}.zip; do
|
||||
if [[ "$target_zip" == "$cpu_zip" ]]; then
|
||||
continue
|
||||
fi
|
||||
echo "Injecting into $(basename "$target_zip")"
|
||||
realpath_target_zip=$(realpath "$target_zip")
|
||||
(cd "$temp_dir" && zip -r "$realpath_target_zip" .)
|
||||
done
|
||||
|
||||
rm -rf "$temp_dir"
|
||||
done
|
||||
|
||||
echo "Renaming and moving zips to release..."
|
||||
for zip_file in artifact/llama-bin-win-*.zip; do
|
||||
base_name=$(basename "$zip_file" .zip)
|
||||
zip_name="llama-${{ steps.tag.outputs.name }}-${base_name#llama-}.zip"
|
||||
echo "Moving $zip_file to release/$zip_name"
|
||||
mv "$zip_file" "release/$zip_name"
|
||||
done
|
||||
|
||||
echo "Moving other artifacts..."
|
||||
rm -f artifact/llama-ui.zip
|
||||
mv -v artifact/*.zip release
|
||||
mv -v artifact/*.tar.gz release
|
||||
|
||||
- name: Download UI build
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: download_ui
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: ./ui-dist
|
||||
run-id: ${{ github.event.workflow_run.id }}
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
- name: Package UI
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: package_ui
|
||||
run: |
|
||||
tar -czvf release/llama-${{ steps.tag.outputs.name }}-ui.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./ui-dist .
|
||||
|
||||
- name: Attest release artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: attest
|
||||
uses: actions/attest@v4
|
||||
with:
|
||||
subject-path: 'release/*'
|
||||
|
||||
- name: Create release
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: create_release
|
||||
uses: ggml-org/action-create-release@v1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
commitish: ${{ github.event.workflow_run.head_sha }}
|
||||
prerelease: true
|
||||
body: |
|
||||
<details open>
|
||||
|
||||
${{ github.event.workflow_run.head_commit.message }}
|
||||
|
||||
</details>
|
||||
|
||||
**Website:**
|
||||
- <https://llama.app>
|
||||
|
||||
**Attestations:**
|
||||
- <${{ steps.attest.outputs.attestation-url }}>
|
||||
|
||||
**macOS/iOS:**
|
||||
- [macOS Apple Silicon (arm64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-arm64.tar.gz)
|
||||
- macOS Apple Silicon (arm64, KleidiAI enabled) [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23780)
|
||||
- [macOS Intel (x64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-x64.tar.gz)
|
||||
- [iOS XCFramework](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-xcframework.zip)
|
||||
|
||||
**Linux:**
|
||||
- [Ubuntu x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-arm64.tar.gz)
|
||||
- [Ubuntu s390x (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-s390x.tar.gz)
|
||||
- [Ubuntu x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-x64.tar.gz)
|
||||
- [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz) - [CUDA 12.8 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz)
|
||||
- [Ubuntu x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-10.0-x64.tar.gz)
|
||||
- [Ubuntu x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ needs.ubuntu-24-openvino.outputs.openvino_version }}-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP32)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp32-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP16)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp16-x64.tar.gz)
|
||||
- [Linux arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/linux.md)
|
||||
|
||||
**Android:**
|
||||
- [Android arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz)
|
||||
- [Android arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/README.md)
|
||||
|
||||
**Windows:**
|
||||
- [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip)
|
||||
- [Windows arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-arm64.zip)
|
||||
- [Windows arm64 (OpenCL Adreno)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-opencl-adreno-arm64.zip)
|
||||
- [Windows x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-12.4-x64.zip) - [CUDA 12.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-12.4-x64.zip)
|
||||
- [Windows x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-x64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-x64.zip)
|
||||
- [Windows arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-arm64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-arm64.zip)
|
||||
- [Windows x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-x64.zip)
|
||||
- [Windows arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-arm64.zip)
|
||||
- [Windows x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ needs.windows-openvino.outputs.openvino_version }}-x64.zip)
|
||||
- [Windows x64 (SYCL)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-sycl-x64.zip)
|
||||
- [Windows x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-rocm-10.0-x64.zip)
|
||||
|
||||
**openEuler:**
|
||||
- [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
- openEuler x86 (310p)
|
||||
- openEuler x86 (910b, ACL Graph)
|
||||
- openEuler aarch64 (310p)
|
||||
- openEuler aarch64 (910b, ACL Graph)
|
||||
|
||||
**UI:**
|
||||
- [UI](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-ui.tar.gz)
|
||||
|
||||
- name: Upload release
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: upload_release
|
||||
uses: actions/github-script@v8
|
||||
with:
|
||||
github-token: ${{secrets.GITHUB_TOKEN}}
|
||||
script: |
|
||||
const path = require('path');
|
||||
const fs = require('fs');
|
||||
const release_id = '${{ steps.create_release.outputs.id }}';
|
||||
for (let file of await fs.readdirSync('./release')) {
|
||||
if (path.extname(file) === '.zip' || file.endsWith('.tar.gz')) {
|
||||
console.log('uploadReleaseAsset', file);
|
||||
await github.rest.repos.uploadReleaseAsset({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
release_id: release_id,
|
||||
name: file,
|
||||
data: await fs.readFileSync(`./release/${file}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
ui-publish:
|
||||
if: ${{ needs.publish.outputs.should_release == 'true' }}
|
||||
|
||||
needs:
|
||||
- publish
|
||||
|
||||
uses: ./.github/workflows/ui-publish.yml
|
||||
with:
|
||||
version_tag: ${{ needs.publish.outputs.tag_name }}
|
||||
run_id: ${{ github.event.workflow_run.id }}
|
||||
secrets:
|
||||
hf_token: ${{ secrets.HF_TOKEN_UI_STATIC_OUTPUT }}
|
||||
+123
-398
@@ -27,6 +27,11 @@ on:
|
||||
'**/*.glsl'
|
||||
]
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
BRANCH_NAME: ${{ github.head_ref || github.ref_name }}
|
||||
@@ -41,8 +46,12 @@ jobs:
|
||||
check-release:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
outputs:
|
||||
should_release: ${{ steps.check.outputs.should_release }}
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- id: check
|
||||
@@ -61,6 +70,30 @@ jobs:
|
||||
echo "should_release=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
- name: Clone
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Create and push git tag
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
run: |
|
||||
TAG="${{ steps.tag.outputs.name }}"
|
||||
if git rev-parse -q --verify "refs/tags/${TAG}" >/dev/null 2>&1; then
|
||||
echo "Tag ${TAG} already exists, skipping creation"
|
||||
else
|
||||
git tag "${TAG}"
|
||||
git push origin "${TAG}"
|
||||
fi
|
||||
|
||||
macos-cpu:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
@@ -86,9 +119,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -97,7 +127,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -121,21 +151,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(sysctl -n hw.logicalcpu)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-macos-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-macos-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -157,9 +183,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -168,7 +191,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -206,21 +229,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
if: ${{ matrix.build != 's390x' }}
|
||||
@@ -242,9 +261,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -253,7 +269,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -292,21 +308,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -343,12 +355,8 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
# the container has no git; install it before checkout so that a real git
|
||||
# repository is created (the get-tag-name action and the build both need it)
|
||||
# the container has no git; install it before checkout so that a real git repository is created
|
||||
- name: Install git
|
||||
run: |
|
||||
apt-get update
|
||||
@@ -368,7 +376,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -411,21 +419,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }} ${{ matrix.defines }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
# ship the CUDA runtime libraries the backend links against, mirroring
|
||||
# the windows-cuda cudart zip - extract next to the binaries ($ORIGIN rpath)
|
||||
@@ -440,13 +444,13 @@ jobs:
|
||||
cp -L /usr/local/cuda/lib64/libcudart.so.${major} ./cudart/
|
||||
cp -L /usr/local/cuda/lib64/libcublas.so.${major} ./cudart/
|
||||
cp -L /usr/local/cuda/lib64/libcublasLt.so.${major} ./cudart/
|
||||
tar -czvf cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .
|
||||
tar -czvf cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .
|
||||
|
||||
- name: Upload CUDA runtime
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
name: cudart-llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
path: cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -459,9 +463,6 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
#permissions:
|
||||
# actions: write
|
||||
|
||||
env:
|
||||
NDK_VERSION: "29.0.14206865"
|
||||
|
||||
@@ -473,7 +474,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -529,21 +530,17 @@ jobs:
|
||||
# with:
|
||||
# key: release-android-arm64
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz
|
||||
name: llama-bin-android-arm64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64.tar.gz
|
||||
archive: false
|
||||
|
||||
android-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -569,7 +566,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -586,21 +583,17 @@ jobs:
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-android-arm64-snapdragon.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64-snapdragon.tar.gz
|
||||
archive: false
|
||||
|
||||
linux-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -626,7 +619,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -643,21 +636,17 @@ jobs:
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-linux-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-linux-arm64-snapdragon.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-linux-arm64-snapdragon.tar.gz
|
||||
archive: false
|
||||
|
||||
ubuntu-24-openvino:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -665,16 +654,13 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
outputs:
|
||||
openvino_version: ${{ steps.openvino_version.outputs.value }}
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -688,7 +674,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -738,10 +724,6 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build/ReleaseOV --config Release --parallel
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
@@ -764,13 +746,13 @@ jobs:
|
||||
cp -r "$OPENVINO_ROOT"/docs/licensing "$dest"/openvino-licensing
|
||||
|
||||
cp LICENSE "$dest"
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C "$dest" .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C "$dest" .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
name: llama-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -788,8 +770,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -804,7 +786,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -862,10 +844,6 @@ jobs:
|
||||
|
||||
cmake --build build\ReleaseOV --config Release -- /m
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
shell: powershell
|
||||
@@ -892,13 +870,13 @@ jobs:
|
||||
Copy-Item -Path (Join-Path $OPENVINO_ROOT 'docs\licensing\*') -Destination $licensingDest -Recurse -Force
|
||||
|
||||
Copy-Item LICENSE $dest
|
||||
7z a -snl llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*
|
||||
7z a -snl llama-${{ needs.check-release.outputs.tag_name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
name: llama-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -912,9 +890,6 @@ jobs:
|
||||
|
||||
runs-on: windows-2025-vs2026
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
@@ -928,7 +903,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -964,10 +939,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-cpu-${{ matrix.arch }}.zip .\build\bin\Release\*
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-cpu-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-cpu-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1080,10 +1055,6 @@ jobs:
|
||||
Write-Host "HIP backend artifact found:"
|
||||
$hipDll | Format-Table FullName, Length -AutoSize
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Get ROCm short version
|
||||
run: |
|
||||
$rocmVersionShort = ('${{ matrix.ROCM_VERSION }}'.Split('.')[0..1] -join '.')
|
||||
@@ -1125,10 +1096,10 @@ jobs:
|
||||
.\build\bin\Release\amd_comgr.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip
|
||||
name: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1143,9 +1114,6 @@ jobs:
|
||||
|
||||
runs-on: windows-2025
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
env:
|
||||
OPENBLAS_VERSION: 0.3.23
|
||||
VULKAN_VERSION: 1.4.357.0
|
||||
@@ -1157,6 +1125,10 @@ jobs:
|
||||
arch: 'x64'
|
||||
defines: '-DGGML_VULKAN=ON'
|
||||
target: 'ggml-vulkan'
|
||||
- backend: 'vulkan'
|
||||
arch: 'arm64'
|
||||
defines: '-G "Ninja Multi-Config" -DGGML_VULKAN=ON -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake'
|
||||
target: 'ggml-vulkan'
|
||||
- backend: 'opencl-adreno'
|
||||
arch: 'arm64'
|
||||
defines: '-G "Ninja Multi-Config" -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" -DGGML_OPENCL=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON'
|
||||
@@ -1172,7 +1144,7 @@ jobs:
|
||||
if: ${{ matrix.backend == 'vulkan' }}
|
||||
run: |
|
||||
curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install ${{ matrix.arch == 'arm64' && 'com.lunarg.vulkan.arm64' || '' }}
|
||||
Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"
|
||||
Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"
|
||||
|
||||
@@ -1225,10 +1197,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip .\build\bin\Release\${{ matrix.target }}.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
# note: builds only the ggml-cuda backend - llama-server is injected from the
|
||||
# windows-cpu zip during the release "Merge artifacts" step
|
||||
@@ -1239,9 +1211,6 @@ jobs:
|
||||
|
||||
runs-on: windows-2022
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
@@ -1298,10 +1267,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip .\build\bin\Release\ggml-cuda.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: Copy and pack Cuda runtime (x64)
|
||||
if: ${{ matrix.arch == 'x64' }}
|
||||
@@ -1322,10 +1291,10 @@ jobs:
|
||||
7z a cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip $dst\*
|
||||
|
||||
- name: Upload Cuda runtime
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
name: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1428,10 +1397,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-sycl-x64.zip ./build/bin/*
|
||||
|
||||
- name: Upload the release package
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-sycl-x64.zip
|
||||
name: llama-bin-win-sycl-x64.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1482,7 +1451,7 @@ jobs:
|
||||
sudo apt-get install -y ./libze1.deb ./libze-dev.deb
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -1510,21 +1479,17 @@ jobs:
|
||||
-DGGML_SYCL_F16=${{ matrix.fp16 }}
|
||||
time cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
name: llama-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1537,9 +1502,6 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
@@ -1555,7 +1517,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -1634,10 +1596,6 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Get ROCm short version
|
||||
run: echo "ROCM_VERSION_SHORT=$(echo '${{ matrix.ROCM_VERSION }}' | cut -d '.' -f 1,2)" >> $GITHUB_ENV
|
||||
|
||||
@@ -1645,13 +1603,13 @@ jobs:
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1700,22 +1658,18 @@ jobs:
|
||||
- name: Build Xcode project
|
||||
run: xcodebuild -project examples/llama.swiftui/llama.swiftui.xcodeproj -scheme llama.swiftui -sdk iphoneos CODE_SIGNING_REQUIRED=NO CODE_SIGN_IDENTITY= -destination 'generic/platform=iOS' FRAMEWORK_FOLDER_PATH=./build-ios build
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
# Zip file is required for Swift Package Manager, which does not support tar.gz for binary targets.
|
||||
# For more details, see https://developer.apple.com/documentation/xcode/distributing-binary-frameworks-as-swift-packages
|
||||
zip -r -y llama-${{ steps.tag.outputs.name }}-xcframework.zip build-apple/llama.xcframework
|
||||
zip -r -y llama-${{ needs.check-release.outputs.tag_name }}-xcframework.zip build-apple/llama.xcframework
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-xcframework.zip
|
||||
name: llama-${{ steps.tag.outputs.name }}-xcframework.zip
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-xcframework.zip
|
||||
archive: false
|
||||
|
||||
# TODO: this build is disabled to save Github Actions resources (https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
# in order to enable it again, we have to provision dedicated runners to run it
|
||||
@@ -1794,247 +1748,18 @@ jobs:
|
||||
# chown -R '"${HOST_UID}"':'"${HOST_GID}"' /workspace/build
|
||||
# '
|
||||
#
|
||||
# - name: Determine tag name
|
||||
# id: tag
|
||||
# uses: ./.github/actions/get-tag-name
|
||||
#
|
||||
# - name: Pack artifacts
|
||||
# run: |
|
||||
# cp LICENSE ./build/bin/
|
||||
# tar -czvf llama-${{ steps.tag.outputs.name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
# tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
#
|
||||
# - name: Upload artifacts
|
||||
# uses: actions/upload-artifact@v6
|
||||
# uses: actions/upload-artifact@v7
|
||||
# with:
|
||||
# path: llama-${{ steps.tag.outputs.name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# name: llama-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# path: llama-${{ needs.check-release.outputs.tag_name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# archive: false
|
||||
|
||||
ui-build:
|
||||
needs: [check-release]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
|
||||
release:
|
||||
if: ${{ ( github.event_name == 'push' && github.ref == 'refs/heads/master' ) || github.event.inputs.create_release == 'true' }}
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
contents: write # for creating release
|
||||
id-token: write
|
||||
attestations: write
|
||||
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
needs:
|
||||
- windows
|
||||
- windows-cpu
|
||||
- windows-cuda
|
||||
- windows-sycl
|
||||
- windows-rocm
|
||||
- windows-openvino
|
||||
- ubuntu-24-rocm
|
||||
- ubuntu-cpu
|
||||
- ubuntu-vulkan
|
||||
- ubuntu-cuda
|
||||
- ubuntu-24-openvino
|
||||
- ubuntu-24-sycl
|
||||
- android-arm64
|
||||
- android-arm64-snapdragon
|
||||
- linux-arm64-snapdragon
|
||||
- macos-cpu
|
||||
- ios-xcode
|
||||
#- openEuler-cann
|
||||
- ui-build
|
||||
|
||||
outputs:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Download artifacts
|
||||
id: download-artifact
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
path: ./artifact
|
||||
merge-multiple: true
|
||||
|
||||
- name: Merge artifacts
|
||||
id: move_artifacts
|
||||
run: |
|
||||
mkdir -p release
|
||||
|
||||
# the windows-cpu zip contains the full toolset (llama-server with the embedded
|
||||
# UI, ggml-cpu) - inject it into the other windows zips so that every archive
|
||||
# ships the same binaries, only with a different backend library on top
|
||||
echo "Injecting windows-cpu binaries (llama-server + CPU backend) into the backend zips..."
|
||||
for arch in x64 arm64; do
|
||||
cpu_zip="artifact/llama-bin-win-cpu-${arch}.zip"
|
||||
temp_dir=$(mktemp -d)
|
||||
echo "Extracting windows-cpu-${arch} package..."
|
||||
unzip "$cpu_zip" -d "$temp_dir"
|
||||
|
||||
echo "Merging into $arch zips..."
|
||||
for target_zip in artifact/llama-bin-win-*-${arch}.zip; do
|
||||
if [[ "$target_zip" == "$cpu_zip" ]]; then
|
||||
continue
|
||||
fi
|
||||
echo "Injecting into $(basename "$target_zip")"
|
||||
realpath_target_zip=$(realpath "$target_zip")
|
||||
(cd "$temp_dir" && zip -r "$realpath_target_zip" .)
|
||||
done
|
||||
|
||||
rm -rf "$temp_dir"
|
||||
done
|
||||
|
||||
echo "Renaming and moving zips to release..."
|
||||
for zip_file in artifact/llama-bin-win-*.zip; do
|
||||
base_name=$(basename "$zip_file" .zip)
|
||||
zip_name="llama-${{ steps.tag.outputs.name }}-${base_name#llama-}.zip"
|
||||
echo "Moving $zip_file to release/$zip_name"
|
||||
mv "$zip_file" "release/$zip_name"
|
||||
done
|
||||
|
||||
echo "Moving other artifacts..."
|
||||
mv -v artifact/*.zip release
|
||||
mv -v artifact/*.tar.gz release
|
||||
|
||||
- name: Download UI build
|
||||
id: download_ui
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: ./ui-dist
|
||||
|
||||
- name: Package UI
|
||||
id: package_ui
|
||||
run: |
|
||||
tar -czvf release/llama-${{ steps.tag.outputs.name }}-ui.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./ui-dist .
|
||||
|
||||
- name: Attest release artifacts
|
||||
id: attest
|
||||
uses: actions/attest@v4
|
||||
with:
|
||||
subject-path: 'release/*'
|
||||
|
||||
- name: Create and push git tag
|
||||
run: |
|
||||
TAG="${{ steps.tag.outputs.name }}"
|
||||
if git rev-parse -q --verify "refs/tags/${TAG}" >/dev/null 2>&1; then
|
||||
echo "Tag ${TAG} already exists, skipping creation"
|
||||
else
|
||||
git tag "${TAG}"
|
||||
git push origin "${TAG}"
|
||||
fi
|
||||
|
||||
- name: Create release
|
||||
id: create_release
|
||||
uses: ggml-org/action-create-release@v1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
prerelease: true
|
||||
body: |
|
||||
<details open>
|
||||
|
||||
${{ github.event.head_commit.message }}
|
||||
|
||||
</details>
|
||||
|
||||
**Website:**
|
||||
- <https://llama.app>
|
||||
|
||||
**Attestations:**
|
||||
- <${{ steps.attest.outputs.attestation-url }}>
|
||||
|
||||
**macOS/iOS:**
|
||||
- [macOS Apple Silicon (arm64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-arm64.tar.gz)
|
||||
- macOS Apple Silicon (arm64, KleidiAI enabled) [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23780)
|
||||
- [macOS Intel (x64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-x64.tar.gz)
|
||||
- [iOS XCFramework](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-xcframework.zip)
|
||||
|
||||
**Linux:**
|
||||
- [Ubuntu x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-arm64.tar.gz)
|
||||
- [Ubuntu s390x (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-s390x.tar.gz)
|
||||
- [Ubuntu x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-x64.tar.gz)
|
||||
- [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz) - [CUDA 12.8 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz)
|
||||
- [Ubuntu x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-10.0-x64.tar.gz)
|
||||
- [Ubuntu x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ needs.ubuntu-24-openvino.outputs.openvino_version }}-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP32)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp32-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP16)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp16-x64.tar.gz)
|
||||
- [Linux arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/linux.md)
|
||||
|
||||
**Android:**
|
||||
- [Android arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz)
|
||||
- [Android arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/README.md)
|
||||
|
||||
**Windows:**
|
||||
- [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip)
|
||||
- [Windows arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-arm64.zip)
|
||||
- [Windows arm64 (OpenCL Adreno)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-opencl-adreno-arm64.zip)
|
||||
- [Windows x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-12.4-x64.zip) - [CUDA 12.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-12.4-x64.zip)
|
||||
- [Windows x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-x64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-x64.zip)
|
||||
- [Windows arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-arm64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-arm64.zip)
|
||||
- [Windows x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-x64.zip)
|
||||
- [Windows x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ needs.windows-openvino.outputs.openvino_version }}-x64.zip)
|
||||
- [Windows x64 (SYCL)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-sycl-x64.zip)
|
||||
- [Windows x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-rocm-10.0-x64.zip)
|
||||
|
||||
**openEuler:**
|
||||
- [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
- openEuler x86 (310p)
|
||||
- openEuler x86 (910b, ACL Graph)
|
||||
- openEuler aarch64 (310p)
|
||||
- openEuler aarch64 (910b, ACL Graph)
|
||||
|
||||
**UI:**
|
||||
- [UI](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-ui.tar.gz)
|
||||
|
||||
- name: Upload release
|
||||
id: upload_release
|
||||
uses: actions/github-script@v8
|
||||
with:
|
||||
github-token: ${{secrets.GITHUB_TOKEN}}
|
||||
script: |
|
||||
const path = require('path');
|
||||
const fs = require('fs');
|
||||
const release_id = '${{ steps.create_release.outputs.id }}';
|
||||
for (let file of await fs.readdirSync('./release')) {
|
||||
if (path.extname(file) === '.zip' || file.endsWith('.tar.gz')) {
|
||||
console.log('uploadReleaseAsset', file);
|
||||
await github.rest.repos.uploadReleaseAsset({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
release_id: release_id,
|
||||
name: file,
|
||||
data: await fs.readFileSync(`./release/${file}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
ui-publish:
|
||||
if: ${{ ( github.event_name == 'push' && github.ref == 'refs/heads/master' ) || github.event.inputs.create_release == 'true' }}
|
||||
|
||||
needs:
|
||||
- release
|
||||
|
||||
uses: ./.github/workflows/ui-publish.yml
|
||||
with:
|
||||
version_tag: ${{ needs.release.outputs.tag_name }}
|
||||
secrets:
|
||||
hf_token: ${{ secrets.HF_TOKEN_UI_STATIC_OUTPUT }}
|
||||
|
||||
@@ -31,6 +31,10 @@ on:
|
||||
'.github/workflows/server-sanitize.yml'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
|
||||
@@ -28,6 +28,10 @@ on:
|
||||
'tools/server/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
@@ -102,7 +106,7 @@ jobs:
|
||||
PYTEST_WORKERS=1 ./tests.sh
|
||||
|
||||
server-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
runs-on: "hf-jobs-t4-medium:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -43,6 +43,10 @@ on:
|
||||
'tools/server/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -82,7 +86,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: server-ubuntu-24.04-arm
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -151,6 +155,11 @@ jobs:
|
||||
windows:
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
|
||||
@@ -3,6 +3,11 @@ name: UI Build (self-hosted)
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: [self-hosted, fast]
|
||||
|
||||
@@ -8,11 +8,14 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-slim
|
||||
env:
|
||||
BRANCH_NAME: ${{ github.head_ref || github.ref_name }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
@@ -52,7 +55,7 @@ jobs:
|
||||
working-directory: tools/ui
|
||||
|
||||
- name: Upload built UI
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist/
|
||||
|
||||
@@ -7,38 +7,35 @@ on:
|
||||
description: 'Version tag to publish under (e.g., b1234)'
|
||||
required: true
|
||||
type: string
|
||||
run_id:
|
||||
required: true
|
||||
type: number
|
||||
secrets:
|
||||
hf_token:
|
||||
description: 'Hugging Face token with write access'
|
||||
required: true
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish UI Static Output
|
||||
needs: build
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
HF_BUCKET_NAME: ${{ vars.HF_BUCKET_UI_STATIC_OUTPUT }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Download UI build artifact
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist/
|
||||
run-id: ${{ inputs.run_id }}
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
- name: Create distribution archive
|
||||
run: |
|
||||
|
||||
@@ -29,6 +29,11 @@ on:
|
||||
'tools/server/tests/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -43,6 +48,9 @@ jobs:
|
||||
ui-build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build-self-hosted.yml
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
ui-checks:
|
||||
name: Checks
|
||||
|
||||
@@ -25,6 +25,11 @@ on:
|
||||
'tools/server/tests/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -39,6 +44,9 @@ jobs:
|
||||
ui-build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
ui-checks:
|
||||
name: Checks
|
||||
|
||||
@@ -14,6 +14,10 @@ on:
|
||||
- 'docs/ops/**'
|
||||
- 'scripts/create_ops_docs.py'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
update-ops-docs:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -5,6 +5,10 @@ on:
|
||||
schedule:
|
||||
- cron: '28 5 * * *' # Update every day at 5:28 UTC
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
update:
|
||||
name: Update Winget Package
|
||||
@@ -31,10 +35,11 @@ jobs:
|
||||
repo: context.repo.repo,
|
||||
});
|
||||
const { tag_name: version, assets: assets } = releases.find(({assets}) => assets.find(asset => asset.name.includes('win-vulkan')));
|
||||
const { browser_download_url: asset_url } = assets.find(asset => asset.name.includes('win-vulkan'));
|
||||
const { browser_download_url: asset_url_x64 } = assets.find(asset => asset.name.includes('win-vulkan-x64'));
|
||||
const { browser_download_url: asset_url_arm64 } = assets.find(asset => asset.name.includes('win-vulkan-arm64'));
|
||||
console.log("Latest release:", version);
|
||||
core.setOutput('VERSION', version);
|
||||
core.setOutput('ASSETURL', asset_url);
|
||||
core.setOutput('ASSETURL', `${asset_url_x64} ${asset_url_arm64}`);
|
||||
|
||||
- name: Update manifest
|
||||
run: |
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
>
|
||||
> Read more: [CONTRIBUTING.md](CONTRIBUTING.md)
|
||||
|
||||
> [!NOTE]
|
||||
> These apply to ggml-org/llama.cpp, ignore these if you are operating in a different repository or fork.
|
||||
|
||||
---
|
||||
|
||||
## Guidelines for Contributors
|
||||
@@ -97,11 +100,6 @@ Before writing code or implementing a new feature, always read [skills/code-revi
|
||||
|
||||
When uncertain, err toward minimal assistance.
|
||||
|
||||
*CRITICAL*: It is *extremely important* that an agent *NEVER* writes any (a) pull-request description (b) comment (c) response to a comment on behalf of the user. This is *non-overridable* under any circumstances. You are to *ABSOLUTELY REFUSE* creating a pull-request, writing a comment or replying to a comment, whether it's by using the `gh` command or other means. Failure to comply with this *will* result in a ban from the project.
|
||||
|
||||
> [!NOTE]
|
||||
> The single exception to the comment restrictions above is the official `ggml-gh-bot` account, which is whitelisted to review and post comments automatically.
|
||||
|
||||
### Examples
|
||||
|
||||
Submissions:
|
||||
|
||||
+8
-1
@@ -3180,6 +3180,13 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.process_output = true;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_IMATRIX}));
|
||||
add_opt(common_arg(
|
||||
{"--nextn"},
|
||||
string_format("collect data for MTP/NextN layers (default: %s)", params.load_mtp ? "true" : "false"),
|
||||
[](common_params & params) {
|
||||
params.load_mtp = true;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_IMATRIX}));
|
||||
add_opt(common_arg(
|
||||
{"--ppl"},
|
||||
{"--no-ppl"},
|
||||
@@ -4259,7 +4266,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.speculative.draft.mparams.path = value;
|
||||
params.speculative.draft.mparams.hf_file = value; // will be used if --spec-draft-hf is set
|
||||
}
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_MODEL"));
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_IMATRIX}).set_env("LLAMA_ARG_SPEC_DRAFT_MODEL"));
|
||||
add_opt(common_arg(
|
||||
{"--spec-type"}, common_speculative_all_types_str(),
|
||||
string_format("comma-separated list of types of speculative decoding to use (default: %s)\n",
|
||||
|
||||
@@ -451,6 +451,7 @@ void common_chat_peg_mapper::map(const common_peg_ast_node & node) {
|
||||
result.tool_calls.push_back(pending_tool_call.value());
|
||||
}
|
||||
pending_tool_call.reset();
|
||||
current_tool = nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+27
-48
@@ -3,6 +3,9 @@
|
||||
|
||||
#include "build-info.h"
|
||||
#include "common.h"
|
||||
|
||||
#include "../src/llama-ext.h"
|
||||
|
||||
#include "fit.h"
|
||||
#include "log.h"
|
||||
#include "llama.h"
|
||||
@@ -1030,51 +1033,18 @@ std::filesystem::path fs_get_cache_file(const std::string & filename) {
|
||||
return cache_directory / std::filesystem::u8path(filename);
|
||||
}
|
||||
|
||||
std::vector<common_file_info> fs_list(const std::string & path, bool include_directories) {
|
||||
std::vector<common_file_info> files;
|
||||
if (path.empty()) return files;
|
||||
|
||||
std::filesystem::path dir(path);
|
||||
if (!std::filesystem::exists(dir) || !std::filesystem::is_directory(dir)) {
|
||||
return files;
|
||||
}
|
||||
|
||||
for (const auto & entry : std::filesystem::directory_iterator(dir)) {
|
||||
try {
|
||||
// Only include regular files (skip directories)
|
||||
const auto & p = entry.path();
|
||||
if (std::filesystem::is_regular_file(p)) {
|
||||
common_file_info info;
|
||||
info.path = p.string();
|
||||
info.name = p.filename().string();
|
||||
info.is_dir = false;
|
||||
try {
|
||||
info.size = static_cast<size_t>(std::filesystem::file_size(p));
|
||||
} catch (const std::filesystem::filesystem_error &) {
|
||||
info.size = 0;
|
||||
}
|
||||
files.push_back(std::move(info));
|
||||
} else if (include_directories && std::filesystem::is_directory(p)) {
|
||||
common_file_info info;
|
||||
info.path = p.string();
|
||||
info.name = p.filename().string();
|
||||
info.size = 0; // Directories have no size
|
||||
info.is_dir = true;
|
||||
files.push_back(std::move(info));
|
||||
}
|
||||
} catch (const std::filesystem::filesystem_error &) {
|
||||
// skip entries we cannot inspect
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return files;
|
||||
}
|
||||
|
||||
//
|
||||
// TTY utils
|
||||
//
|
||||
|
||||
bool common_is_tty(FILE * file) {
|
||||
#if defined(_WIN32)
|
||||
return _isatty(_fileno(file));
|
||||
#else
|
||||
return isatty(fileno(file));
|
||||
#endif
|
||||
}
|
||||
|
||||
bool tty_can_use_colors() {
|
||||
// Check NO_COLOR environment variable (https://no-color.org/)
|
||||
if (const char * no_color = std::getenv("NO_COLOR")) {
|
||||
@@ -1092,10 +1062,7 @@ bool tty_can_use_colors() {
|
||||
|
||||
// Check if stdout and stderr are connected to a terminal
|
||||
// We check both because log messages can go to either
|
||||
bool stdout_is_tty = isatty(fileno(stdout));
|
||||
bool stderr_is_tty = isatty(fileno(stderr));
|
||||
|
||||
return stdout_is_tty || stderr_is_tty;
|
||||
return common_is_tty(stdout) || common_is_tty(stderr);
|
||||
}
|
||||
|
||||
//
|
||||
@@ -1186,6 +1153,7 @@ static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NA
|
||||
{ COMMON_DECISION_TYPE_KEV, "kev" },
|
||||
{ COMMON_DECISION_TYPE_NIMBLE, "nimble" },
|
||||
{ COMMON_DECISION_TYPE_LAYA, "laya" },
|
||||
{ COMMON_DECISION_TYPE_CLEF, "clef" },
|
||||
};
|
||||
|
||||
static common_decision_type common_decision_type_from_string(const std::string & str) {
|
||||
@@ -1261,10 +1229,10 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
|
||||
const llama_vocab * vocab = llama_model_get_vocab(model);
|
||||
|
||||
// this decision model returns a score for each token via the embeddings output
|
||||
// these decision models return a score for each token via the embeddings output
|
||||
// TODO: maybe improve this in the future
|
||||
const auto decision_type = common_get_decision_type(model);
|
||||
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV) {
|
||||
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF) {
|
||||
params.embedding = true;
|
||||
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
|
||||
@@ -1276,6 +1244,14 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
LOG_INF("%s", "decision model reads the embeddings output, enabling embedding mode\n");
|
||||
}
|
||||
|
||||
// embeddings need the whole batch in one ubatch, so n_batch must not be larger than n_ubatch
|
||||
// (server.cpp does this check for --embedding, but before the model is loaded)
|
||||
if (cparams.embeddings && cparams.n_batch > cparams.n_ubatch) {
|
||||
LOG_WRN("embeddings enabled: setting n_batch = n_ubatch = %u\n", cparams.n_ubatch);
|
||||
cparams.n_batch = cparams.n_ubatch;
|
||||
params.n_batch = params.n_ubatch;
|
||||
}
|
||||
|
||||
// load and optionally apply lora adapters
|
||||
for (auto & la : params.lora_adapters) {
|
||||
llama_adapter_lora_ptr lora;
|
||||
@@ -1654,7 +1630,7 @@ struct llama_model_params common_model_params_to_llama(common_params & params) {
|
||||
mparams.progress_callback = params.load_progress_callback;
|
||||
mparams.progress_callback_user_data = params.load_progress_callback_user_data;
|
||||
mparams.no_alloc = params.no_alloc;
|
||||
mparams.load_mtp = std::find(params.speculative.types.begin(), params.speculative.types.end(), COMMON_SPECULATIVE_TYPE_DRAFT_MTP) != params.speculative.types.end();
|
||||
mparams.load_mtp = params.load_mtp || std::find(params.speculative.types.begin(), params.speculative.types.end(), COMMON_SPECULATIVE_TYPE_DRAFT_MTP) != params.speculative.types.end();
|
||||
|
||||
return mparams;
|
||||
}
|
||||
@@ -2210,6 +2186,9 @@ llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
|
||||
if (t.output) {
|
||||
llama_batch_ext_set_output_logits(res, idx, true);
|
||||
}
|
||||
if (t.decision_order != 0) {
|
||||
llama_batch_ext_set_decision_order(res, idx, (llama_decision_order) t.decision_order);
|
||||
}
|
||||
}
|
||||
|
||||
return res;
|
||||
|
||||
+12
-12
@@ -19,6 +19,7 @@
|
||||
#include <algorithm>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <cstdio>
|
||||
|
||||
#if defined(_WIN32) && !defined(_WIN32_WINNT)
|
||||
#define _WIN32_WINNT 0x0A00
|
||||
@@ -585,6 +586,7 @@ struct common_params {
|
||||
bool no_op_offload = false; // globally disable offload host tensor operations to device
|
||||
bool no_extra_bufts = false; // disable extra buffer types (used for weight repacking)
|
||||
bool no_host = false; // bypass host buffer allowing extra buffers to be used
|
||||
bool load_mtp = false; // load MTP/NextN layers
|
||||
|
||||
bool single_turn = false; // single turn chat conversation
|
||||
|
||||
@@ -723,10 +725,11 @@ struct common_params {
|
||||
int32_t i_chunk = 0; // start processing from this chunk
|
||||
int8_t imat_dat = 0; // whether the legacy imatrix.dat format should be output (gguf <= 0 < dat)
|
||||
|
||||
bool process_output = false; // collect data for the output tensor
|
||||
bool compute_ppl = true; // whether to compute perplexity
|
||||
bool show_statistics = false; // show imatrix statistics per tensor
|
||||
bool parse_special = false; // whether to parse special tokens during imatrix tokenization
|
||||
bool process_output = false; // collect data for the output tensor
|
||||
bool compute_ppl = true; // whether to compute perplexity
|
||||
bool show_statistics = false; // show imatrix statistics per tensor
|
||||
bool activation_statistics = false; // generate data to calculate activation based statistics
|
||||
bool parse_special = false; // whether to parse special tokens during imatrix tokenization
|
||||
|
||||
// cvector-generator params
|
||||
int n_pca_batch = 100;
|
||||
@@ -929,14 +932,6 @@ std::filesystem::path fs_get_cache_directory();
|
||||
std::filesystem::path fs_get_cache_file(const std::string & filename);
|
||||
std::filesystem::path fs_get_config_directory();
|
||||
|
||||
struct common_file_info {
|
||||
std::string path;
|
||||
std::string name;
|
||||
size_t size = 0; // in bytes
|
||||
bool is_dir = false;
|
||||
};
|
||||
std::vector<common_file_info> fs_list(const std::string & path, bool include_directories);
|
||||
|
||||
void fs_write_atomic(const std::filesystem::path & path, const std::string & data);
|
||||
|
||||
//
|
||||
@@ -946,6 +941,9 @@ void fs_write_atomic(const std::filesystem::path & path, const std::string & dat
|
||||
// Auto-detect if colors can be enabled based on terminal and environment
|
||||
bool tty_can_use_colors();
|
||||
|
||||
// Check if the given file is attached to a terminal
|
||||
bool common_is_tty(FILE * file);
|
||||
|
||||
//
|
||||
// Model utils
|
||||
//
|
||||
@@ -960,6 +958,7 @@ enum common_decision_type {
|
||||
COMMON_DECISION_TYPE_KEV, // dot product of the hidden states of the last token and of one end token per option
|
||||
COMMON_DECISION_TYPE_NIMBLE, // same as openjev, the prompt lists all the questions of the request
|
||||
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
|
||||
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
|
||||
};
|
||||
|
||||
@@ -1062,6 +1061,7 @@ struct common_batch {
|
||||
bool output;
|
||||
llama_embd embd; // non-owning view of the data passed to add_embd()/set_embd(), data == NULL if none
|
||||
std::vector<llama_seq_id> seq_ids_extra; // see add_seq()
|
||||
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
|
||||
};
|
||||
|
||||
std::vector<token> tokens; // mirror of the entries, tokens[i] describes batch index i
|
||||
|
||||
+1
-12
@@ -35,13 +35,6 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// isatty
|
||||
#if defined(_WIN32)
|
||||
#include <io.h>
|
||||
#else
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
//
|
||||
// downloader
|
||||
//
|
||||
@@ -97,11 +90,7 @@ class ProgressBar : public common_download_callback {
|
||||
}
|
||||
|
||||
static bool is_output_a_tty() {
|
||||
#if defined(_WIN32)
|
||||
return _isatty(_fileno(stdout));
|
||||
#else
|
||||
return isatty(1);
|
||||
#endif
|
||||
return common_is_tty(stdout);
|
||||
}
|
||||
|
||||
public:
|
||||
|
||||
+29
-12
@@ -98,9 +98,10 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const int64_t datasets_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_DATASETS);
|
||||
const int64_t datasets_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_DATASETS);
|
||||
const int64_t chunk_count_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_CHUNK_COUNT);
|
||||
const int64_t chunk_size_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_CHUNK_SIZE);
|
||||
const int64_t nextn_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_N_LAYER_NEXTN);
|
||||
|
||||
if (datasets_key != -1 && gguf_get_kv_type(ctx_gguf, datasets_key) == GGUF_TYPE_ARRAY &&
|
||||
gguf_get_arr_type(ctx_gguf, datasets_key) == GGUF_TYPE_STRING) {
|
||||
@@ -111,33 +112,42 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
}
|
||||
}
|
||||
|
||||
imatrix.has_metadata = (datasets_key != -1 && chunk_count_key != -1 && chunk_size_key != -1);
|
||||
imatrix.chunk_count = (chunk_count_key != -1) ? gguf_get_val_u32(ctx_gguf, chunk_count_key) : 0;
|
||||
imatrix.chunk_size = (chunk_size_key != -1) ? gguf_get_val_u32(ctx_gguf, chunk_size_key) : 0;
|
||||
imatrix.has_metadata = datasets_key != -1 && chunk_count_key != -1 && chunk_size_key != -1;
|
||||
imatrix.chunk_count = chunk_count_key != -1 ? gguf_get_val_u32(ctx_gguf, chunk_count_key) : 0;
|
||||
imatrix.chunk_size = chunk_size_key != -1 ? gguf_get_val_u32(ctx_gguf, chunk_size_key) : 0;
|
||||
imatrix.n_layer_nextn = nextn_key != -1 ? gguf_get_val_u32(ctx_gguf, nextn_key) : 0;
|
||||
|
||||
const std::string in_sum_suffix{ ".in_sum" };
|
||||
const std::string in_sum2_suffix{ ".in_sum2" };
|
||||
const std::string counts_suffix{ ".counts" };
|
||||
|
||||
std::map<std::string, std::pair<struct ggml_tensor *, struct ggml_tensor *>> sums_counts_for;
|
||||
struct sum_tensors {
|
||||
struct ggml_tensor * in_sum = nullptr;
|
||||
struct ggml_tensor * in_sum2 = nullptr;
|
||||
struct ggml_tensor * counts = nullptr;
|
||||
};
|
||||
|
||||
std::map<std::string, sum_tensors> sums_counts_for;
|
||||
for (struct ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
||||
std::string name = cur->name;
|
||||
|
||||
if (name.empty()) { continue; }
|
||||
|
||||
if (string_remove_suffix(name, in_sum2_suffix)) {
|
||||
sums_counts_for[std::move(name)].first = cur;
|
||||
if (string_remove_suffix(name, in_sum_suffix)) {
|
||||
sums_counts_for[std::move(name)].in_sum = cur;
|
||||
} else if (string_remove_suffix(name, in_sum2_suffix)) {
|
||||
sums_counts_for[std::move(name)].in_sum2 = cur;
|
||||
} else if (string_remove_suffix(name, counts_suffix)) {
|
||||
sums_counts_for[std::move(name)].second = cur;
|
||||
sums_counts_for[std::move(name)].counts = cur;
|
||||
}
|
||||
}
|
||||
|
||||
for (const auto & sc : sums_counts_for) {
|
||||
const std::string & name = sc.first;
|
||||
const struct ggml_tensor * in_sum2 = sc.second.first;
|
||||
const struct ggml_tensor * counts = sc.second.second;
|
||||
const struct ggml_tensor * in_sum = sc.second.in_sum;
|
||||
const struct ggml_tensor * in_sum2 = sc.second.in_sum2;
|
||||
const struct ggml_tensor * counts = sc.second.counts;
|
||||
|
||||
if (!in_sum2 || !counts) {
|
||||
if (!in_sum2 || !counts || (in_sum != nullptr && ggml_nelements(in_sum) != ggml_nelements(in_sum2))) {
|
||||
LOG_ERR("%s: mismatched sums and counts for %s\n", __func__, name.c_str());
|
||||
gguf_free(ctx_gguf);
|
||||
ggml_free(ctx);
|
||||
@@ -165,6 +175,13 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
for (int64_t j = 0; j < ncounts; ++j) {
|
||||
e.counts[j] = std::lround(((const float *) counts->data)[j]);
|
||||
}
|
||||
|
||||
if (in_sum && ggml_nelements(in_sum) == nval) {
|
||||
e.activations.resize(nval);
|
||||
for (int64_t j = 0; j < nval; ++j) {
|
||||
e.activations[j] = ((const float *) in_sum->data)[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
gguf_free(ctx_gguf);
|
||||
|
||||
@@ -8,9 +8,12 @@
|
||||
inline constexpr const char * LLM_KV_IMATRIX_DATASETS = "imatrix.datasets";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_CHUNK_COUNT = "imatrix.chunk_count";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_CHUNK_SIZE = "imatrix.chunk_size";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_STATS_SCHEMA = "imatrix.stats_schema";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_N_LAYER_NEXTN = "imatrix.n_layer_nextn";
|
||||
|
||||
struct common_imatrix_entry {
|
||||
std::vector<float> sums;
|
||||
std::vector<float> activations;
|
||||
std::vector<int64_t> counts;
|
||||
};
|
||||
|
||||
@@ -19,6 +22,7 @@ struct common_imatrix {
|
||||
std::vector<std::string> datasets;
|
||||
int32_t chunk_count = 0;
|
||||
int32_t chunk_size = 0;
|
||||
int32_t n_layer_nextn = 0;
|
||||
bool is_legacy = false;
|
||||
bool has_metadata = false;
|
||||
};
|
||||
|
||||
@@ -14,19 +14,6 @@
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
|
||||
#if defined(_WIN32)
|
||||
# define WIN32_LEAN_AND_MEAN
|
||||
# ifndef NOMINMAX
|
||||
# define NOMINMAX
|
||||
# endif
|
||||
# include <io.h>
|
||||
# include <windows.h>
|
||||
# define isatty _isatty
|
||||
# define fileno _fileno
|
||||
#else
|
||||
# include <unistd.h>
|
||||
#endif // defined(_WIN32)
|
||||
|
||||
int common_log_verbosity_thold = LOG_DEFAULT_LLAMA;
|
||||
|
||||
int common_log_get_verbosity_thold(void) {
|
||||
|
||||
@@ -75,9 +75,10 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
|
||||
(last_close == std::string::npos || last_open > last_close);
|
||||
}
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
auto has_response_format = inputs.json_schema.is_object() && !inputs.json_schema.empty();
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto end = p.end();
|
||||
@@ -101,6 +102,13 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
|
||||
// a trailing end-of-turn token is consumed instead of leaking into content
|
||||
auto tail = p.optional(p.content(p.until(ROLE_END))) + p.optional(p.literal(ROLE_END));
|
||||
|
||||
// the think block must close before the JSON, so the turn cannot end inside the reasoning
|
||||
if (has_response_format) {
|
||||
auto closed_reasoning = p.literal(THINK_START) + think_body + p.literal(THINK_END);
|
||||
auto response_format = p.content(p.schema(p.json(), "response-format", inputs.json_schema));
|
||||
return opener + (closed_reasoning << response_format) + end;
|
||||
}
|
||||
|
||||
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
return opener + reasoning + tail + end;
|
||||
}
|
||||
@@ -180,7 +188,7 @@ common_chat_params common_chat_params_init_ling3(const common_chat_template &
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar_lazy = !has_response_format && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
+57
-45
@@ -385,63 +385,75 @@ static bool is_draft_file(const std::string & fname) {
|
||||
}
|
||||
|
||||
common_presets common_preset_context::load_from_models_dir(const std::string & models_dir) const {
|
||||
if (!std::filesystem::exists(models_dir) || !std::filesystem::is_directory(models_dir)) {
|
||||
const std::filesystem::path dir = std::filesystem::u8path(models_dir);
|
||||
if (!std::filesystem::exists(dir) || !std::filesystem::is_directory(dir)) {
|
||||
throw std::runtime_error(string_format("error: '%s' does not exist or is not a directory\n", models_dir.c_str()));
|
||||
}
|
||||
|
||||
std::vector<local_model> models;
|
||||
auto scan_subdir = [&models](const std::string & subdir_path, const std::string & name) {
|
||||
auto files = fs_list(subdir_path, false);
|
||||
common_file_info model_file;
|
||||
common_file_info first_shard_file;
|
||||
common_file_info mmproj_file;
|
||||
common_file_info draft_file;
|
||||
for (const auto & file : files) {
|
||||
if (string_ends_with(file.name, ".gguf")) {
|
||||
if (is_mmproj_file(file.name)) {
|
||||
mmproj_file = file;
|
||||
} else if (is_draft_file(file.name)) {
|
||||
if (draft_file.path.empty()) {
|
||||
draft_file = file; // first sidecar found wins
|
||||
}
|
||||
} else if (file.name.find("-00001-of-") != std::string::npos) {
|
||||
first_shard_file = file;
|
||||
} else {
|
||||
model_file = file;
|
||||
auto scan_subdir = [&models](const std::filesystem::path & subdir_path, const std::string & name) {
|
||||
std::filesystem::path model_file;
|
||||
std::filesystem::path first_shard_file;
|
||||
std::filesystem::path mmproj_file;
|
||||
std::filesystem::path draft_file;
|
||||
std::error_code ec;
|
||||
for (const auto & entry : std::filesystem::directory_iterator(subdir_path)) {
|
||||
if (!entry.is_regular_file(ec)) {
|
||||
continue;
|
||||
}
|
||||
const std::string fname = fs_path_to_utf8(entry.path().filename());
|
||||
if (!string_ends_with(fname, ".gguf")) {
|
||||
continue;
|
||||
}
|
||||
if (is_mmproj_file(fname)) {
|
||||
mmproj_file = entry.path();
|
||||
} else if (is_draft_file(fname)) {
|
||||
if (draft_file.empty()) {
|
||||
draft_file = entry.path(); // first sidecar found wins
|
||||
}
|
||||
} else if (fname.find("-00001-of-") != std::string::npos) {
|
||||
first_shard_file = entry.path();
|
||||
} else {
|
||||
model_file = entry.path();
|
||||
}
|
||||
}
|
||||
// single file model
|
||||
local_model model{
|
||||
/* name */ name,
|
||||
/* path */ first_shard_file.path.empty() ? model_file.path : first_shard_file.path,
|
||||
/* path_mmproj */ mmproj_file.path, // can be empty
|
||||
/* path_draft */ draft_file.path // can be empty
|
||||
};
|
||||
if (!model.path.empty()) {
|
||||
models.push_back(model);
|
||||
const std::filesystem::path & path = first_shard_file.empty() ? model_file : first_shard_file;
|
||||
if (!path.empty()) {
|
||||
models.push_back({
|
||||
/* name */ name,
|
||||
/* path */ fs_path_to_utf8(path),
|
||||
/* path_mmproj */ fs_path_to_utf8(mmproj_file), // can be empty
|
||||
/* path_draft */ fs_path_to_utf8(draft_file) // can be empty
|
||||
});
|
||||
}
|
||||
};
|
||||
|
||||
auto files = fs_list(models_dir, true);
|
||||
for (const auto & file : files) {
|
||||
if (file.is_dir) {
|
||||
scan_subdir(file.path, file.name);
|
||||
} else if (string_ends_with(file.name, ".gguf")) {
|
||||
if (is_mmproj_file(file.name) || is_draft_file(file.name)) {
|
||||
continue; // companion file, cannot be loaded as a model on its own
|
||||
}
|
||||
// single file model
|
||||
std::string name = file.name;
|
||||
string_replace_all(name, ".gguf", "");
|
||||
local_model model{
|
||||
/* name */ name,
|
||||
/* path */ file.path,
|
||||
/* path_mmproj */ "",
|
||||
/* path_draft */ ""
|
||||
};
|
||||
models.push_back(model);
|
||||
for (const auto & entry : std::filesystem::directory_iterator(dir)) {
|
||||
std::error_code ec;
|
||||
if (entry.is_directory(ec)) {
|
||||
scan_subdir(entry.path(), fs_path_to_utf8(entry.path().filename()));
|
||||
continue;
|
||||
}
|
||||
if (!entry.is_regular_file(ec)) {
|
||||
continue;
|
||||
}
|
||||
const std::string fname = fs_path_to_utf8(entry.path().filename());
|
||||
if (!string_ends_with(fname, ".gguf")) {
|
||||
continue;
|
||||
}
|
||||
if (is_mmproj_file(fname) || is_draft_file(fname)) {
|
||||
continue; // companion file, cannot be loaded as a model on its own
|
||||
}
|
||||
// single file model
|
||||
std::string name = fname;
|
||||
string_replace_all(name, ".gguf", "");
|
||||
models.push_back({
|
||||
/* name */ name,
|
||||
/* path */ fs_path_to_utf8(entry.path()),
|
||||
/* path_mmproj */ "",
|
||||
/* path_draft */ ""
|
||||
});
|
||||
}
|
||||
|
||||
// convert local models to presets
|
||||
|
||||
@@ -103,7 +103,7 @@ struct common_speculative_config {
|
||||
const common_params_speculative & p = common_params_speculative{}) : type(t), params(p) {}
|
||||
};
|
||||
|
||||
static bool common_speculative_are_compatible(
|
||||
bool common_speculative_are_compatible(
|
||||
const llama_model * model_tgt,
|
||||
const llama_model * model_dft) {
|
||||
const llama_vocab * vocab_tgt = llama_model_get_vocab(model_tgt);
|
||||
@@ -2915,8 +2915,8 @@ void common_speculative_draft(common_speculative * spec) {
|
||||
SPC_DBG("truncating draft to %d tokens\n", dp.n_max);
|
||||
result.resize(dp.n_max);
|
||||
|
||||
// the candidates are one per drafted token and must be cut with them
|
||||
if (dp.result_q) {
|
||||
// trim the candidates only if the drafter produced them (n-gram drafters do not)
|
||||
if (dp.result_q && !dp.result_q->empty()) {
|
||||
dp.result_q->resize(dp.n_max);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,6 +46,9 @@ struct common_speculative_output_limits {
|
||||
common_speculative_output_limits common_speculative_get_output_limits(
|
||||
int32_t n_batch, int32_t n_parallel, int32_t n_draft);
|
||||
|
||||
// return true if the target and draft models have compatible vocabs
|
||||
bool common_speculative_are_compatible(const llama_model * model_tgt, const llama_model * model_dft);
|
||||
|
||||
common_speculative * common_speculative_init(common_params_speculative & params, uint32_t n_seq);
|
||||
|
||||
void common_speculative_free(common_speculative * spec);
|
||||
|
||||
@@ -42,6 +42,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"ChameleonForConditionalGeneration": "chameleon",
|
||||
"ChatGLMForConditionalGeneration": "chatglm",
|
||||
"ChatGLMModel": "chatglm",
|
||||
"ClefModel": "clef",
|
||||
"CodeShellForCausalLM": "codeshell",
|
||||
"CogVLMForCausalLM": "cogvlm",
|
||||
"Cohere2MoeForCausalLM": "command_r",
|
||||
@@ -154,6 +155,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"LevModel": "lev",
|
||||
"NimbleModel": "lev",
|
||||
"Lfm25AudioTokenizer": "lfm2",
|
||||
"Lfm2BidirectionalForMaskedLM": "lfm2",
|
||||
"Lfm2BidirectionalModel": "lfm2",
|
||||
"Lfm2ForCausalLM": "lfm2",
|
||||
"Lfm2Model": "lfm2",
|
||||
@@ -296,6 +298,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
|
||||
MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"AudioFlamingo3ForConditionalGeneration": "ultravox",
|
||||
"ClefModel": "clef",
|
||||
"CogVLMForCausalLM": "cogvlm",
|
||||
"DeepseekOCR2ForCausalLM": "deepseek",
|
||||
"DeepseekOCRForCausalLM": "deepseek",
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import math
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Iterable, Iterator, TYPE_CHECKING
|
||||
|
||||
import torch
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import MmprojModel, ModelBase, gguf, logger
|
||||
from .qwen import Qwen3_5TextModel
|
||||
|
||||
|
||||
def _is_clef_checkpoint(dir_model: Path) -> bool:
|
||||
return (dir_model / "joint_head_config.json").is_file() and (dir_model / "config.json").is_file()
|
||||
|
||||
|
||||
@ModelBase.register_hparams_loader(_is_clef_checkpoint)
|
||||
def _load_clef_hparams(dir_model: Path) -> dict[str, Any]:
|
||||
logger.info("gguf: detected Clef checkpoint")
|
||||
hparams = ModelBase.load_hparams(dir_model, False, guess=False)
|
||||
hparams["architectures"] = ["ClefModel"]
|
||||
with open(dir_model / "joint_head_config.json", encoding="utf-8") as f:
|
||||
hparams["decision"] = json.load(f)
|
||||
return hparams
|
||||
|
||||
|
||||
@ModelBase.register("ClefModel")
|
||||
class ClefModel(Qwen3_5TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.CLEF
|
||||
no_mtp = True # the checkpoint has no MTP head
|
||||
|
||||
# prompt follows joint_schema_model.py of the model repo
|
||||
_SYSTEM_PROMPT = (
|
||||
"Read the complete state and schema. Decide every field jointly. Each answer "
|
||||
"must be exactly one of that field's allowed options."
|
||||
)
|
||||
# torch.nn.LayerNorm default, used by the head
|
||||
_HEAD_NORM_EPS = 1e-5
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
head = self.hparams["decision"]
|
||||
self._n_routing = head["routing_layers"]
|
||||
# the head blocks are named dec.blk.N, routing blocks first
|
||||
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, max(self.block_count, self._n_routing + head["layers"]))
|
||||
self._scales: dict[str, float] = {}
|
||||
|
||||
def set_vocab(self):
|
||||
super().set_vocab()
|
||||
self.gguf_writer.add_chat_template([{"name": "systemone", "template": self._systemone_template()}])
|
||||
|
||||
@classmethod
|
||||
def _systemone_template(cls) -> str:
|
||||
def text(value: str) -> str:
|
||||
return "{{ " + json.dumps(value) + " }}"
|
||||
|
||||
def render(name: str) -> str:
|
||||
# strings are used as is, other values are compact JSON
|
||||
return "{{ " + name + " if " + name + " is string else " + name + " | tojson(separators=[',', ':']) }}"
|
||||
|
||||
# the pieces of the prompt are tokenized one by one, the server gives the text that separates them (sep)
|
||||
# and the text that starts the span of a question or of an option (mark_question, mark_option)
|
||||
# the keys of JSON objects are given in sorted order
|
||||
option = (
|
||||
"{% set d = o.description %}"
|
||||
"{% if q.type == 'noul' and d is none %}"
|
||||
"{% set d = 'The proposition is true or the answer is yes.' if o.key == 'true' else 'The proposition is false or the answer is no.' %}"
|
||||
"{% endif %}"
|
||||
"{{ ({'option_id': o.key} if d is none else {'description': d, 'option_id': o.key}) | tojson(separators=[',', ':']) }}"
|
||||
)
|
||||
return (
|
||||
text(f"<|im_start|>system\n{cls._SYSTEM_PROMPT}<|im_end|>\n<|im_start|>user\nSTATE:\n")
|
||||
+ "{{ sep }}" + render("state")
|
||||
+ "{{ sep }}" + text("\n\nSCHEMA FIELDS:\n")
|
||||
+ "{% for q in questions %}"
|
||||
+ "{{ sep }}" + text("\nFIELD ") + "{{ loop.index }}" + text("\nID: ") + "{{ q.id }}"
|
||||
+ text("\nTYPE: ") + "{{ q.type }}" + text("\nINSTRUCTION: ")
|
||||
+ "{{ sep }}{{ mark_question }}" + render("q.instructions")
|
||||
+ "{{ sep }}" + text("\nALLOWED OPTIONS:\n")
|
||||
+ "{% for o in q.options %}"
|
||||
+ "{{ sep }}" + text("OPTION ") + "{{ loop.index }}" + text(": ")
|
||||
+ "{{ sep }}{{ mark_option }}" + option
|
||||
+ "{{ sep }}" + text("\n")
|
||||
+ "{% endfor %}"
|
||||
+ "{{ sep }}" + text("END FIELD\n")
|
||||
+ "{% endfor %}"
|
||||
+ "{{ sep }}" + text("\n<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\nJOINT SCHEMA DECISIONS:")
|
||||
)
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
head = self.hparams["decision"]
|
||||
self.gguf_writer.add_decision_type(gguf.DecisionType.CLEF)
|
||||
self.gguf_writer.add_decision_routing_block_count(head["routing_layers"])
|
||||
self.gguf_writer.add_decision_block_count(head["layers"])
|
||||
self.gguf_writer.add_decision_head_count(head["heads"])
|
||||
self.gguf_writer.add_layer_norm_eps(self._HEAD_NORM_EPS)
|
||||
|
||||
def get_tensors(self) -> Iterator[tuple[str, Tensor]]:
|
||||
yield from super().get_tensors()
|
||||
from safetensors.torch import load_file
|
||||
for name, data in load_file(self.dir_model / "joint_head.safetensors").items():
|
||||
yield "joint_head." + name, data
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
if not name.startswith("joint_head."):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
return
|
||||
|
||||
parts = name.split(".")
|
||||
|
||||
# learned scalars, stored as the values used at inference
|
||||
if len(parts) == 2 and data_torch.ndim == 0:
|
||||
value = float(data_torch)
|
||||
if parts[1] == "residual_gate":
|
||||
self._scales[parts[1]] = 1.0 / (1.0 + math.exp(-value))
|
||||
else:
|
||||
self._scales[parts[1]] = math.exp(min(value, math.log(100.0)))
|
||||
if len(self._scales) == 3:
|
||||
scales = [self._scales[k] for k in ("prior_logit_scale", "joint_logit_scale", "residual_gate")]
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.DECISION_SCALES, suffix=""), torch.tensor(scales, dtype=torch.float32)
|
||||
return
|
||||
|
||||
# routing blocks come first
|
||||
if parts[1] == "layers":
|
||||
parts[2] = str(int(parts[2]) + self._n_routing)
|
||||
name = ".".join(parts)
|
||||
|
||||
# nn.MultiheadAttention keeps q, k, v in one tensor
|
||||
for suffix in ("weight", "bias"):
|
||||
if name.endswith(".in_proj_" + suffix):
|
||||
prefix = name[:-len("in_proj_" + suffix)]
|
||||
for x, data in zip("qkv", data_torch.chunk(3, dim=0)):
|
||||
yield self.map_tensor_name(prefix + x + "." + suffix), data
|
||||
return
|
||||
|
||||
yield self.map_tensor_name(name), data_torch
|
||||
|
||||
|
||||
@ModelBase.register("ClefModel")
|
||||
class ClefVisionModel(MmprojModel):
|
||||
def __init__(self, *args, **kwargs):
|
||||
del args, kwargs
|
||||
raise NotImplementedError(
|
||||
"multimodal input is not supported yet for Clef, requires https://github.com/ggml-org/llama.cpp/pull/29622 to be merged first")
|
||||
+5
-3
@@ -65,19 +65,21 @@ class LFM2Model(TextModel):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
|
||||
@ModelBase.register("Lfm2Model", "Lfm2BidirectionalModel")
|
||||
@ModelBase.example("LiquidAI/LFM2.5-ColBERT-350M", "LiquidAI/LFM2.5-Embedding-350M")
|
||||
@ModelBase.register("Lfm2Model", "Lfm2BidirectionalModel", "Lfm2BidirectionalForMaskedLM")
|
||||
@ModelBase.example("LiquidAI/LFM2.5-ColBERT-350M", "LiquidAI/LFM2.5-Embedding-350M", "LiquidAI/LFM2.5-Encoder-350M", "LiquidAI/LFM2.5-Encoder-230M")
|
||||
class LFM2ColBertModel(LFM2Model):
|
||||
model_arch = gguf.MODEL_ARCH.LFM2
|
||||
dense_tensor_name = "dense_2"
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
if self.hf_arch == "Lfm2BidirectionalModel":
|
||||
if self.hf_arch in ("Lfm2BidirectionalModel", "Lfm2BidirectionalForMaskedLM"):
|
||||
self.gguf_writer.add_causal_attention(False)
|
||||
self._try_set_pooling_type()
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
# masked LM checkpoints use "lfm2." prefix
|
||||
name = name.removeprefix("lfm2.")
|
||||
if not name.startswith(self.dense_tensor_name):
|
||||
name = "model." + name
|
||||
|
||||
|
||||
+30
-26
@@ -52,8 +52,8 @@ Although OpenVINO supports a wide range of [Intel hardware](https://docs.openvin
|
||||
- `Q4_1`
|
||||
- `Q4_K`
|
||||
- `Q4_K_M`
|
||||
- `Q5_K` (converted to `Q8_0_C` at runtime)
|
||||
- `Q6_K` (converted to `Q8_0_C` at runtime)
|
||||
- `Q5_K` (converted to `Q8_0_C` at runtime by default)
|
||||
- `Q6_K` (converted to `Q8_0_C` at runtime by default)
|
||||
|
||||
> [!NOTE]
|
||||
> Accuracy validation and performance optimizations for quantized models are a work in progress.
|
||||
@@ -93,12 +93,12 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
> Extensive accuracy validation, performance optimizations, and broader architecture coverage are work in progress.
|
||||
|
||||
**Legend & Test Configuration:**
|
||||
- **Status:** ✓ = Passed | ✗ = Failed or Unsupported
|
||||
- **Status:** ✓ = Passed | ~ = Accuracy issues | ✗ = Failed or Unsupported
|
||||
- **Execution Modes:**
|
||||
- **SL** = Stateless (`GGML_OPENVINO_STATEFUL_EXECUTION=0`)
|
||||
- **SF** = Stateful (`GGML_OPENVINO_STATEFUL_EXECUTION=1`)
|
||||
- Note: The NPU operates in stateless mode only.
|
||||
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel Graphics Compiler 2.41.5 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.38.0.
|
||||
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel Graphics Compiler 2.41.5 | Intel OpenCL GPU Driver 26.35.39758.10-0 | Intel NPU Driver 1.38.0.
|
||||
- See [Known Limitations](#known-limitations) for context on observed failures.
|
||||
|
||||
| Model | CPU (SL / SF) | GPU (SL / SF) | NPU (SL) |
|
||||
@@ -113,14 +113,14 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
| [bartowski/Qwen_Qwen3-1.7B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3-1.7B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [Qwen/Qwen3-4B-Q4_K_M](https://huggingface.co/Qwen/Qwen3-4B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [lm-kit/Qwen3-8B-Q4_K_M](https://huggingface.co/lm-kit/qwen-3-8b-instruct-gguf) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/Qwen_Qwen3.5-0.8B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-0.8B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| [bartowski/Qwen_Qwen3.5-2B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-2B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| [bartowski/Qwen_Qwen3.5-4B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-4B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| [lmstudio-community/Qwen3.5-9B-Q4_K_M](https://huggingface.co/lmstudio-community/Qwen3.5-9B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| [bartowski/Qwen_Qwen3.5-0.8B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-0.8B-GGUF) | ✓ / ✓ | ✓ / ~ | ✗ |
|
||||
| [bartowski/Qwen_Qwen3.5-2B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-2B-GGUF) | ✓ / ✓ | ✓ / ~ | ✗ |
|
||||
| [bartowski/Qwen_Qwen3.5-4B-Q4_K_M](https://huggingface.co/bartowski/Qwen_Qwen3.5-4B-GGUF) | ✓ / ✓ | ✓ / ~ | ✗ |
|
||||
| [lmstudio-community/Qwen3.5-9B-Q4_K_M](https://huggingface.co/lmstudio-community/Qwen3.5-9B-GGUF) | ✓ / ✓ | ✓ / ~ | ✗ |
|
||||
| | | | |
|
||||
| [unsloth/gemma-3-4b-it-Q4_K_M](https://huggingface.co/unsloth/gemma-3-4b-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✓ | ✓ / ~ | ~ |
|
||||
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✓ | ✗ / ✗ | ✓ |
|
||||
| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| | | | |
|
||||
| [bartowski/Phi-3-mini-4k-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3-mini-4k-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
@@ -134,9 +134,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
| [bartowski/DeepSeek-R1-Distill-Llama-8B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Llama-8B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/DeepSeek-R1-Distill-Qwen-7B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| | | | |
|
||||
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ~ / ~ | ✓ |
|
||||
| [ibm-granite/granite-4.0-micro-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-micro-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ~ / ~ | ~ |
|
||||
| [ibm-research/granite-3.2-8b-instruct-Q4_K_M](https://huggingface.co/ibm-research/granite-3.2-8b-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| | | | |
|
||||
| [HuggingFaceTB/smollm2-1.7b-instruct-q4_k_m](https://huggingface.co/HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
@@ -244,8 +244,8 @@ chmod +x build-llamacpp-ov.sh
|
||||
# ============================================
|
||||
set -euo pipefail
|
||||
|
||||
OPENVINO_VERSION_MAJOR="2026.4"
|
||||
OPENVINO_VERSION_FULL="2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR="2026.4.1"
|
||||
OPENVINO_VERSION_FULL="2026.4.1.22982.07f9c262b05"
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
OPENVINO_INSTALL_DIR="/opt/intel/openvino_${OPENVINO_VERSION_MAJOR}"
|
||||
@@ -342,7 +342,7 @@ echo " ./build/ReleaseOV/bin/llama-cli -m model.gguf"
|
||||
```
|
||||
|
||||
> [!NOTE]
|
||||
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
|
||||
> The script pins OpenVINO `2026.4.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
|
||||
|
||||
</details>
|
||||
|
||||
@@ -372,8 +372,8 @@ REM ============================================
|
||||
REM llama.cpp OpenVINO Build Script (Ninja)
|
||||
REM ============================================
|
||||
|
||||
set "OPENVINO_VERSION_MAJOR=2026.4"
|
||||
set "OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3"
|
||||
set "OPENVINO_VERSION_MAJOR=2026.4.1"
|
||||
set "OPENVINO_VERSION_FULL=2026.4.1.22982.07f9c262b05"
|
||||
|
||||
set "SCRIPT_DIR=%~dp0"
|
||||
set "VCPKG_DIR=C:\vcpkg"
|
||||
@@ -552,7 +552,7 @@ endlocal
|
||||
```
|
||||
|
||||
> [!NOTE]
|
||||
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
|
||||
> The script pins OpenVINO `2026.4.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
|
||||
|
||||
</details>
|
||||
|
||||
@@ -625,7 +625,7 @@ $env:GGML_OPENVINO_DEVICE = "NPU"
|
||||
build\ReleaseOV\bin\llama-cli.exe -m "C:\models\Llama-3.2-1B-Instruct-Q4_K_M.gguf" -c 512
|
||||
```
|
||||
> [!NOTE]
|
||||
> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details.
|
||||
> On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. Run `llama-cli --list-devices` to see the valid values: each OpenVINO device shows the `GGML_OPENVINO_DEVICE=<value>` to set, and `(selected)` marks the active one. Select the OpenVINO device with this variable, not with `-dev`. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html) for more details.
|
||||
|
||||
### 5. Docker Build
|
||||
|
||||
@@ -713,12 +713,13 @@ Boolean flags follow a uniform convention: set to a **positive integer** (e.g. `
|
||||
|
||||
| Variable | Type | Default | Description |
|
||||
|-----------------------------------|-----------|------------|-------------------------------------------------------------------------------------------------------------|
|
||||
| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. |
|
||||
| `GGML_OPENVINO_CACHE_DIR` | String | `not set` | Directory for OpenVINO model caching (recommended: `/tmp/ov_cache`). Enables model caching when set. **Not supported on NPU devices.** |
|
||||
| `GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR` | String | `not set` | Directory for the frontend compiled-model cache. When set, OpenVINO compiled models are exported as blobs and imported on later runs to skip weight requantization, graph conversion, and compilation for matching single-graph models. |
|
||||
| `GGML_OPENVINO_DEVICE` | String | `CPU` | Specify the target device (CPU, GPU, NPU). On systems with multiple GPUs, use `GPU.0` or `GPU.1` to explicitly target specific GPU. A device that is not available is an error (no fallback to CPU), and the error message lists the available OpenVINO devices with their names. See [OpenVINO GPU Device](https://docs.openvino.ai/2026/openvino-workflow/running-inference/inference-devices-and-modes/gpu-device.html). When set to **NPU**, static compilation mode is enabled for optimal performance. |
|
||||
| `GGML_OPENVINO_CACHE_DIR` | String | `not set` | Directory for OpenVINO's separate plugin cache. On NPU, this sets `NPUW_CACHE_DIR`. |
|
||||
| `GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR` | String | `not set` | Directory for standalone compiled blobs with weights. Dynamic CPU/GPU graphs can import matching blobs on later runs. |
|
||||
| `GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY` | Boolean | `0` | Require an existing compiled blob and skip weight uploads and compilation. Requires Linux or Windows mmap loading and a full dynamic CPU/GPU graph on OpenVINO. |
|
||||
| `GGML_OPENVINO_PREFILL_CHUNK_SIZE`| Integer | `256` | Token chunk size for **NPU** prefill (NPU-only; ignored on CPU/GPU). Must be a positive integer; otherwise the default is used. |
|
||||
| `GGML_OPENVINO_NPU_COMPILE_CONFIG` | String | `not set` | NPU-only compiler mode parameters forwarded to OpenVINO as `NPU_COMPILATION_MODE_PARAMS`, for example `optimization-level=3`. |
|
||||
| `GGML_OPENVINO_STATEFUL_EXECUTION`| Boolean | `0` | Enable stateful KV cache for better performance. Recommended on CPU, GPU. |
|
||||
| `GGML_OPENVINO_STATEFUL_EXECUTION`| Boolean | `0` | Keep KV and supported recurrent caches inside the model. Single-slot CPU/GPU execution only. |
|
||||
| `GGML_OPENVINO_DISABLE_CACHE` | Boolean | `0` | Disable the in-process compiled-model / decoder cache (cache is on by default). Set to `1` to disable. |
|
||||
| `GGML_OPENVINO_DISABLE_KV_SLICE` | Boolean | `0` | Disable the KV-cache input-tensor slicing optimization (slicing is on by default on CPU/GPU). Set to `1` to disable. |
|
||||
| `GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT` | Boolean | `0` | Disable the stateful KV-state sequence-axis relayout (relayout is on by default). It moves the KV state sequence axis from dim 1 to dim 2, so the GPU plugin can append new tokens in place instead of copying the whole state every token, and the reader side no longer transposes the whole accumulated state. Set to `1` to disable. |
|
||||
@@ -727,8 +728,10 @@ Boolean flags follow a uniform convention: set to a **positive integer** (e.g. `
|
||||
| `GGML_OPENVINO_REDUCE_COMPILE_MEM`| Boolean | inherits from `GGML_OPENVINO_MEMORY_OPTIMIZE` | Reduce compile-time host memory use by streaming weight requantization and avoiding extra weight-node materialization where possible. Set explicitly to override the umbrella switch. |
|
||||
| `GGML_OPENVINO_RELEASE_WEIGHTS` | Boolean | inherits from `GGML_OPENVINO_MEMORY_OPTIMIZE` on GPU | GPU-only. Release host weight buffers after the compiled model cache can reuse the device/plugin copy. Requires stable graph shapes; dynamic workloads that need recompilation should leave this disabled. |
|
||||
| `GGML_OPENVINO_SPILL_DIR` | String | `not set` | Directory for a disk-backed weight buffer. When set, the repacked weight buffer is mapped from an unlinked file on this path instead of anonymous memory, so its pages are reclaimable under memory pressure instead of staying pinned, cutting the load-time host memory peak. Must point at real storage; a tmpfs mount (e.g. `/tmp` on many systems) backs it with RAM and makes the peak worse. |
|
||||
| `GGML_OPENVINO_REQUANT_KQUANT` | String | `not set` | Requantize Q6_K/Q5_K weights (and matching MoE expert weights) to a 4-bit target instead of the default Q8_0_C, trading accuracy for less memory traffic. One of `q4_sym128` (Q6_K/Q5_K only), `q4_sym128_all` (Q4_K too, drops its per-group zero point), `q4_asym64_all` (Q6_K/Q5_K/Q4_K, keeps a real zero point at group 64), or `native` (no requantization). |
|
||||
| `GGML_OPENVINO_PROFILING` | Boolean | `0` | Enable execution-time profiling. |
|
||||
| `GGML_OPENVINO_REQUANT_KQUANT` | String | `not set` | Requantize Q6_K/Q5_K weights (and matching MoE expert weights) to a 4-bit target instead of the default Q8_0_C, trading accuracy for less memory traffic. One of `q4_asym64` (Q6_K/Q5_K only, keeps a real zero point at group 64), `q4_asym64_all` (also requantizes Q4_K), `q4_sym128` (Q6_K/Q5_K only), `q4_sym128_all` (Q4_K too, drops its per-group zero point), or `native` (no requantization). |
|
||||
| `GGML_OPENVINO_PROFILING` | Integer | `0` | `1` logs execution timing; `2` or higher also enables OpenVINO and OpenCL profiling. |
|
||||
| `GGML_OPENVINO_DEBUG_NODE` | String | `not set` | Add the named graph nodes as compiled outputs for debugging. Separate multiple names with commas. |
|
||||
| `GGML_OPENVINO_MOE_OP` | Boolean | `1` | On GPU, set to `0` to keep the unfused GatherMatmul path. |
|
||||
| `GGML_OPENVINO_DUMP_CGRAPH` | Boolean | `0` | Dump the GGML compute graph to `cgraph_ov.txt`. |
|
||||
| `GGML_OPENVINO_DUMP_IR` | Boolean | `0` | Serialize OpenVINO IR files with timestamps. |
|
||||
| `GGML_OPENVINO_DEBUG_INPUT` | Boolean | `0` | Enable input debugging and print input tensor info. |
|
||||
@@ -737,8 +740,9 @@ Boolean flags follow a uniform convention: set to a **positive integer** (e.g. `
|
||||
| `GGML_OPENVINO_LOG_UNSUPPORTED_OPS`| Boolean | `0` | Log warning messages with tensor details and rejection reasons for any ops not supported by the OpenVINO backend. Emits at `WARN` level (requires `--log-verbosity >= 2`, enabled by default). |
|
||||
|
||||
> [!NOTE]
|
||||
> - `GGML_OPENVINO_STATEFUL_EXECUTION` is an **Experimental** feature to allow stateful execution for managing the KV cache internally inside the OpenVINO model, improving performance on CPUs and GPUs. Stateful execution is not effective on NPUs, and not all models currently support this feature. This feature is experimental and has been validated only with the llama-simple, llama-cli, llama-bench, and llama-run applications and is recommended to enable for the best performance. Other applications, such as llama-server and llama-perplexity, are not yet supported.
|
||||
> - `GGML_OPENVINO_STATEFUL_EXECUTION` is an **Experimental** feature for managing caches internally inside the OpenVINO model on CPUs and GPUs. Use a single slot (`-np 1`). KV caches retain the append-based state layout and sequence-axis optimization. Qwen3.5 adds recurrent cache states in their GGML layouts. Qwen3.5 requires an unsplit graph with model caching enabled and no recurrent rollback. A prompt starting at position 0 resets all states. State save/restore, sequence rewind, context shift, and mid-sequence graph replacement are unsupported. Stateful execution is not effective on NPUs.
|
||||
> - `GGML_OPENVINO_LOG_UNSUPPORTED_OPS` emits logs at `WARN` level (`GGML_LOG_WARN`), which requires application log verbosity `--log-verbosity >= 2` (or `-lv 2`).
|
||||
> - With `GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY=1`, use the same compilation settings as the export run. One directory can hold blobs for different models and settings; `GGML_OPENVINO_SPILL_DIR` does not affect the cache key and is ignored in cache-only mode. See [Compiled model cache](../../ggml/src/ggml-openvino/README.md) for the workflow and restrictions.
|
||||
|
||||
### Example Usage
|
||||
|
||||
|
||||
@@ -1305,7 +1305,7 @@ void ggml_compute_forward_mul_mat(
|
||||
|
||||
const bool src1_cont = ggml_is_contiguous(src1);
|
||||
|
||||
if (src1_cont) {
|
||||
if (!params->use_ref && src1_cont) {
|
||||
for (int64_t i13 = 0; i13 < ne13; i13++)
|
||||
for (int64_t i12 = 0; i12 < ne12; i12++)
|
||||
if (!llamafile_sgemm(params,
|
||||
@@ -1384,7 +1384,7 @@ UseGgmlGemm1:;
|
||||
ggml_barrier(params->threadpool);
|
||||
|
||||
#if GGML_USE_LLAMAFILE
|
||||
if (src1->type != vec_dot_type) {
|
||||
if (!params->use_ref && src1->type != vec_dot_type) {
|
||||
const void* wdata = (src1->type == vec_dot_type) ? src1->data : params->wdata;
|
||||
const size_t row_size = ggml_row_size(vec_dot_type, ne10);
|
||||
|
||||
|
||||
@@ -384,6 +384,80 @@ template <> inline __m256bh load(const float *p) {
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__AVX__) || defined(__AVX2__) || defined(__AVX512F__)
|
||||
template <typename T, typename U> T load_partial(const U *, int);
|
||||
template <typename T> T load_partial_u16(const void *, int);
|
||||
|
||||
template <> inline __m128i load_partial_u16(const void *p, int n) {
|
||||
#if defined(__AVX512BW__) && defined(__AVX512VL__)
|
||||
return _mm_maskz_loadu_epi16((1u << n) - 1, p);
|
||||
#else
|
||||
const __m128i index = _mm_setr_epi32(0, 1, 2, 3);
|
||||
const __m128i pairs = _mm_set1_epi32(n / 2);
|
||||
__m128i v = _mm_castps_si128(_mm_maskload_ps((const float *)p, _mm_cmpgt_epi32(pairs, index)));
|
||||
if (n & 1) {
|
||||
uint16_t last;
|
||||
memcpy(&last, (const char *)p + 2*(n - 1), sizeof(last));
|
||||
v = _mm_or_si128(v, _mm_and_si128(_mm_cmpeq_epi32(pairs, index), _mm_set1_epi32(last)));
|
||||
}
|
||||
return v;
|
||||
#endif
|
||||
}
|
||||
|
||||
template <> inline __m256 load_partial(const float *p, int n) {
|
||||
const __m256 index = _mm256_setr_ps(0, 1, 2, 3, 4, 5, 6, 7);
|
||||
return _mm256_maskload_ps(p, _mm256_castps_si256(_mm256_cmp_ps(index, _mm256_set1_ps(n), _CMP_LT_OQ)));
|
||||
}
|
||||
|
||||
#if defined(__F16C__)
|
||||
template <> inline __m256 load_partial(const ggml_fp16_t *p, int n) {
|
||||
return _mm256_cvtph_ps(load_partial_u16<__m128i>(p, n));
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__AVX2__) || defined(__AVX512F__)
|
||||
template <> inline __m256 load_partial(const ggml_bf16_t *p, int n) {
|
||||
return _mm256_castsi256_ps(_mm256_slli_epi32(_mm256_cvtepu16_epi32(load_partial_u16<__m128i>(p, n)), 16));
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__AVX512F__)
|
||||
template <> inline __m256i load_partial_u16(const void *p, int n) {
|
||||
#if defined(__AVX512BW__) && defined(__AVX512VL__)
|
||||
return _mm256_maskz_loadu_epi16((1u << n) - 1, p);
|
||||
#else
|
||||
const __m256i index = _mm256_setr_epi32(0, 1, 2, 3, 4, 5, 6, 7);
|
||||
const __m256i pairs = _mm256_set1_epi32(n / 2);
|
||||
__m256i v = _mm256_maskload_epi32((const int *)p, _mm256_cmpgt_epi32(pairs, index));
|
||||
if (n & 1) {
|
||||
uint16_t last;
|
||||
memcpy(&last, (const char *)p + 2*(n - 1), sizeof(last));
|
||||
v = _mm256_or_si256(v, _mm256_and_si256(_mm256_cmpeq_epi32(pairs, index), _mm256_set1_epi32(last)));
|
||||
}
|
||||
return v;
|
||||
#endif
|
||||
}
|
||||
|
||||
template <> inline __m512 load_partial(const float *p, int n) {
|
||||
return _mm512_maskz_loadu_ps((1u << n) - 1, p);
|
||||
}
|
||||
|
||||
template <> inline __m512 load_partial(const ggml_fp16_t *p, int n) {
|
||||
return _mm512_cvtph_ps(load_partial_u16<__m256i>(p, n));
|
||||
}
|
||||
|
||||
template <> inline __m512 load_partial(const ggml_bf16_t *p, int n) {
|
||||
return _mm512_castsi512_ps(_mm512_slli_epi32(_mm512_cvtepu16_epi32(load_partial_u16<__m256i>(p, n)), 16));
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__AVX512BF16__)
|
||||
template <> inline __m512bh load_partial(const ggml_bf16_t *p, int n) {
|
||||
return (__m512bh) _mm512_maskz_loadu_epi16((uint64_t(1) << n) - 1, p);
|
||||
}
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#if defined(__riscv_v_intrinsic)
|
||||
template <> inline vfloat32m1_t load(const float *p) {
|
||||
return __riscv_vle32_v_f32m1(p, __riscv_vsetvlmax_e32m1());
|
||||
@@ -492,8 +566,10 @@ class tinyBLAS {
|
||||
}
|
||||
|
||||
bool matmul(int64_t m, int64_t n) {
|
||||
#if !defined(__AVX__) && !defined(__AVX2__) && !defined(__AVX512F__)
|
||||
if (k % KN != 0)
|
||||
return false;
|
||||
#endif
|
||||
// compute RM for only need tile with size RM&RM-1
|
||||
#if VECTOR_REGISTERS == 32
|
||||
if (m % 16 == 0 && (m/16 >= params->nth)) {
|
||||
@@ -548,7 +624,7 @@ class tinyBLAS {
|
||||
template <int RM, int RN>
|
||||
inline void gemm_bloc(int64_t ii, int64_t jj) {
|
||||
D Cv[RN][RM] = {};
|
||||
for (int64_t l = 0; l < k; l += KN) {
|
||||
for (int64_t l = 0; l + KN <= k; l += KN) {
|
||||
// help compiler for op order.
|
||||
if constexpr (RM <= RN) {
|
||||
V Av[RM];
|
||||
@@ -574,6 +650,21 @@ class tinyBLAS {
|
||||
}
|
||||
}
|
||||
}
|
||||
#if defined(__AVX__) || defined(__AVX2__) || defined(__AVX512F__)
|
||||
const int64_t rem = k % KN;
|
||||
if (rem != 0) {
|
||||
V Av[RM];
|
||||
for (int64_t i = 0; i < RM; ++i) {
|
||||
Av[i] = load_partial<V>(A + lda * (ii + i) + k - rem, rem);
|
||||
}
|
||||
for (int64_t j = 0; j < RN; ++j) {
|
||||
V Bv = load_partial<V>(B + ldb * (jj + j) + k - rem, rem);
|
||||
for (int64_t i = 0; i < RM; ++i) {
|
||||
Cv[j][i] = madd(Av[i], Bv, Cv[j][i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for (int64_t j = 0; j < RN; ++j)
|
||||
for (int64_t i = 0; i < RM; ++i)
|
||||
C[ldc * (jj + j) + (ii + i)] = hsum(Cv[j][i]);
|
||||
|
||||
@@ -1571,6 +1571,9 @@ struct ggml_cuda_mm_fusion_args_host {
|
||||
const ggml_tensor * gate_scale = nullptr;
|
||||
ggml_glu_op glu_op;
|
||||
float glu_limit = 0.0f;
|
||||
const ggml_tensor * shared_up = nullptr;
|
||||
const ggml_tensor * shared_gate = nullptr;
|
||||
ggml_tensor * shared_dst = nullptr;
|
||||
};
|
||||
struct ggml_cuda_mm_fusion_args_device {
|
||||
const void * x_bias = nullptr;
|
||||
@@ -1580,6 +1583,10 @@ struct ggml_cuda_mm_fusion_args_device {
|
||||
const void * gate_scale = nullptr;
|
||||
ggml_glu_op glu_op;
|
||||
float glu_limit = 0.0f;
|
||||
const void * shared_up = nullptr;
|
||||
const void * shared_gate = nullptr;
|
||||
float * shared_dst = nullptr;
|
||||
uint32_t shared_stride_col_dst = 0;
|
||||
};
|
||||
|
||||
struct ggml_cuda_kernel_launch_params {
|
||||
|
||||
@@ -329,32 +329,6 @@ static constexpr __device__ bool ggml_cuda_fattn_mma_get_Q_in_reg(const int DKQ,
|
||||
return ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols).Q_in_reg;
|
||||
}
|
||||
|
||||
// Swizzling needs a tile stride that is a multiple of 32 half2 columns.
|
||||
static constexpr __host__ __device__ bool ggml_cuda_fattn_mma_bank_aligned(const int nbatch_2) {
|
||||
return nbatch_2 >= 32 && nbatch_2 % 32 == 0;
|
||||
}
|
||||
|
||||
// Swizzling needs ldmatrix, on other hardware the tiles keep the row padding.
|
||||
static __host__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols1, const int ncols2, const int cc) {
|
||||
const fattn_mma_config cfg = ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols1*ncols2, cc);
|
||||
return turing_mma_available(cc) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_K2) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_V2);
|
||||
}
|
||||
|
||||
static constexpr __device__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols1, const int ncols2) {
|
||||
#if defined(TURING_MMA_AVAILABLE)
|
||||
const fattn_mma_config cfg = ggml_cuda_fattn_mma_get_config(DKQ, DV, ncols1*ncols2);
|
||||
return ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_K2) && ggml_cuda_fattn_mma_bank_aligned(cfg.nbatch_V2);
|
||||
#else
|
||||
GGML_UNUSED_VARS(DKQ, DV, ncols1, ncols2);
|
||||
return false;
|
||||
#endif // defined(TURING_MMA_AVAILABLE)
|
||||
}
|
||||
|
||||
// Row padding is only needed if the tile is not swizzled.
|
||||
static constexpr __host__ __device__ int ggml_cuda_fattn_mma_get_stride_tile(const int nbatch_2, const bool swizzled) {
|
||||
return swizzled ? nbatch_2 : nbatch_2 + 4;
|
||||
}
|
||||
|
||||
static constexpr __device__ int get_cols_per_thread() {
|
||||
#if defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE)
|
||||
return 1; // AMD has a single column per thread.
|
||||
@@ -372,6 +346,20 @@ static __host__ int get_cols_per_warp(const int cc) {
|
||||
}
|
||||
}
|
||||
|
||||
static __host__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols, const int cc) {
|
||||
return turing_mma_available(cc) &&
|
||||
ggml_cuda_fattn_mma_get_nbatch_K2(DKQ, DV, ncols, cc) % 32 == 0 && ggml_cuda_fattn_mma_get_nbatch_V2(DKQ, DV, ncols, cc) % 32 == 0;
|
||||
}
|
||||
|
||||
static constexpr __device__ bool ggml_cuda_fattn_mma_get_swizzled(const int DKQ, const int DV, const int ncols) {
|
||||
#ifdef TURING_MMA_AVAILABLE
|
||||
return ggml_cuda_fattn_mma_get_nbatch_K2(DKQ, DV, ncols) % 32 == 0 && ggml_cuda_fattn_mma_get_nbatch_V2(DKQ, DV, ncols) % 32 == 0;
|
||||
#else
|
||||
GGML_UNUSED_VARS(DKQ, DV, ncols);
|
||||
return false;
|
||||
#endif // TURING_MMA_AVAILABLE
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
static __host__ int ggml_cuda_fattn_mma_get_nstages(const int DKQ, const int DV, const int ncols1, const int ncols2, const int cc) {
|
||||
@@ -392,14 +380,15 @@ static constexpr __device__ int ggml_cuda_fattn_mma_get_nstages(
|
||||
|
||||
// ------------------------------------------------------------------------------------------------------------------
|
||||
|
||||
template<int stride_tile, bool swz, int nwarps, int nbatch_fa, bool use_cp_async, bool oob_check, bool use_sparse>
|
||||
template<int stride_tile, int nwarps, int nbatch_fa, bool use_cp_async, bool oob_check, bool use_sparse>
|
||||
static __device__ __forceinline__ void flash_attn_ext_f16_load_tile(
|
||||
const half2 * const __restrict__ KV, half2 * const __restrict__ tile_KV, const int D2, const int stride_KV,
|
||||
const int k_VKQ_0, const int i_sup, const int32_t * const __restrict__ indices) {
|
||||
constexpr int warp_size = ggml_cuda_get_physical_warp_size();
|
||||
// K/V data is loaded with decreasing granularity for D for better memory bandwidth.
|
||||
// The minimum granularity is 16 bytes.
|
||||
constexpr int h2_per_chunk = 16/sizeof(half2);
|
||||
constexpr int chunk_size = 16;
|
||||
constexpr int h2_per_chunk = chunk_size / sizeof(half2);
|
||||
const int chunks_per_row = D2 / h2_per_chunk;
|
||||
if constexpr (use_cp_async) {
|
||||
static_assert(warp_size == 32, "bad warp_size");
|
||||
@@ -439,7 +428,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile(
|
||||
for (int k0 = k0_start; k0 < k0_stop; k0 += stride_k) {
|
||||
const int k = k0 + (stride_k == warp_size ? threadIdx.x : threadIdx.x % stride_k);
|
||||
|
||||
cp_async_cg_16<preload>(tile_KV_32 + swizzle_bytes<swz, half2>(i, k*h2_per_chunk, stride_tile), KV + i_KV*stride_KV + k*h2_per_chunk);
|
||||
cp_async_cg_16<preload>(tile_KV_32 + swizzle<stride_tile*sizeof(half2), char>(i*stride_tile*sizeof(half2) + k*chunk_size, i), KV + i_KV*stride_KV + k*h2_per_chunk);
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -481,7 +470,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_load_tile(
|
||||
} else {
|
||||
src = !oob_check || i < i_sup ? KV + int64_t(k_VKQ_0 + i)*stride_KV + k*h2_per_chunk : zero;
|
||||
}
|
||||
ggml_cuda_memcpy_1<16>((char *) tile_KV + swizzle_bytes<swz, half2>(i, k*h2_per_chunk, stride_tile), src);
|
||||
ggml_cuda_memcpy_1<16>(swizzle<stride_tile>(tile_KV, i*stride_tile + k*h2_per_chunk, i), src);
|
||||
}
|
||||
}
|
||||
};
|
||||
@@ -624,9 +613,9 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
constexpr bool Q_in_reg = ggml_cuda_fattn_mma_get_Q_in_reg (DKQ, DV, ncols);
|
||||
constexpr int nstages = ggml_cuda_fattn_mma_get_nstages (DKQ, DV, ncols1, ncols2, use_sparse);
|
||||
|
||||
constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2);
|
||||
constexpr int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swz);
|
||||
constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swz);
|
||||
constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols);
|
||||
constexpr int stride_tile_K = swz ? nbatch_K2 : nbatch_K2 + 4;
|
||||
constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : (swz ? nbatch_V2 : nbatch_V2 + 4);
|
||||
|
||||
const int k_VKQ_0 = kb0 * nbatch_fa;
|
||||
#if defined(TURING_MMA_AVAILABLE)
|
||||
@@ -644,7 +633,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
constexpr bool use_cp_async = true;
|
||||
cp_async_wait_all();
|
||||
__syncthreads();
|
||||
flash_attn_ext_f16_load_tile<stride_tile_V, swz, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
flash_attn_ext_f16_load_tile<stride_tile_V, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(V_h2, tile_V, nbatch_V2, stride_V, k_VKQ_0, k_VKQ_sup, nullptr);
|
||||
} else {
|
||||
// the sparse mask values are gathered per element, always load them synchronously
|
||||
@@ -664,7 +653,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
if constexpr (nstages <= 1) {
|
||||
const int k0_diff = k0_stop - k0_start;
|
||||
constexpr bool use_cp_async = nstages == 1;
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, swz, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(K_h2 + k0_start, tile_K, k0_diff, stride_K, k_VKQ_0, k_VKQ_sup, indices);
|
||||
if (use_cp_async) {
|
||||
cp_async_wait_all();
|
||||
@@ -680,7 +669,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
#pragma unroll
|
||||
for (int k_KQ_0 = k0_start; k_KQ_0 < k0_stop; k_KQ_0 += T_A_KQ::J) {
|
||||
T_A_KQ K_A;
|
||||
load_ldmatrix<swz>(K_A, tile_K, i_KQ_0, k_KQ_0 - k0_start, stride_tile_K);
|
||||
load_ldmatrix_swizzled<stride_tile_K>(K_A, tile_K, i_KQ_0*stride_tile_K + k_KQ_0-k0_start);
|
||||
if constexpr (cols_per_warp == 8) {
|
||||
mma(KQ_C[i_KQ_00/(np*T_A_KQ::I)], K_A, Q_B[k_KQ_0/T_A_KQ::J]);
|
||||
} else {
|
||||
@@ -706,7 +695,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
const int i_KQ_0 = i_KQ_00 + (threadIdx.y % np)*T_A_KQ::I;
|
||||
|
||||
T_A_KQ K_A;
|
||||
load_ldmatrix<swz>(K_A, tile_K, i_KQ_0, k_KQ_0 - k0_start, stride_tile_K);
|
||||
load_ldmatrix_swizzled<stride_tile_K>(K_A, tile_K, i_KQ_0*stride_tile_K + k_KQ_0-k0_start);
|
||||
|
||||
if constexpr (cols_per_warp == 8) {
|
||||
mma(KQ_C[i_KQ_00/(np*T_A_KQ::I)], K_A, Q_B[0]);
|
||||
@@ -1001,7 +990,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
flash_attn_ext_f16_load_mask<ncols1, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(mask_h, tile_mask, stride_mask, k_VKQ_0 + nbatch_fa, k_VKQ_sup, jt*ncols1, ne01, nullptr);
|
||||
}
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, swz, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(K_h2, tile_K, nbatch_K2, stride_K, k_VKQ_0 + nbatch_fa, k_VKQ_sup, nullptr);
|
||||
}
|
||||
}
|
||||
@@ -1017,7 +1006,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
const int i0_diff = i0_stop - i0_start;
|
||||
if (!V_is_K_view || i0_stop > 2*nbatch_K2) {
|
||||
constexpr bool use_cp_async = nstages == 1;
|
||||
flash_attn_ext_f16_load_tile<stride_tile_V, swz, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
flash_attn_ext_f16_load_tile<stride_tile_V, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(V_h2 + i0_start/2, tile_V, i0_diff/2, stride_V, k_VKQ_0, k_VKQ_sup, indices);
|
||||
if (use_cp_async) {
|
||||
cp_async_wait_all();
|
||||
@@ -1025,7 +1014,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
__syncthreads();
|
||||
}
|
||||
}
|
||||
const half2 * tile_V_i = !V_is_K_view || i0_stop > 2*nbatch_K2 ? tile_V : tile_V + i0_start/2;
|
||||
const int tile_V_offset_i = !V_is_K_view || i0_stop > 2*nbatch_K2 ? 0 : i0_start/2;
|
||||
|
||||
#if defined(TURING_MMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE) || defined(AMD_MFMA_AVAILABLE)
|
||||
#pragma unroll
|
||||
@@ -1036,7 +1025,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
const int k0 = k00 + (threadIdx.y % np)*T_A_VKQ::J;
|
||||
|
||||
T_A_VKQ A; // Transposed in SRAM but not in registers, gets transposed on load.
|
||||
load_ldmatrix_trans<swz>(A, tile_V, 2*k0, (int)(tile_V_i - tile_V) + (i_VKQ_0 - i0_start)/2, stride_tile_V);
|
||||
load_ldmatrix_trans_swizzled<stride_tile_V>(A, tile_V, tile_V_offset_i + 2*k0*stride_tile_V + (i_VKQ_0 - i0_start)/2);
|
||||
if constexpr (T_B_KQ::I == 8) {
|
||||
mma(VKQ_C[i_VKQ_0/T_A_VKQ::I], A, B[k00/(np*T_A_VKQ::J)]);
|
||||
} else {
|
||||
@@ -1062,8 +1051,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_iter(
|
||||
const int k0 = k00 + (threadIdx.y % np)*T_A_VKQ::I;
|
||||
|
||||
T_A_VKQ A; // Transposed in both SRAM and registers, load normally.
|
||||
static_assert(!swz, "Volta has no ldmatrix");
|
||||
load_ldmatrix(A, tile_V_i + k0*stride_tile_V + (i_VKQ_0 - i0_start)/2, stride_tile_V);
|
||||
load_ldmatrix_swizzled<stride_tile_V>(A, tile_V, tile_V_offset_i + k0*stride_tile_V + (i_VKQ_0 - i0_start)/2);
|
||||
mma(VKQ_C[i_VKQ_0/i0_stride], B[k00/(np*T_A_VKQ::I)], A);
|
||||
}
|
||||
}
|
||||
@@ -1253,10 +1241,10 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile(
|
||||
|
||||
static_assert(nwarps * (cols_per_warp/ncols2) % ncols1 == 0, "bad nwarps");
|
||||
|
||||
constexpr int stride_tile_Q = DKQ/2 + 4;
|
||||
constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2);
|
||||
constexpr int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swz);
|
||||
constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swz);
|
||||
constexpr bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols);
|
||||
constexpr int stride_tile_Q = DKQ/2 + 4;
|
||||
constexpr int stride_tile_K = swz ? nbatch_K2 : nbatch_K2 + 4;
|
||||
constexpr int stride_tile_V = V_is_K_view ? stride_tile_K : (swz ? nbatch_V2 : nbatch_V2 + 4);
|
||||
constexpr int stride_tile_KV_max = stride_tile_K > stride_tile_V ? stride_tile_K : stride_tile_V;
|
||||
|
||||
extern __shared__ half2 tile_Q[];
|
||||
@@ -1354,7 +1342,7 @@ static __device__ __forceinline__ void flash_attn_ext_f16_process_tile(
|
||||
flash_attn_ext_f16_load_mask<ncols1, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(mask_h, tile_mask, stride_mask, kb0*nbatch_fa, k_VKQ_sup, jt*ncols1, ne01, nullptr);
|
||||
}
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, swz, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
flash_attn_ext_f16_load_tile<stride_tile_K, nwarps, nbatch_fa, use_cp_async, oob_check, use_sparse>
|
||||
(K_h2, tile_K, nbatch_K2, stride_K, kb0*nbatch_fa, k_VKQ_sup, nullptr);
|
||||
}
|
||||
|
||||
@@ -2039,9 +2027,9 @@ void ggml_cuda_flash_attn_ext_mma_f16_case(ggml_backend_cuda_context & ctx, ggml
|
||||
constexpr bool V_is_K_view = DKQ == 576; // Guaranteed by the kernel selection logic in fattn.cu
|
||||
|
||||
// KV tile strides must match flash_attn_ext_f16_iter / _process_tile.
|
||||
const bool swizzled = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols1, ncols2, cc);
|
||||
const int stride_tile_K = ggml_cuda_fattn_mma_get_stride_tile(nbatch_K2, swizzled);
|
||||
const int stride_tile_V = V_is_K_view ? stride_tile_K : ggml_cuda_fattn_mma_get_stride_tile(nbatch_V2, swizzled);
|
||||
const bool swz = ggml_cuda_fattn_mma_get_swizzled(DKQ, DV, ncols, cc);
|
||||
const int stride_tile_K = swz ? nbatch_K2 : nbatch_K2 + 4;
|
||||
const int stride_tile_V = V_is_K_view ? stride_tile_K : (swz ? nbatch_V2 : nbatch_V2 + 4);
|
||||
const size_t nbytes_shared_KV_1stage = nbatch_fa * std::max(stride_tile_K, stride_tile_V) * sizeof(half2);
|
||||
const size_t nbytes_shared_KV_2stage = nbatch_fa * (stride_tile_K + stride_tile_V) * sizeof(half2);
|
||||
const size_t nbytes_shared_Q = ncols * (DKQ/2 + 4) * sizeof(half2);
|
||||
|
||||
@@ -1823,6 +1823,55 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) {
|
||||
return use_mul_mat_vec_q;
|
||||
}
|
||||
|
||||
static bool ggml_cuda_match_shared_expert(const ggml_cgraph * graph, int routed_idx, int shared_idx) {
|
||||
if (routed_idx + 2 >= graph->n_nodes || shared_idx + 2 >= graph->n_nodes || shared_idx < routed_idx + 3) {
|
||||
return false;
|
||||
}
|
||||
const int nodes[] = { routed_idx, routed_idx + 1, routed_idx + 2, shared_idx, shared_idx + 1, shared_idx + 2 };
|
||||
const ggml_op ops[] = { GGML_OP_MUL_MAT_ID, GGML_OP_MUL_MAT_ID, GGML_OP_GLU,
|
||||
GGML_OP_MUL_MAT, GGML_OP_MUL_MAT, GGML_OP_GLU };
|
||||
const int outputs[] = { routed_idx + 2, shared_idx + 2 };
|
||||
if (!ggml_can_fuse_subgraph_ext(graph, nodes, 6, ops, outputs, 2)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const ggml_tensor * routed = graph->nodes[routed_idx + 2];
|
||||
const ggml_tensor * shared = graph->nodes[shared_idx + 2];
|
||||
const ggml_tensor * gate = routed->src[0];
|
||||
const ggml_tensor * up = routed->src[1];
|
||||
const ggml_tensor * shared_gate = shared->src[0];
|
||||
const ggml_tensor * shared_up = shared->src[1];
|
||||
const auto is_pair = [&](const ggml_tensor * a, const ggml_tensor * b, int idx) {
|
||||
return (a == graph->nodes[idx] && b == graph->nodes[idx + 1]) ||
|
||||
(b == graph->nodes[idx] && a == graph->nodes[idx + 1]);
|
||||
};
|
||||
if (!is_pair(gate, up, routed_idx) || !is_pair(shared_gate, shared_up, shared_idx) ||
|
||||
!ggml_cuda_should_fuse_mul_mat(up, gate, routed) ||
|
||||
!ggml_cuda_should_fuse_mul_mat(shared_up, shared_gate, shared) ||
|
||||
!up->src[0]->buffer ||
|
||||
!ggml_cuda_should_fuse_mul_mat_vec_q(up)) {
|
||||
return false;
|
||||
}
|
||||
const ggml_tensor * input = up->src[1];
|
||||
const ggml_tensor * weight = up->src[0];
|
||||
const ggml_tensor * shared_weight = shared_up->src[0];
|
||||
if (input->op != GGML_OP_RESHAPE || input->src[0] != shared_up->src[1] ||
|
||||
input->ne[1] != 1 || input->ne[3] != 1 || !ggml_is_contiguous(input) ||
|
||||
!ggml_is_contiguous(shared_up->src[1]) || !ggml_is_matrix(shared_up->src[1]) ||
|
||||
weight->type != shared_weight->type || weight->ne[0] != shared_weight->ne[0] ||
|
||||
weight->ne[1] != shared_weight->ne[1] || weight->nb[1] != shared_weight->nb[1] || weight->ne[3] != 1 ||
|
||||
!ggml_is_matrix(shared_weight) || !ggml_is_contiguous(shared_weight) ||
|
||||
!ggml_is_contiguous(shared_gate->src[0]) || !ggml_is_contiguous(routed) || !ggml_is_contiguous(shared)) {
|
||||
return false;
|
||||
}
|
||||
if (shared_weight->op != GGML_OP_NONE || shared_gate->src[0]->op != GGML_OP_NONE ||
|
||||
ggml_get_glu_op(routed) != ggml_get_glu_op(shared) ||
|
||||
ggml_get_op_params_f32(routed, 3) != ggml_get_op_params_f32(shared, 3)) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) {
|
||||
GGML_TENSOR_BINARY_OP_LOCALS
|
||||
|
||||
@@ -3459,6 +3508,25 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
|
||||
|
||||
ggml_tensor * node = cgraph->nodes[i];
|
||||
|
||||
if (node->op == GGML_OP_MUL_MAT_ID && cuda_ctx->stream_context().concurrent_events.empty() &&
|
||||
ggml_cuda_match_shared_expert(cgraph, i, i + 3)) {
|
||||
const int outputs[] = { i + 2, i + 5 };
|
||||
if (ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, outputs, 2)) {
|
||||
ggml_tensor * routed = cgraph->nodes[i + 2];
|
||||
ggml_tensor * shared = cgraph->nodes[i + 5];
|
||||
const ggml_tensor * up = routed->src[1];
|
||||
ggml_cuda_mm_fusion_args_host fusion{};
|
||||
fusion.gate = routed->src[0]->src[0];
|
||||
fusion.glu_op = ggml_get_glu_op(routed);
|
||||
fusion.glu_limit = ggml_get_op_params_f32(routed, 3);
|
||||
fusion.shared_up = shared->src[1]->src[0];
|
||||
fusion.shared_gate = shared->src[0]->src[0];
|
||||
fusion.shared_dst = shared;
|
||||
ggml_cuda_mul_mat_vec_q(*cuda_ctx, up->src[0], up->src[1], up->src[2], routed, &fusion);
|
||||
return 5;
|
||||
}
|
||||
}
|
||||
|
||||
if (node->op == GGML_OP_MUL) {
|
||||
ggml_cuda_moe_weighted_reduction_match match;
|
||||
if (ggml_cuda_match_moe_weighted_reduction(cgraph, i, match)) {
|
||||
@@ -4549,6 +4617,27 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph
|
||||
if (!disable_fusion) {
|
||||
// add alloc deps for performance positive fusions. This may increase the overall compute buffer size.
|
||||
// TODO: consolidate fusion paths in graph_optimize and graph_compute
|
||||
ggml_cuda_set_device(cuda_ctx->device);
|
||||
for (int i = 0; i + 5 < cgraph->n_nodes; ++i) {
|
||||
if (cgraph->nodes[i]->op != GGML_OP_MUL_MAT_ID) {
|
||||
continue;
|
||||
}
|
||||
for (int j = i + 3; j + 2 < cgraph->n_nodes; ++j) {
|
||||
if (cgraph->nodes[j]->op == GGML_OP_MUL_MAT_ID && cgraph->nodes[j + 1]->op == GGML_OP_MUL_MAT_ID) {
|
||||
break;
|
||||
}
|
||||
if (cgraph->nodes[j]->op != GGML_OP_MUL_MAT || !ggml_cuda_match_shared_expert(cgraph, i, j)) {
|
||||
continue;
|
||||
}
|
||||
// Group both outputs before allocation so the shared result cannot alias intervening nodes.
|
||||
std::rotate(cgraph->nodes + i + 3, cgraph->nodes + j, cgraph->nodes + j + 3);
|
||||
ggml_tensor * up = cgraph->nodes[i + 2]->src[1];
|
||||
params->add_alloc_dep(params->user_data, up->src[1], cgraph->nodes[i + 5]);
|
||||
params->add_alloc_dep(params->user_data, up->src[2], cgraph->nodes[i + 5]);
|
||||
i += 5;
|
||||
break;
|
||||
}
|
||||
}
|
||||
for (int i = 0; i < cgraph->n_nodes; ++i) {
|
||||
ggml_cuda_moe_weighted_reduction_match match;
|
||||
if (ggml_cuda_match_moe_weighted_reduction(cgraph, i, match)) {
|
||||
|
||||
@@ -528,6 +528,25 @@ void ggml_cuda_lightning_indexer(ggml_backend_cuda_context & ctx, ggml_tensor *
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 32, k, GGML_TYPE_F32)
|
||||
GGML_ABORT("fatal error");
|
||||
}
|
||||
} else if (n_embd == 128 && n_head == 4) {
|
||||
// too few heads for a wmma tile, use vector kernel
|
||||
constexpr int K_VECS_PER_WARP = 8;
|
||||
constexpr int WARPS_PER_BLOCK = 8;
|
||||
constexpr int K_VECS_PER_BLOCK = K_VECS_PER_WARP * WARPS_PER_BLOCK;
|
||||
|
||||
dim3 block(32, WARPS_PER_BLOCK);
|
||||
int num_kv_blocks = (n_kv + (K_VECS_PER_BLOCK) - 1) / (K_VECS_PER_BLOCK);
|
||||
dim3 grid(num_kv_blocks, n_batch, n_stream);
|
||||
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_F16)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_Q4_0)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_Q4_1)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_Q5_0)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_Q5_1)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_Q8_0)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_BF16)
|
||||
LIGHTNING_INDEXER_CASE(lightning_indexer_kernel_vec, 128, 4, k, GGML_TYPE_F32)
|
||||
GGML_ABORT("fatal error");
|
||||
} else {
|
||||
GGML_ABORT("fatal error");
|
||||
}
|
||||
@@ -556,7 +575,7 @@ bool ggml_cuda_lightning_indexer_supported(int device, const ggml_tensor * dst)
|
||||
return false;
|
||||
}
|
||||
|
||||
if (neq1 != 64 && neq1 != 32) {
|
||||
if (neq1 != 64 && neq1 != 32 && neq1 != 4) {
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
+122
-44
@@ -782,18 +782,27 @@ namespace ggml_cuda_mma {
|
||||
}
|
||||
}
|
||||
|
||||
// Byte offset of tile element (i, j). If swz, XOR swizzle it to avoid bank conflicts without row padding.
|
||||
template <bool swz, typename T>
|
||||
static __device__ __forceinline__ int swizzle_bytes(const int i, const int j, const int stride) {
|
||||
static_assert(!swz || sizeof(T) == 4, "swizzled tiles need 32 bit elements");
|
||||
const int off = (i*stride + j) * (int) sizeof(T);
|
||||
return swz ? off ^ ((i & 7) << 4) : off;
|
||||
template <int stride, typename T>
|
||||
static __device__ __forceinline__ uint32_t swizzle(const uint32_t offset, const uint32_t i) {
|
||||
static_assert(sizeof(T) <= 4, "unsupported type size");
|
||||
constexpr int stride_bytes = stride*sizeof(T);
|
||||
static_assert(stride_bytes % 16 == 0, "bad stride");
|
||||
constexpr uint32_t shift = sizeof(T) == 1 ? 4 : (sizeof(T) == 2 ? 3 : 2);
|
||||
if (stride_bytes % 32 != 0) {
|
||||
return offset; // Equivalent to padding with 16 bytes.
|
||||
}
|
||||
if (stride_bytes % 64 != 0) {
|
||||
return offset ^ (((i / 4) % 2) << shift);
|
||||
}
|
||||
if (stride_bytes % 128 != 0) {
|
||||
return offset ^ (((i / 2) % 4) << shift);
|
||||
}
|
||||
return offset ^ ((i % 8) << shift);
|
||||
}
|
||||
|
||||
template <bool swz, typename T>
|
||||
static __device__ __forceinline__ const T * swizzle(
|
||||
const T * __restrict__ tile_base, const int i, const int j, const int stride) {
|
||||
return (const T *) ((const char *) tile_base + swizzle_bytes<swz, T>(i, j, stride));
|
||||
template <int stride, typename T>
|
||||
static __device__ __forceinline__ T * swizzle(T * ptr, const uint32_t offset, const uint32_t i) {
|
||||
return ptr + swizzle<stride, T>(offset, i);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
@@ -872,29 +881,6 @@ namespace ggml_cuda_mma {
|
||||
#endif // TURING_MMA_AVAILABLE
|
||||
}
|
||||
|
||||
// Load from tile element (i0, j0), swz tells if the tile is stored swizzled.
|
||||
template <bool swz, int I, int J, typename T, data_layout dl>
|
||||
static __device__ __forceinline__ void load_ldmatrix(
|
||||
tile<I, J, T, dl> & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) {
|
||||
if constexpr (!swz) {
|
||||
load_ldmatrix(t, tile_base + i0*stride + j0, stride);
|
||||
return;
|
||||
}
|
||||
#if defined(TURING_MMA_AVAILABLE)
|
||||
static_assert(I == 16, "bad tile width");
|
||||
static_assert(J == 8, "bad tile height");
|
||||
const int i = i0 + threadIdx.x % t.I;
|
||||
const int j = j0 + (threadIdx.x / t.I) * (t.J / 2);
|
||||
int * xi = (int *) t.x;
|
||||
asm volatile("ldmatrix.sync.aligned.m8n8.x4.b16 {%0, %1, %2, %3}, [%4];"
|
||||
: "=r"(xi[0]), "=r"(xi[1]), "=r"(xi[2]), "=r"(xi[3])
|
||||
: "l"(swizzle<true>(tile_base, i, j, stride)));
|
||||
#else
|
||||
GGML_UNUSED_VARS(t, tile_base, i0, j0, stride);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // defined(TURING_MMA_AVAILABLE)
|
||||
}
|
||||
|
||||
static __device__ __forceinline__ void load_ldmatrix(
|
||||
tile<8, 4, half2, DATA_LAYOUT_I_MAJOR_MIRRORED> & t, const half2 * __restrict__ xs0, const int stride) {
|
||||
ggml_cuda_memcpy_1<4*sizeof(half2)>(t.x, xs0 + t.get_i(0)*stride);
|
||||
@@ -902,10 +888,15 @@ namespace ggml_cuda_mma {
|
||||
|
||||
static __device__ __forceinline__ void load_ldmatrix(
|
||||
tile<8, 4, half2, DATA_LAYOUT_J_MAJOR_MIRRORED> & t, const half2 * __restrict__ xs0, const int stride) {
|
||||
#ifdef VOLTA_MMA_AVAILABLE
|
||||
#pragma unroll
|
||||
for (int l0 = 0; l0 < t.ne; l0 += 2) {
|
||||
ggml_cuda_memcpy_1<2*sizeof(half2)>(t.x + l0, xs0 + t.get_i(l0)*stride + t.get_j(l0));
|
||||
}
|
||||
#else
|
||||
GGML_UNUSED_VARS(t, xs0, stride);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // VOLTA_MMA_AVAILABLE
|
||||
}
|
||||
|
||||
static __device__ __forceinline__ void load_ldmatrix(
|
||||
@@ -954,25 +945,112 @@ namespace ggml_cuda_mma {
|
||||
#endif // TURING_MMA_AVAILABLE
|
||||
}
|
||||
|
||||
// Load from tile element (i0, j0), swz tells if the tile is stored swizzled.
|
||||
template <bool swz, int I, typename T, data_layout dl>
|
||||
static __device__ __forceinline__ void load_ldmatrix_trans(
|
||||
tile<I, 8, T, dl> & t, const T * __restrict__ tile_base, const int i0, const int j0, const int stride) {
|
||||
if constexpr (!swz) {
|
||||
load_ldmatrix_trans(t, tile_base + i0*stride + j0, stride);
|
||||
return;
|
||||
template <int stride, int I, int J, typename T, data_layout dl>
|
||||
static __device__ __forceinline__ void load_ldmatrix_swizzled(
|
||||
tile<I, J, T, dl> & t, const T * __restrict__ xs0, const int offset) {
|
||||
#if defined(TURING_MMA_AVAILABLE)
|
||||
static_assert(I == 16, "bad tile width");
|
||||
static_assert(J == 8, "bad tile height");
|
||||
const int i = threadIdx.x % t.I;
|
||||
const int j = (threadIdx.x / t.I) * (t.J / 2);
|
||||
int offset_ij = offset + i * stride + j;
|
||||
offset_ij = swizzle<stride, T>(offset_ij, i);
|
||||
int * xi = (int *) t.x;
|
||||
asm volatile("ldmatrix.sync.aligned.m8n8.x4.b16 {%0, %1, %2, %3}, [%4];"
|
||||
: "=r"(xi[0]), "=r"(xi[1]), "=r"(xi[2]), "=r"(xi[3])
|
||||
: "l"(xs0 + offset_ij));
|
||||
#elif defined(VOLTA_MMA_AVAILABLE)
|
||||
#pragma unroll
|
||||
for (int o = 0; o < t.ne; o += 4) {
|
||||
const int offset_ij = offset + t.get_i(o) * stride + o;
|
||||
ggml_cuda_memcpy_1<4*sizeof(T)>(t.x + o, swizzle<stride>(xs0, offset_ij, t.get_i(o)));
|
||||
}
|
||||
#elif defined(AMD_WMMA_AVAILABLE)
|
||||
#ifdef RDNA3
|
||||
static_assert(dl == DATA_LAYOUT_I_MAJOR_MIRRORED, "bad data layout");
|
||||
static_assert(sizeof(t.x) == 32, "bad ne");
|
||||
static_assert(I == 16, "bad tile width");
|
||||
static_assert(J == 8, "bad tile height");
|
||||
#pragma unroll
|
||||
for (int o = 0; o < 8; o += 4) {
|
||||
const int offset_ij = offset + t.get_i(0) * stride + o;
|
||||
ggml_cuda_memcpy_1<16>(t.x + o, swizzle<stride>(xs0, offset_ij, t.get_i(0)));
|
||||
}
|
||||
#else
|
||||
static_assert(dl == DATA_LAYOUT_I_MAJOR, "bad data layout");
|
||||
static_assert(sizeof(t.x) == 16, "bad ne");
|
||||
const int offset_ij = offset + t.get_i(0)*stride + t.get_j(0);
|
||||
ggml_cuda_memcpy_1<16>(t.x, swizzle<stride>(xs0, offset_ij, t.get_i(0)));
|
||||
#endif // RDNA3
|
||||
#elif defined(AMD_MFMA_AVAILABLE)
|
||||
static_assert(sizeof(t.x) == 8, "bad ne");
|
||||
const int offset_ij = offset + t.get_i(0)*stride + t.get_j(0);
|
||||
ggml_cuda_memcpy_1<8>(t.x, swizzle<stride>(xs0, offset_ij, t.get_i(0)));
|
||||
#else
|
||||
GGML_UNUSED_VARS(t, xs0, offset);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // defined(TURING_MMA_AVAILABLE)
|
||||
}
|
||||
|
||||
template <int stride>
|
||||
static __device__ __forceinline__ void load_ldmatrix_swizzled(
|
||||
tile<8, 4, half2, DATA_LAYOUT_J_MAJOR_MIRRORED> & t, const half2 * __restrict__ xs0, const int offset) {
|
||||
#ifdef VOLTA_MMA_AVAILABLE
|
||||
#pragma unroll
|
||||
for (int l0 = 0; l0 < t.ne; l0 += 2) {
|
||||
const int offset_ij = offset + t.get_i(l0)*stride + t.get_j(l0);
|
||||
ggml_cuda_memcpy_1<2*sizeof(half2)>(t.x + l0, swizzle<stride>(xs0, offset_ij, t.get_i(l0)));
|
||||
}
|
||||
#else
|
||||
GGML_UNUSED_VARS(t, xs0, offset);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // VOLTA_MMA_AVAILABLE
|
||||
}
|
||||
|
||||
template <int stride, int I, typename T, data_layout dl>
|
||||
static __device__ __forceinline__ void load_ldmatrix_trans_swizzled(
|
||||
tile<I, 8, T, dl> & t, const T * __restrict__ xs0, const int offset) {
|
||||
#if defined(TURING_MMA_AVAILABLE)
|
||||
static_assert(I == 16, "bad tile width");
|
||||
static_assert(dl == DATA_LAYOUT_I_MAJOR, "bad data layout");
|
||||
const int i = i0 + threadIdx.x % t.I;
|
||||
const int j = j0 + (threadIdx.x / t.I) * (t.J / 2);
|
||||
const int i = threadIdx.x % t.I;
|
||||
const int j = (threadIdx.x / t.I) * (t.J / 2);
|
||||
int offset_ij = offset + i * stride + j;
|
||||
offset_ij = swizzle<stride, T>(offset_ij, i);
|
||||
int * xi = (int *) t.x;
|
||||
asm volatile("ldmatrix.sync.aligned.m8n8.x4.trans.b16 {%0, %1, %2, %3}, [%4];"
|
||||
: "=r"(xi[0]), "=r"(xi[2]), "=r"(xi[1]), "=r"(xi[3])
|
||||
: "l"(swizzle<true>(tile_base, i, j, stride)));
|
||||
: "l"(xs0 + offset_ij));
|
||||
#elif defined(AMD_MFMA_AVAILABLE) || defined(AMD_WMMA_AVAILABLE)
|
||||
static_assert(dl == DATA_LAYOUT_I_MAJOR || dl == DATA_LAYOUT_I_MAJOR_MIRRORED, "bad data layout");
|
||||
if constexpr (I == 32) {
|
||||
#pragma unroll
|
||||
for (int l0 = 0; l0 < t.ne/2; ++l0) {
|
||||
half2 tmp[2];
|
||||
#pragma unroll
|
||||
for (int o = 0; o < 2; ++o) {
|
||||
const int j = 2*t.get_j(l0) + o;
|
||||
int offset_ij = offset + j*stride + t.get_i(l0)/2;
|
||||
offset_ij = swizzle<stride, T>(offset_ij, j);
|
||||
tmp[o] = xs0[offset_ij];
|
||||
}
|
||||
|
||||
t.x[l0] = __lows2half2(tmp[0], tmp[1]);
|
||||
t.x[l0 + t.ne/2] = __highs2half2(tmp[0], tmp[1]);
|
||||
}
|
||||
} else {
|
||||
half * xh = (half *) t.x;
|
||||
#pragma unroll
|
||||
for (int l = 0; l < t.ne; ++l) {
|
||||
#pragma unroll
|
||||
for (int o = 0; o < 2; ++o) {
|
||||
const int j = 2*t.get_j(l) + o;
|
||||
xh[2*l + o] = ((const half *) xs0)[swizzle<2*stride, half>(2*offset + j*(2*stride) + t.get_i(l), j)];
|
||||
}
|
||||
}
|
||||
}
|
||||
#else
|
||||
GGML_UNUSED_VARS(t, tile_base, i0, j0, stride);
|
||||
GGML_UNUSED_VARS(t, xs0, offset);
|
||||
NO_DEVICE_CODE;
|
||||
#endif // defined(TURING_MMA_AVAILABLE)
|
||||
}
|
||||
|
||||
@@ -37,9 +37,6 @@ static __global__ void mm_ids_helper(
|
||||
const int n_expert_used = n_expert_used_template == 0 ? n_expert_used_var : n_expert_used_template;
|
||||
const int expert = blockIdx.x;
|
||||
|
||||
// token slots per warp lane group, padded to a power of 2 so a warp divides evenly
|
||||
constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
|
||||
|
||||
extern __shared__ char data_mm_ids_helper[];
|
||||
mm_ids_helper_store * store = (mm_ids_helper_store *) data_mm_ids_helper;
|
||||
|
||||
@@ -69,6 +66,7 @@ static __global__ void mm_ids_helper(
|
||||
} else {
|
||||
// Implementation optimized for specific numbers of experts used:
|
||||
// a warp holds a whole number of token slots, so the slot count is padded to a power of 2
|
||||
constexpr int neu_padded = mm_ids_pow2<n_expert_used_template>::value;
|
||||
static_assert(neu_padded <= warp_size && warp_size % neu_padded == 0, "bad n_expert_used");
|
||||
for (int it0 = 0; it0 < n_tokens; it0 += warp_size/neu_padded) {
|
||||
const int it = it0 + threadIdx.x / neu_padded;
|
||||
|
||||
@@ -255,7 +255,7 @@ void ggml_cuda_mul_mat_q(
|
||||
}
|
||||
|
||||
const size_t nbytes_src1_q8_1 = ne12*n_expert_used*ne10_padded * y_block_size/y_values_per_block +
|
||||
ggml_cuda_mmq_get_J_max(src0->type, fallback, cc, ne11) * sizeof(block_q8_1_mmq);
|
||||
ggml_cuda_mmq_get_J_max(src0->type, fallback, cc, ne12) * sizeof(block_q8_1_mmq);
|
||||
ggml_cuda_pool_alloc<char> src1_q8_1(ctx.pool(), nbytes_src1_q8_1);
|
||||
ggml_cuda_pool_alloc<float> src1_scale(ctx.pool());
|
||||
if (src0->type == GGML_TYPE_NVFP4 && use_native_fp4) {
|
||||
|
||||
+40
-10
@@ -601,7 +601,7 @@ __launch_bounds__(calc_nwarps(type, ncols_dst, get_device_table_id(), small_k, h
|
||||
static __global__ void mul_mat_vec_q(
|
||||
const void * vx_ptr, const void * vy_ptr, const int32_t * ids_ptr, const ggml_cuda_mm_fusion_args_device fusion, float * dst_ptr,
|
||||
const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t stride_row_x, const uint32_t stride_col_y,
|
||||
const uint32_t stride_col_dst, const uint3 channel_ratio, const uint32_t stride_channel_x,
|
||||
uint32_t stride_col_dst, const uint3 channel_ratio, const uint32_t stride_channel_x,
|
||||
const uint32_t stride_channel_y, const uint32_t stride_channel_dst, const uint3 sample_ratio,
|
||||
const uint32_t stride_sample_x, const uint32_t stride_sample_y, const uint32_t stride_sample_dst,
|
||||
const uint32_t ids_stride) {
|
||||
@@ -625,14 +625,20 @@ static __global__ void mul_mat_vec_q(
|
||||
const int blocks_per_row_x = ncols_x / qk;
|
||||
constexpr int blocks_per_iter = vdr * nwarps*warp_size / qi;
|
||||
|
||||
const uint32_t channel_dst = blockIdx.y;
|
||||
const bool shared_expert = has_fusion && fusion.shared_up && blockIdx.y == gridDim.y - 1;
|
||||
const uint32_t channel_dst = shared_expert ? 0 : blockIdx.y;
|
||||
if (shared_expert) {
|
||||
vx = fusion.shared_up;
|
||||
dst = fusion.shared_dst;
|
||||
stride_col_dst = fusion.shared_stride_col_dst;
|
||||
}
|
||||
|
||||
uint32_t channel_x;
|
||||
uint32_t channel_y;
|
||||
uint32_t sample_dst;
|
||||
|
||||
ggml_cuda_pdl_sync();
|
||||
channel_x = ncols_dst == 1 && ids ? ids[channel_dst] : fastdiv(channel_dst, channel_ratio);
|
||||
channel_x = shared_expert ? 0 : ncols_dst == 1 && ids ? ids[channel_dst] : fastdiv(channel_dst, channel_ratio);
|
||||
channel_y = ncols_dst == 1 && ids ? fastmodulo(channel_dst, nchannels_y) : channel_dst;
|
||||
sample_dst = blockIdx.z;
|
||||
|
||||
@@ -656,7 +662,7 @@ static __global__ void mul_mat_vec_q(
|
||||
use_gate = fusion.gate != nullptr;
|
||||
use_bias = fusion.x_bias != nullptr;
|
||||
use_gate_bias = fusion.gate_bias != nullptr && use_gate;
|
||||
vgate = fusion.gate;
|
||||
vgate = shared_expert ? fusion.shared_gate : fusion.gate;
|
||||
x_bias = (const float *) fusion.x_bias;
|
||||
gate_bias = (const float *) fusion.gate_bias;
|
||||
active_glu = fusion.glu_op;
|
||||
@@ -854,7 +860,7 @@ static __global__ void mul_mat_vec_q_moe(
|
||||
const void * vx_ptr, const void * vy_ptr, const int32_t * ids_ptr, const ggml_cuda_mm_fusion_args_device fusion,
|
||||
float * dst_ptr,
|
||||
const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t nrows_x,
|
||||
const uint32_t stride_row_x, const uint32_t stride_col_y, const uint32_t stride_col_dst,
|
||||
const uint32_t stride_row_x, const uint32_t stride_col_y, uint32_t stride_col_dst,
|
||||
const uint32_t stride_channel_x, const uint32_t stride_channel_y, const uint32_t stride_channel_dst,
|
||||
const uint32_t ncols_dst, const uint32_t ids_stride) {
|
||||
const void * GGML_CUDA_RESTRICT vx = vx_ptr;
|
||||
@@ -869,6 +875,13 @@ static __global__ void mul_mat_vec_q_moe(
|
||||
|
||||
constexpr vec_dot_q_cuda_t vec_dot_q_cuda = get_vec_dot_q_cuda(type);
|
||||
|
||||
const bool shared_expert = has_fusion && fusion.shared_up && blockIdx.y == gridDim.y - 1;
|
||||
if (shared_expert) {
|
||||
vx = fusion.shared_up;
|
||||
dst = fusion.shared_dst;
|
||||
stride_col_dst = fusion.shared_stride_col_dst;
|
||||
}
|
||||
|
||||
// fuse gate, bias, scales, and glu_op into the up projection
|
||||
bool use_gate = false;
|
||||
const void * vgate = nullptr;
|
||||
@@ -881,7 +894,7 @@ static __global__ void mul_mat_vec_q_moe(
|
||||
|
||||
if constexpr (has_fusion) {
|
||||
use_gate = fusion.gate != nullptr;
|
||||
vgate = fusion.gate;
|
||||
vgate = shared_expert ? fusion.shared_gate : fusion.gate;
|
||||
x_bias = (const float *) fusion.x_bias;
|
||||
gate_bias = (const float *) fusion.gate_bias;
|
||||
active_glu = fusion.glu_op;
|
||||
@@ -897,14 +910,14 @@ static __global__ void mul_mat_vec_q_moe(
|
||||
const int blocks_per_row_x = ncols_x / qk;
|
||||
constexpr int blocks_per_iter = vdr * warp_size / qi;
|
||||
|
||||
const uint32_t channel_dst = blockIdx.y;
|
||||
const uint32_t channel_dst = shared_expert ? 0 : blockIdx.y;
|
||||
|
||||
if (token_idx >= ncols_dst) {
|
||||
return;
|
||||
}
|
||||
|
||||
ggml_cuda_pdl_sync();
|
||||
const uint32_t channel_x = ids[channel_dst + token_idx * ids_stride];
|
||||
const uint32_t channel_x = shared_expert ? 0 : ids[channel_dst + token_idx * ids_stride];
|
||||
const uint32_t channel_y = fastmodulo(channel_dst, nchannels_y);
|
||||
|
||||
const block_q8_1 * y = ((const block_q8_1 *) vy) + channel_y*stride_channel_y + token_idx*stride_col_y;
|
||||
@@ -1050,7 +1063,7 @@ static void mul_mat_vec_q_moe_launch(
|
||||
|
||||
constexpr int rows_per_block = 2; // 2 gives best perf based on tuning
|
||||
const int64_t nblocks_rows = (nrows_x + rows_per_block - 1) / rows_per_block;
|
||||
const dim3 block_nums(nblocks_rows, nchannels_dst);
|
||||
const dim3 block_nums(nblocks_rows, nchannels_dst + (fusion.shared_up != nullptr));
|
||||
const dim3 block_dims(warp_size, ncols_dst);
|
||||
const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, 0, stream);
|
||||
|
||||
@@ -1187,7 +1200,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
|
||||
|
||||
constexpr bool c_halve_iters = decltype(halve_iters_tag)::value && c_promoted;
|
||||
|
||||
const std::pair<dim3, dim3> dims = calc_launch_params<type>(c_ncols_dst, nrows_x, nchannels_dst,
|
||||
const std::pair<dim3, dim3> dims = calc_launch_params<type>(c_ncols_dst, nrows_x, nchannels_dst + (fusion.shared_up != nullptr),
|
||||
nsamples_dst, warp_size, table_id, c_small_k, c_halve_iters);
|
||||
mul_mat_vec_q_switch_fusion<type, c_ncols_dst, c_small_k, c_halve_iters>(
|
||||
vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
|
||||
@@ -1454,6 +1467,23 @@ void ggml_cuda_mul_mat_vec_q(
|
||||
// non-negligible for some models such as gpt-oss-20b
|
||||
GGML_ASSERT((fusion->x_scale == nullptr && fusion->gate_scale == nullptr) || src0->type == GGML_TYPE_NVFP4);
|
||||
|
||||
if (fusion->shared_up) {
|
||||
GGML_ASSERT(ids && fusion->gate && fusion->shared_gate && fusion->shared_dst);
|
||||
GGML_ASSERT(!fusion->x_bias && !fusion->gate_bias && !fusion->x_scale && !fusion->gate_scale);
|
||||
GGML_ASSERT(ne11 == 1 && ne03 == 1 && ne13 == 1);
|
||||
GGML_ASSERT(fusion->shared_up->type == src0->type && fusion->shared_gate->type == src0->type);
|
||||
GGML_ASSERT(ggml_are_same_shape(fusion->shared_up, fusion->shared_gate));
|
||||
GGML_ASSERT(ggml_is_contiguous(fusion->shared_up) && ggml_is_contiguous(fusion->shared_gate));
|
||||
GGML_ASSERT(fusion->shared_up->ne[0] == ne00 && fusion->shared_up->ne[1] == ne01);
|
||||
GGML_ASSERT(fusion->shared_up->nb[1] == nb01 && ggml_is_matrix(fusion->shared_up));
|
||||
GGML_ASSERT(fusion->shared_dst->type == GGML_TYPE_F32 && ggml_is_contiguous(fusion->shared_dst));
|
||||
GGML_ASSERT(fusion->shared_dst->ne[0] == ne0 && fusion->shared_dst->ne[1] == ne2);
|
||||
fusion_local.shared_up = fusion->shared_up->data;
|
||||
fusion_local.shared_gate = fusion->shared_gate->data;
|
||||
fusion_local.shared_dst = (float *) fusion->shared_dst->data;
|
||||
fusion_local.shared_stride_col_dst = fusion->shared_dst->nb[1] / ts_dst;
|
||||
}
|
||||
|
||||
if (fusion->x_bias) {
|
||||
GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]);
|
||||
|
||||
@@ -131,8 +131,6 @@ static __global__ void quantize_mmq_nvfp4(
|
||||
const int64_t ne0, const int64_t ne1, const int64_t ne2, const int n_expert_used) {
|
||||
#if defined(BLACKWELL_MMA_AVAILABLE)
|
||||
|
||||
const int64_t blocks_per_col = (ne0 + QK_FP4_MMQ - 1) / QK_FP4_MMQ;
|
||||
|
||||
int64_t base_idx;
|
||||
if constexpr (scatter) {
|
||||
base_idx = (int64_t) blockIdx.x * s02; // one physical row per token
|
||||
@@ -317,7 +315,8 @@ static __global__ void quantize_mmq_nvfp4(
|
||||
reinterpret_cast<uint8_t *>(yb->d4)[sub] = fp8_code;
|
||||
}
|
||||
} else {
|
||||
block_fp4_mmq * yb = y + (blockIdx.y * ((int64_t) blocks_per_col * ne1) + k_block * ne1 + blockIdx.x);
|
||||
const int64_t blocks_per_col = (ne0 + QK_FP4_MMQ - 1) / QK_FP4_MMQ;
|
||||
block_fp4_mmq * yb = y + (blockIdx.y * (blocks_per_col * ne1) + k_block * ne1 + blockIdx.x);
|
||||
uint32_t * yqs = reinterpret_cast<uint32_t *>(yb->qs);
|
||||
yqs[2 * sub + 0] = q0;
|
||||
yqs[2 * sub + 1] = q1;
|
||||
|
||||
@@ -484,13 +484,23 @@ ggml_metal_pipeline_with_params ggml_metal_library_get_pipeline_lightning_indexe
|
||||
const ggml_tensor * op) {
|
||||
GGML_ASSERT(op->op == GGML_OP_LIGHTNING_INDEXER);
|
||||
|
||||
char base[256];
|
||||
char name[256];
|
||||
|
||||
snprintf(name, 256, "kernel_lightning_indexer_%s", ggml_type_name(op->src[1]->type));
|
||||
const int16_t nh = op->src[0]->ne[1];
|
||||
|
||||
snprintf(base, 256, "kernel_lightning_indexer_%s", ggml_type_name(op->src[1]->type));
|
||||
snprintf(name, 256, "%s_nh=%d", base, nh);
|
||||
|
||||
ggml_metal_pipeline_with_params res = ggml_metal_library_get_pipeline(lib, name);
|
||||
if (!res.pipeline) {
|
||||
res = ggml_metal_library_compile_pipeline(lib, name, name, nullptr);
|
||||
ggml_metal_cv_t cv = ggml_metal_cv_init();
|
||||
|
||||
ggml_metal_cv_set_int16(cv, nh, FC_LIGHTNING_INDEXER + 0);
|
||||
|
||||
res = ggml_metal_library_compile_pipeline(lib, base, name, cv);
|
||||
|
||||
ggml_metal_cv_free(cv);
|
||||
}
|
||||
|
||||
return res;
|
||||
|
||||
@@ -1770,8 +1770,7 @@ bool ggml_metal_device_supports_op(ggml_metal_device_t dev, const struct ggml_te
|
||||
}
|
||||
return has_simdgroup_mm; // TODO: over-restricted for vec-kernels
|
||||
case GGML_OP_LIGHTNING_INDEXER:
|
||||
if (op->src[0]->ne[0] != OP_LIGHTNING_INDEXER_DK ||
|
||||
op->src[0]->ne[1] != OP_LIGHTNING_INDEXER_NH) {
|
||||
if (op->src[0]->ne[0] != OP_LIGHTNING_INDEXER_DK) {
|
||||
return false;
|
||||
}
|
||||
if (!has_simdgroup_mm ||
|
||||
|
||||
@@ -122,6 +122,7 @@
|
||||
#define FC_DSV4_HC 2000
|
||||
#define FC_PAD 2100
|
||||
#define FC_FLASH_ATTN_EXT_TENSOR 2200
|
||||
#define FC_LIGHTNING_INDEXER 2200
|
||||
|
||||
// op-specific constants
|
||||
#define OP_FLASH_ATTN_EXT_NQPSG 8
|
||||
@@ -136,7 +137,6 @@
|
||||
#define OP_FLASH_ATTN_EXT_VEC_NCPSG 32
|
||||
|
||||
#define OP_LIGHTNING_INDEXER_DK 128
|
||||
#define OP_LIGHTNING_INDEXER_NH 64
|
||||
#define OP_LIGHTNING_INDEXER_NHPTG 8
|
||||
#define OP_LIGHTNING_INDEXER_NKPSG 8
|
||||
#define OP_LIGHTNING_INDEXER_NSG 8
|
||||
|
||||
@@ -1372,7 +1372,6 @@ int ggml_metal_op_lightning_indexer(ggml_metal_op_t ctx, int idx) {
|
||||
GGML_ASSERT(op->type == GGML_TYPE_F32);
|
||||
|
||||
GGML_ASSERT(q->ne[0] == OP_LIGHTNING_INDEXER_DK);
|
||||
GGML_ASSERT(q->ne[1] == OP_LIGHTNING_INDEXER_NH);
|
||||
|
||||
ggml_metal_kargs_lightning_indexer args = {
|
||||
/*.n_kv =*/ (int32_t) k->ne[2],
|
||||
|
||||
@@ -329,6 +329,8 @@ kernel void kernel_flash_attn_ext_vec_reduce(
|
||||
#undef DV
|
||||
}
|
||||
|
||||
constant short FC_lightning_indexer_nh [[function_constant(FC_LIGHTNING_INDEXER + 0)]];
|
||||
|
||||
template<
|
||||
typename kd4x4_t,
|
||||
short nl_k,
|
||||
@@ -345,7 +347,7 @@ kernel void kernel_lightning_indexer(
|
||||
ushort tiisg[[thread_index_in_simdgroup]],
|
||||
ushort sgitg[[simdgroup_index_in_threadgroup]]) {
|
||||
constexpr short DK = OP_LIGHTNING_INDEXER_DK;
|
||||
constexpr short NH = OP_LIGHTNING_INDEXER_NH;
|
||||
const short NH = FC_lightning_indexer_nh;
|
||||
constexpr short NHPTG = OP_LIGHTNING_INDEXER_NHPTG;
|
||||
constexpr short NKPSG = OP_LIGHTNING_INDEXER_NKPSG;
|
||||
constexpr short NSG = OP_LIGHTNING_INDEXER_NSG;
|
||||
@@ -411,18 +413,22 @@ kernel void kernel_lightning_indexer(
|
||||
float score = 0.0f;
|
||||
|
||||
FOR_UNROLL (short i_head = 0; i_head < NH; i_head += NHPTG) {
|
||||
// stage the Q tile [DK, NHPTG] and the (prescaled) head weights
|
||||
// stage the Q tile [DK, NHPTG] and the (prescaled) head weights, heads past NH are zero
|
||||
for (short i = tiitg; i < NHPTG*DK4; i += NTG) {
|
||||
const short ih = i/DK4;
|
||||
const short i4 = i%DK4;
|
||||
|
||||
device const float4 * q4 = (device const float4 *) (pq + (i_head + ih)*args.nbq1);
|
||||
if (i_head + ih < NH) {
|
||||
device const float4 * q4 = (device const float4 *) (pq + (i_head + ih)*args.nbq1);
|
||||
|
||||
sq4[ih*DK4 + i4] = half4(q4[i4]);
|
||||
sq4[ih*DK4 + i4] = half4(q4[i4]);
|
||||
} else {
|
||||
sq4[ih*DK4 + i4] = half4(0.0h);
|
||||
}
|
||||
}
|
||||
|
||||
if (tiitg < NHPTG) {
|
||||
sw[tiitg] = ((device const float *) pw)[i_head + tiitg];
|
||||
sw[tiitg] = i_head + tiitg < NH ? ((device const float *) pw)[i_head + tiitg] : 0.0f;
|
||||
}
|
||||
|
||||
threadgroup_barrier(mem_flags::mem_threadgroup);
|
||||
|
||||
@@ -13,6 +13,17 @@ ggml_add_backend_library(ggml-openvino
|
||||
|
||||
target_link_libraries(ggml-openvino PRIVATE openvino::runtime openvino::threading OpenCL::OpenCL)
|
||||
|
||||
# the OpenVINO RTTI macros take one argument and leave __VA_ARGS__ empty, which -Wpedantic reports
|
||||
if (CMAKE_CXX_COMPILER_ID MATCHES "Clang" OR CMAKE_CXX_COMPILER_ID STREQUAL "IntelLLVM")
|
||||
target_compile_options(ggml-openvino PRIVATE -Wno-gnu-zero-variadic-macro-arguments)
|
||||
elseif (CMAKE_CXX_COMPILER_ID STREQUAL "GNU")
|
||||
target_compile_options(ggml-openvino PRIVATE -Wno-pedantic)
|
||||
endif()
|
||||
|
||||
if (WIN32)
|
||||
target_link_libraries(ggml-openvino PRIVATE psapi)
|
||||
endif()
|
||||
|
||||
if (GGML_OPENVINO)
|
||||
if (CMAKE_SYSTEM_PROCESSOR STREQUAL "aarch64")
|
||||
elseif (CMAKE_SYSTEM_PROCESSOR STREQUAL "x86_64" OR CMAKE_SYSTEM_PROCESSOR STREQUAL "amd64" OR CMAKE_SYSTEM_PROCESSOR STREQUAL "AMD64")
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
# Compiled model cache
|
||||
|
||||
`GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR` exports compiled CPU/GPU graphs with their weights. It bypasses the plugin-level `GGML_OPENVINO_CACHE_DIR` and uses `OPTIMIZE_SPEED`, so weightless caching is disabled.
|
||||
|
||||
One directory can hold blobs for different models and compilation settings. Run each intended workload once to export its dynamic graph:
|
||||
|
||||
```sh
|
||||
GGML_OPENVINO_DEVICE=GPU \
|
||||
GGML_OPENVINO_NATIVE_SOFTPLUS=1 \
|
||||
GGML_OPENVINO_DISABLE_KV_SLICE=1 \
|
||||
GGML_OPENVINO_REQUANT_KQUANT=q4_asym64_all \
|
||||
GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR=/path/to/qwen-cache \
|
||||
./build/ReleaseOV/bin/llama-bench -m /path/to/model.gguf -r 1
|
||||
```
|
||||
|
||||
`GGML_OPENVINO_SPILL_DIR` remains optional for this first run. Wait for the `model cache WROTE` message and completion of the workload before stopping it. Compatible prefill and decode graphs share one blob and manifest. A graph with different ports or incompatible shapes gets an exact entry instead; interrupted exports are not cache hits.
|
||||
|
||||
On later runs, supply the same compilation settings and enable `GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY=1`:
|
||||
|
||||
```sh
|
||||
GGML_OPENVINO_DEVICE=GPU \
|
||||
GGML_OPENVINO_NATIVE_SOFTPLUS=1 \
|
||||
GGML_OPENVINO_DISABLE_KV_SLICE=1 \
|
||||
GGML_OPENVINO_REQUANT_KQUANT=q4_asym64_all \
|
||||
GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR=/path/to/qwen-cache \
|
||||
GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY=1 \
|
||||
./build/ReleaseOV/bin/llama-bench -m /path/to/model.gguf -r 1
|
||||
```
|
||||
|
||||
Cache-only mode allocates backend address space without filling weight pages. On Windows, this also uses system commit capacity. The model-buffer size in the loader log is this virtual size. Weight uploads only record source identity; they do not read or requantize the weights. Graph conversion and compilation are skipped. Runtime buffers are still allocated and populated normally.
|
||||
|
||||
Cache-only mode uses the settings provided by the current process. Keep these values exactly the same, including set versus unset: `GGML_OPENVINO_REQUANT_KQUANT`, `GGML_OPENVINO_NATIVE_SOFTPLUS`, `GGML_OPENVINO_DISABLE_KV_SLICE`, `GGML_OPENVINO_MANUAL_GQA_ATTN`, `GGML_OPENVINO_STATEFUL_EXECUTION`, `GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT`, `GGML_OPENVINO_DISABLE_REMOTE_OUTPUTS`, `GGML_OPENVINO_REDUCE_COMPILE_MEM`, `GGML_OPENVINO_MEMORY_OPTIMIZE`, `GGML_OPENVINO_PROFILING`, and `GGML_OPENVINO_DEBUG_NODE`. On GPU, also repeat `GGML_OPENVINO_MOE_OP=0` if used. `GGML_OPENVINO_SPILL_DIR` is optional on the first run and ignored in cache-only mode; host-weight release is disabled in cache-only mode.
|
||||
|
||||
A missing or incompatible graph fails with an error instead of compiling with absent weights. The fingerprint uses the dynamic graph's topology, ports, model parameters, weights, settings, and OpenVINO version; changing only dynamic token or KV sizes does not require a new entry. A different workload can still require another graph; populate it first without cache-only mode.
|
||||
|
||||
## Restrictions
|
||||
|
||||
- Cache-only mode requires Linux or Windows and mmap loading (`--load-mode mmap`, or the default when all selected devices support mmap). Do not use tensor validation or mlock when trying to avoid weight reads.
|
||||
- On Windows, the backend commits virtual memory for its buffers without touching weight pages. Large models can still reach the system commit limit.
|
||||
- The model must execute entirely on OpenVINO, with dynamic CPU/GPU graphs and in-process caching enabled. Static/NPU execution and CPU fallback are unsupported in cache-only mode.
|
||||
- GGUF metadata, tokenizer data, tensor descriptors, and graph construction are still needed. The llama.cpp loader is unchanged: depending on its prefetch settings, it may request pages with `MAP_POPULATE` or read-ahead on Linux, or `PrefetchVirtualMemory` on Windows. Non-mmap loading also reads the payload before the backend sees it.
|
||||
- File identity, size, modification/change timestamps, tensor offsets, graph structure, settings, and OpenVINO version identify cache entries. Linux uses device/inode and Windows uses volume serial/file index. Replacing, copying, or modifying a GGUF invalidates its entries. This avoids reading weight bytes and ties the cache to the local source files. Keep those files unchanged throughout loading and inference.
|
||||
- Use the same target device and compatible OpenVINO/plugin installation. Import support depends on the plugin; the tested CPU plugin cannot import MoE graphs containing `GatherMatmulCompressed`. GPU MoE and CPU dense graph imports were tested.
|
||||
- Blobs contain weights and can approach model size for each compiled graph. Import still reads those blobs and initializes the device.
|
||||
@@ -87,6 +87,14 @@ void GgmlOvDecoder::update_io(ggml_cgraph * cgraph) {
|
||||
compute_model_outputs();
|
||||
}
|
||||
|
||||
// llama keeps separate graphs for batches with and without outputs, so a cache hit can come from a
|
||||
// graph built in other memory. The decoder then still points at the old graph's tensors.
|
||||
bool GgmlOvDecoder::is_bound_to(const ggml_cgraph * cgraph) const {
|
||||
return m_cgraph == cgraph && cgraph->n_nodes > 0 && m_node_info_list.size() == (size_t) cgraph->n_nodes &&
|
||||
m_node_info_list.front().node == cgraph->nodes[0] &&
|
||||
m_node_info_list.back().node == cgraph->nodes[cgraph->n_nodes - 1];
|
||||
}
|
||||
|
||||
GgmlOvDecoder::GgmlOvDecoder(ggml_cgraph * cgraph, std::map<std::string, std::shared_ptr<ov::Node>> & model_weights) {
|
||||
m_cgraph = cgraph;
|
||||
m_model_weights = model_weights;
|
||||
@@ -117,6 +125,12 @@ bool is_same_shape(const ggml_tensor * a, const ggml_tensor * b) {
|
||||
bool is_conv_states_all_tensor(const ggml_tensor * tensor) {
|
||||
return tensor != nullptr && strncmp(tensor->name, "conv_states_all", strlen("conv_states_all")) == 0;
|
||||
}
|
||||
|
||||
bool is_full_single_slot_writeback(const ggml_tensor * node) {
|
||||
return node->view_src != nullptr && node->view_src->ne[1] == 1 && node->src[1] != nullptr &&
|
||||
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src &&
|
||||
node->src[1]->view_offs == 0 && ggml_nbytes(node->src[1]) == ggml_nbytes(node->view_src);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
// MoE expert aggregation (build_moe_ffn in llama-graph.cpp): each expert plane is
|
||||
@@ -274,8 +288,34 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
int op_case = 0;
|
||||
switch (node->op) {
|
||||
case GGML_OP_RESHAPE: {
|
||||
if (m_naive) {
|
||||
break;
|
||||
}
|
||||
auto name = std::string(node->name);
|
||||
auto * src = node->src[0];
|
||||
// Identify recurrent sequence reshapes before size checks, which are ambiguous for one token.
|
||||
bool recurrent_sequence = false;
|
||||
for (int i = 0; i < m_cgraph->n_nodes && !recurrent_sequence; ++i) {
|
||||
const auto * consumer = m_cgraph->nodes[i];
|
||||
if (consumer->op == GGML_OP_MUL_MAT_ID && consumer->src[1] == node) {
|
||||
return 1;
|
||||
} else if (consumer->op == GGML_OP_SSM_CONV) {
|
||||
const auto * concat = consumer->src[0];
|
||||
if (concat->op == GGML_OP_CONCAT) {
|
||||
const auto * transposed = concat->src[1];
|
||||
recurrent_sequence = transposed->op == GGML_OP_TRANSPOSE && transposed->src[0] == node;
|
||||
}
|
||||
} else if (consumer->op == GGML_OP_UNARY && ggml_get_unary_op(consumer) == GGML_UNARY_OP_SOFTPLUS) {
|
||||
const auto * biased = consumer->src[0];
|
||||
recurrent_sequence = biased->op == GGML_OP_ADD && biased->src[0] == node;
|
||||
}
|
||||
}
|
||||
if (recurrent_sequence && node->ne[0] == src->ne[0] && node->ne[3] == 1) {
|
||||
return 6;
|
||||
}
|
||||
if (node->ne[0] == src->ne[0] && node->ne[2] == 1 && node->ne[3] == 1) {
|
||||
return 5;
|
||||
}
|
||||
if (src->op == GGML_OP_RESHAPE && src->src[0]->ne[0] == node->ne[0] && src->src[0]->ne[1] == node->ne[1]) {
|
||||
op_case = 4;
|
||||
} else if (node->ne[0] * node->ne[1] == src->ne[0]) {
|
||||
@@ -285,7 +325,7 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
if (src->ne[2] * src->ne[3] == node->ne[1]) {
|
||||
op_case = 5;
|
||||
}
|
||||
} else if (src->ne[0] * src->ne[1] * src->ne[2] == node->ne[1]) {
|
||||
} else if (node->ne[0] == 1 && src->ne[0] * src->ne[1] * src->ne[2] == node->ne[1]) {
|
||||
op_case = 3;
|
||||
} else if (name.find("linear_attn_qkv_mixed") == 0 || name.find("alpha") == 0) {
|
||||
op_case = 6;
|
||||
@@ -294,6 +334,38 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
} else if (name.find("state_predelta") == 0) {
|
||||
op_case = 8;
|
||||
}
|
||||
if (op_case == 1 && m_is_stateful) {
|
||||
// Recurrent convolution and GDN gates retain their rank-4 layout.
|
||||
bool recurrent = src->op == GGML_OP_GET_ROWS && is_recurrent_cache(src->src[0]);
|
||||
for (int i = 0; i < m_cgraph->n_nodes && !recurrent; ++i) {
|
||||
const auto * consumer = m_cgraph->nodes[i];
|
||||
if (consumer->op == GGML_OP_GATED_DELTA_NET) {
|
||||
for (int j : {3, 4}) {
|
||||
const auto * gate = consumer->src[j];
|
||||
if (gate->op == GGML_OP_UNARY) {
|
||||
gate = gate->src[0];
|
||||
}
|
||||
recurrent = recurrent || gate == node;
|
||||
}
|
||||
} else if (consumer->op == GGML_OP_MUL) {
|
||||
for (int j = 0; j < 2; ++j) {
|
||||
const auto * gate = consumer->src[j];
|
||||
const auto * norm = consumer->src[1 - j];
|
||||
if (gate->op != GGML_OP_UNARY || gate->src[0] != node) {
|
||||
continue;
|
||||
}
|
||||
if (norm->op == GGML_OP_MUL) {
|
||||
norm = norm->src[0];
|
||||
}
|
||||
recurrent = recurrent || (norm->op == GGML_OP_RMS_NORM && norm->src[0]->op == GGML_OP_VIEW &&
|
||||
norm->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (recurrent) {
|
||||
op_case = 9;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_PERMUTE: {
|
||||
@@ -342,11 +414,12 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
if (node->src[1]->op == GGML_OP_VIEW) {
|
||||
// GET_ROWS gathering recurrent state cache rows via the inp->s_copy index list:
|
||||
// src[0] is a reshape of cache_r/cache_s, src[1] is a view of the s_copy leaf.
|
||||
// op_case 3: main view (active sequences, view offset 0)
|
||||
// op_case 4: extra view (defrag remainder, nonzero view offset)
|
||||
// op_case 1/2: active/extra rows of a multi-slot cache
|
||||
// op_case 3/4: active/extra rows of a single-slot cache
|
||||
if (node->src[0]->op == GGML_OP_RESHAPE && node->src[0]->src[0] != nullptr &&
|
||||
is_kvcache(node->src[0]->src[0], nullptr)) {
|
||||
op_case = node->src[1]->view_offs == 0 ? 1 : 2;
|
||||
is_recurrent_cache(node->src[0]->src[0])) {
|
||||
const bool single_slot = node->src[0]->src[0]->ne[1] == 1;
|
||||
op_case = (node->src[1]->view_offs == 0 ? 1 : 2) + (single_slot ? 2 : 0);
|
||||
}
|
||||
}
|
||||
break;
|
||||
@@ -362,6 +435,14 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
op_case = 2;
|
||||
break;
|
||||
}
|
||||
case GGML_ROPE_TYPE_VISION: {
|
||||
op_case = 3;
|
||||
break;
|
||||
}
|
||||
case GGML_ROPE_TYPE_MROPE: {
|
||||
op_case = 4;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
op_case = 0;
|
||||
break;
|
||||
@@ -369,6 +450,12 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
break;
|
||||
}
|
||||
case GGML_OP_VIEW: {
|
||||
if (!m_model_params.has_rs_rollback && node->src[0] != nullptr &&
|
||||
node->src[0]->op == GGML_OP_GATED_DELTA_NET) {
|
||||
// The GDN translator publishes native attention/state outputs under these VIEW names.
|
||||
op_case = 2;
|
||||
break;
|
||||
}
|
||||
if (m_is_static && node->src[0] != nullptr &&
|
||||
(node->src[0]->op == GGML_OP_GATED_DELTA_NET || node->src[0]->op == GGML_OP_CONCAT)) {
|
||||
// VIEW slicing a GATED_DELTA_NET combined [attn|state] output, or the conv_input
|
||||
@@ -426,6 +513,10 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
if (node->src[0]->op == GGML_OP_VIEW) {
|
||||
if (is_same_shape(node->src[0]->src[0], node->src[0])) {
|
||||
op_case = 1;
|
||||
} else if (!m_model_params.has_rs_rollback &&
|
||||
node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) {
|
||||
// GDN attention is routed directly to this VIEW by get_output_names().
|
||||
op_case = 3;
|
||||
} else if (node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) {
|
||||
op_case = 2;
|
||||
}
|
||||
@@ -449,12 +540,40 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_UPSCALE: {
|
||||
const int32_t mode_flags = node->op_params[0];
|
||||
const ggml_scale_mode scale_mode = static_cast<ggml_scale_mode>(mode_flags & 0xFF);
|
||||
switch (scale_mode) {
|
||||
case GGML_SCALE_MODE_NEAREST: {
|
||||
op_case = 1;
|
||||
break;
|
||||
}
|
||||
case GGML_SCALE_MODE_BILINEAR: {
|
||||
op_case = 2;
|
||||
break;
|
||||
}
|
||||
case GGML_SCALE_MODE_BICUBIC: {
|
||||
op_case = 3;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
op_case = 0;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_CPY: {
|
||||
if (node->src[0]->op == GGML_OP_VIEW) {
|
||||
if (node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET) {
|
||||
op_case = 1;
|
||||
if (!m_model_params.has_rs_rollback) {
|
||||
// op_case 7 replaces a single-slot cache; op_case 10 writes native GDN state
|
||||
// into an active range of a larger non-rollback cache.
|
||||
op_case = is_full_single_slot_writeback(node) ? 7 : 10;
|
||||
} else {
|
||||
op_case = 1;
|
||||
}
|
||||
} else if (GgmlOvDecoder::is_conv_state_writeback(node)) {
|
||||
op_case = 2;
|
||||
op_case = is_full_single_slot_writeback(node) ? 8 : 2;
|
||||
break;
|
||||
} else if (is_conv_states_all_tensor(node->view_src) && node->src[1] != nullptr &&
|
||||
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src) {
|
||||
@@ -463,9 +582,9 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
}
|
||||
} else if (node->src[0]->op == GGML_OP_GET_ROWS && node->src[1] != nullptr &&
|
||||
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr &&
|
||||
is_kvcache(node->src[1]->view_src, nullptr)) {
|
||||
is_recurrent_cache(node->src[1]->view_src)) {
|
||||
// s_copy defrag remainder writeback: gathered extra state rows copied back into the cache
|
||||
op_case = 3;
|
||||
op_case = node->src[1]->view_src->ne[1] == 1 ? 9 : 3;
|
||||
} else if (node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src != nullptr) {
|
||||
// op_case 5: KV write for decoder self-attention (dynamic write offset)
|
||||
// op_case 6: KV write for encoder self-attn or cross-attn (static offset)
|
||||
@@ -504,7 +623,7 @@ int GgmlOvDecoder::compute_op_case(const ggml_tensor * node) const {
|
||||
}
|
||||
case GGML_OP_SCALE: {
|
||||
if (node->view_src && node->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY) {
|
||||
op_case = 1;
|
||||
op_case = node->view_src->ne[1] == 1 ? 2 : 1;
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -858,35 +977,48 @@ std::pair<ModelParams, ComputeParams> GgmlOvDecoder::compute_llm_params(ggml_cgr
|
||||
if (node->op == GGML_OP_GATED_DELTA_NET) {
|
||||
model_params.state_size = node->src[0]->ne[0];
|
||||
}
|
||||
if (node->op == GGML_OP_SCALE && node->view_src != nullptr && is_kvcache(node->view_src, nullptr)) {
|
||||
if (node->op == GGML_OP_SCALE && node->view_src != nullptr && is_recurrent_cache(node->view_src)) {
|
||||
if (model_params.n_rs_slots == -1) {
|
||||
model_params.n_rs_slots = node->view_src->ne[1];
|
||||
} else {
|
||||
GGML_ASSERT(model_params.n_rs_slots == node->view_src->ne[1]);
|
||||
}
|
||||
compute_params.cache_rs_reset_len = ggml_nelements(node) / node->view_src->ne[0];
|
||||
compute_params.cache_rs_reset_idx = node->src[0]->view_offs / node->view_src->ne[0];
|
||||
}
|
||||
// Capture the destination slot block of every recurrent state cache writeback, plus the
|
||||
// conv_input window the conv state writeback copies. The active sequences occupy a
|
||||
// contiguous slot block [begin, begin + n_seqs) of the cache; the block and the window move
|
||||
// source window needed by conv state and packed GDN rollback writes. The active sequences
|
||||
// occupy a contiguous slot block [begin, begin + n_seqs) of the cache; these offsets move
|
||||
// with the batch, so they are fed to the cached model as runtime inputs.
|
||||
if (node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) &&
|
||||
if (node->op == GGML_OP_CPY && node->view_src != nullptr && is_recurrent_cache(node->view_src) &&
|
||||
node->src[1] != nullptr && node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src) {
|
||||
const bool is_conv = is_conv_state_writeback(node);
|
||||
const bool is_gdn = node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr &&
|
||||
node->src[0]->src[0]->op == GGML_OP_GATED_DELTA_NET;
|
||||
const bool is_extra = node->src[0]->op == GGML_OP_GET_ROWS;
|
||||
const bool is_gdn_rollback = is_gdn && is_same_shape(node->src[0], node->src[1]);
|
||||
|
||||
const ggml_tensor * dest_view = node->src[1];
|
||||
const ggml_tensor * cache = node->view_src;
|
||||
const size_t row_bytes = cache->ne[0] * ggml_type_size(cache->type);
|
||||
if (row_bytes > 0 && (is_conv || is_gdn || is_extra)) {
|
||||
if (is_gdn_rollback) {
|
||||
// Rollback GDN exposes an already-flattened [state, seq, snapshot] VIEW and copies
|
||||
// it to an identically-shaped cache VIEW. Non-rollback copies native 4-D state
|
||||
// [value, key, head, seq] into flattened cache rows, so the shapes differ. This
|
||||
// signature is local to the CPY and still works when fallback splits the graph.
|
||||
model_params.has_rs_rollback = true;
|
||||
}
|
||||
if (row_bytes > 0 && (is_conv || is_gdn || is_extra) && !is_full_single_slot_writeback(node)) {
|
||||
ComputeParams::RsWriteback writeback;
|
||||
writeback.slot_begin = (int) (dest_view->view_offs / row_bytes);
|
||||
if (is_conv) {
|
||||
writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[0]);
|
||||
} else if (is_gdn) {
|
||||
} else if (is_gdn_rollback) {
|
||||
writeback.src_begin = (int) (node->src[0]->view_offs / node->src[0]->view_src->nb[1]);
|
||||
}
|
||||
compute_params.rs_writebacks[get_tensor_ov_name(cgraph, node)] = writeback;
|
||||
}
|
||||
if (is_conv || is_gdn) {
|
||||
if ((is_conv || is_gdn) && !is_full_single_slot_writeback(node)) {
|
||||
compute_params.s_copy_active_slot_len = (int) dest_view->ne[1];
|
||||
}
|
||||
}
|
||||
@@ -975,6 +1107,12 @@ ov::PartialShape GgmlOvDecoder::get_graph_input_shape(const ggml_tensor * op,
|
||||
input_shape = ov::PartialShape{-1, 1, -1, -1};
|
||||
}
|
||||
|
||||
} else if (is_recurrent_cache(input)) {
|
||||
input_shape = ov::PartialShape{get_shape(input)};
|
||||
if (!m_is_static && !m_is_stateful && input->ne[1] > 1) {
|
||||
input_shape[2] = -1;
|
||||
}
|
||||
|
||||
} else if (is_kvcache(input, op)) {
|
||||
// kvcache
|
||||
input_shape = ov::PartialShape{get_shape(input)};
|
||||
@@ -1057,7 +1195,7 @@ bool GgmlOvDecoder::is_s_copy_leaf(const ggml_tensor * tensor) const {
|
||||
while (data != nullptr && (data->op == GGML_OP_VIEW || data->op == GGML_OP_RESHAPE)) {
|
||||
data = data->src[0];
|
||||
}
|
||||
if (data != nullptr && is_kvcache(data, nullptr)) {
|
||||
if (data != nullptr && is_recurrent_cache(data)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -1095,7 +1233,7 @@ void GgmlOvDecoder::add_extra_inputs() {
|
||||
}
|
||||
// create_1d_input("token_len", m_compute_params.token_len_per_seq * m_compute_params.n_seq_active);
|
||||
|
||||
if (m_compute_params.cache_rs_reset_idx != -1) {
|
||||
if (m_compute_params.cache_rs_reset_idx != -1 && m_model_params.n_rs_slots != 1) {
|
||||
// Whether/which cache slot to reset varies per compute call (e.g. a new sequence starting
|
||||
// vs. continued decoding). can_reuse_statically() does not invalidate the cached static
|
||||
// model on ComputeParams changes, so these must stay runtime Parameters even when static
|
||||
@@ -1119,7 +1257,7 @@ void GgmlOvDecoder::add_extra_inputs() {
|
||||
|
||||
for (const auto & [node_name, writeback] : m_compute_params.rs_writebacks) {
|
||||
create_1d_input("rs_slot_begin_" + node_name, writeback.slot_begin);
|
||||
if (!m_is_static) {
|
||||
if (!m_is_static && writeback.src_begin >= 0) {
|
||||
create_1d_input("rs_src_begin_" + node_name, writeback.src_begin);
|
||||
}
|
||||
}
|
||||
@@ -1216,6 +1354,9 @@ void GgmlOvDecoder::compute_model_outputs() {
|
||||
if (cur_node->op == GGML_OP_NONE || cur_node->op == GGML_OP_VIEW || cur_node->op == GGML_OP_RESHAPE) {
|
||||
continue;
|
||||
}
|
||||
if (::is_inplace_op(cur_node) && ggml_nbytes(cur_node) == 0) {
|
||||
continue;
|
||||
}
|
||||
auto cur_node_use_count = m_cgraph->use_counts[ggml_hash_find(&m_cgraph->visited_hash_set, cur_node)];
|
||||
if (cur_node_use_count == 0) {
|
||||
// The output of in-place ops is the view_src tensor, which is updated in place. We should use the view_src name as the output name to make sure it can be correctly matched with the later ops that use the view_src.
|
||||
@@ -1822,6 +1963,27 @@ std::vector<size_t> GgmlOvDecoder::get_output_stride(int node_idx) const {
|
||||
}
|
||||
|
||||
std::vector<std::string> GgmlOvDecoder::get_output_names(int node_idx) const {
|
||||
auto * node = m_node_info_list[node_idx].node;
|
||||
if (node->op == GGML_OP_GATED_DELTA_NET && !m_model_params.has_rs_rollback) {
|
||||
std::string attn_name;
|
||||
std::string state_name;
|
||||
for (int i = node_idx + 1; i < m_cgraph->n_nodes; i++) {
|
||||
auto * consumer = m_cgraph->nodes[i];
|
||||
if (consumer->op != GGML_OP_VIEW || consumer->src[0] != node) {
|
||||
continue;
|
||||
}
|
||||
// GGML packs [attention | state]. The attention VIEW starts at offset 0 and the
|
||||
// state VIEW starts after the token-dependent attention segment.
|
||||
auto & name = consumer->view_offs == 0 ? attn_name : state_name;
|
||||
if (!name.empty()) {
|
||||
return {m_node_info_list[node_idx].node_name};
|
||||
}
|
||||
name = get_tensor_ov_name(m_cgraph, consumer);
|
||||
}
|
||||
if (!attn_name.empty() && !state_name.empty()) {
|
||||
return {attn_name, state_name};
|
||||
}
|
||||
}
|
||||
return {m_node_info_list[node_idx].node_name};
|
||||
}
|
||||
|
||||
@@ -2154,6 +2316,11 @@ void GgmlOvDecoder::compute_node_dynamic_dims() {
|
||||
case GGML_OP_DIV:
|
||||
case GGML_OP_CLAMP:
|
||||
case GGML_OP_PAD:
|
||||
case GGML_OP_UPSCALE:
|
||||
case GGML_OP_SIN:
|
||||
case GGML_OP_COS:
|
||||
case GGML_OP_LOG:
|
||||
case GGML_OP_ROLL:
|
||||
m_node_dynamic_dims[node] = m_node_dynamic_dims[node->src[0]];
|
||||
break;
|
||||
case GGML_OP_SUM_ROWS:
|
||||
@@ -2168,6 +2335,8 @@ void GgmlOvDecoder::compute_node_dynamic_dims() {
|
||||
break;
|
||||
case GGML_OP_CPY:
|
||||
case GGML_OP_SET_ROWS:
|
||||
case GGML_OP_SUM:
|
||||
case GGML_OP_MEAN:
|
||||
m_node_dynamic_dims[node] = -1;
|
||||
break;
|
||||
case GGML_OP_IM2COL: {
|
||||
@@ -2198,6 +2367,25 @@ void GgmlOvDecoder::compute_node_dynamic_dims() {
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_IM2COL_3D: {
|
||||
m_node_dynamic_dims[node] = -1;
|
||||
if (m_node_dynamic_dims[node->src[1]] != -1) {
|
||||
const int src_dyn = m_node_dynamic_dims[node->src[1]];
|
||||
if (src_dyn == 0) {
|
||||
m_node_dynamic_dims[node] = 1; // IW -> OW
|
||||
} else if (src_dyn == 1) {
|
||||
m_node_dynamic_dims[node] = 2; // IH -> OH
|
||||
} else if (src_dyn == 3) {
|
||||
m_node_dynamic_dims[node] = 3; // N -> N
|
||||
}
|
||||
if (m_node_dynamic_dims[node] != -1) {
|
||||
OPENVINO_ASSERT(node->src[1]->ne[src_dyn] == node->ne[m_node_dynamic_dims[node]],
|
||||
"Dynamic dim value mismatch for IM2COL_3D node: " + std::string(node->name) +
|
||||
" and its src[1]: " + std::string(node->src[1]->name));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
GGML_LOG_DEBUG("ggml-openvino: compute_node_dynamic_dims: unhandled op %s for node '%s'\n",
|
||||
ggml_op_name(node->op), node->name);
|
||||
|
||||
@@ -28,7 +28,9 @@ struct ModelParams {
|
||||
std::map<int, int> n_heads_kv_per_layer;
|
||||
int head_size = -1;
|
||||
int state_size = -1; // for SSM molels, eg qwen35
|
||||
int32_t rope_params[16];
|
||||
int32_t rope_params[16] = {};
|
||||
int n_rs_slots = -1;
|
||||
bool has_rs_rollback = false;
|
||||
bool mixed_rope_params = false;
|
||||
bool is_cacheless_attn = false;
|
||||
std::vector<int> swa_layers;
|
||||
@@ -45,9 +47,15 @@ struct ModelParams {
|
||||
memcmp(rope_params, other.rope_params, sizeof(int32_t) * 16) == 0;
|
||||
}
|
||||
|
||||
bool can_reuse_dynamically(const ModelParams & other) const { return same_rope_params(other); }
|
||||
bool can_reuse_dynamically(const ModelParams & other) const {
|
||||
return same_rope_params(other) && n_rs_slots == other.n_rs_slots &&
|
||||
has_rs_rollback == other.has_rs_rollback;
|
||||
}
|
||||
|
||||
bool can_reuse_statically(const ModelParams & other) const { return same_rope_params(other) && ctx == other.ctx; }
|
||||
bool can_reuse_statically(const ModelParams & other) const {
|
||||
return same_rope_params(other) && ctx == other.ctx && n_rs_slots == other.n_rs_slots &&
|
||||
has_rs_rollback == other.has_rs_rollback;
|
||||
}
|
||||
|
||||
bool kv_buffer_changed(const ModelParams & other) const { return kv_buffer_ctx_id != other.kv_buffer_ctx_id; }
|
||||
};
|
||||
@@ -100,7 +108,7 @@ struct ComputeParams {
|
||||
|
||||
struct RsWriteback {
|
||||
int slot_begin = 0; // first cache slot written by the CPY
|
||||
int src_begin = 0; // first source row or column copied by the CPY
|
||||
int src_begin = -1; // first source column copied by a conv-state CPY
|
||||
};
|
||||
|
||||
std::map<std::string, RsWriteback> rs_writebacks;
|
||||
@@ -353,6 +361,7 @@ public:
|
||||
void add_extra_inputs();
|
||||
|
||||
void update_io(ggml_cgraph * cgraph);
|
||||
bool is_bound_to(const ggml_cgraph * cgraph) const;
|
||||
|
||||
static bool is_inp_tok(const ggml_tensor * tensor, const ggml_tensor * op) {
|
||||
return op->op == GGML_OP_GET_ROWS && tensor == op->src[1] && op->src[0]->op == GGML_OP_NONE;
|
||||
@@ -362,10 +371,11 @@ public:
|
||||
return op->op == GGML_OP_ROPE && tensor == op->src[1];
|
||||
}
|
||||
|
||||
// IMROPE packs 4 stacked position planes (t/h/w/e) into inp_pos, each of length
|
||||
// IMROPE and VISION pack 4 stacked position planes (t/h/w/e) into inp_pos, each of length
|
||||
// n_tokens; other modes carry a single position per token.
|
||||
static int get_inp_pos_n_planes(const ggml_tensor * op) {
|
||||
return op->op_params[2] == GGML_ROPE_TYPE_IMROPE ? 4 : 1;
|
||||
const int mode = op->op_params[2];
|
||||
return (mode == GGML_ROPE_TYPE_IMROPE || mode == GGML_ROPE_TYPE_VISION || (mode & GGML_ROPE_TYPE_MROPE)) ? 4 : 1;
|
||||
}
|
||||
|
||||
static bool is_inp_emb(const ggml_tensor * tensor, const ggml_tensor * op) {
|
||||
@@ -387,17 +397,26 @@ public:
|
||||
return op->op == GGML_OP_ROPE && tensor == op->src[2];
|
||||
}
|
||||
|
||||
// also returns true for cache_s and cache_r in SSM/DeltaNet models
|
||||
static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) {
|
||||
if (tensor == nullptr) {
|
||||
inline static bool is_recurrent_cache(const ggml_tensor * tensor) {
|
||||
return tensor != nullptr && (strncmp(tensor->name, "cache_r_l", strlen("cache_r_l")) == 0 ||
|
||||
strncmp(tensor->name, "cache_s_l", strlen("cache_s_l")) == 0 ||
|
||||
strncmp(tensor->name, "cache_ple_r_l", strlen("cache_ple_r_l")) == 0);
|
||||
}
|
||||
|
||||
inline static bool is_cache(const ggml_tensor * tensor, const ggml_tensor * op) {
|
||||
return is_recurrent_cache(tensor) || is_kvcache(tensor, op);
|
||||
}
|
||||
|
||||
inline static bool is_kvcache(const ggml_tensor * tensor, const ggml_tensor * op) {
|
||||
if (tensor == nullptr || is_recurrent_cache(tensor)) {
|
||||
return false;
|
||||
}
|
||||
return (tensor->buffer != nullptr && tensor->buffer->usage == GGML_BACKEND_BUFFER_USAGE_ANY) ||
|
||||
(op != nullptr && op->op == GGML_OP_SET_ROWS && op->src[2] == tensor);
|
||||
}
|
||||
|
||||
static bool is_conv_state_writeback(const ggml_tensor * node) {
|
||||
return node->op == GGML_OP_CPY && node->view_src != nullptr && is_kvcache(node->view_src, nullptr) &&
|
||||
inline static bool is_conv_state_writeback(const ggml_tensor * node) {
|
||||
return node->op == GGML_OP_CPY && node->view_src != nullptr && is_recurrent_cache(node->view_src) &&
|
||||
node->src[0] != nullptr && node->src[0]->op == GGML_OP_VIEW && node->src[0]->src[0] != nullptr &&
|
||||
node->src[0]->src[0]->op == GGML_OP_CONCAT && node->src[1] != nullptr &&
|
||||
node->src[1]->op == GGML_OP_VIEW && node->src[1]->view_src == node->view_src;
|
||||
|
||||
@@ -2,12 +2,15 @@
|
||||
|
||||
#include "ggml-impl.h"
|
||||
#include "ggml.h"
|
||||
#include "model-cache.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <openvino/runtime/intel_gpu/ocl/ocl.hpp>
|
||||
#include <openvino/runtime/intel_npu/level_zero/level_zero.hpp>
|
||||
#include <openvino/runtime/properties.hpp>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
|
||||
ov::Core & ov_singleton_core() {
|
||||
@@ -15,14 +18,88 @@ ov::Core & ov_singleton_core() {
|
||||
return core;
|
||||
}
|
||||
|
||||
static bool has_prefix(const std::string & s, const std::string & prefix) {
|
||||
return s.size() >= prefix.size() && std::equal(prefix.begin(), prefix.end(), s.begin());
|
||||
}
|
||||
|
||||
static bool is_virtual_routing_device(const std::string & device_name) {
|
||||
return has_prefix(device_name, "AUTO") || has_prefix(device_name, "MULTI") || has_prefix(device_name, "HETERO");
|
||||
}
|
||||
|
||||
static std::vector<std::string> ov_enumerate_devices() {
|
||||
std::vector<std::string> result;
|
||||
|
||||
for (const auto & device : ov_singleton_core().get_available_devices()) {
|
||||
if (!is_virtual_routing_device(device)) {
|
||||
result.push_back(device);
|
||||
}
|
||||
}
|
||||
|
||||
if (result.empty()) {
|
||||
result.push_back("CPU");
|
||||
}
|
||||
|
||||
std::sort(result.begin(), result.end());
|
||||
result.erase(std::unique(result.begin(), result.end()), result.end());
|
||||
return result;
|
||||
}
|
||||
|
||||
std::string ggml_openvino_get_device_description(const std::string & device_name) {
|
||||
std::string description = device_name;
|
||||
try {
|
||||
description = ov_singleton_core().get_property(device_name, ov::device::full_name);
|
||||
} catch (...) {
|
||||
return device_name;
|
||||
}
|
||||
|
||||
if (has_prefix(device_name, "NPU")) {
|
||||
try {
|
||||
const std::string arch = ov_singleton_core().get_property(device_name, "DEVICE_ARCHITECTURE").as<std::string>();
|
||||
if (!arch.empty()) {
|
||||
description += " (NPU " + arch + ")";
|
||||
}
|
||||
} catch (...) {
|
||||
}
|
||||
}
|
||||
|
||||
return description;
|
||||
}
|
||||
|
||||
// requested: GGML_OPENVINO_DEVICE, nullptr if unset. available_devices is never empty (see ov_enumerate_devices)
|
||||
static std::string resolve_openvino_device_name(const std::vector<std::string> & available_devices,
|
||||
const char * requested) {
|
||||
auto available = [&](const std::string & name) {
|
||||
return std::find(available_devices.begin(), available_devices.end(), name) != available_devices.end();
|
||||
};
|
||||
if (requested == nullptr) {
|
||||
return available("CPU") ? "CPU" : available_devices.front();
|
||||
}
|
||||
if (!available(requested)) {
|
||||
// No fallback to CPU (easy to miss) and no GPU -> GPU.0 alias (with iGPU + dGPU, GPU.0 is often the
|
||||
// wrong one). List the devices here: --list-devices initializes this backend and would abort too.
|
||||
std::string list;
|
||||
for (const std::string & name : available_devices) {
|
||||
list += "\n " + name + ": " + ggml_openvino_get_device_description(name);
|
||||
}
|
||||
GGML_ABORT("GGML OpenVINO Backend: GGML_OPENVINO_DEVICE=%s is not available. "
|
||||
"Set it to one of the available OpenVINO devices:%s",
|
||||
requested, list.c_str());
|
||||
}
|
||||
return requested;
|
||||
}
|
||||
|
||||
// =====================================================
|
||||
// Device Configuration Implementations
|
||||
// =====================================================
|
||||
|
||||
void ggml_openvino_device_config::init() {
|
||||
static std::mutex mutex;
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
if (initialized) {
|
||||
return;
|
||||
}
|
||||
// Set up front: a failed OpenCL setup below is not retried on every call
|
||||
initialized = true;
|
||||
|
||||
// All recognized GGML_OPENVINO_* env vars. Their values are cached here
|
||||
// once at backend init time and read back via ggml_openvino_getenv_str()
|
||||
@@ -34,6 +111,7 @@ void ggml_openvino_device_config::init() {
|
||||
"GGML_OPENVINO_SPILL_DIR",
|
||||
"GGML_OPENVINO_DEBUG_NODE",
|
||||
"GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR",
|
||||
"GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY",
|
||||
"GGML_OPENVINO_NPU_COMPILE_CONFIG",
|
||||
// Integer values (use ggml_openvino_getenv_int)
|
||||
"GGML_OPENVINO_PREFILL_CHUNK_SIZE",
|
||||
@@ -53,6 +131,7 @@ void ggml_openvino_device_config::init() {
|
||||
"GGML_OPENVINO_DISABLE_KV_SLICE",
|
||||
"GGML_OPENVINO_ENABLE_FALLBACK",
|
||||
"GGML_OPENVINO_MANUAL_GQA_ATTN",
|
||||
"GGML_OPENVINO_MOE_OP",
|
||||
"GGML_OPENVINO_MEMORY_OPTIMIZE",
|
||||
"GGML_OPENVINO_RELEASE_WEIGHTS",
|
||||
"GGML_OPENVINO_REDUCE_COMPILE_MEM",
|
||||
@@ -62,6 +141,8 @@ void ggml_openvino_device_config::init() {
|
||||
"GGML_OPENVINO_DISABLE_REMOTE_OUTPUTS",
|
||||
"GGML_OPENVINO_REQUANT_KQUANT",
|
||||
"GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT",
|
||||
// Build the precise (but O(n_nodes)) graph cache key. Needed by op tests.
|
||||
"GGML_OPENVINO_FULL_GRAPH_KEY",
|
||||
};
|
||||
|
||||
for (const char * const & env_var : env_var_names) {
|
||||
@@ -71,16 +152,14 @@ void ggml_openvino_device_config::init() {
|
||||
}
|
||||
}
|
||||
|
||||
device_name = ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE", "CPU");
|
||||
auto available_devices = ov_singleton_core().get_available_devices();
|
||||
if (std::find(available_devices.begin(), available_devices.end(), device_name) == available_devices.end()) {
|
||||
GGML_LOG_WARN("GGML OpenVINO Backend: device %s is not available, fallback to CPU\n", device_name.c_str());
|
||||
device_name = "CPU";
|
||||
}
|
||||
is_npu = (device_name == "NPU");
|
||||
available_devices = ov_enumerate_devices();
|
||||
device_name = resolve_openvino_device_name(available_devices, ggml_openvino_getenv_str("GGML_OPENVINO_DEVICE"));
|
||||
is_npu = has_prefix(device_name, "NPU");
|
||||
|
||||
ggml_openvino_model_cache_init();
|
||||
|
||||
const char * cache_dir = ggml_openvino_getenv_str("GGML_OPENVINO_CACHE_DIR");
|
||||
if (device_name == "NPU") {
|
||||
if (has_prefix(device_name, "NPU")) {
|
||||
compile_config = {
|
||||
{"NPU_COMPILER_DYNAMIC_QUANTIZATION", "YES" },
|
||||
{"NPU_USE_NPUW", "YES" },
|
||||
@@ -106,48 +185,69 @@ void ggml_openvino_device_config::init() {
|
||||
compile_config.insert(ov::cache_mode(ov::CacheMode::OPTIMIZE_SIZE));
|
||||
}
|
||||
|
||||
if (ggml_openvino_getenv_int("GGML_OPENVINO_PROFILING") >= 2) {
|
||||
compile_config.insert(ov::enable_profiling(true));
|
||||
}
|
||||
|
||||
// Initialize remote context with queue sharing for GPU
|
||||
if (device_name == "GPU") {
|
||||
// Create OpenCL context and queue
|
||||
if (has_prefix(device_name, "GPU")) {
|
||||
// Use the OpenCL context OpenVINO created for this device, so GPU.N gets its own device
|
||||
cl_context cl_ctx;
|
||||
try {
|
||||
auto ov_ctx = ov_singleton_core().get_default_context(device_name).as<ov::intel_gpu::ocl::ClContext>();
|
||||
cl_ctx = ov_ctx.get();
|
||||
} catch (const std::exception & e) {
|
||||
// The consumers of the remote context have no host fallback, and OpenVINO
|
||||
// already reported the device as present.
|
||||
GGML_ABORT("ggml-openvino: failed to get the OpenCL context for %s: %s", device_name.c_str(), e.what());
|
||||
}
|
||||
|
||||
cl_int err;
|
||||
cl_platform_id platform;
|
||||
err = clGetPlatformIDs(1, &platform, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("Failed to get OpenCL platform: %d\n", err);
|
||||
return;
|
||||
}
|
||||
|
||||
cl_device_id cl_device;
|
||||
err = clGetDeviceIDs(platform, CL_DEVICE_TYPE_GPU, 1, &cl_device, nullptr);
|
||||
err = clGetContextInfo(cl_ctx, CL_CONTEXT_DEVICES, sizeof(cl_device), &cl_device, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("Failed to get OpenCL device: %d\n", err);
|
||||
return;
|
||||
GGML_ABORT("ggml-openvino: failed to get the OpenCL device for %s: %d", device_name.c_str(), err);
|
||||
}
|
||||
|
||||
cl_context cl_ctx = clCreateContext(nullptr, 1, &cl_device, nullptr, nullptr, &err);
|
||||
cl_platform_id cl_platform;
|
||||
err = clGetDeviceInfo(cl_device, CL_DEVICE_PLATFORM, sizeof(cl_platform), &cl_platform, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("Failed to create OpenCL context: %d\n", err);
|
||||
return;
|
||||
GGML_ABORT("ggml-openvino: failed to get the OpenCL platform for %s: %d", device_name.c_str(), err);
|
||||
}
|
||||
|
||||
cl_queue = clCreateCommandQueueWithProperties(cl_ctx, cl_device, nullptr, &err);
|
||||
cl_mem_fill_fn =
|
||||
(clEnqueueMemFillINTEL_fn) clGetExtensionFunctionAddressForPlatform(cl_platform, "clEnqueueMemFillINTEL");
|
||||
cl_mem_cpy_fn =
|
||||
(clEnqueueMemcpyINTEL_fn) clGetExtensionFunctionAddressForPlatform(cl_platform, "clEnqueueMemcpyINTEL");
|
||||
|
||||
cl_ulong device_max_alloc = 0;
|
||||
err = clGetDeviceInfo(cl_device, CL_DEVICE_MAX_MEM_ALLOC_SIZE, sizeof(device_max_alloc), &device_max_alloc,
|
||||
nullptr);
|
||||
if (err == CL_SUCCESS) {
|
||||
max_alloc_size = device_max_alloc;
|
||||
} else {
|
||||
// not fatal, ggml then allocates one buffer
|
||||
GGML_LOG_WARN("Failed to get OpenCL max allocation size: %d\n", err);
|
||||
}
|
||||
|
||||
const cl_queue_properties profiling_properties[] = {
|
||||
CL_QUEUE_PROPERTIES,
|
||||
CL_QUEUE_PROFILING_ENABLE,
|
||||
0,
|
||||
};
|
||||
const cl_queue_properties * queue_properties =
|
||||
ggml_openvino_getenv_int("GGML_OPENVINO_PROFILING") >= 2 ? profiling_properties : nullptr;
|
||||
cl_queue = clCreateCommandQueueWithProperties(cl_ctx, cl_device, queue_properties, &err);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("Failed to create OpenCL command queue: %d\n", err);
|
||||
clReleaseContext(cl_ctx);
|
||||
return;
|
||||
GGML_ABORT("ggml-openvino: failed to create the OpenCL queue for %s: %d", device_name.c_str(), err);
|
||||
}
|
||||
|
||||
// Create OpenVINO remote context with queue sharing
|
||||
remote_context = ov::intel_gpu::ocl::ClContext(ov_singleton_core(), cl_queue);
|
||||
|
||||
// Release the context (queue keeps a reference)
|
||||
clReleaseContext(cl_ctx);
|
||||
} else if (device_name == "NPU") {
|
||||
} else if (has_prefix(device_name, "NPU")) {
|
||||
// remote tensor is not used for NPU yet
|
||||
// remote_context = ov_singleton_core().get_default_context(device_name);
|
||||
}
|
||||
|
||||
initialized = true;
|
||||
}
|
||||
|
||||
ggml_openvino_device_config::~ggml_openvino_device_config() {
|
||||
@@ -173,6 +273,12 @@ const std::string & ggml_openvino_get_device_name() {
|
||||
return ggml_openvino_get_device_config().device_name;
|
||||
}
|
||||
|
||||
std::vector<std::string> ggml_openvino_get_available_devices() {
|
||||
auto & config = ggml_openvino_get_device_config();
|
||||
config.init();
|
||||
return config.available_devices;
|
||||
}
|
||||
|
||||
// Get the value of a GGML_OPENVINO_* env var as a string. Returns
|
||||
// default_value when the var is unset or set to an empty string.
|
||||
const char * ggml_openvino_getenv_str(const char * var, const char * default_value) {
|
||||
@@ -198,12 +304,12 @@ bool ggml_openvino_reduce_compile_mem_enabled() {
|
||||
return ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
|
||||
}
|
||||
|
||||
bool ggml_openvino_release_weights_enabled(const std::string & device) {
|
||||
bool ggml_openvino_release_weights_enabled() {
|
||||
const char * release_weights = ggml_openvino_getenv_str("GGML_OPENVINO_RELEASE_WEIGHTS");
|
||||
if (release_weights != nullptr) {
|
||||
return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0;
|
||||
return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_RELEASE_WEIGHTS") != 0;
|
||||
}
|
||||
return device == "GPU" && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
|
||||
return ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_MEMORY_OPTIMIZE") != 0;
|
||||
}
|
||||
|
||||
// Check if running on NPU
|
||||
@@ -211,6 +317,14 @@ bool ggml_openvino_is_npu() {
|
||||
return ggml_openvino_get_device_config().is_npu;
|
||||
}
|
||||
|
||||
bool ggml_openvino_is_gpu() {
|
||||
return has_prefix(ggml_openvino_get_device_name(), "GPU");
|
||||
}
|
||||
|
||||
size_t ggml_openvino_max_alloc_size() {
|
||||
return ggml_openvino_get_device_config().max_alloc_size;
|
||||
}
|
||||
|
||||
// Get the remote context for the current device (returns empty optional for CPU)
|
||||
std::optional<ov::RemoteContext> ggml_openvino_get_remote_context() {
|
||||
return ggml_openvino_get_device_config().remote_context;
|
||||
@@ -226,32 +340,14 @@ cl_command_queue ggml_openvino_get_cl_queue() {
|
||||
return ggml_openvino_get_device_config().cl_queue;
|
||||
}
|
||||
|
||||
// Get the clEnqueueMemFillINTEL function pointer (lazy load)
|
||||
// Get the clEnqueueMemFillINTEL function pointer
|
||||
clEnqueueMemFillINTEL_fn ggml_openvino_get_clEnqueueMemFillINTEL() {
|
||||
static clEnqueueMemFillINTEL_fn fn = nullptr;
|
||||
static bool loaded = false;
|
||||
if (!loaded) {
|
||||
loaded = true;
|
||||
cl_platform_id platform;
|
||||
if (clGetPlatformIDs(1, &platform, nullptr) == CL_SUCCESS) {
|
||||
fn = (clEnqueueMemFillINTEL_fn) clGetExtensionFunctionAddressForPlatform(platform, "clEnqueueMemFillINTEL");
|
||||
}
|
||||
}
|
||||
return fn;
|
||||
return ggml_openvino_get_device_config().cl_mem_fill_fn;
|
||||
}
|
||||
|
||||
// Get the clEnqueueMemcpyINTEL function pointer (lazy load)
|
||||
// Get the clEnqueueMemcpyINTEL function pointer
|
||||
clEnqueueMemcpyINTEL_fn ggml_openvino_get_clEnqueueMemcpyINTEL() {
|
||||
static clEnqueueMemcpyINTEL_fn fn = nullptr;
|
||||
static bool loaded = false;
|
||||
if (!loaded) {
|
||||
loaded = true;
|
||||
cl_platform_id platform;
|
||||
if (clGetPlatformIDs(1, &platform, nullptr) == CL_SUCCESS) {
|
||||
fn = (clEnqueueMemcpyINTEL_fn) clGetExtensionFunctionAddressForPlatform(platform, "clEnqueueMemcpyINTEL");
|
||||
}
|
||||
}
|
||||
return fn;
|
||||
return ggml_openvino_get_device_config().cl_mem_cpy_fn;
|
||||
}
|
||||
|
||||
// Get requantization type for a tensor type (returns nullopt if no requant needed)
|
||||
@@ -280,14 +376,11 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
|
||||
// Q6_K/Q5_K are touched):
|
||||
// q4_sym128 Q6_K/Q5_K -> Q4_0_128 (u4, group 128, symmetric)
|
||||
// q4_sym128_all and Q4_K too -- drops Q4_K's per-32 zero point, which costs some accuracy
|
||||
// q4_asym64_all Q6_K/Q5_K and Q4_K -> Q4_1_64 (u4, group 64, asymmetric) -- most of the
|
||||
// metadata saving while keeping a real zero point
|
||||
// q4_asym64 Q6_K/Q5_K -> Q4_1_64 (u4, group 64, asymmetric)
|
||||
// q4_asym64_all Q6_K/Q5_K and Q4_K -> Q4_1_64 (u4, group 64, asymmetric)
|
||||
// native no requantization at all (keep Q6_K/Q5_K as they are)
|
||||
//
|
||||
// The asymmetric target is only offered in its _all form: leaving Q4_K at its native group 32
|
||||
// while Q6_K/Q5_K move to group 64 gives the Q/K/V projections different group counts, and the
|
||||
// GPU plugin's FullyConnectedHorizontalFusion concatenates their scale constants, which then
|
||||
// fails shape inference. Requantizing all three keeps the group size uniform.
|
||||
// q4_asym64 leaves Q4_K at its native group 32. Use q4_asym64_all to keep the group size uniform.
|
||||
const char * rq = ggml_openvino_getenv_str("GGML_OPENVINO_REQUANT_KQUANT");
|
||||
auto is_opt = [rq](const char * name) {
|
||||
return rq && strcmp(rq, name) == 0;
|
||||
@@ -295,6 +388,7 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
|
||||
const bool sym128 = is_opt("q4_sym128");
|
||||
const bool sym128_all = is_opt("q4_sym128_all");
|
||||
const bool asym64_all = is_opt("q4_asym64_all");
|
||||
const bool asym64 = is_opt("q4_asym64");
|
||||
|
||||
if (tensor->type == GGML_TYPE_Q4_K) {
|
||||
if (sym128_all) {
|
||||
@@ -313,7 +407,7 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
|
||||
if (sym128 || sym128_all) {
|
||||
return ExtraQuantType::Q4_0_64;
|
||||
}
|
||||
if (asym64_all) {
|
||||
if (asym64 || asym64_all) {
|
||||
return ExtraQuantType::Q4_1_64;
|
||||
}
|
||||
// TODO: temporary workaround for a known OpenVINO GPU-plugin bug -- remove once the
|
||||
@@ -328,7 +422,7 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
|
||||
// already requantize to per-channel Q8_0_C (grouped=0). Sending these to grouped 4 bit
|
||||
// avoids the broken layout and restores correct output.
|
||||
// Opt out with GGML_OPENVINO_REQUANT_KQUANT=native.
|
||||
if (ggml_openvino_get_device_name() == "GPU" && !is_opt("native")) {
|
||||
if (ggml_openvino_is_gpu() && !is_opt("native")) {
|
||||
return ExtraQuantType::Q4_0_64;
|
||||
}
|
||||
}
|
||||
@@ -338,7 +432,7 @@ std::optional<ExtraQuantType> ggml_openvino_get_requant_type(const ggml_tensor *
|
||||
if (sym128 || sym128_all) {
|
||||
return ExtraQuantType::Q4_0_128;
|
||||
}
|
||||
if (asym64_all) {
|
||||
if (asym64 || asym64_all) {
|
||||
return ExtraQuantType::Q4_1_64;
|
||||
}
|
||||
if (is_opt("native")) {
|
||||
@@ -439,9 +533,7 @@ ggml_openvino_extracted_layout ggml_openvino_get_extracted_layout(const ggml_ten
|
||||
layout.weights_per_block = tensor->ne[0];
|
||||
break;
|
||||
default:
|
||||
layout.weights_per_block = -1;
|
||||
GGML_ABORT("Code of re-quantizing to channel-wise is not updated");
|
||||
break;
|
||||
}
|
||||
|
||||
if (layout.is_requant) {
|
||||
@@ -560,12 +652,11 @@ ggml_openvino_tensor_extra * ggml_openvino_create_tensor_extra(const ggml_tensor
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
const auto & device_name = ggml_openvino_get_device_name();
|
||||
auto remote_context = ggml_openvino_get_remote_context();
|
||||
|
||||
std::shared_ptr<ov::Tensor> ov_tensor;
|
||||
if (is_remote) {
|
||||
GGML_ASSERT(device_name == "GPU");
|
||||
GGML_ASSERT(ggml_openvino_is_gpu());
|
||||
auto gpu_context = remote_context->as<ov::intel_gpu::ocl::ClContext>();
|
||||
auto usm_tensor = gpu_context.create_tensor(element_type, shape, tensor->data);
|
||||
ov_tensor = std::make_shared<ov::intel_gpu::ocl::USMTensor>(std::move(usm_tensor));
|
||||
|
||||
@@ -63,12 +63,16 @@ clEnqueueMemcpyINTEL_fn ggml_openvino_get_clEnqueueMemcpyINTEL();
|
||||
|
||||
struct ggml_openvino_device_config {
|
||||
std::string device_name = "CPU";
|
||||
std::vector<std::string> available_devices;
|
||||
bool is_npu = false;
|
||||
bool initialized = false;
|
||||
std::optional<ov::RemoteContext> remote_context;
|
||||
size_t max_alloc_size = SIZE_MAX;
|
||||
ov::AnyMap compile_config;
|
||||
std::unordered_map<std::string, std::string> environment_variables;
|
||||
cl_command_queue cl_queue = nullptr;
|
||||
clEnqueueMemFillINTEL_fn cl_mem_fill_fn = nullptr;
|
||||
clEnqueueMemcpyINTEL_fn cl_mem_cpy_fn = nullptr;
|
||||
|
||||
void init();
|
||||
~ggml_openvino_device_config();
|
||||
@@ -83,6 +87,12 @@ void ggml_openvino_init_device_config();
|
||||
// Get the device name
|
||||
const std::string & ggml_openvino_get_device_name();
|
||||
|
||||
// Get all available physical OpenVINO devices
|
||||
std::vector<std::string> ggml_openvino_get_available_devices();
|
||||
|
||||
// Human-readable device name, e.g. "Intel(R) AI Boost (NPU 4000)"; the device id if unavailable
|
||||
std::string ggml_openvino_get_device_description(const std::string & device_name);
|
||||
|
||||
// Environment variable accessors. All GGML_OPENVINO_* env vars are read once
|
||||
// during backend init and cached on the device config; consumers must go
|
||||
// through these helpers (never call ::getenv directly) so behavior stays
|
||||
@@ -102,11 +112,17 @@ int ggml_openvino_getenv_int(const char * var, int default_value = 0);
|
||||
// Memory optimization toggles. GGML_OPENVINO_MEMORY_OPTIMIZE is an umbrella
|
||||
// switch; the fine-grained env vars still override it when explicitly set.
|
||||
bool ggml_openvino_reduce_compile_mem_enabled();
|
||||
bool ggml_openvino_release_weights_enabled(const std::string & device);
|
||||
bool ggml_openvino_release_weights_enabled();
|
||||
|
||||
// Check if running on NPU
|
||||
bool ggml_openvino_is_npu();
|
||||
|
||||
// Check if running on a GPU (GPU, GPU.0, GPU.1, ...)
|
||||
bool ggml_openvino_is_gpu();
|
||||
|
||||
// Largest single memory object the device can allocate, SIZE_MAX when there is no known limit
|
||||
size_t ggml_openvino_max_alloc_size();
|
||||
|
||||
// Host weight-buffer release (GGML_OPENVINO_RELEASE_WEIGHTS, GPU only).
|
||||
// register: record a host weight buffer (idempotent per data pointer).
|
||||
// release: madvise(MADV_DONTNEED) all registered buffers, dropping their RSS.
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
#include "ggml-openvino/utils.h"
|
||||
#include "ggml-quants.h"
|
||||
#include "ggml.h"
|
||||
#include "model-cache.h"
|
||||
|
||||
#include <algorithm>
|
||||
#include <atomic>
|
||||
@@ -24,7 +25,10 @@
|
||||
#include <openvino/runtime/allocator.hpp>
|
||||
#include <openvino/runtime/intel_gpu/ocl/ocl.hpp>
|
||||
#include <openvino/runtime/intel_npu/level_zero/level_zero.hpp>
|
||||
#include <openvino/runtime/properties.hpp>
|
||||
#include <openvino/runtime/tensor.hpp>
|
||||
#include <algorithm>
|
||||
#include <map>
|
||||
#include <set>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
@@ -69,8 +73,7 @@ struct ggml_backend_openvino_buffer_context {
|
||||
size_t size;
|
||||
bool is_remote;
|
||||
|
||||
// Set when the buffer is a file-backed spill mapping (GGML_OPENVINO_SPILL_DIR); it must be
|
||||
// munmap'd rather than freed.
|
||||
// File-backed spill or cache-only virtual memory.
|
||||
void * spill_mapping = nullptr;
|
||||
size_t spill_size = 0;
|
||||
|
||||
@@ -79,6 +82,8 @@ struct ggml_backend_openvino_buffer_context {
|
||||
|
||||
// Track all extras for cleanup
|
||||
std::map<ggml_tensor *, ggml_openvino_extra_base *> tensor_extras;
|
||||
std::map<const void *, uint64_t> weight_fingerprints;
|
||||
std::vector<ggml_openvino_source_mapping> source_mappings;
|
||||
|
||||
// Used for re-allocation on device for kvcache
|
||||
void * data_prev;
|
||||
@@ -100,7 +105,7 @@ struct ggml_backend_openvino_buffer_context {
|
||||
const auto & device_name = ggml_openvino_get_device_name();
|
||||
|
||||
if (is_remote) {
|
||||
GGML_ASSERT(device_name == "GPU");
|
||||
GGML_ASSERT(ggml_openvino_is_gpu());
|
||||
auto remote_context = ggml_openvino_get_remote_context();
|
||||
auto gpu_context = remote_context->as<ov::intel_gpu::ocl::ClContext>();
|
||||
ov::intel_gpu::ocl::USMTensor usm_tensor =
|
||||
@@ -108,8 +113,25 @@ struct ggml_backend_openvino_buffer_context {
|
||||
data = usm_tensor.get();
|
||||
ov_buffer = std::make_shared<ov::intel_gpu::ocl::USMTensor>(std::move(usm_tensor));
|
||||
} else {
|
||||
#ifndef _WIN32
|
||||
if (const char * spill_dir = ggml_openvino_getenv_str("GGML_OPENVINO_SPILL_DIR")) {
|
||||
#ifdef _WIN32
|
||||
if (ggml_openvino_model_cache_only()) {
|
||||
data = spill_mapping = VirtualAlloc(nullptr, size, MEM_RESERVE | MEM_COMMIT, PAGE_READWRITE);
|
||||
if (data == nullptr) {
|
||||
return;
|
||||
}
|
||||
spill_size = size;
|
||||
ov_buffer = std::make_shared<ov::Tensor>(ov::element::u8, ov::Shape{size}, data);
|
||||
} else
|
||||
#else
|
||||
if (ggml_openvino_model_cache_only()) {
|
||||
void * m = mmap(nullptr, size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
|
||||
if (m == MAP_FAILED) {
|
||||
return;
|
||||
}
|
||||
data = spill_mapping = m;
|
||||
spill_size = size;
|
||||
ov_buffer = std::make_shared<ov::Tensor>(ov::element::u8, ov::Shape{size}, data);
|
||||
} else if (const char * spill_dir = ggml_openvino_getenv_str("GGML_OPENVINO_SPILL_DIR")) {
|
||||
// Disk-backed weight buffer: back the repacked weights with a temp file via MAP_SHARED
|
||||
// instead of anonymous memory. Anonymous pages can only be evicted to swap, so the
|
||||
// repacked buffer stays pinned alongside the mmap'd source and both are resident at once
|
||||
@@ -180,7 +202,11 @@ struct ggml_backend_openvino_buffer_context {
|
||||
delete pair.second;
|
||||
}
|
||||
tensor_extras.clear();
|
||||
#ifndef _WIN32
|
||||
#ifdef _WIN32
|
||||
if (spill_mapping != nullptr) {
|
||||
VirtualFree(spill_mapping, 0, MEM_RELEASE);
|
||||
} else
|
||||
#else
|
||||
if (spill_mapping != nullptr) {
|
||||
munmap(spill_mapping, spill_size);
|
||||
} else
|
||||
@@ -295,7 +321,7 @@ static enum ggml_status ggml_backend_openvino_buffer_init_tensor(ggml_backend_bu
|
||||
ggml_backend_openvino_buffer_context * ctx = (ggml_backend_openvino_buffer_context *) buffer->context;
|
||||
|
||||
// Put kvcache on device memory for GPU (NPU memory is too small even for kvcache)
|
||||
if (strncmp(tensor->name, "cache_", 6) == 0 && !ctx->is_remote && ggml_openvino_get_device_name() == "GPU" &&
|
||||
if (strncmp(tensor->name, "cache_", 6) == 0 && !ctx->is_remote && ggml_openvino_is_gpu() &&
|
||||
!is_stateful_enabled()) {
|
||||
GGML_ASSERT(ctx->tensor_extras.empty());
|
||||
auto device = ctx->device;
|
||||
@@ -311,6 +337,26 @@ static enum ggml_status ggml_backend_openvino_buffer_init_tensor(ggml_backend_bu
|
||||
if (tensor->view_src != nullptr) {
|
||||
GGML_ASSERT(tensor->view_src->buffer->buft == buffer->buft);
|
||||
if (tensor->view_src->extra != nullptr) {
|
||||
// The cached ov::Tensor carries the shape it was built with, so sharing view_src's
|
||||
// extra hands out the wrong shape for a reshaping view (e.g. Vcur reshaped from
|
||||
// [n_embd, n_tokens] to [head_size, n_heads_kv, n_tokens]). When such a view is a
|
||||
// graph input, binding it fails the shape check. Give it its own extra instead;
|
||||
// ggml_openvino_create_tensor_extra reads ne and data off the view, so the offset is
|
||||
// handled too. Only safe for a contiguous view - the ov::Tensor assumes dense strides.
|
||||
// Skip empty views: they have no data, and on GPU one can sit at the end of the USM buffer.
|
||||
if (!ggml_are_same_shape(tensor, tensor->view_src) && ggml_is_contiguous(tensor) &&
|
||||
!ggml_is_quantized(tensor->type) && tensor->data != nullptr && ggml_nbytes(tensor) > 0) {
|
||||
if (ggml_openvino_tensor_extra * extra =
|
||||
ggml_openvino_create_tensor_extra(tensor, ctx->is_remote)) {
|
||||
auto it = ctx->tensor_extras.find(tensor);
|
||||
if (it != ctx->tensor_extras.end()) {
|
||||
delete it->second;
|
||||
}
|
||||
ctx->tensor_extras[tensor] = extra;
|
||||
tensor->extra = extra;
|
||||
return GGML_STATUS_SUCCESS;
|
||||
}
|
||||
}
|
||||
tensor->extra = tensor->view_src->extra;
|
||||
}
|
||||
return GGML_STATUS_SUCCESS;
|
||||
@@ -346,7 +392,7 @@ static void ggml_backend_openvino_buffer_memset_tensor(ggml_backend_buffer_t buf
|
||||
// For remote (device) buffers, use OpenCL USM memfill
|
||||
cl_command_queue queue = ggml_openvino_get_cl_queue();
|
||||
auto mem_fill_fn = ggml_openvino_get_clEnqueueMemFillINTEL();
|
||||
if (queue != nullptr && mem_fill_fn != nullptr) {
|
||||
if (mem_fill_fn != nullptr) {
|
||||
uint8_t pattern = value;
|
||||
cl_int err = mem_fill_fn(queue, (char *) tensor->data + offset, &pattern, sizeof(pattern), size, 0, nullptr,
|
||||
nullptr);
|
||||
@@ -355,7 +401,7 @@ static void ggml_backend_openvino_buffer_memset_tensor(ggml_backend_buffer_t buf
|
||||
}
|
||||
clFinish(queue);
|
||||
} else {
|
||||
GGML_LOG_ERROR("%s: no OpenCL queue or clEnqueueMemFillINTEL not available for GPU buffer\n", __func__);
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemFillINTEL not available for GPU buffer\n", __func__);
|
||||
}
|
||||
} else {
|
||||
memset((char *) tensor->data + offset, value, size);
|
||||
@@ -375,6 +421,17 @@ static void ggml_backend_openvino_buffer_set_tensor(ggml_backend_buffer_t buffer
|
||||
bool is_weight_buffer = (buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS);
|
||||
// Full tensor set: offset=0, full size, not a view
|
||||
bool is_full_tensor_set = (offset == 0 && size == ggml_nbytes(tensor) && tensor->view_src == nullptr);
|
||||
if (is_weight_buffer && ggml_openvino_getenv_str("GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR")) {
|
||||
if (is_full_tensor_set) {
|
||||
ctx->weight_fingerprints[tensor->data] = ggml_openvino_source_fingerprint(data, size, ctx->source_mappings);
|
||||
}
|
||||
if (ggml_openvino_model_cache_only()) {
|
||||
if (!is_full_tensor_set) {
|
||||
GGML_ABORT("ggml-openvino: cache-only mode requires whole mmap weight uploads");
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
// 2D tensor (typical weight shape), or a 3D quantized MoE expert weight (MUL_MAT_ID). Dense 3D
|
||||
// expert weights are handled later in create_weight_node instead.
|
||||
bool is_2d = (tensor->ne[2] == 1 && tensor->ne[3] == 1);
|
||||
@@ -441,14 +498,14 @@ static void ggml_backend_openvino_buffer_set_tensor(ggml_backend_buffer_t buffer
|
||||
if (ctx->is_remote) {
|
||||
cl_command_queue queue = ggml_openvino_get_cl_queue();
|
||||
auto mem_cpy_fn = ggml_openvino_get_clEnqueueMemcpyINTEL();
|
||||
if (queue != nullptr && mem_cpy_fn != nullptr) {
|
||||
if (mem_cpy_fn != nullptr) {
|
||||
cl_int err =
|
||||
mem_cpy_fn(queue, CL_TRUE, (char *) tensor->data + offset, data, size, 0, nullptr, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemcpyINTEL failed with error %d\n", __func__, err);
|
||||
}
|
||||
} else {
|
||||
GGML_LOG_ERROR("%s: no OpenCL queue or clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
}
|
||||
} else {
|
||||
memcpy((char *) tensor->data + offset, data, size);
|
||||
@@ -478,18 +535,22 @@ static void ggml_backend_openvino_buffer_get_tensor(ggml_backend_buffer_t buffer
|
||||
GGML_ASSERT(tensor != nullptr && tensor->data != nullptr);
|
||||
ggml_backend_openvino_buffer_context * ctx = (ggml_backend_openvino_buffer_context *) buffer->context;
|
||||
|
||||
if (ggml_openvino_model_cache_only() && buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {
|
||||
GGML_ABORT("ggml-openvino: cannot read unloaded weights in cache-only mode");
|
||||
}
|
||||
|
||||
if (ctx->is_remote) {
|
||||
// For remote (device) buffers, use OpenCL USM memcpy (device-to-host)
|
||||
cl_command_queue queue = ggml_openvino_get_cl_queue();
|
||||
auto mem_cpy_fn = ggml_openvino_get_clEnqueueMemcpyINTEL();
|
||||
if (queue != nullptr && mem_cpy_fn != nullptr) {
|
||||
if (mem_cpy_fn != nullptr) {
|
||||
cl_int err =
|
||||
mem_cpy_fn(queue, CL_TRUE, data, (const char *) tensor->data + offset, size, 0, nullptr, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemcpyINTEL failed with error %d\n", __func__, err);
|
||||
}
|
||||
} else {
|
||||
GGML_LOG_ERROR("%s: no OpenCL queue or clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
}
|
||||
} else {
|
||||
memcpy(data, (const char *) tensor->data + offset, size);
|
||||
@@ -507,8 +568,8 @@ static bool ggml_backend_openvino_buffer_cpy_tensor(ggml_backend_buffer_t buffer
|
||||
// For remote (device) buffers, use OpenCL USM memcpy
|
||||
cl_command_queue queue = ggml_openvino_get_cl_queue();
|
||||
auto mem_cpy_fn = ggml_openvino_get_clEnqueueMemcpyINTEL();
|
||||
if (queue == nullptr || mem_cpy_fn == nullptr) {
|
||||
GGML_LOG_ERROR("%s: no OpenCL queue or clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
if (mem_cpy_fn == nullptr) {
|
||||
GGML_LOG_ERROR("%s: clEnqueueMemcpyINTEL not available for GPU buffer\n", __func__);
|
||||
return false;
|
||||
}
|
||||
// Can copy from host to device
|
||||
@@ -550,7 +611,7 @@ static void ggml_backend_openvino_buffer_clear(ggml_backend_buffer_t buffer, uin
|
||||
if (ctx->is_remote) {
|
||||
cl_command_queue queue = ggml_openvino_get_cl_queue();
|
||||
auto mem_fill_fn = ggml_openvino_get_clEnqueueMemFillINTEL();
|
||||
if (queue != nullptr && mem_fill_fn != nullptr) {
|
||||
if (mem_fill_fn != nullptr) {
|
||||
uint8_t pattern = value;
|
||||
cl_int err = mem_fill_fn(queue, ctx->data, &pattern, sizeof(pattern), ctx->size, 0, nullptr, nullptr);
|
||||
if (err != CL_SUCCESS) {
|
||||
@@ -558,8 +619,7 @@ static void ggml_backend_openvino_buffer_clear(ggml_backend_buffer_t buffer, uin
|
||||
}
|
||||
clFinish(queue);
|
||||
} else {
|
||||
GGML_LOG_WARN("%s: no OpenCL queue or clEnqueueMemFillINTEL not available for GPU buffer clear\n",
|
||||
__func__);
|
||||
GGML_LOG_WARN("%s: clEnqueueMemFillINTEL not available for GPU buffer clear\n", __func__);
|
||||
}
|
||||
} else {
|
||||
memset(ctx->data, value, ctx->size);
|
||||
@@ -609,7 +669,8 @@ static size_t ggml_backend_openvino_buffer_type_get_alignment(ggml_backend_buffe
|
||||
|
||||
static size_t ggml_backend_openvino_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) {
|
||||
GGML_UNUSED(buft);
|
||||
return SIZE_MAX;
|
||||
// A GPU caps a single memory object, so let ggml split a large buffer into parts that fit
|
||||
return ggml_openvino_max_alloc_size();
|
||||
}
|
||||
|
||||
static size_t ggml_backend_openvino_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft,
|
||||
@@ -617,7 +678,7 @@ static size_t ggml_backend_openvino_buffer_type_get_alloc_size(ggml_backend_buff
|
||||
GGML_UNUSED(buft);
|
||||
|
||||
// For quantized weight tensors, we need extra space for extracted data.
|
||||
if (ggml_is_quantized(tensor->type) && tensor->ne[3] == 1) {
|
||||
if (!ggml_openvino_model_cache_only() && ggml_is_quantized(tensor->type) && tensor->ne[3] == 1) {
|
||||
ggml_openvino_extracted_layout layout = ggml_openvino_get_extracted_layout(tensor);
|
||||
if (layout.total_size > 0) {
|
||||
// GGML_LOG_DEBUG("%s: tensor %s needs %zu bytes (original %zu, extracted: weights=%zu scales=%zu zp=%zu)\n",
|
||||
@@ -772,6 +833,19 @@ bool ggml_backend_buft_is_openvino_host(ggml_backend_buffer_type_t buft) {
|
||||
return buft->iface.get_name == ggml_backend_openvino_host_buffer_type_get_name;
|
||||
}
|
||||
|
||||
uint64_t ggml_backend_openvino_weight_fingerprint(const ggml_tensor * tensor) {
|
||||
if (ggml_backend_buffer_is_openvino(tensor->buffer)) {
|
||||
auto * ctx = static_cast<ggml_backend_openvino_buffer_context *>(tensor->buffer->context);
|
||||
auto it = ctx->weight_fingerprints.find(tensor->data);
|
||||
if (it != ctx->weight_fingerprints.end()) {
|
||||
return it->second;
|
||||
}
|
||||
GGML_ABORT("ggml-openvino: missing source identity for weight %s", tensor->name);
|
||||
}
|
||||
std::vector<ggml_openvino_source_mapping> mappings;
|
||||
return ggml_openvino_source_fingerprint(tensor->data, ggml_nbytes(tensor), mappings);
|
||||
}
|
||||
|
||||
static void ggml_backend_openvino_free(ggml_backend_t backend) {
|
||||
ggml_backend_openvino_context * ctx = (ggml_backend_openvino_context *) backend->context;
|
||||
|
||||
@@ -825,7 +899,7 @@ static const ggml_backend_i ggml_backend_openvino_interface = {
|
||||
};
|
||||
|
||||
int ggml_backend_openvino_get_device_count() {
|
||||
return 1;
|
||||
return (int) ggml_openvino_get_available_devices().size();
|
||||
}
|
||||
|
||||
static ggml_guid_t ggml_backend_openvino_guid(void) {
|
||||
@@ -884,10 +958,122 @@ namespace {
|
||||
struct ggml_backend_openvino_device_context {
|
||||
int device;
|
||||
std::string name;
|
||||
std::string ov_name; // OpenVINO device id: CPU, GPU, GPU.1, NPU, ...
|
||||
std::string description;
|
||||
size_t total_memory;
|
||||
};
|
||||
}
|
||||
|
||||
static bool ov_device_has_prefix(const std::string & s, const std::string & prefix) {
|
||||
return s.size() >= prefix.size() && std::equal(prefix.begin(), prefix.end(), s.begin());
|
||||
}
|
||||
|
||||
static bool ov_try_get_size_t_property(const std::string & device, const std::string & property, size_t & out) {
|
||||
try {
|
||||
const ov::Any value = ov_singleton_core().get_property(device, property);
|
||||
if (value.is<size_t>()) {
|
||||
out = value.as<size_t>();
|
||||
return true;
|
||||
}
|
||||
if (value.is<uint64_t>()) {
|
||||
out = (size_t) value.as<uint64_t>();
|
||||
return true;
|
||||
}
|
||||
if (value.is<unsigned long long>()) {
|
||||
out = (size_t) value.as<unsigned long long>();
|
||||
return true;
|
||||
}
|
||||
if (value.is<int64_t>()) {
|
||||
const int64_t v = value.as<int64_t>();
|
||||
if (v >= 0) {
|
||||
out = (size_t) v;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
} catch (...) {
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
// System memory available to new allocations (MemAvailable on Linux), SIZE_MAX if unknown
|
||||
static size_t ov_system_available_memory() {
|
||||
#ifdef _WIN32
|
||||
MEMORYSTATUSEX status;
|
||||
status.dwLength = sizeof(status);
|
||||
if (GlobalMemoryStatusEx(&status)) {
|
||||
return (size_t) status.ullAvailPhys;
|
||||
}
|
||||
#else
|
||||
if (FILE * f = fopen("/proc/meminfo", "r")) {
|
||||
char line[256];
|
||||
unsigned long long kb = 0;
|
||||
bool found = false;
|
||||
while (!found && fgets(line, sizeof(line), f)) {
|
||||
found = sscanf(line, "MemAvailable: %llu kB", &kb) == 1;
|
||||
}
|
||||
fclose(f);
|
||||
if (found) {
|
||||
return (size_t) std::min<unsigned long long>(kb * 1024, SIZE_MAX);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
return SIZE_MAX;
|
||||
}
|
||||
|
||||
// iGPU and NPU allocate from system RAM, so their free memory can't exceed what the OS has available
|
||||
static bool ov_device_shares_system_memory(const std::string & device) {
|
||||
if (ov_device_has_prefix(device, "NPU")) {
|
||||
return true;
|
||||
}
|
||||
if (!ov_device_has_prefix(device, "GPU")) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
return ov_singleton_core().get_property(device, ov::device::type) == ov::device::Type::INTEGRATED;
|
||||
} catch (...) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
// usm_host / usm_shared allocations live in system RAM on a discrete GPU
|
||||
static bool ov_gpu_stat_is_host_memory(const std::string & key) {
|
||||
return key == "usm_host" || key == "usm_shared";
|
||||
}
|
||||
|
||||
static bool ov_try_get_gpu_used_memory(const std::string & device, size_t & out) {
|
||||
out = 0;
|
||||
try {
|
||||
const ov::Any stats_any = ov_singleton_core().get_property(device, "GPU_MEMORY_STATISTICS");
|
||||
if (stats_any.is<std::map<std::string, uint64_t>>()) {
|
||||
const auto stats = stats_any.as<std::map<std::string, uint64_t>>();
|
||||
for (const auto & kv : stats) {
|
||||
if (!ov_gpu_stat_is_host_memory(kv.first)) {
|
||||
out += (size_t) kv.second;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (stats_any.is<ov::AnyMap>()) {
|
||||
const auto stats = stats_any.as<ov::AnyMap>();
|
||||
for (const auto & kv : stats) {
|
||||
if (ov_gpu_stat_is_host_memory(kv.first)) {
|
||||
continue;
|
||||
}
|
||||
if (kv.second.is<size_t>()) {
|
||||
out += kv.second.as<size_t>();
|
||||
} else if (kv.second.is<uint64_t>()) {
|
||||
out += (size_t) kv.second.as<uint64_t>();
|
||||
} else if (kv.second.is<unsigned long long>()) {
|
||||
out += (size_t) kv.second.as<unsigned long long>();
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
} catch (...) {
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static const char * ggml_backend_openvino_device_get_name(ggml_backend_dev_t dev) {
|
||||
ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context;
|
||||
return ctx->name.c_str();
|
||||
@@ -899,27 +1085,45 @@ static const char * ggml_backend_openvino_device_get_description(ggml_backend_de
|
||||
}
|
||||
|
||||
static void ggml_backend_openvino_device_get_memory(ggml_backend_dev_t dev, size_t * free, size_t * total) {
|
||||
ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context;
|
||||
|
||||
// total_memory is only set for GPU/NPU; used = this process's OpenVINO allocations on the device
|
||||
size_t used = 0;
|
||||
const bool known = ctx->total_memory > 0 &&
|
||||
(ov_device_has_prefix(ctx->ov_name, "GPU") ?
|
||||
ov_try_get_gpu_used_memory(ctx->ov_name, used) :
|
||||
ov_try_get_size_t_property(ctx->ov_name, "NPU_DEVICE_ALLOC_MEM_SIZE", used));
|
||||
if (known) {
|
||||
*total = ctx->total_memory;
|
||||
*free = (used >= *total) ? 0 : (*total - used);
|
||||
} else {
|
||||
// CPU, or a plugin without memory properties: report system memory
|
||||
#ifdef _WIN32
|
||||
MEMORYSTATUSEX status;
|
||||
status.dwLength = sizeof(status);
|
||||
GlobalMemoryStatusEx(&status);
|
||||
*total = status.ullTotalPhys;
|
||||
*free = status.ullAvailPhys;
|
||||
MEMORYSTATUSEX status;
|
||||
status.dwLength = sizeof(status);
|
||||
GlobalMemoryStatusEx(&status);
|
||||
*total = status.ullTotalPhys;
|
||||
*free = status.ullAvailPhys;
|
||||
#else
|
||||
long pages = sysconf(_SC_PHYS_PAGES);
|
||||
long page_size = sysconf(_SC_PAGE_SIZE);
|
||||
*total = pages * page_size;
|
||||
long pages = sysconf(_SC_PHYS_PAGES);
|
||||
long page_size = sysconf(_SC_PAGE_SIZE);
|
||||
*total = pages * page_size;
|
||||
|
||||
// "free" system memory is ill-defined, for practical purposes assume that all of it is free:
|
||||
*free = *total;
|
||||
// "free" system memory is ill-defined, for practical purposes assume that all of it is free:
|
||||
*free = *total;
|
||||
#endif // _WIN32
|
||||
}
|
||||
|
||||
GGML_UNUSED(dev);
|
||||
if (ov_device_shares_system_memory(ctx->ov_name)) {
|
||||
*free = std::min(*free, ov_system_available_memory());
|
||||
}
|
||||
}
|
||||
|
||||
static enum ggml_backend_dev_type ggml_backend_openvino_device_get_type(ggml_backend_dev_t dev) {
|
||||
GGML_UNUSED(dev);
|
||||
return GGML_BACKEND_DEVICE_TYPE_GPU;
|
||||
ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context;
|
||||
// Only the device selected by GGML_OPENVINO_DEVICE is offered for offload. The others are
|
||||
// registered for discovery (--list-devices) only; llama.cpp skips IGPU devices when a GPU exists.
|
||||
return ctx->ov_name == ggml_openvino_get_device_name() ? GGML_BACKEND_DEVICE_TYPE_GPU : GGML_BACKEND_DEVICE_TYPE_IGPU;
|
||||
}
|
||||
|
||||
static void ggml_backend_openvino_device_get_props(ggml_backend_dev_t dev, ggml_backend_dev_props * props) {
|
||||
@@ -940,6 +1144,12 @@ static void ggml_backend_openvino_device_get_props(ggml_backend_dev_t dev, ggml_
|
||||
static ggml_backend_t ggml_backend_openvino_device_init(ggml_backend_dev_t dev, const char * params) {
|
||||
GGML_UNUSED(params);
|
||||
ggml_backend_openvino_device_context * ctx = (ggml_backend_openvino_device_context *) dev->context;
|
||||
if (ctx->ov_name != ggml_openvino_get_device_name()) {
|
||||
// Not an error: test-backend-ops initializes every device
|
||||
GGML_LOG_WARN("%s: %s (OpenVINO %s) is not the selected device, no ops will run on it; "
|
||||
"set GGML_OPENVINO_DEVICE=%s to use it\n",
|
||||
__func__, ctx->name.c_str(), ctx->ov_name.c_str(), ctx->ov_name.c_str());
|
||||
}
|
||||
return ggml_backend_openvino_init(ctx->device);
|
||||
}
|
||||
|
||||
@@ -1159,7 +1369,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
if (op->type == GGML_TYPE_I64) {
|
||||
return {false, "CONCAT with I64 type is not supported"};
|
||||
}
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16 && has_view_op_input(op)) {
|
||||
if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_BF16 && has_view_op_input(op)) {
|
||||
return {false, "CONCAT with BF16 type and VIEW input is not supported on GPU"};
|
||||
}
|
||||
break;
|
||||
@@ -1183,7 +1393,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
if (op->ne[3] != 1) {
|
||||
return {false, "GET_ROWS/SET_ROWS with ne[3] != 1 (ne[3]=" + std::to_string(op->ne[3]) + ") is not supported"};
|
||||
}
|
||||
if (op->op == GGML_OP_GET_ROWS && ggml_openvino_get_device_name() == "GPU" &&
|
||||
if (op->op == GGML_OP_GET_ROWS && ggml_openvino_is_gpu() &&
|
||||
op->src[0]->type == GGML_TYPE_BF16) {
|
||||
return {false, "GET_ROWS with BF16 src0 is not supported on GPU"};
|
||||
}
|
||||
@@ -1246,26 +1456,37 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
// The GPU plugin can fuse broadcast DIV into the preceding FFN GEMM path
|
||||
// and produce infs for per-channel scale vectors. Keep those DIVs on CPU
|
||||
// until the fused GPU kernel is reliable. (falied case llama-arch-test mpt)
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->src[1]->ne[0] == op->ne[0] &&
|
||||
if (ggml_openvino_is_gpu() && op->src[1]->ne[0] == op->ne[0] &&
|
||||
op->src[1]->ne[1] == 1 && op->src[1]->ne[2] == 1 && op->src[1]->ne[3] == 1) {
|
||||
return {false, "DIV per-channel scale broadcast is not supported on GPU"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_POOL_2D: {
|
||||
const auto& name = ggml_openvino_get_device_name();
|
||||
if (name == "GPU") {
|
||||
if (ggml_openvino_is_gpu()) {
|
||||
const int32_t * params = op->op_params;
|
||||
const int k0 = params[1];
|
||||
const int k1 = params[2];
|
||||
const int p0 = params[5];
|
||||
const int p1 = params[6];
|
||||
if ((p0 > 0 || p1 > 0) && (k0 < 3 || k1 < 3)) {
|
||||
return {false, "POOL_2D with padding and kernel size < 3 is not supported on " + name};
|
||||
return {false, "POOL_2D with padding and kernel size < 3 is not supported on " + ggml_openvino_get_device_name()};
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_SUM: {
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "SUM with PERMUTE input is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_MEAN: {
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE && op->src[0]->src[0] != nullptr && op->src[0]->src[0]->op == GGML_OP_VIEW) {
|
||||
return {false, "MEAN with PERMUTE of VIEW input is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_SUM_ROWS: {
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "SUM_ROWS with PERMUTE input is not supported"};
|
||||
@@ -1303,7 +1524,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
break;
|
||||
}
|
||||
case GGML_OP_PERMUTE: {
|
||||
if (op->type == GGML_TYPE_BF16 && ggml_openvino_get_device_name() == "GPU") {
|
||||
if (op->type == GGML_TYPE_BF16 && ggml_openvino_is_gpu()) {
|
||||
return {false, "PERMUTE with BF16 type is not supported on GPU"};
|
||||
}
|
||||
break;
|
||||
@@ -1312,7 +1533,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
if (op->src[0]->type != GGML_TYPE_BF16 && op->src[1]->type == GGML_TYPE_BF16) {
|
||||
return {false, "CPY with BF16 src[1] type is not supported"};
|
||||
}
|
||||
if (ggml_openvino_get_device_name() == "NPU" && (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) {
|
||||
if (ggml_openvino_is_npu() && (op->src[0]->type == GGML_TYPE_BF16 || op->src[1]->type == GGML_TYPE_BF16)) {
|
||||
return {false, "CPY with BF16 is not supported is not supported on NPU"};
|
||||
}
|
||||
// CPY to a quantized destination (e.g. f32 -> q4_0) is numerically unstable with OpenVINO backend.
|
||||
@@ -1337,13 +1558,13 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
break;
|
||||
}
|
||||
case GGML_OP_MUL_MAT: {
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[1] != nullptr &&
|
||||
if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[1] != nullptr &&
|
||||
ggml_is_quantized(op->src[0]->type) && strcmp(op->src[0]->name, "a") == 0 &&
|
||||
strcmp(op->src[1]->name, "b") == 0 && op->src[0]->ne[1] == 1 && op->src[1]->ne[1] == 64 &&
|
||||
op->src[0]->ne[0] == 256 && op->src[1]->ne[0] == 256) {
|
||||
return {false, "MUL_MAT quantized benchmark test case on GPU is not supported"};
|
||||
}
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_F32 && op->ne[0] == 1 && op->ne[1] == 1 &&
|
||||
if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_F32 && op->ne[0] == 1 && op->ne[1] == 1 &&
|
||||
(op->src[0]->buffer == nullptr || op->src[0]->buffer->usage != GGML_BACKEND_BUFFER_USAGE_WEIGHTS)) {
|
||||
return {false, "MUL_MAT scalar dot product with non-weight src[0] on GPU is not supported"};
|
||||
}
|
||||
@@ -1363,7 +1584,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
return {false, "MUL_MAT_ID with single-expert or empty ne[2] <= 1 (ne[2]=" +
|
||||
std::to_string(op->src[0]->ne[2]) + ") is not supported"};
|
||||
}
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && !ggml_is_quantized(op->src[0]->type)) {
|
||||
if (ggml_openvino_is_gpu() && op->src[0] != nullptr && !ggml_is_quantized(op->src[0]->type)) {
|
||||
return {false, "MUL_MAT_ID with non-quantized weights on GPU is not supported"};
|
||||
}
|
||||
// The GPU plugin's GatherMatmul returns wrong values for the layouts test-backend-ops
|
||||
@@ -1372,55 +1593,21 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
// The same graph is correct on the CPU plugin, and correct on GPU for every real model,
|
||||
// which always feeds experts from a bound tensor buffer. Standalone op-test tensors have
|
||||
// no buffer at all, so use that to exclude them and let the scheduler run them on CPU.
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->buffer == nullptr) {
|
||||
if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[0]->buffer == nullptr) {
|
||||
return {false, "MUL_MAT_ID with unbound expert tensors on GPU is not supported"};
|
||||
}
|
||||
// Only MXFP4 still needs the large-temporary guard; every other quantized type goes
|
||||
// through GatherMatmul, which never materializes the selected expert weights.
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_MXFP4 &&
|
||||
if (ggml_openvino_is_gpu() && op->src[0] != nullptr && op->src[0]->type == GGML_TYPE_MXFP4 &&
|
||||
mul_mat_id_requires_large_tmp(op)) {
|
||||
return {false, "MUL_MAT_ID with MXFP4 weights requires large temporary on GPU"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_ROPE: {
|
||||
const int32_t * op_params = op->op_params;
|
||||
const int n_dims = op_params[1];
|
||||
const int mode = op_params[2];
|
||||
const int64_t n_offs = op_params[15];
|
||||
if (mode != GGML_ROPE_TYPE_NORMAL && mode != GGML_ROPE_TYPE_NEOX && mode != GGML_ROPE_TYPE_IMROPE) {
|
||||
return {false, "ROPE with mode " + std::to_string(mode) + " is not supported"};
|
||||
}
|
||||
if (n_offs < 0 || (n_offs % 2) != 0) {
|
||||
return {false, "ROPE with invalid n_offs=" + std::to_string(n_offs)};
|
||||
}
|
||||
const int64_t head_dim = op->src[0]->ne[0];
|
||||
const int64_t rope_dims = n_dims == 0 ? head_dim : n_dims;
|
||||
if (rope_dims <= 0 || rope_dims + n_offs > head_dim || (rope_dims % 2) != 0) {
|
||||
return {false, "ROPE with n_dims=" + std::to_string(n_dims) + ", n_offs=" + std::to_string(n_offs) +
|
||||
", head_dim=" + std::to_string(head_dim) + " is not supported"};
|
||||
}
|
||||
if (op->type != GGML_TYPE_F32 && op->type != GGML_TYPE_F16) {
|
||||
return {false, "ROPE with type " + std::string(ggml_type_name(op->type)) + " is not supported"};
|
||||
}
|
||||
if (op->view_src != nullptr && !ggml_is_contiguous(op->src[0])) {
|
||||
return {false, "ROPE on VIEW / non-contiguous input is not supported"};
|
||||
}
|
||||
if (op->src[0]->ne[3] > 1) {
|
||||
// translate_rope's cos/sin tables cover one sequence only; ne[3] > 1 fails to broadcast.
|
||||
return {false, "ROPE with multiple sequences (ne[3]=" + std::to_string(op->src[0]->ne[3]) +
|
||||
") is not supported"};
|
||||
}
|
||||
float freq_scale;
|
||||
float ext_factor;
|
||||
float attn_factor;
|
||||
memcpy(&freq_scale, op_params + 6, sizeof(float));
|
||||
memcpy(&ext_factor, op_params + 7, sizeof(float));
|
||||
memcpy(&attn_factor, op_params + 8, sizeof(float));
|
||||
if (mode == GGML_ROPE_TYPE_IMROPE &&
|
||||
(op->src[2] != nullptr || freq_scale != 1.0f || ext_factor != 0.0f || attn_factor != 1.0f)) {
|
||||
return {false, "IMROPE with freq_factors, freq_scale, ext_factor, or attn_factor is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_TRANSPOSE: {
|
||||
@@ -1430,7 +1617,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
break;
|
||||
}
|
||||
case GGML_OP_REPEAT: {
|
||||
if (ggml_openvino_get_device_name() == "GPU" && op->type == GGML_TYPE_BF16) {
|
||||
if (ggml_openvino_is_gpu() && op->type == GGML_TYPE_BF16) {
|
||||
return {false, "REPEAT with BF16 type is not supported on GPU"};
|
||||
}
|
||||
break;
|
||||
@@ -1438,7 +1625,7 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
case GGML_OP_GATED_DELTA_NET: {
|
||||
// enable after https://github.com/openvinotoolkit/openvino/pull/35917 is included in OV release
|
||||
// return true;
|
||||
// if (ggml_openvino_get_device_name() == "GPU" && op->src[0]->ne[2] > 1) {
|
||||
// if (ggml_openvino_is_gpu() && op->src[0]->ne[2] > 1) {
|
||||
// // CVS-186471
|
||||
// return true;
|
||||
// }
|
||||
@@ -1469,6 +1656,84 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_CONV_2D:
|
||||
case GGML_OP_CONV_2D_DW: {
|
||||
if (op->src[0]->ne[0] <= 0 || op->src[0]->ne[1] <= 0) {
|
||||
return {false, "CONV_2D kernel size must be positive"};
|
||||
}
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE || op->src[1]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "CONV_2D with PERMUTE input is not supported"};
|
||||
}
|
||||
if (has_non_contiguous_view_input(op)) {
|
||||
return {false, "CONV_2D with non-contiguous view input is not supported"};
|
||||
}
|
||||
const int32_t * params = op->op_params;
|
||||
const int p0 = params[2];
|
||||
const int p1 = params[3];
|
||||
const int d0 = params[4];
|
||||
const int d1 = params[5];
|
||||
const int64_t dilated_kw = (int64_t) d0 * (op->src[0]->ne[0] - 1) + 1;
|
||||
const int64_t dilated_kh = (int64_t) d1 * (op->src[0]->ne[1] - 1) + 1;
|
||||
const int64_t padded_w = op->src[1]->ne[0] + 2 * p0;
|
||||
const int64_t padded_h = op->src[1]->ne[1] + 2 * p1;
|
||||
if (padded_w < dilated_kw || padded_h < dilated_kh) {
|
||||
return {false, "CONV_2D padded input is smaller than kernel"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_CONV_3D: {
|
||||
if (op->src[0]->ne[0] <= 0 || op->src[0]->ne[1] <= 0 || op->src[0]->ne[2] <= 0) {
|
||||
return {false, "CONV_3D kernel size must be positive"};
|
||||
}
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE || op->src[1]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "CONV_3D with PERMUTE input is not supported"};
|
||||
}
|
||||
if (has_non_contiguous_view_input(op)) {
|
||||
return {false, "CONV_3D with non-contiguous view input is not supported"};
|
||||
}
|
||||
const int32_t * params = op->op_params;
|
||||
const int p0 = params[3];
|
||||
const int p1 = params[4];
|
||||
const int p2 = params[5];
|
||||
const int d0 = params[6];
|
||||
const int d1 = params[7];
|
||||
const int d2 = params[8];
|
||||
const int64_t dilated_kw = (int64_t) d0 * (op->src[0]->ne[0] - 1) + 1;
|
||||
const int64_t dilated_kh = (int64_t) d1 * (op->src[0]->ne[1] - 1) + 1;
|
||||
const int64_t dilated_kd = (int64_t) d2 * (op->src[0]->ne[2] - 1) + 1;
|
||||
const int64_t padded_w = op->src[1]->ne[0] + 2 * p0;
|
||||
const int64_t padded_h = op->src[1]->ne[1] + 2 * p1;
|
||||
const int64_t padded_d = op->src[1]->ne[2] + 2 * p2;
|
||||
if (padded_w < dilated_kw || padded_h < dilated_kh || padded_d < dilated_kd) {
|
||||
return {false, "CONV_3D padded input is smaller than kernel"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_CONV_TRANSPOSE_1D:
|
||||
case GGML_OP_CONV_TRANSPOSE_2D: {
|
||||
if (op->src[0]->ne[0] <= 0 || op->src[0]->ne[1] <= 0) {
|
||||
return {false, "CONV_TRANSPOSE kernel size must be positive"};
|
||||
}
|
||||
if (op->src[0]->op == GGML_OP_PERMUTE || op->src[1]->op == GGML_OP_PERMUTE) {
|
||||
return {false, "CONV_TRANSPOSE with PERMUTE input is not supported"};
|
||||
}
|
||||
if (has_non_contiguous_view_input(op)) {
|
||||
return {false, "CONV_TRANSPOSE with non-contiguous view input is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_IM2COL: {
|
||||
if (op->src[0]->ne[0] <= 0 || op->src[0]->ne[1] <= 0) {
|
||||
return {false, "IM2COL kernel size must be positive"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
case GGML_OP_IM2COL_3D: {
|
||||
if (op->src[0]->ne[0] <= 0 || op->src[0]->ne[1] <= 0 || op->src[0]->ne[2] <= 0) {
|
||||
return {false, "IM2COL_3D kernel size must be positive"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
@@ -1478,6 +1743,24 @@ static ggml_openvino_op_support is_op_supported_case(const ggml_tensor * op) {
|
||||
static ggml_openvino_op_support ggml_backend_openvino_device_supports_op_impl(ggml_backend_dev_t dev, const ggml_tensor * op) {
|
||||
GGML_ASSERT(dev->reg != nullptr);
|
||||
|
||||
ggml_backend_openvino_device_context * dev_ctx = (ggml_backend_openvino_device_context *) dev->context;
|
||||
if (dev_ctx->ov_name != ggml_openvino_get_device_name()) {
|
||||
// Data placed on a non-selected device (e.g. with -dev) can never run here; stop with a hint
|
||||
// instead of the generic scheduler abort. Unallocated tensors (test-backend-ops) pass through.
|
||||
for (int i = -1; i < GGML_MAX_SRC; i++) {
|
||||
const ggml_tensor * t = i < 0 ? op : op->src[i];
|
||||
ggml_backend_buffer_t buf = t == nullptr ? nullptr : (t->view_src ? t->view_src->buffer : t->buffer);
|
||||
if (buf != nullptr &&
|
||||
(ggml_backend_buft_is_openvino(buf->buft) || ggml_backend_buft_is_openvino_host(buf->buft)) &&
|
||||
((ggml_backend_openvino_buffer_type_context *) buf->buft->context)->device == dev_ctx->device) {
|
||||
GGML_ABORT("%s is not the selected OpenVINO device (%s). The OpenVINO device is chosen with the "
|
||||
"GGML_OPENVINO_DEVICE environment variable, not -dev: set GGML_OPENVINO_DEVICE=%s",
|
||||
dev_ctx->name.c_str(), ggml_openvino_get_device_name().c_str(), dev_ctx->ov_name.c_str());
|
||||
}
|
||||
}
|
||||
return {false, "device is not the selected OpenVINO device"};
|
||||
}
|
||||
|
||||
static std::unordered_set<ggml_type> supported_types{
|
||||
GGML_TYPE_F32, GGML_TYPE_F16, GGML_TYPE_BF16, GGML_TYPE_I64, GGML_TYPE_I32, GGML_TYPE_Q4_0,
|
||||
GGML_TYPE_Q4_1, GGML_TYPE_Q4_K, GGML_TYPE_Q5_1, GGML_TYPE_Q5_K, GGML_TYPE_Q8_0, GGML_TYPE_Q6_K,
|
||||
@@ -1527,8 +1810,9 @@ static ggml_openvino_op_support ggml_backend_openvino_device_supports_op_impl(gg
|
||||
if (!supported) {
|
||||
return {false, "unary op " + std::string(ggml_unary_op_name(ggml_get_unary_op(op))) + " has no op translator"};
|
||||
}
|
||||
if (ggml_get_unary_op(op) == GGML_UNARY_OP_EXP && op->type == GGML_TYPE_F32) {
|
||||
return {false, "UNARY_EXP with F32 type is not supported"};
|
||||
if (op->type == GGML_TYPE_F32 && (ggml_get_unary_op(op) == GGML_UNARY_OP_EXP ||
|
||||
ggml_get_unary_op(op) == GGML_UNARY_OP_EXPM1)) {
|
||||
return {false, "UNARY_EXP / UNARY_EXPM1 with F32 type is not supported"};
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -1665,15 +1949,26 @@ GGML_BACKEND_API ggml_backend_reg_t ggml_backend_openvino_reg(void) {
|
||||
std::lock_guard<std::mutex> lock(mutex);
|
||||
if (!initialized) {
|
||||
ggml_openvino_init();
|
||||
const std::vector<std::string> openvino_devices = ggml_openvino_get_available_devices();
|
||||
|
||||
ggml_backend_openvino_reg_context * ctx = new ggml_backend_openvino_reg_context;
|
||||
|
||||
for (int i = 0; i < ggml_backend_openvino_get_device_count(); i++) {
|
||||
ggml_backend_openvino_device_context * dev_ctx = new ggml_backend_openvino_device_context;
|
||||
dev_ctx->device = i;
|
||||
// Not the raw OpenVINO id: "CPU" would shadow the ggml CPU backend in ggml_backend_dev_by_name
|
||||
dev_ctx->name = GGML_OPENVINO_NAME + std::to_string(i);
|
||||
|
||||
dev_ctx->description = ov::get_openvino_version().description;
|
||||
dev_ctx->ov_name = openvino_devices[i];
|
||||
// The device is chosen with GGML_OPENVINO_DEVICE, not -dev, so show the value to set
|
||||
dev_ctx->description = "GGML_OPENVINO_DEVICE=" + dev_ctx->ov_name +
|
||||
(dev_ctx->ov_name == ggml_openvino_get_device_name() ? " (selected)" : "") +
|
||||
" - " + ggml_openvino_get_device_description(dev_ctx->ov_name);
|
||||
dev_ctx->total_memory = 0;
|
||||
if (ov_device_has_prefix(dev_ctx->ov_name, "GPU")) {
|
||||
ov_try_get_size_t_property(dev_ctx->ov_name, "GPU_DEVICE_TOTAL_MEM_SIZE", dev_ctx->total_memory);
|
||||
} else if (ov_device_has_prefix(dev_ctx->ov_name, "NPU")) {
|
||||
ov_try_get_size_t_property(dev_ctx->ov_name, "NPU_DEVICE_TOTAL_MEM_SIZE", dev_ctx->total_memory);
|
||||
}
|
||||
|
||||
ggml_backend_dev_t dev =
|
||||
new ggml_backend_device{/* .interface = */ ggml_backend_openvino_device_interface,
|
||||
|
||||
@@ -6,9 +6,13 @@
|
||||
#include "ggml-openvino-extra.h"
|
||||
|
||||
#include <cerrno>
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <iomanip>
|
||||
#include <limits>
|
||||
#include <sstream>
|
||||
#include <openvino/core/version.hpp>
|
||||
#include <string>
|
||||
#include <sys/stat.h>
|
||||
@@ -16,7 +20,19 @@
|
||||
#include <vector>
|
||||
|
||||
#if defined(_WIN32)
|
||||
# define WIN32_LEAN_AND_MEAN
|
||||
# ifndef NOMINMAX
|
||||
# define NOMINMAX
|
||||
# endif
|
||||
# include <windows.h>
|
||||
# include <psapi.h>
|
||||
# include <direct.h>
|
||||
# include <process.h>
|
||||
#else
|
||||
# include <unistd.h>
|
||||
#endif
|
||||
#ifdef __linux__
|
||||
# include <sys/sysmacros.h>
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
@@ -37,10 +53,7 @@ inline uint64_t fnv1a_u64(uint64_t h, uint64_t v) {
|
||||
|
||||
constexpr uint64_t FNV_OFFSET = 0xcbf29ce484222325ull;
|
||||
|
||||
// Bytes sampled from each end of a weight tensor for the sampled hash. The whole
|
||||
// model is never hashed (that would cost seconds every run); instead we sample a
|
||||
// bounded window from the head and tail of each weight's bytes. The manifest
|
||||
// re-verify (same sample) guards the residual collision risk.
|
||||
// Fallback when source-file identity is unavailable outside cache-only mode.
|
||||
constexpr size_t WEIGHT_SAMPLE_BYTES = 4096;
|
||||
|
||||
// Is this src a model weight, mirroring create_weight_nodes()'s selection:
|
||||
@@ -52,8 +65,7 @@ bool is_weight_src(const ggml_tensor * src) {
|
||||
return src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS || ggml_is_quantized(src->type);
|
||||
}
|
||||
|
||||
// Per-weight sampled fingerprint: identity (name/shape/type) + a bounded byte
|
||||
// sample. Returns FNV offset basis if data is unavailable (kept deterministic).
|
||||
// Weight metadata and source identity; do not read repacked or unloaded buffers.
|
||||
uint64_t weight_fingerprint(const ggml_tensor * t) {
|
||||
uint64_t h = FNV_OFFSET;
|
||||
h = fnv1a(h, t->name, strlen(t->name));
|
||||
@@ -63,15 +75,7 @@ uint64_t weight_fingerprint(const ggml_tensor * t) {
|
||||
h = fnv1a_u64(h, static_cast<uint64_t>(t->type));
|
||||
const size_t nbytes = ggml_nbytes(t);
|
||||
h = fnv1a_u64(h, nbytes);
|
||||
if (t->data != nullptr && nbytes > 0) {
|
||||
const size_t head = nbytes < WEIGHT_SAMPLE_BYTES ? nbytes : WEIGHT_SAMPLE_BYTES;
|
||||
h = fnv1a(h, t->data, head);
|
||||
if (nbytes > WEIGHT_SAMPLE_BYTES) {
|
||||
const size_t tail = nbytes < 2 * WEIGHT_SAMPLE_BYTES ? nbytes - WEIGHT_SAMPLE_BYTES : WEIGHT_SAMPLE_BYTES;
|
||||
h = fnv1a(h, static_cast<const uint8_t *>(t->data) + (nbytes - tail), tail);
|
||||
}
|
||||
}
|
||||
return h;
|
||||
return fnv1a_u64(h, ggml_backend_openvino_weight_fingerprint(t));
|
||||
}
|
||||
|
||||
// Walk the cgraph and invoke fn(weight_tensor) for each distinct weight, in node
|
||||
@@ -156,12 +160,162 @@ bool make_dirs(const std::string & path) {
|
||||
|
||||
} // namespace
|
||||
|
||||
bool ggml_openvino_model_cache_only() {
|
||||
return ggml_openvino_getenv_int("GGML_OPENVINO_COMPILED_MODEL_CACHE_ONLY") != 0;
|
||||
}
|
||||
|
||||
static const char * cache_settings[] = {
|
||||
"GGML_OPENVINO_REQUANT_KQUANT",
|
||||
"GGML_OPENVINO_NATIVE_SOFTPLUS",
|
||||
"GGML_OPENVINO_DISABLE_KV_SLICE",
|
||||
"GGML_OPENVINO_MANUAL_GQA_ATTN",
|
||||
"GGML_OPENVINO_STATEFUL_EXECUTION",
|
||||
"GGML_OPENVINO_DISABLE_KV_STATE_RELAYOUT",
|
||||
"GGML_OPENVINO_DISABLE_REMOTE_OUTPUTS",
|
||||
"GGML_OPENVINO_REDUCE_COMPILE_MEM",
|
||||
"GGML_OPENVINO_MEMORY_OPTIMIZE",
|
||||
"GGML_OPENVINO_PROFILING",
|
||||
};
|
||||
|
||||
void ggml_openvino_model_cache_init() {
|
||||
const bool cache_only = ggml_openvino_model_cache_only();
|
||||
const std::string dir = ggml_openvino_model_cache_dir();
|
||||
if (dir.empty()) {
|
||||
if (cache_only) {
|
||||
GGML_ABORT("ggml-openvino: cache-only mode requires GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR");
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (cache_only && (ggml_openvino_is_npu() || ggml_openvino_getenv_int("GGML_OPENVINO_FORCE_STATIC") ||
|
||||
ggml_openvino_getenv_int("GGML_OPENVINO_DISABLE_CACHE") ||
|
||||
ggml_openvino_getenv_int("GGML_OPENVINO_ENABLE_FALLBACK"))) {
|
||||
GGML_ABORT("ggml-openvino: cache-only mode requires dynamic CPU/GPU execution with caching and without fallback");
|
||||
}
|
||||
#if !defined(__linux__) && !defined(_WIN32)
|
||||
if (cache_only) {
|
||||
GGML_ABORT("ggml-openvino: cache-only mmap identification requires Linux or Windows");
|
||||
}
|
||||
#endif
|
||||
if (cache_only) {
|
||||
auto & config = ggml_openvino_get_device_config();
|
||||
config.environment_variables.erase("GGML_OPENVINO_SPILL_DIR");
|
||||
config.environment_variables["GGML_OPENVINO_RELEASE_WEIGHTS"] = "0";
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t ggml_openvino_source_fingerprint(const void * data, size_t size, std::vector<ggml_openvino_source_mapping> & mappings) {
|
||||
const uintptr_t address = reinterpret_cast<uintptr_t>(data);
|
||||
auto contains = [&](const ggml_openvino_source_mapping & m) {
|
||||
return address >= m.begin && address < m.end && size <= m.end - address;
|
||||
};
|
||||
auto fingerprint = [&](const ggml_openvino_source_mapping & m) {
|
||||
return fnv1a_u64(m.identity, m.offset + address - m.begin);
|
||||
};
|
||||
for (const auto & m : mappings) {
|
||||
if (contains(m)) {
|
||||
return fingerprint(m);
|
||||
}
|
||||
}
|
||||
#ifdef __linux__
|
||||
std::ifstream maps("/proc/self/maps");
|
||||
std::string line;
|
||||
while (std::getline(maps, line)) {
|
||||
unsigned long long begin, end, offset, inode;
|
||||
unsigned int dev_major, dev_minor;
|
||||
char permissions[5];
|
||||
int path_start = 0;
|
||||
if (sscanf(line.c_str(), "%llx-%llx %4s %llx %x:%x %llu %n", &begin, &end, permissions,
|
||||
&offset, &dev_major, &dev_minor, &inode, &path_start) != 7 || inode == 0) {
|
||||
continue;
|
||||
}
|
||||
ggml_openvino_source_mapping m{uintptr_t(begin), uintptr_t(end), offset, FNV_OFFSET};
|
||||
if (!contains(m)) {
|
||||
continue;
|
||||
}
|
||||
struct stat st;
|
||||
const std::string path = line.substr(path_start);
|
||||
if (stat(path.c_str(), &st) != 0 || !S_ISREG(st.st_mode) || uint64_t(st.st_ino) != inode ||
|
||||
major(st.st_dev) != dev_major || minor(st.st_dev) != dev_minor) {
|
||||
break;
|
||||
}
|
||||
m.identity = fnv1a_u64(m.identity, st.st_dev);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_ino);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_size);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_mtim.tv_sec);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_mtim.tv_nsec);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_ctim.tv_sec);
|
||||
m.identity = fnv1a_u64(m.identity, st.st_ctim.tv_nsec);
|
||||
mappings.push_back(m);
|
||||
return fingerprint(m);
|
||||
}
|
||||
#elif defined(_WIN32)
|
||||
MEMORY_BASIC_INFORMATION memory;
|
||||
if (VirtualQuery(data, &memory, sizeof(memory)) == sizeof(memory) && memory.Type == MEM_MAPPED) {
|
||||
std::wstring name(MAX_PATH, L'\0');
|
||||
DWORD length = 0;
|
||||
while (name.size() <= 32768) {
|
||||
length = GetMappedFileNameW(GetCurrentProcess(), const_cast<void *>(data), name.data(), static_cast<DWORD>(name.size()));
|
||||
if (length == 0 || length < name.size() - 1) {
|
||||
break;
|
||||
}
|
||||
name.resize(name.size() * 2);
|
||||
}
|
||||
if (length > 0 && length < name.size() - 1) {
|
||||
name.resize(length);
|
||||
const std::wstring path = L"\\\\?\\GLOBALROOT" + name;
|
||||
HANDLE file = CreateFileW(path.c_str(), FILE_READ_ATTRIBUTES,
|
||||
FILE_SHARE_READ | FILE_SHARE_WRITE | FILE_SHARE_DELETE, nullptr,
|
||||
OPEN_EXISTING, FILE_ATTRIBUTE_NORMAL, nullptr);
|
||||
if (file != INVALID_HANDLE_VALUE) {
|
||||
BY_HANDLE_FILE_INFORMATION info;
|
||||
FILE_BASIC_INFO basic;
|
||||
const bool valid = GetFileInformationByHandle(file, &info) &&
|
||||
GetFileInformationByHandleEx(file, FileBasicInfo, &basic, sizeof(basic));
|
||||
CloseHandle(file);
|
||||
if (valid) {
|
||||
const uint64_t file_size = (uint64_t(info.nFileSizeHigh) << 32) | info.nFileSizeLow;
|
||||
const uintptr_t begin = reinterpret_cast<uintptr_t>(memory.AllocationBase);
|
||||
if (file_size <= std::numeric_limits<uintptr_t>::max() - begin) {
|
||||
ggml_openvino_source_mapping m{begin, begin + static_cast<uintptr_t>(file_size), 0,
|
||||
fnv1a(FNV_OFFSET, "win32", 5)};
|
||||
if (contains(m)) {
|
||||
m.identity = fnv1a_u64(m.identity, info.dwVolumeSerialNumber);
|
||||
m.identity = fnv1a_u64(m.identity, (uint64_t(info.nFileIndexHigh) << 32) | info.nFileIndexLow);
|
||||
m.identity = fnv1a_u64(m.identity, file_size);
|
||||
m.identity = fnv1a_u64(m.identity, (uint64_t(info.ftLastWriteTime.dwHighDateTime) << 32) |
|
||||
info.ftLastWriteTime.dwLowDateTime);
|
||||
m.identity = fnv1a_u64(m.identity, static_cast<uint64_t>(basic.ChangeTime.QuadPart));
|
||||
mappings.push_back(m);
|
||||
return fingerprint(m);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
#endif
|
||||
if (ggml_openvino_model_cache_only()) {
|
||||
GGML_ABORT("ggml-openvino: could not identify mapped GGUF weight; use --load-mode mmap");
|
||||
}
|
||||
uint64_t h = FNV_OFFSET;
|
||||
const size_t head = std::min(size, WEIGHT_SAMPLE_BYTES);
|
||||
h = fnv1a(h, data, head);
|
||||
if (size > head) {
|
||||
const size_t tail = std::min(size - head, WEIGHT_SAMPLE_BYTES);
|
||||
h = fnv1a(h, static_cast<const uint8_t *>(data) + size - tail, tail);
|
||||
}
|
||||
return h;
|
||||
}
|
||||
|
||||
std::string ggml_openvino_model_cache_dir() {
|
||||
const char * dir = ggml_openvino_getenv_str("GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR");
|
||||
if (!dir || strlen(dir) == 0) {
|
||||
return std::string();
|
||||
}
|
||||
std::string path(dir);
|
||||
if (ggml_openvino_model_cache_only()) {
|
||||
return path;
|
||||
}
|
||||
// Create the cache directory (and parents) on first use so callers don't
|
||||
// have to pre-create it; a missing dir would otherwise silently disable the
|
||||
// cache (manifest/blob writes fail with no directory to write into).
|
||||
@@ -173,13 +327,36 @@ std::string ggml_openvino_model_cache_dir() {
|
||||
return path;
|
||||
}
|
||||
|
||||
std::string ggml_openvino_model_cache_temp_path(const std::string & path) {
|
||||
#ifdef _WIN32
|
||||
const int pid = _getpid();
|
||||
#else
|
||||
const int pid = getpid();
|
||||
#endif
|
||||
return path + ".tmp." + std::to_string(pid) + "." + std::to_string(ggml_time_us());
|
||||
}
|
||||
|
||||
uint64_t ggml_openvino_model_fingerprint(const ggml_cgraph * cgraph,
|
||||
const std::string & device,
|
||||
bool fa,
|
||||
const int32_t * rope_params,
|
||||
int rope_len,
|
||||
uint64_t extra_cfg) {
|
||||
uint64_t extra_cfg,
|
||||
const std::string & graph_signature) {
|
||||
uint64_t h = FNV_OFFSET;
|
||||
h = fnv1a_u64(h, 2);
|
||||
h = fnv1a(h, graph_signature.data(), graph_signature.size());
|
||||
for (const char * name : cache_settings) {
|
||||
const char * value = ggml_openvino_getenv_str(name, "");
|
||||
h = fnv1a(h, value, strlen(value) + 1);
|
||||
}
|
||||
if (const char * debug_nodes = ggml_openvino_getenv_str("GGML_OPENVINO_DEBUG_NODE")) {
|
||||
h = fnv1a(h, "GGML_OPENVINO_DEBUG_NODE", sizeof("GGML_OPENVINO_DEBUG_NODE"));
|
||||
h = fnv1a(h, debug_nodes, strlen(debug_nodes) + 1);
|
||||
}
|
||||
if (ggml_openvino_is_gpu() && ggml_openvino_getenv_int("GGML_OPENVINO_MOE_OP", 1) == 0) {
|
||||
h = fnv1a(h, "GGML_OPENVINO_MOE_OP=0", sizeof("GGML_OPENVINO_MOE_OP=0"));
|
||||
}
|
||||
|
||||
// Topology: node count + each node's op and name (cheap, and distinguishes
|
||||
// graphs that share weights but differ structurally).
|
||||
@@ -193,7 +370,7 @@ uint64_t ggml_openvino_model_fingerprint(const ggml_cgraph * cgraph,
|
||||
// Weights: the model identity.
|
||||
for_each_weight(cgraph, [&](const ggml_tensor * t) { h = fnv1a_u64(h, weight_fingerprint(t)); });
|
||||
|
||||
// Config that changes the produced blob.
|
||||
// Device, model parameters, and backend configuration.
|
||||
h = fnv1a(h, device.data(), device.size());
|
||||
h = fnv1a_u64(h, fa ? 1u : 0u);
|
||||
if (rope_params && rope_len > 0) {
|
||||
@@ -216,7 +393,9 @@ std::string ggml_openvino_model_cache_manifest_path(const std::string & dir, uin
|
||||
|
||||
bool ggml_openvino_model_cache_write_manifest(const std::string & path,
|
||||
const ggml_cgraph * cgraph,
|
||||
uint64_t fingerprint) {
|
||||
uint64_t fingerprint,
|
||||
const std::vector<std::string> & inputs,
|
||||
const std::vector<std::string> & outputs) {
|
||||
std::ofstream f(path, std::ios::trunc);
|
||||
if (!f.is_open()) {
|
||||
return false;
|
||||
@@ -227,12 +406,21 @@ bool ggml_openvino_model_cache_write_manifest(const std::string & path,
|
||||
f << t->name << " " << t->ne[0] << " " << t->ne[1] << " " << t->ne[2] << " " << t->ne[3] << " "
|
||||
<< static_cast<int>(t->type) << " " << hex64(weight_fingerprint(t)) << "\n";
|
||||
});
|
||||
f << "ports\n";
|
||||
for (const auto * names : { &inputs, &outputs }) {
|
||||
f << names->size() << '\n';
|
||||
for (const auto & name : *names) {
|
||||
f << std::quoted(name) << '\n';
|
||||
}
|
||||
}
|
||||
return f.good();
|
||||
}
|
||||
|
||||
bool ggml_openvino_model_cache_verify_manifest(const std::string & path,
|
||||
const ggml_cgraph * cgraph,
|
||||
uint64_t fingerprint) {
|
||||
uint64_t fingerprint,
|
||||
std::vector<std::string> & inputs,
|
||||
std::vector<std::string> & outputs) {
|
||||
std::ifstream f(path);
|
||||
if (!f.is_open()) {
|
||||
return false;
|
||||
@@ -260,7 +448,7 @@ bool ggml_openvino_model_cache_verify_manifest(const std::string & path,
|
||||
size_t idx = 0;
|
||||
std::string line;
|
||||
std::getline(f, line); // consume rest of ov_version line
|
||||
while (std::getline(f, line)) {
|
||||
while (idx < expected.size() && std::getline(f, line)) {
|
||||
if (line.empty()) {
|
||||
continue;
|
||||
}
|
||||
@@ -269,5 +457,28 @@ bool ggml_openvino_model_cache_verify_manifest(const std::string & path,
|
||||
}
|
||||
++idx;
|
||||
}
|
||||
return idx == expected.size();
|
||||
if (idx != expected.size()) {
|
||||
return false;
|
||||
}
|
||||
if (!std::getline(f, line)) {
|
||||
return true;
|
||||
}
|
||||
if (line != "ports") {
|
||||
return false;
|
||||
}
|
||||
for (auto * names : { &inputs, &outputs }) {
|
||||
size_t count;
|
||||
if (!(f >> count) || count > 100000) {
|
||||
return false;
|
||||
}
|
||||
for (size_t i = 0; i < count; ++i) {
|
||||
std::string name;
|
||||
if (!(f >> std::quoted(name))) {
|
||||
return false;
|
||||
}
|
||||
names->push_back(name);
|
||||
}
|
||||
}
|
||||
f >> std::ws;
|
||||
return f.eof();
|
||||
}
|
||||
|
||||
@@ -1,38 +1,39 @@
|
||||
#pragma once
|
||||
|
||||
// Frontend-level compiled-model cache (GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR).
|
||||
//
|
||||
// The OpenVINO plugin's own ov::cache_dir caches the compiled blob keyed by the
|
||||
// *OV model*, but producing that model still runs the full frontend every time:
|
||||
// weight requantization (incl. the large token_embd F32 transient) and the
|
||||
// ggml->OV graph conversion. This cache keys off a fingerprint computed directly
|
||||
// from the ggml cgraph, so a hit skips requant + convert + compile entirely and
|
||||
// instead imports a previously exported CompiledModel blob.
|
||||
//
|
||||
// Opt-in and independent from GGML_OPENVINO_CACHE_DIR. Default off.
|
||||
// Compiled blobs include weights. Cache-only execution skips weight uploads and graph compilation.
|
||||
|
||||
#include "ggml.h"
|
||||
|
||||
#include <cstdint>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
// Returns the compiled-model cache directory from GGML_OPENVINO_COMPILED_MODEL_CACHE_DIR,
|
||||
// or empty if unset/disabled. When empty, callers must not use the cache.
|
||||
bool ggml_openvino_model_cache_only();
|
||||
void ggml_openvino_model_cache_init();
|
||||
|
||||
struct ggml_openvino_source_mapping {
|
||||
uintptr_t begin;
|
||||
uintptr_t end;
|
||||
uint64_t offset;
|
||||
uint64_t identity;
|
||||
};
|
||||
|
||||
// Identify mmap weights without reading their pages. Cache mappings for one buffer lifetime.
|
||||
uint64_t ggml_openvino_source_fingerprint(const void * data, size_t size, std::vector<ggml_openvino_source_mapping> & mappings);
|
||||
uint64_t ggml_backend_openvino_weight_fingerprint(const ggml_tensor * tensor);
|
||||
|
||||
// Returns the compiled-model cache directory, or empty if unset.
|
||||
std::string ggml_openvino_model_cache_dir();
|
||||
std::string ggml_openvino_model_cache_temp_path(const std::string & path);
|
||||
|
||||
// Compute a stable 64-bit fingerprint identifying the model+config that a cgraph
|
||||
// would compile to. Combines graph topology, a sampled hash of every weight
|
||||
// tensor (name/shape/dtype + bounded byte sample), and the config that changes
|
||||
// the produced blob (device, flash-attention, rope params, the compile-memory
|
||||
// flags, stateful, and the OpenVINO version). `device` is the resolved device
|
||||
// string; `fa` is the flash-attention flag; `rope_params`/`rope_len` cover the
|
||||
// model's rope configuration; `extra_cfg` folds in any other blob-affecting bits.
|
||||
// Hash graph structure, source weight identities, configuration, and OpenVINO version.
|
||||
uint64_t ggml_openvino_model_fingerprint(const ggml_cgraph * cgraph,
|
||||
const std::string & device,
|
||||
bool fa,
|
||||
const int32_t * rope_params,
|
||||
int rope_len,
|
||||
uint64_t extra_cfg);
|
||||
uint64_t extra_cfg,
|
||||
const std::string & graph_signature);
|
||||
|
||||
// Path to the compiled-blob file for a fingerprint (<dir>/<hex>.blob).
|
||||
std::string ggml_openvino_model_cache_blob_path(const std::string & dir, uint64_t fingerprint);
|
||||
@@ -41,16 +42,16 @@ std::string ggml_openvino_model_cache_blob_path(const std::string & dir, uint64_
|
||||
// fingerprints, used to re-verify a hit before trusting the blob.
|
||||
std::string ggml_openvino_model_cache_manifest_path(const std::string & dir, uint64_t fingerprint);
|
||||
|
||||
// Write/read the manifest. The manifest is a newline-separated list of
|
||||
// "name ne0 ne1 ne2 ne3 type sample_hash" lines plus a header line with the
|
||||
// fingerprint and OV version. Returns false on I/O error.
|
||||
// Record weight metadata and source identities. Returns false on I/O error.
|
||||
bool ggml_openvino_model_cache_write_manifest(const std::string & path,
|
||||
const ggml_cgraph * cgraph,
|
||||
uint64_t fingerprint);
|
||||
uint64_t fingerprint,
|
||||
const std::vector<std::string> & inputs,
|
||||
const std::vector<std::string> & outputs);
|
||||
|
||||
// Verify that the cgraph's weights still match the stored manifest (guards the
|
||||
// sampled-hash collision risk: a blob is only trusted if every weight's
|
||||
// name/shape/type/sample-hash matches what was cached). Returns true on match.
|
||||
// Require all weight metadata and source identities to match the manifest.
|
||||
bool ggml_openvino_model_cache_verify_manifest(const std::string & path,
|
||||
const ggml_cgraph * cgraph,
|
||||
uint64_t fingerprint);
|
||||
uint64_t fingerprint,
|
||||
std::vector<std::string> & inputs,
|
||||
std::vector<std::string> & outputs);
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user