mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-04 03:47:27 -05:00
Compare commits
198
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7d8a964eb7 | ||
|
|
ee2ddd0938 | ||
|
|
043d5ea3d8 | ||
|
|
7b1c77be19 | ||
|
|
e1d0dd6b17 | ||
|
|
f805c57a2d | ||
|
|
4de0926596 | ||
|
|
ed319febb1 | ||
|
|
84e76d8a23 | ||
|
|
cdc06426e7 | ||
|
|
bced4595b8 | ||
|
|
a02c7f58c1 | ||
|
|
5cf3a35287 | ||
|
|
07fc586e38 | ||
|
|
97a418bdf4 | ||
|
|
a72e04abe0 | ||
|
|
8212c78024 | ||
|
|
945064fcea | ||
|
|
fc343a84bb | ||
|
|
308883b335 | ||
|
|
70596c4dcb | ||
|
|
70c4e1582e | ||
|
|
6b790a9c29 | ||
|
|
3423f940e8 | ||
|
|
53ed051ce5 | ||
|
|
f830688e91 | ||
|
|
2b70583997 | ||
|
|
4c5957c277 | ||
|
|
9710a32175 | ||
|
|
013b31c03c | ||
|
|
bd4f514db1 | ||
|
|
b9ae43a5d4 | ||
|
|
d2e54583c7 | ||
|
|
6e60f35608 | ||
|
|
fee39dd926 | ||
|
|
7fe450e193 | ||
|
|
177cd8cc70 | ||
|
|
e4e2f62325 | ||
|
|
66fba63af1 | ||
|
|
bddf8263c3 | ||
|
|
9575389609 | ||
|
|
dc9879cf66 | ||
|
|
42916d83f4 | ||
|
|
4e416ee730 | ||
|
|
ee3ecce05c | ||
|
|
057494f93f | ||
|
|
bcbc936a87 | ||
|
|
26758d38f9 | ||
|
|
18f9f7bef9 | ||
|
|
633733d0ae | ||
|
|
86b2daa730 | ||
|
|
183d2a04c2 | ||
|
|
45062d4056 | ||
|
|
503549c5f4 | ||
|
|
e97545d916 | ||
|
|
b1ff4ca236 | ||
|
|
94256114c2 | ||
|
|
1a679828f3 | ||
|
|
384a534ce3 | ||
|
|
5e48b31000 | ||
|
|
4d7d7703fe | ||
|
|
08b1d2aea5 | ||
|
|
441df11f65 | ||
|
|
e6ab7c1a41 | ||
|
|
f46bc30cb6 | ||
|
|
709fe755df | ||
|
|
d5f66492e6 | ||
|
|
9919911185 | ||
|
|
bbf99b1b33 | ||
|
|
4098fdc922 | ||
|
|
4ceb171910 | ||
|
|
73c941b111 | ||
|
|
0f8a414b75 | ||
|
|
f95b0d9539 | ||
|
|
c350a40bbd | ||
|
|
9b421fa946 | ||
|
|
348f853b7a | ||
|
|
217f81c266 | ||
|
|
828fdf282e | ||
|
|
bfd73a876e | ||
|
|
a60f9aead0 | ||
|
|
7ab4ee7baa | ||
|
|
0ee9435b8f | ||
|
|
8cfc315a8a | ||
|
|
ec5a12b85a | ||
|
|
c550d2f60b | ||
|
|
58367713a6 | ||
|
|
ff0dbb975e | ||
|
|
fb34fc262c | ||
|
|
c641dfa833 | ||
|
|
9655061365 | ||
|
|
b1c2863e2c | ||
|
|
f4e276a206 | ||
|
|
e6cef8152f | ||
|
|
c21284cdf5 | ||
|
|
6f41ac59e0 | ||
|
|
ec91ab5add | ||
|
|
bb3c853c30 | ||
|
|
af911149c5 | ||
|
|
1884824fda | ||
|
|
161755f29e | ||
|
|
1d72b05d38 | ||
|
|
542e9202d7 | ||
|
|
e0dff58475 | ||
|
|
982a3329af | ||
|
|
711f60beeb | ||
|
|
335b21fcbd | ||
|
|
26394b4e67 | ||
|
|
1aa2954bde | ||
|
|
8034c1d1f1 | ||
|
|
6ad1af5603 | ||
|
|
0c3626ec06 | ||
|
|
68d9053afd | ||
|
|
8aa161b54a | ||
|
|
932a68e068 | ||
|
|
62668d6b26 | ||
|
|
ce8caa6e60 | ||
|
|
a894dae939 | ||
|
|
3d82ef62d4 | ||
|
|
3cf03257f2 | ||
|
|
b23efaa2ef | ||
|
|
4260903678 | ||
|
|
9a9f939b80 | ||
|
|
f072b10371 | ||
|
|
59657a613a | ||
|
|
e613ef2c81 | ||
|
|
851cb34f21 | ||
|
|
7d4b92bb9b | ||
|
|
1af554f8fc | ||
|
|
eb1e1f495f | ||
|
|
5b59b83f4e | ||
|
|
60b06ab9a9 | ||
|
|
efa28e950e | ||
|
|
59fc5a1ca3 | ||
|
|
b23701f77d | ||
|
|
60081bb2b5 | ||
|
|
2b1847030c | ||
|
|
50631b3d2c | ||
|
|
18a04f09c2 | ||
|
|
ec92815050 | ||
|
|
4fea119de3 | ||
|
|
5b335f413e | ||
|
|
542348a35c | ||
|
|
d663dd3f3a | ||
|
|
44be98f057 | ||
|
|
911f6cdc8a | ||
|
|
bbd488c42a | ||
|
|
dc85f89c7e | ||
|
|
8ed1a55efc | ||
|
|
bb11ebb682 | ||
|
|
f03cf3e9b8 | ||
|
|
bdcbaaf6e7 | ||
|
|
5c53396b89 | ||
|
|
972d2313bc | ||
|
|
c77ae695c9 | ||
|
|
b49650adb3 | ||
|
|
7076180486 | ||
|
|
ebbb185227 | ||
|
|
4ff829ec2e | ||
|
|
f172be756a | ||
|
|
87f9c82f2d | ||
|
|
7f6f0c2a9d | ||
|
|
81aeaeb74b | ||
|
|
c9a5eeeb34 | ||
|
|
7490357f22 | ||
|
|
817e5f83eb | ||
|
|
c57da6fd81 | ||
|
|
79bfc1d43a | ||
|
|
05f2dcfdba | ||
|
|
35822afe58 | ||
|
|
aa39d7a3e1 | ||
|
|
4bc272fd72 | ||
|
|
fb27a525d2 | ||
|
|
c6824a9e42 | ||
|
|
2f3fd02526 | ||
|
|
1ec8188094 | ||
|
|
82324fc508 | ||
|
|
7ceed8737f | ||
|
|
7d6f5d02bb | ||
|
|
83078fec0d | ||
|
|
f266648fa9 | ||
|
|
60199339bc | ||
|
|
b04d4e567c | ||
|
|
37b53fd454 | ||
|
|
fccf7166fb | ||
|
|
0bec16e388 | ||
|
|
d4365d9554 | ||
|
|
0a8b29a607 | ||
|
|
583926e3ac | ||
|
|
e13469a323 | ||
|
|
930e2fa599 | ||
|
|
72b590d65f | ||
|
|
38a5b42d9a | ||
|
|
9f31776c37 | ||
|
|
d1d3c3396a | ||
|
|
6011c34ce6 | ||
|
|
7609846557 | ||
|
|
5431581326 |
@@ -1,5 +1,5 @@
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.3.1
|
||||
ARG OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.4
|
||||
ARG OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3
|
||||
ARG UBUNTU_VERSION=24.04
|
||||
|
||||
# Intel GPU driver versions. https://github.com/intel/compute-runtime/releases
|
||||
@@ -10,9 +10,9 @@ ARG COMPUTE_RUNTIME_VERSION_FULL=26.31.39395.13-0
|
||||
ARG IGDGMM_VERSION=22.10.0
|
||||
|
||||
# Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases
|
||||
ARG NPU_DRIVER_VERSION=v1.35.0
|
||||
ARG NPU_DRIVER_FULL=v1.35.0.20260722-29947505341
|
||||
ARG LIBZE1_VERSION=1.28.2-1~24.04~ppa1
|
||||
ARG NPU_DRIVER_VERSION=v1.38.0
|
||||
ARG NPU_DRIVER_FULL=v1.38.0.20260910-34487311128
|
||||
ARG LIBZE1_VERSION=1.32.0-1~24.04~ppa1
|
||||
|
||||
# Optional proxy build arguments
|
||||
ARG http_proxy=
|
||||
@@ -173,7 +173,7 @@ RUN --mount=type=cache,target=/var/cache/intel-npu,sharing=locked \
|
||||
fi; \
|
||||
DEB=/var/cache/intel-npu/libze1_${LIBZE1_VERSION}_amd64.deb; \
|
||||
if [ ! -f "$DEB" ]; then \
|
||||
wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260606T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \
|
||||
wget -q -O "$DEB" https://snapshot.ppa.launchpadcontent.net/kobuk-team/intel-graphics/ubuntu/20260830T100000Z/pool/main/l/level-zero-loader/libze1_${LIBZE1_VERSION}_amd64.deb; \
|
||||
fi; \
|
||||
mkdir /tmp/npu/ && cd /tmp/npu/ && tar -xf "$TGZ" && cp "$DEB" .; \
|
||||
apt-get update; \
|
||||
|
||||
@@ -29,7 +29,7 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
android-ndk-snapdragon:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
container:
|
||||
image: 'ghcr.io/snapdragon-toolchain/arm64-android:v0.7'
|
||||
defaults:
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
path: pkg-snapdragon/llama.cpp
|
||||
|
||||
linux-iot-snapdragon:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
container:
|
||||
image: 'ghcr.io/snapdragon-toolchain/arm64-linux:v0.7'
|
||||
defaults:
|
||||
|
||||
@@ -33,7 +33,7 @@ env:
|
||||
|
||||
jobs:
|
||||
default:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -49,7 +49,7 @@ jobs:
|
||||
distribution: zulu
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
uses: android-actions/setup-android@be39fa834029ff78f1a44aa3bb0819b8fc2bd8fd # v4.0.4
|
||||
with:
|
||||
log-accepted-android-sdk-licenses: false
|
||||
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
./gradlew build --no-daemon
|
||||
|
||||
ndk:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
container:
|
||||
image: 'ghcr.io/snapdragon-toolchain/arm64-android:v0.3'
|
||||
defaults:
|
||||
@@ -93,7 +93,7 @@ jobs:
|
||||
path: pkg-adb/llama.cpp
|
||||
|
||||
arm64:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
env:
|
||||
NDK_VERSION: "29.0.14206865"
|
||||
@@ -123,7 +123,7 @@ jobs:
|
||||
distribution: temurin
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
uses: android-actions/setup-android@be39fa834029ff78f1a44aa3bb0819b8fc2bd8fd # v4.0.4
|
||||
with:
|
||||
log-accepted-android-sdk-licenses: false
|
||||
|
||||
|
||||
@@ -41,8 +41,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -69,8 +69,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -5,7 +5,7 @@ on:
|
||||
|
||||
jobs:
|
||||
linux:
|
||||
runs-on: [self-hosted, Linux]
|
||||
runs-on: [self-hosted, Linux, CPU]
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
run: |
|
||||
export PIP_BREAK_SYSTEM_PACKAGES="1"
|
||||
python3 -m pip install --upgrade pip setuptools
|
||||
pip3 install ./gguf-py
|
||||
pip3 install ./gguf-py jinja2==3.1.6
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
@@ -124,7 +124,7 @@ jobs:
|
||||
id: cmake_test
|
||||
run: |
|
||||
cd build
|
||||
ctest -L main --verbose --timeout 900
|
||||
ctest -L 'main|python' --verbose --timeout 900
|
||||
|
||||
- name: Test llama2c conversion
|
||||
id: llama2c_test
|
||||
|
||||
@@ -177,7 +177,8 @@ jobs:
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build -S . \
|
||||
-DGGML_MUSA=ON
|
||||
-DGGML_MUSA=ON \
|
||||
-DMUSA_ARCHITECTURES=21
|
||||
time cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
|
||||
@@ -41,8 +41,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -96,8 +96,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -1,447 +0,0 @@
|
||||
name: CI (self-hosted)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/build-self-hosted.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.cu',
|
||||
'**/*.cuh',
|
||||
'**/*.swift',
|
||||
'**/*.m',
|
||||
'**/*.metal',
|
||||
'**/*.comp',
|
||||
'**/*.glsl',
|
||||
'**/*.wgsl'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/build-self-hosted.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.cu',
|
||||
'**/*.cuh',
|
||||
'**/*.swift',
|
||||
'**/*.m',
|
||||
'**/*.metal',
|
||||
'**/*.comp',
|
||||
'**/*.glsl',
|
||||
'**/*.wgsl'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: self-hosted-gpu-cuda
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
nvidia-smi
|
||||
GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: self-hosted-gpu-cuda
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
gpu-rocm:
|
||||
runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
# HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
|
||||
# issue on integrated RDNA3.5 (gfx1151) where batched inference returns
|
||||
# incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
|
||||
# restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
|
||||
env:
|
||||
HIP_LAUNCH_BLOCKING: "1"
|
||||
run: |
|
||||
rocminfo
|
||||
GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-nvidia-cm:
|
||||
runs-on: [self-hosted, Linux, NVIDIA]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-nvidia-cm2:
|
||||
runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-webgpu-nvidia:
|
||||
runs-on: [self-hosted, Linux, NVIDIA, X64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dawn Dependency
|
||||
id: dawn-depends
|
||||
run: |
|
||||
DAWN_VERSION="v20260908.214631"
|
||||
DAWN_OWNER="google"
|
||||
DAWN_REPO="dawn"
|
||||
DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"
|
||||
echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
curl -L -o artifact.tar.gz \
|
||||
"https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
mkdir dawn
|
||||
tar -xvf artifact.tar.gz -C dawn --strip-components=1
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_WEBGPU=1 \
|
||||
GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
|
||||
GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMX-compatible machine
|
||||
#cpu-amx:
|
||||
# runs-on: [self-hosted, Linux, CPU, AMX]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMD GPU machine
|
||||
# amd-vulkan:
|
||||
# runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# vulkaninfo --summary
|
||||
# GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMD GPU machine
|
||||
# amd-rocm:
|
||||
# runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# amd-smi static
|
||||
# GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-metal:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-webgpu-apple:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dawn Dependency
|
||||
id: dawn-depends
|
||||
run: |
|
||||
DAWN_VERSION="v20260908.214631"
|
||||
DAWN_OWNER="google"
|
||||
DAWN_REPO="dawn"
|
||||
DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release"
|
||||
echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
curl -L -o artifact.tar.gz \
|
||||
"https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
mkdir dawn
|
||||
tar -xvf artifact.tar.gz -C dawn --strip-components=1
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-apple:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-intel-linux:
|
||||
runs-on: [self-hosted, Linux, Intel]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-intel-windows:
|
||||
runs-on: [self-hosted, Windows, X64, Intel]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}"
|
||||
env:
|
||||
MSYSTEM: UCRT64
|
||||
CHERE_INVOKING: 1
|
||||
PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }}
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
# Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create
|
||||
# a valid python environment for testing
|
||||
LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp
|
||||
|
||||
gpu-openvino-low-perf:
|
||||
runs-on: [self-hosted, Linux, Intel, OpenVINO]
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Setup OpenVINO Toolkit
|
||||
uses: ./.github/actions/linux-setup-openvino
|
||||
with:
|
||||
path: ./openvino_toolkit
|
||||
version_major: ${{ env.OPENVINO_VERSION_MAJOR }}
|
||||
version_full: ${{ env.OPENVINO_VERSION_FULL }}
|
||||
|
||||
- name: Install OpenVINO dependencies
|
||||
run: |
|
||||
cd ./openvino_toolkit
|
||||
chmod +x ./install_dependencies/install_openvino_dependencies.sh
|
||||
echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
source ./openvino_toolkit/setupvars.sh
|
||||
GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
cpu-x64-high-perf:
|
||||
runs-on: [self-hosted, Linux, X64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
cpu-arm64-high-perf-graviton4:
|
||||
runs-on: ah-ubuntu_24_04-c8g_8x
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dependencies
|
||||
id: depends
|
||||
run: |
|
||||
set -euxo pipefail
|
||||
sudo apt-get update
|
||||
sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
|
||||
apt-get install -y \
|
||||
build-essential \
|
||||
python3-venv \
|
||||
gpg \
|
||||
wget \
|
||||
time \
|
||||
git-lfs
|
||||
|
||||
git lfs install
|
||||
|
||||
# install the latest cmake
|
||||
sudo install -d /usr/share/keyrings
|
||||
wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
|
||||
| gpg --dearmor \
|
||||
| sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
|
||||
echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
|
||||
| sudo tee /etc/apt/sources.list.d/kitware.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y cmake
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) \
|
||||
GG_BUILD_HIGH_PERF=1 \
|
||||
GG_BUILD_NO_BF16=1 \
|
||||
GG_BUILD_EXTRA_TESTS_0=1 \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
cpu-arm64-graviton4-kleidiai:
|
||||
runs-on: ah-ubuntu_24_04-c8g_8x
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dependencies
|
||||
id: depends
|
||||
run: |
|
||||
set -euxo pipefail
|
||||
sudo apt-get update
|
||||
sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
|
||||
apt-get install -y \
|
||||
build-essential \
|
||||
python3-venv \
|
||||
gpg \
|
||||
wget \
|
||||
time \
|
||||
git-lfs
|
||||
|
||||
git lfs install
|
||||
|
||||
# install the latest cmake
|
||||
sudo install -d /usr/share/keyrings
|
||||
wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
|
||||
| gpg --dearmor \
|
||||
| sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
|
||||
echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
|
||||
| sudo tee /etc/apt/sources.list.d/kitware.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y cmake
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) \
|
||||
GG_BUILD_KLEIDIAI=1 \
|
||||
GG_BUILD_EXTRA_TESTS_0=1 \
|
||||
GG_BUILD_HIGH_PERF=1 \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -49,7 +49,7 @@ jobs:
|
||||
env:
|
||||
ONEAPI_ROOT: /opt/intel/oneapi/
|
||||
ONEAPI_INSTALLER_VERSION: "2025.3.3"
|
||||
LEVEL_ZERO_VERSION: "1.28.2"
|
||||
LEVEL_ZERO_VERSION: "1.33.1"
|
||||
LEVEL_ZERO_UBUNTU_VERSION: "u24.04"
|
||||
|
||||
continue-on-error: true
|
||||
@@ -70,9 +70,10 @@ jobs:
|
||||
shell: bash
|
||||
run: |
|
||||
cd /tmp
|
||||
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero.deb
|
||||
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero-devel_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero-devel.deb
|
||||
sudo apt-get install -y ./level-zero.deb ./level-zero-devel.deb
|
||||
# v1.33.x renamed the Debian packages to libze1 / libze-dev
|
||||
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/libze1_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O libze1.deb
|
||||
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/libze-dev_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O libze-dev.deb
|
||||
sudo apt-get install -y ./libze1.deb ./libze-dev.deb
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
@@ -101,7 +102,11 @@ jobs:
|
||||
-DCMAKE_CXX_COMPILER=icpx \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_NATIVE=OFF \
|
||||
-DGGML_SYCL_F16=${{ matrix.fp16 }}
|
||||
-DGGML_SYCL_F16=${{ matrix.fp16 }} \
|
||||
-DGGML_SYCL_SUPPORT_LEVEL_ZERO_API=ON \
|
||||
-DGGML_SYCL_DNN=ON \
|
||||
-DCMAKE_CXX_FLAGS="-fsycl-unnamed-lambda" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsycl-unnamed-lambda"
|
||||
time cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
@@ -126,7 +131,7 @@ jobs:
|
||||
env:
|
||||
WINDOWS_BASEKIT_URL: https://registrationcenter-download.intel.com/akdlm/IRC_NAS/b60765d1-2b85-4e85-86b6-cb0e9563a699/intel-deep-learning-essentials-2025.3.3.18_offline.exe
|
||||
WINDOWS_DPCPP_MKL: intel.oneapi.win.cpp-dpcpp-common:intel.oneapi.win.mkl.devel:intel.oneapi.win.dnnl:intel.oneapi.win.tbb.devel
|
||||
LEVEL_ZERO_SDK_URL: https://github.com/oneapi-src/level-zero/releases/download/v1.28.2/level-zero-win-sdk-1.28.2.zip
|
||||
LEVEL_ZERO_SDK_URL: https://github.com/oneapi-src/level-zero/releases/download/v1.33.1/level-zero-win-sdk-1.33.1.zip
|
||||
ONEAPI_ROOT: "C:/Program Files (x86)/Intel/oneAPI"
|
||||
ONEAPI_INSTALLER_VERSION: "2025.3.3"
|
||||
steps:
|
||||
|
||||
@@ -19,7 +19,7 @@ on:
|
||||
|
||||
jobs:
|
||||
check-vendor:
|
||||
runs-on: [self-hosted, fast]
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
name: CI (self-hosted CPU backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-cpu.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-cpu.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
cpu-x64-high-perf:
|
||||
runs-on: [self-hosted, Linux, X64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
cpu-arm64-high-perf-graviton4:
|
||||
runs-on: ah-ubuntu_24_04-c8g_8x
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dependencies
|
||||
id: depends
|
||||
run: |
|
||||
set -euxo pipefail
|
||||
sudo apt-get update
|
||||
sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
|
||||
apt-get install -y \
|
||||
build-essential \
|
||||
python3-venv \
|
||||
gpg \
|
||||
wget \
|
||||
time \
|
||||
git-lfs
|
||||
|
||||
git lfs install
|
||||
|
||||
# install the latest cmake
|
||||
sudo install -d /usr/share/keyrings
|
||||
wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
|
||||
| gpg --dearmor \
|
||||
| sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
|
||||
echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
|
||||
| sudo tee /etc/apt/sources.list.d/kitware.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y cmake
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) \
|
||||
GG_BUILD_HIGH_PERF=1 \
|
||||
GG_BUILD_NO_BF16=1 \
|
||||
GG_BUILD_EXTRA_TESTS_0=1 \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMX-compatible machine
|
||||
#cpu-amx:
|
||||
# runs-on: [self-hosted, Linux, CPU, AMX]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,124 @@
|
||||
name: CI (self-hosted CUDA backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-cuda.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.cu',
|
||||
'**/*.cuh'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-cuda.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-cuda/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: self-hosted-gpu-cuda
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
nvidia-smi
|
||||
GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: self-hosted-gpu-cuda
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
gpu-rocm:
|
||||
runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
# HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness
|
||||
# issue on integrated RDNA3.5 (gfx1151) where batched inference returns
|
||||
# incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches
|
||||
# restores correctness. Remove once the underlying ROCm/HIP issue is fixed.
|
||||
env:
|
||||
HIP_LAUNCH_BLOCKING: "1"
|
||||
run: |
|
||||
rocminfo
|
||||
GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMD GPU machine
|
||||
# amd-rocm:
|
||||
# runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# amd-smi static
|
||||
# GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,85 @@
|
||||
name: CI (self-hosted KleidiAI backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-kleidiai.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-kleidiai.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
cpu-arm64-graviton4-kleidiai:
|
||||
runs-on: ah-ubuntu_24_04-c8g_8x
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dependencies
|
||||
id: depends
|
||||
run: |
|
||||
set -euxo pipefail
|
||||
sudo apt-get update
|
||||
sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \
|
||||
apt-get install -y \
|
||||
build-essential \
|
||||
python3-venv \
|
||||
gpg \
|
||||
wget \
|
||||
time \
|
||||
git-lfs
|
||||
|
||||
git lfs install
|
||||
|
||||
# install the latest cmake
|
||||
sudo install -d /usr/share/keyrings
|
||||
wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \
|
||||
| gpg --dearmor \
|
||||
| sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null
|
||||
echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \
|
||||
| sudo tee /etc/apt/sources.list.d/kitware.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y cmake
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
LLAMA_ARG_THREADS=$(nproc) \
|
||||
GG_BUILD_KLEIDIAI=1 \
|
||||
GG_BUILD_EXTRA_TESTS_0=1 \
|
||||
GG_BUILD_HIGH_PERF=1 \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,59 @@
|
||||
name: CI (self-hosted Metal backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-metal.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.swift',
|
||||
'**/*.m',
|
||||
'**/*.metal'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-metal.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-metal/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-metal:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,75 @@
|
||||
name: CI (self-hosted OpenVINO backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-openvino.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-openvino.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-openvino/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-openvino-low-perf:
|
||||
runs-on: [self-hosted, Linux, Intel, OpenVINO]
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Setup OpenVINO Toolkit
|
||||
uses: ./.github/actions/linux-setup-openvino
|
||||
with:
|
||||
path: ./openvino_toolkit
|
||||
version_major: ${{ env.OPENVINO_VERSION_MAJOR }}
|
||||
version_full: ${{ env.OPENVINO_VERSION_FULL }}
|
||||
|
||||
- name: Install OpenVINO dependencies
|
||||
run: |
|
||||
cd ./openvino_toolkit
|
||||
chmod +x ./install_dependencies/install_openvino_dependencies.sh
|
||||
echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
source ./openvino_toolkit/setupvars.sh
|
||||
GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,201 @@
|
||||
name: CI (self-hosted Vulkan backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-vulkan.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.comp',
|
||||
'**/*.glsl'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-vulkan.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-vulkan/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-vulkan-nvidia-cm:
|
||||
# runs-on: "hf-jobs-t4-small:ubuntu26_04"
|
||||
runs-on: [self-hosted, Linux, NVIDIA]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
# - name: Install dependencies
|
||||
# run: |
|
||||
# sudo apt update
|
||||
# sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
|
||||
|
||||
# - name: ccache
|
||||
# uses: ggml-org/ccache-action@v1.2.24
|
||||
# with:
|
||||
# restore: false
|
||||
# save: false
|
||||
|
||||
# - name: ccache-buckets-restore
|
||||
# uses: ./.github/actions/ccache-buckets
|
||||
# with:
|
||||
# key: self-hosted-vulkan-nvidia-cm
|
||||
# folder: llama.cpp
|
||||
# hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# - name: ccache-buckets-save
|
||||
# if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
# uses: ./.github/actions/ccache-buckets
|
||||
# env:
|
||||
# HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
# with:
|
||||
# key: self-hosted-vulkan-nvidia-cm
|
||||
# folder: llama.cpp
|
||||
# evict-old-files: 1d
|
||||
# hf_bucket: ggml-org/cache
|
||||
# save: true
|
||||
|
||||
gpu-vulkan-nvidia-cm2:
|
||||
# runs-on: "hf-jobs-t4-small:ubuntu26_04"
|
||||
runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
# - name: Install dependencies
|
||||
# run: |
|
||||
# sudo apt update
|
||||
# sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
|
||||
|
||||
# - name: ccache
|
||||
# uses: ggml-org/ccache-action@v1.2.24
|
||||
# with:
|
||||
# restore: false
|
||||
# save: false
|
||||
|
||||
# - name: ccache-buckets-restore
|
||||
# uses: ./.github/actions/ccache-buckets
|
||||
# with:
|
||||
# key: self-hosted-vulkan-nvidia-cm2
|
||||
# folder: llama.cpp
|
||||
# hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
# - name: ccache-buckets-save
|
||||
# if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
# uses: ./.github/actions/ccache-buckets
|
||||
# env:
|
||||
# HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
# with:
|
||||
# key: self-hosted-vulkan-nvidia-cm2
|
||||
# folder: llama.cpp
|
||||
# evict-old-files: 1d
|
||||
# hf_bucket: ggml-org/cache
|
||||
# save: true
|
||||
|
||||
gpu-vulkan-apple:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-intel-linux:
|
||||
runs-on: [self-hosted, Linux, Intel]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
gpu-vulkan-intel-windows:
|
||||
runs-on: [self-hosted, Windows, X64, Intel]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}"
|
||||
env:
|
||||
MSYSTEM: UCRT64
|
||||
CHERE_INVOKING: 1
|
||||
PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }}
|
||||
run: |
|
||||
vulkaninfo --summary
|
||||
# Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create
|
||||
# a valid python environment for testing
|
||||
LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp
|
||||
|
||||
# TODO: provision AMD GPU machine
|
||||
# amd-vulkan:
|
||||
# runs-on: [self-hosted, Linux, AMD]
|
||||
|
||||
# steps:
|
||||
# - name: Clone
|
||||
# id: checkout
|
||||
# uses: actions/checkout@v6
|
||||
|
||||
# - name: Test
|
||||
# id: ggml-ci
|
||||
# run: |
|
||||
# vulkaninfo --summary
|
||||
# GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -0,0 +1,130 @@
|
||||
name: CI (self-hosted WebGPU backend)
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-webgpu.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'**/*.h',
|
||||
'**/*.hpp',
|
||||
'**/*.c',
|
||||
'**/*.cpp',
|
||||
'**/*.wgsl'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/ci-self-hosted-webgpu.yml',
|
||||
'ci/run.sh',
|
||||
'**/CMakeLists.txt',
|
||||
'**/.cmake',
|
||||
'ggml/src/*',
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-webgpu/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
gpu-webgpu-nvidia:
|
||||
runs-on: "hf-jobs-t4-small:ubuntu26_04"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: self-hosted-webgpu-nvidia
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Dawn Dependency
|
||||
id: dawn-depends
|
||||
run: |
|
||||
DAWN_VERSION="v20260908.214631"
|
||||
DAWN_OWNER="google"
|
||||
DAWN_REPO="dawn"
|
||||
DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"
|
||||
echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
curl -L -o artifact.tar.gz \
|
||||
"https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
mkdir dawn
|
||||
tar -xvf artifact.tar.gz -C dawn --strip-components=1
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_WEBGPU=1 \
|
||||
GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
|
||||
GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: self-hosted-webgpu-nvidia
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
gpu-webgpu-apple:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Dawn Dependency
|
||||
id: dawn-depends
|
||||
run: |
|
||||
DAWN_VERSION="v20260908.214631"
|
||||
DAWN_OWNER="google"
|
||||
DAWN_REPO="dawn"
|
||||
DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release"
|
||||
echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
curl -L -o artifact.tar.gz \
|
||||
"https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
mkdir dawn
|
||||
tar -xvf artifact.tar.gz -C dawn --strip-components=1
|
||||
|
||||
- name: Test
|
||||
id: ggml-ci
|
||||
run: |
|
||||
GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \
|
||||
bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp
|
||||
@@ -11,10 +11,12 @@ on:
|
||||
paths:
|
||||
- .github/workflows/copilot-setup-steps.yml
|
||||
|
||||
cache-mode: none
|
||||
|
||||
jobs:
|
||||
# The job MUST be called `copilot-setup-steps` or it will not be picked up by Copilot.
|
||||
copilot-setup-steps:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
# Set the permissions to the lowest permissions possible needed for your steps.
|
||||
# Copilot will be given its own token for its operations.
|
||||
@@ -31,8 +33,8 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: copilot-setup-steps
|
||||
evict-old-files: 1d
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Dependencies
|
||||
id: depends
|
||||
|
||||
@@ -90,8 +90,8 @@ jobs:
|
||||
{ "tag": "cpu", "dockerfile": ".devops/s390x.Dockerfile", "platforms": "linux/s390x", "full": true, "light": true, "server": true, "free_disk_space": false, "runs_on": "ubuntu-24.04-s390x", "prebuilt_ui": true },
|
||||
{ "tag": "cuda cuda12", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "12.8.1", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04" },
|
||||
{ "tag": "cuda cuda12", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "12.8.1", "platforms": "linux/arm64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04-arm" },
|
||||
{ "tag": "cuda13", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "13.3.0", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04" },
|
||||
{ "tag": "cuda13", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "13.3.0", "platforms": "linux/arm64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04-arm" },
|
||||
{ "tag": "cuda13", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "13.4.1", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04" },
|
||||
{ "tag": "cuda13", "dockerfile": ".devops/cuda.Dockerfile", "cuda_version": "13.4.1", "platforms": "linux/arm64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04-arm" },
|
||||
{ "tag": "musa", "dockerfile": ".devops/musa.Dockerfile", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04" },
|
||||
{ "tag": "intel", "dockerfile": ".devops/intel.Dockerfile", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": true, "runs_on": "ubuntu-24.04" },
|
||||
{ "tag": "vulkan", "dockerfile": ".devops/vulkan.Dockerfile", "platforms": "linux/amd64", "full": true, "light": true, "server": true, "free_disk_space": false, "runs_on": "ubuntu-24.04" },
|
||||
|
||||
@@ -21,7 +21,7 @@ on:
|
||||
jobs:
|
||||
deploy:
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
|
||||
@@ -13,6 +13,16 @@ on:
|
||||
required: true
|
||||
type: boolean
|
||||
default: true
|
||||
skip_apiabi_check:
|
||||
description: 'Skip API/ABI compatibility check'
|
||||
required: false
|
||||
type: boolean
|
||||
default: false
|
||||
apiabi_compare_tag:
|
||||
description: 'Tag to compare against for API/ABI check (default: latest release)'
|
||||
required: false
|
||||
type: string
|
||||
default: ''
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
@@ -23,7 +33,7 @@ permissions:
|
||||
|
||||
jobs:
|
||||
make-release:
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -33,12 +43,18 @@ jobs:
|
||||
ref: ${{ inputs.commit != '' && inputs.commit || github.ref_name }}
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Install API/ABI check tools
|
||||
if: ${{ github.event.inputs.skip_apiabi_check != 'true' }}
|
||||
run: sudo apt-get install -y abi-compliance-checker abigail-tools
|
||||
|
||||
- name: Run release checks
|
||||
id: checks
|
||||
run: bash scripts/make-release-checks.sh ${{ github.event.inputs.dry_run == 'true' && '--dry-run' || '' }}
|
||||
env:
|
||||
GITHUB_REPOSITORY: ${{ github.repository }}
|
||||
RELEASE_BRANCH: ${{ github.ref_name }}
|
||||
SKIP_APIABI_CHECK: ${{ github.event.inputs.skip_apiabi_check }}
|
||||
APIABI_COMPARE_TAG: ${{ github.event.inputs.apiabi_compare_tag }}
|
||||
|
||||
- name: Create release tag
|
||||
if: ${{ github.event.inputs.dry_run == 'false' }}
|
||||
|
||||
@@ -12,7 +12,7 @@ on:
|
||||
|
||||
jobs:
|
||||
pre-tokenizer-hashes:
|
||||
runs-on: [self-hosted, fast]
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
|
||||
@@ -20,7 +20,7 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
python-check-requirements:
|
||||
runs-on: [self-hosted, CPU, fast]
|
||||
runs-on: ${{ 'ubuntu-24.04-arm' || 'ubuntu-24.04' }}
|
||||
name: check-requirements
|
||||
steps:
|
||||
- name: Check out source repository
|
||||
|
||||
@@ -21,7 +21,7 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
flake8-lint:
|
||||
runs-on: [self-hosted, fast]
|
||||
runs-on: ubuntu-slim
|
||||
name: Lint
|
||||
steps:
|
||||
- name: Check out source repository
|
||||
|
||||
@@ -22,7 +22,7 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
python-type-check:
|
||||
runs-on: [self-hosted, fast]
|
||||
runs-on: ubuntu-slim
|
||||
name: python type-check
|
||||
steps:
|
||||
- name: Check out source repository
|
||||
|
||||
+140
-13
@@ -106,6 +106,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-${{ matrix.os }}-${{ matrix.arch }}
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -190,6 +191,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-${{ matrix.os }}-cpu
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -275,6 +277,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-${{ matrix.os }}-vulkan
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -327,13 +330,13 @@ jobs:
|
||||
defines: '-DGGML_CUDA_CUB_3DOT2=ON'
|
||||
- build: 'x64'
|
||||
os: ubuntu-24.04
|
||||
cuda: '13.3.1'
|
||||
label: '13.3'
|
||||
cuda: '13.4.1'
|
||||
label: '13.4'
|
||||
defines: ''
|
||||
- build: 'arm64'
|
||||
os: ubuntu-24.04-arm
|
||||
cuda: '13.3.1'
|
||||
label: '13.3'
|
||||
cuda: '13.4.1'
|
||||
label: '13.4'
|
||||
defines: ''
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
@@ -453,7 +456,7 @@ jobs:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
#permissions:
|
||||
# actions: write
|
||||
@@ -481,10 +484,9 @@ jobs:
|
||||
distribution: temurin
|
||||
|
||||
- name: Setup Android SDK
|
||||
uses: android-actions/setup-android@40fd30fb8d7440372e1316f5d1809ec01dcd3699 # v4.0.1
|
||||
uses: android-actions/setup-android@be39fa834029ff78f1a44aa3bb0819b8fc2bd8fd # v4.0.4
|
||||
with:
|
||||
log-accepted-android-sdk-licenses: false
|
||||
packages: 'platform-tools'
|
||||
|
||||
- name: Install NDK
|
||||
run: |
|
||||
@@ -501,6 +503,7 @@ jobs:
|
||||
# uses: ggml-org/ccache-action@v1.2.24
|
||||
# with:
|
||||
# key: release-android-arm64
|
||||
# evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -541,6 +544,120 @@ jobs:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz
|
||||
name: llama-bin-android-arm64.tar.gz
|
||||
|
||||
android-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
container: 'ghcr.io/snapdragon-toolchain/arm64-android:v0.7'
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# checkout runs as the host user; in-container steps run as root, so git
|
||||
# refuses to touch a repo it does not own. Mark the workspace as safe.
|
||||
- name: Git safe directory
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cp docs/backend/snapdragon/CMakeUserPresets.json .
|
||||
cmake --preset arm64-android-snapdragon-release -B build \
|
||||
-DCMAKE_INSTALL_RPATH='$ORIGIN;$ORIGIN/../lib' \
|
||||
-DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \
|
||||
-DLLAMA_BUILD_BORINGSSL=ON \
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-android-arm64-snapdragon.tar.gz
|
||||
|
||||
linux-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
container: 'ghcr.io/snapdragon-toolchain/arm64-linux:v0.7'
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# checkout runs as the host user; in-container steps run as root, so git
|
||||
# refuses to touch a repo it does not own. Mark the workspace as safe.
|
||||
- name: Git safe directory
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cp docs/backend/snapdragon/CMakeUserPresets.json .
|
||||
cmake --preset arm64-linux-snapdragon-release -B build -DGGML_OPENCL=ON \
|
||||
-DCMAKE_INSTALL_RPATH='$ORIGIN;$ORIGIN/../lib' \
|
||||
-DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \
|
||||
-DLLAMA_BUILD_BORINGSSL=ON \
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-linux-arm64-snapdragon.tar.gz
|
||||
|
||||
ubuntu-24-openvino:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
@@ -555,8 +672,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -579,6 +696,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-ubuntu-24.04-openvino-release-no-preset-v1
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Dependencies
|
||||
run: |
|
||||
@@ -669,8 +787,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.3.1"
|
||||
OPENVINO_VERSION_FULL: "2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -822,6 +940,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
shell: cmd
|
||||
@@ -1066,6 +1185,7 @@ jobs:
|
||||
# uses: ggml-org/ccache-action@v1.2.24
|
||||
# with:
|
||||
# key: release-windows-2025-${{ matrix.arch }}-${{ matrix.backend }}
|
||||
# evict-old-files: 1d
|
||||
|
||||
- name: Install OpenCL Headers and Libs
|
||||
id: install_opencl
|
||||
@@ -1154,6 +1274,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-windows-2022-${{ matrix.arch }}-cuda-${{ matrix.cuda }}
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -1250,6 +1371,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-windows-2022-x64-sycl
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -1368,6 +1490,7 @@ jobs:
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: release-ubuntu-24.04-sycl-${{ matrix.build }}
|
||||
evict-old-files: 1d
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
@@ -1716,6 +1839,8 @@ jobs:
|
||||
- ubuntu-24-openvino
|
||||
- ubuntu-24-sycl
|
||||
- android-arm64
|
||||
- android-arm64-snapdragon
|
||||
- linux-arm64-snapdragon
|
||||
- macos-cpu
|
||||
- ios-xcode
|
||||
#- openEuler-cann
|
||||
@@ -1845,15 +1970,17 @@ jobs:
|
||||
- [Ubuntu x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-x64.tar.gz)
|
||||
- [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz) - [CUDA 12.8 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.3-x64.tar.gz) - [CUDA 13.3 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.3-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.3-arm64.tar.gz) - [CUDA 13.3 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.3-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz)
|
||||
- [Ubuntu x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-10.0-x64.tar.gz)
|
||||
- [Ubuntu x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ needs.ubuntu-24-openvino.outputs.openvino_version }}-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP32)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp32-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP16)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp16-x64.tar.gz)
|
||||
- [Linux arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/linux.md)
|
||||
|
||||
**Android:**
|
||||
- [Android arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz)
|
||||
- [Android arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/README.md)
|
||||
|
||||
**Windows:**
|
||||
- [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip)
|
||||
|
||||
@@ -45,7 +45,7 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
server:
|
||||
runs-on: hf-jobs-cpu-upgrade
|
||||
runs-on: hf-jobs-cpu-performance
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -116,7 +116,7 @@ jobs:
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
cd tools/server/tests
|
||||
PYTEST_WORKERS=1 ./tests.sh
|
||||
PYTEST_WORKERS=4 ./tests.sh
|
||||
|
||||
- name: Slow tests
|
||||
id: server_integration_tests_slow
|
||||
@@ -124,4 +124,4 @@ jobs:
|
||||
run: |
|
||||
source .venv/bin/activate
|
||||
cd tools/server/tests
|
||||
PYTEST_WORKERS=1 SLOW_TESTS=1 ./tests.sh
|
||||
PYTEST_WORKERS=4 SLOW_TESTS=1 ./tests.sh
|
||||
|
||||
@@ -16,7 +16,7 @@ on:
|
||||
|
||||
jobs:
|
||||
update-ops-docs:
|
||||
runs-on: [self-hosted, fast, ARM64]
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
|
||||
@@ -8,7 +8,7 @@ on:
|
||||
jobs:
|
||||
update:
|
||||
name: Update Winget Package
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
if: github.repository_owner == 'ggml-org'
|
||||
|
||||
steps:
|
||||
|
||||
@@ -7,6 +7,7 @@ General:
|
||||
- Don't try to build or run the code unless you are explicitly asked to do so
|
||||
- Use the `gh` CLI tool when querying PRs, issues, or other GitHub resources
|
||||
- When [MODEL] is needed, first try to get it from the `PI_MODEL_NAME` env var before asking the user
|
||||
- Never read the `AGENTS.md` file
|
||||
|
||||
Coding:
|
||||
- When in doubt, always refer to the CONTRIBUTING.md file of the project
|
||||
|
||||
+2
-2
@@ -4,8 +4,8 @@ include(CheckIncludeFileCXX)
|
||||
|
||||
### llama.cpp version
|
||||
set(LLAMA_VERSION_MAJOR 0)
|
||||
set(LLAMA_VERSION_MINOR 4)
|
||||
set(LLAMA_VERSION_PATCH 1)
|
||||
set(LLAMA_VERSION_MINOR 5)
|
||||
set(LLAMA_VERSION_PATCH 0)
|
||||
set(LLAMA_VERSION_BASE "${LLAMA_VERSION_MAJOR}.${LLAMA_VERSION_MINOR}.${LLAMA_VERSION_PATCH}")
|
||||
|
||||
# whether this is a development/nightly build
|
||||
|
||||
@@ -97,7 +97,6 @@
|
||||
/src/models/ @CISC
|
||||
/tests/ @ggerganov
|
||||
/tests/test-chat.* @pwilkin
|
||||
/tests/test-llama-archs.cpp @JohannesGaessler
|
||||
/tools/batched-bench/ @ggerganov
|
||||
/tools/cli/ @ngxson
|
||||
/tools/completion/ @ggerganov
|
||||
|
||||
+2
-2
@@ -20,8 +20,8 @@ If AI is used to generate any portion of the code, contributors must adhere to t
|
||||
|
||||
1. Explicitly disclose the manner in which AI was employed.
|
||||
2. Check for an existing PR addressing the same change; if one exists, comment there to work with its author instead of opening a duplicate.
|
||||
3. Perform a comprehensive manual review prior to submitting the pull request.
|
||||
4. Be prepared to explain every line of code they submitted when asked about it by a maintainer.
|
||||
3. Perform a comprehensive manual review prior to submitting the pull request. A proper code review usually takes something like one hour per 200-400 LOC and you should be spending **at least that much time on code review alone**.
|
||||
4. Be prepared to explain every line of code you submit when asked about it by a maintainer.
|
||||
5. It is strictly prohibited to use AI to write your posts for you (bug reports, feature requests, pull request descriptions, Github discussions, responding to humans, ...).
|
||||
|
||||
For more info, please refer to the [AGENTS.md](AGENTS.md) file.
|
||||
|
||||
@@ -17,14 +17,16 @@ find_library(llama_LIBRARY llama
|
||||
NO_CMAKE_FIND_ROOT_PATH
|
||||
)
|
||||
|
||||
add_library(llama UNKNOWN IMPORTED)
|
||||
set_target_properties(llama
|
||||
PROPERTIES
|
||||
INTERFACE_INCLUDE_DIRECTORIES "${LLAMA_INCLUDE_DIR}"
|
||||
INTERFACE_LINK_LIBRARIES "ggml::ggml;ggml::ggml-base;"
|
||||
IMPORTED_LINK_INTERFACE_LANGUAGES "CXX"
|
||||
IMPORTED_LOCATION "${llama_LIBRARY}"
|
||||
INTERFACE_COMPILE_FEATURES c_std_90
|
||||
POSITION_INDEPENDENT_CODE ON)
|
||||
if(NOT TARGET llama)
|
||||
add_library(llama UNKNOWN IMPORTED)
|
||||
set_target_properties(llama
|
||||
PROPERTIES
|
||||
INTERFACE_INCLUDE_DIRECTORIES "${LLAMA_INCLUDE_DIR}"
|
||||
INTERFACE_LINK_LIBRARIES "ggml::ggml;ggml::ggml-base;"
|
||||
IMPORTED_LINK_INTERFACE_LANGUAGES "CXX"
|
||||
IMPORTED_LOCATION "${llama_LIBRARY}"
|
||||
INTERFACE_COMPILE_FEATURES c_std_90
|
||||
POSITION_INDEPENDENT_CODE ON)
|
||||
endif()
|
||||
|
||||
check_required_components(Llama)
|
||||
|
||||
+17
-8
@@ -2016,7 +2016,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.sampling.temp = std::max(params.sampling.temp, 0.0f);
|
||||
params.sampling.user_sampling_config |= common_params_sampling_config::COMMON_PARAMS_SAMPLING_CONFIG_TEMP;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_TEMPERATURE"));
|
||||
add_opt(common_arg(
|
||||
{"--top-k"}, "N",
|
||||
string_format("top-k sampling (default: %d, 0 = disabled)", params.sampling.top_k),
|
||||
@@ -2032,7 +2032,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.sampling.top_p = std::stof(value);
|
||||
params.sampling.user_sampling_config |= common_params_sampling_config::COMMON_PARAMS_SAMPLING_CONFIG_TOP_P;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_TOP_P"));
|
||||
add_opt(common_arg(
|
||||
{"--min-p"}, "N",
|
||||
string_format("min-p sampling (default: %.2f, 0.0 = disabled)", (double)params.sampling.min_p),
|
||||
@@ -2040,7 +2040,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.sampling.min_p = std::stof(value);
|
||||
params.sampling.user_sampling_config |= common_params_sampling_config::COMMON_PARAMS_SAMPLING_CONFIG_MIN_P;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_MIN_P"));
|
||||
add_opt(common_arg(
|
||||
{"--top-nsigma", "--top-n-sigma"}, "N",
|
||||
string_format("top-n-sigma sampling (default: %.2f, -1.0 = disabled)", params.sampling.top_n_sigma),
|
||||
@@ -2096,7 +2096,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.sampling.penalty_repeat = penalty_repeat;
|
||||
params.sampling.user_sampling_config |= common_params_sampling_config::COMMON_PARAMS_SAMPLING_CONFIG_PENALTY_REPEAT;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_REPEAT_PENALTY"));
|
||||
add_opt(common_arg(
|
||||
{"--presence-penalty"}, "N",
|
||||
string_format("repeat alpha presence penalty (default: %.2f, 0.0 = disabled)", (double)params.sampling.penalty_present),
|
||||
@@ -2107,7 +2107,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
}
|
||||
params.sampling.penalty_present = penalty_present;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_PRESENCE_PENALTY"));
|
||||
add_opt(common_arg(
|
||||
{"--frequency-penalty"}, "N",
|
||||
string_format("repeat alpha frequency penalty (default: %.2f, 0.0 = disabled)", (double)params.sampling.penalty_freq),
|
||||
@@ -2118,7 +2118,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
}
|
||||
params.sampling.penalty_freq = penalty_freq;
|
||||
}
|
||||
).set_sampling());
|
||||
).set_sampling().set_env("LLAMA_ARG_FREQUENCY_PENALTY"));
|
||||
add_opt(common_arg(
|
||||
{"--dry-multiplier"}, "N",
|
||||
string_format("set DRY sampling multiplier (default: %.2f, 0.0 = disabled)", (double)params.sampling.dry_multiplier),
|
||||
@@ -3308,9 +3308,18 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
).set_examples({LLAMA_EXAMPLE_EMBEDDING}));
|
||||
add_opt(common_arg(
|
||||
{"--host"}, "HOST",
|
||||
string_format("ip address to listen, or bind to an UNIX socket if the address ends with .sock (default: %s)", params.hostname.c_str()),
|
||||
string_format("IP addresses to listen on, comma-separated, or UNIX socket paths ending in .sock; with multiple TCP addresses, :: binds IPv6 only; overlapping addresses result in undefined behavior (default: %s)", params.hostnames[0].c_str()),
|
||||
[](common_params & params, const std::string & value) {
|
||||
params.hostname = value;
|
||||
params.hostnames.clear();
|
||||
for (auto & host : parse_csv_row(value)) {
|
||||
host = string_strip(host);
|
||||
if (!host.empty()) {
|
||||
params.hostnames.push_back(host);
|
||||
}
|
||||
}
|
||||
if (params.hostnames.empty()) {
|
||||
throw std::invalid_argument("--host requires at least one address");
|
||||
}
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_SERVER}).set_env("LLAMA_ARG_HOST"));
|
||||
add_opt(common_arg(
|
||||
|
||||
@@ -122,6 +122,8 @@ struct common_params_context {
|
||||
|
||||
// parse input arguments from CLI
|
||||
// if one argument has invalid value, it will automatically display usage of the specific argument (and not the full usage message)
|
||||
// TODO: this function can load ggml backend (by calling llama_support_rpc)
|
||||
// this is a side-effect that should be avoided
|
||||
bool common_params_parse(int argc, char ** argv, common_params & params, llama_example ex, void(*print_usage)(int, char **) = nullptr);
|
||||
|
||||
// load all backends and print the list of available (non-CPU) devices to stdout
|
||||
|
||||
@@ -318,13 +318,13 @@ void common_chat_peg_mapper::map(const common_peg_ast_node & node) {
|
||||
bool is_content = node.tag == common_chat_peg_builder::CONTENT;
|
||||
|
||||
if (is_reasoning) { // GPT OSS can have more than 1 reasoning block, so concatenate here
|
||||
result.reasoning_content += std::string(node.text);
|
||||
result.reasoning_content += node.sanitized_text();
|
||||
}
|
||||
|
||||
if (is_content) {
|
||||
// Concatenate content from multiple content nodes (e.g., when reasoning markers
|
||||
// are preserved before content markers in reasoning_format=NONE mode)
|
||||
result.content += std::string(node.text);
|
||||
result.content += node.sanitized_text();
|
||||
}
|
||||
|
||||
// Handle tool-related tags (supporting both JSON and tagged formats)
|
||||
@@ -1058,12 +1058,12 @@ void common_chat_peg_gemma4_mapper::visit(const common_peg_ast_arena & arena, co
|
||||
const auto & node = arena.get(id);
|
||||
|
||||
if (node.tag == "reasoning") {
|
||||
result.reasoning_content += std::string(node.text);
|
||||
result.reasoning_content += node.sanitized_text();
|
||||
return;
|
||||
}
|
||||
|
||||
if (node.tag == "content") {
|
||||
result.content += std::string(node.text);
|
||||
result.content += node.sanitized_text();
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1206,12 +1206,12 @@ void common_chat_peg_minimax_m3_mapper::visit(const common_peg_ast_arena & arena
|
||||
const auto & node = arena.get(id);
|
||||
|
||||
if (node.tag == common_chat_peg_builder::REASONING) {
|
||||
result.reasoning_content += std::string(node.text);
|
||||
result.reasoning_content += node.sanitized_text();
|
||||
return;
|
||||
}
|
||||
|
||||
if (node.tag == common_chat_peg_builder::CONTENT) {
|
||||
result.content += std::string(node.text);
|
||||
result.content += node.sanitized_text();
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
+11
-1
@@ -1133,6 +1133,14 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
return common_chat_params_init_kimi_k3(tmpl, params);
|
||||
}
|
||||
|
||||
// Ling 3.0 / Bailing V3 - <role>X</role> sections with <arg_key>/<arg_value> tagged
|
||||
// tool calls. <role> sections are unique to this family among the tagged-arg templates.
|
||||
if (src.find("<role>ASSISTANT</role>") != std::string::npos &&
|
||||
src.find("<arg_key>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Ling 3.0 (Bailing V3)\n");
|
||||
return common_chat_params_init_ling3(tmpl, params);
|
||||
}
|
||||
|
||||
// Cohere2 MoE / North Code - marker-wrapped format with <|START_TEXT|> content and
|
||||
// <|START_ACTION|> JSON tool calls. <|START_TEXT|> is unique to this template (the older
|
||||
// Command-R templates use <|START_RESPONSE|>).
|
||||
@@ -1204,7 +1212,9 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
// Qwen3-Coder XML tool calls, also used by Nemotron Nano 3, Qwen3.5 and StepFun-3.5-Flash
|
||||
if (src.find("<tool_call>") != std::string::npos &&
|
||||
src.find("<function=") != std::string::npos &&
|
||||
src.find("<parameter=") != std::string::npos) {
|
||||
src.find("<parameter=") != std::string::npos &&
|
||||
// Exclude models that don't use \n between tags
|
||||
src.find("'<tool_call><function=' ~ tool_call.name ~ '>'") == std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Qwen3-Coder\n");
|
||||
return common_chat_params_init_qwen3_coder(tmpl, params);
|
||||
}
|
||||
|
||||
+30
-7
@@ -2198,9 +2198,28 @@ bool common_replay_last_token(struct llama_context * ctx, llama_token last_token
|
||||
return true;
|
||||
}
|
||||
|
||||
llama_batch_ext_ptr common_batch_ext_get_one(llama_context * ctx, const llama_tokens & tokens) {
|
||||
llama_batch_ext_ptr batch(llama_batch_ext_init(ctx));
|
||||
|
||||
auto mem = llama_get_memory(ctx);
|
||||
llama_pos pos = mem ? llama_memory_seq_pos_max(mem, 0) + 1 : 0;
|
||||
|
||||
for (size_t i = 0; i < tokens.size(); ++i) {
|
||||
const int32_t idx = llama_batch_ext_add_token(batch.get(), 0, tokens[i]);
|
||||
llama_batch_ext_set_pos(batch.get(), idx, &pos);
|
||||
pos++;
|
||||
}
|
||||
|
||||
if (!tokens.empty()) {
|
||||
llama_batch_ext_set_output_logits(batch.get(), (int32_t) tokens.size() - 1, true);
|
||||
}
|
||||
|
||||
return batch;
|
||||
}
|
||||
|
||||
bool common_prompt_batch_decode(
|
||||
struct llama_context * ctx,
|
||||
const std::vector<llama_token> & all_tokens,
|
||||
const llama_tokens & all_tokens,
|
||||
int n_new,
|
||||
int & n_past,
|
||||
int n_batch,
|
||||
@@ -2221,7 +2240,9 @@ bool common_prompt_batch_decode(
|
||||
// Memory implementations in recurrent/hybrid models don't support removing tokens from their
|
||||
// memory, so we can't just remove the last token from the memory and replay the last token which
|
||||
// is the reason for this logic.
|
||||
if (llama_decode(ctx, llama_batch_get_one(const_cast<llama_token*>(all_tokens.data() + offset), n_tokens_before_last))) {
|
||||
llama_tokens prefix_tokens(all_tokens.begin() + offset, all_tokens.begin() + offset + n_tokens_before_last);
|
||||
llama_batch_ext_ptr batch_prefix = common_batch_ext_get_one(ctx, prefix_tokens);
|
||||
if (llama_process(ctx, LLAMA_PROCESS_TYPE_DECODE, batch_prefix.get())) {
|
||||
COM_ERR("%s", "failed to eval\n");
|
||||
return false;
|
||||
}
|
||||
@@ -2231,17 +2252,19 @@ bool common_prompt_batch_decode(
|
||||
COM_INF("saved session before last token to %s, n_new = %zu\n", state_path.data(), all_tokens.size());
|
||||
|
||||
llama_token last_token = all_tokens.back();
|
||||
llama_batch batch = llama_batch_get_one(&last_token, 1);
|
||||
int32_t pos = n_past;
|
||||
batch.pos = &pos;
|
||||
llama_batch_ext_ptr batch_last = common_batch_ext_get_one(ctx, { last_token });
|
||||
llama_pos pos = n_past;
|
||||
llama_batch_ext_set_pos(batch_last.get(), 0, &pos);
|
||||
|
||||
if (llama_decode(ctx, batch)) {
|
||||
if (llama_process(ctx, LLAMA_PROCESS_TYPE_DECODE, batch_last.get())) {
|
||||
COM_ERR("%s", "failed to eval last token\n");
|
||||
return false;
|
||||
}
|
||||
n_past++;
|
||||
} else {
|
||||
if (llama_decode(ctx, llama_batch_get_one(const_cast<llama_token*>(all_tokens.data() + offset), n_new))) {
|
||||
llama_tokens new_tokens(all_tokens.begin() + offset, all_tokens.begin() + offset + n_new);
|
||||
llama_batch_ext_ptr batch = common_batch_ext_get_one(ctx, new_tokens);
|
||||
if (llama_process(ctx, LLAMA_PROCESS_TYPE_DECODE, batch.get())) {
|
||||
COM_ERR("%s", "failed to eval\n");
|
||||
return false;
|
||||
}
|
||||
|
||||
+6
-2
@@ -631,10 +631,10 @@ struct common_params {
|
||||
int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints
|
||||
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
|
||||
|
||||
std::string hostname = "127.0.0.1";
|
||||
std::string public_path = ""; // NOLINT
|
||||
std::string api_prefix = ""; // NOLINT
|
||||
std::string chat_template = ""; // NOLINT
|
||||
std::vector<std::string> hostnames = {"127.0.0.1"};
|
||||
bool use_jinja = true; // NOLINT
|
||||
|
||||
// server CORS params
|
||||
@@ -1021,6 +1021,10 @@ void common_batch_add(
|
||||
const std::vector<llama_seq_id> & seq_ids,
|
||||
bool logits);
|
||||
|
||||
// create a single-sequence batch from a list of tokens
|
||||
// last token always have output_logits set to true
|
||||
llama_batch_ext_ptr common_batch_ext_get_one(struct llama_context * ctx, const llama_tokens & tokens);
|
||||
|
||||
// decodes a single batch of tokens for a prompt and manages session tokens
|
||||
//
|
||||
// Note: We save state before the last token so that we can replay it to ensure
|
||||
@@ -1028,7 +1032,7 @@ void common_batch_add(
|
||||
// tokens from memory, so this approach works across all model architectures.
|
||||
bool common_prompt_batch_decode(
|
||||
struct llama_context * ctx,
|
||||
const std::vector<llama_token> & all_tokens,
|
||||
const llama_tokens & all_tokens,
|
||||
int n_new,
|
||||
int & n_past,
|
||||
int n_batch,
|
||||
|
||||
+7
-7
@@ -192,9 +192,9 @@ static void common_params_fit_impl(
|
||||
uint32_t hp_nct = 0; // hparams.n_ctx_train
|
||||
uint32_t hp_nex = 0; // hparams.n_expert
|
||||
|
||||
// with non-unified kv, we need to take into account n_streams
|
||||
// for example, if memory can hold more than model's trained context size, we must extend the n_ctx to hold enough n_streams
|
||||
const uint32_t n_streams = cparams->kv_unified ? 1 : std::max<uint32_t>(1, cparams->n_seq_max);
|
||||
// size the context for all sequences, but keep minimums and alignment per KV stream
|
||||
const uint32_t n_seq_max = std::max<uint32_t>(1, cparams->n_seq_max);
|
||||
const uint32_t n_streams = cparams->kv_unified ? 1 : n_seq_max;
|
||||
const bool n_ctx_auto = cparams->n_ctx == 0;
|
||||
|
||||
dmds_t dmds_extra; // memory of the extra model, laid out on the devices of the main model
|
||||
@@ -264,15 +264,15 @@ static void common_params_fit_impl(
|
||||
dmds_t dmds_full = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);
|
||||
|
||||
// saturate instead of overflowing, this also preserves the UINT32_MAX sentinel of n_ctx_min:
|
||||
const uint32_t n_ctx_max = (uint32_t) std::min<uint64_t>(uint64_t(hp_nct) * n_streams, UINT32_MAX);
|
||||
const uint32_t n_ctx_max = (uint32_t) std::min<uint64_t>(uint64_t(hp_nct) * n_seq_max, UINT32_MAX);
|
||||
const uint32_t n_ctx_min_total = (uint32_t) std::min<uint64_t>(uint64_t(n_ctx_min) * n_streams, UINT32_MAX);
|
||||
|
||||
// llama_context would use only hp_nct in total for n_ctx == 0, resolve the context before measuring anything else:
|
||||
if (n_ctx_auto) {
|
||||
cparams->n_ctx = n_ctx_max;
|
||||
if (n_streams > 1) {
|
||||
LOG_TRC("%s: context size unset and KV cache not unified -> using %" PRIu32 " for %" PRIu32 " sequences:\n",
|
||||
__func__, n_ctx_max, n_streams);
|
||||
if (n_seq_max > 1) {
|
||||
LOG_TRC("%s: context size unset -> using %" PRIu32 " for %" PRIu32 " sequences:\n",
|
||||
__func__, n_ctx_max, n_seq_max);
|
||||
dmds_full = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);
|
||||
}
|
||||
}
|
||||
|
||||
+12
-3
@@ -62,6 +62,15 @@ static fs::path get_cache_directory() {
|
||||
return cache;
|
||||
}
|
||||
|
||||
std::string get_cache_path() {
|
||||
#if defined(__cpp_lib_char8_t)
|
||||
const std::u8string u8str = get_cache_directory().u8string();
|
||||
return std::string(reinterpret_cast<const char *>(u8str.data()), u8str.size());
|
||||
#else
|
||||
return get_cache_directory().u8string();
|
||||
#endif
|
||||
}
|
||||
|
||||
static std::string folder_name_to_repo(const std::string & folder) {
|
||||
constexpr std::string_view prefix = "models--";
|
||||
if (folder.rfind(prefix, 0)) {
|
||||
@@ -393,8 +402,8 @@ static std::string get_cached_ref(const fs::path & repo_path) {
|
||||
}
|
||||
|
||||
hf_files get_cached_files(const std::string & repo_id) {
|
||||
fs::path cache_dir = get_cache_directory();
|
||||
if (!fs::exists(cache_dir)) {
|
||||
const fs::path cache_path = get_cache_directory();
|
||||
if (!fs::exists(cache_path)) {
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -405,7 +414,7 @@ hf_files get_cached_files(const std::string & repo_id) {
|
||||
|
||||
hf_files files;
|
||||
|
||||
for (const auto & repo : fs::directory_iterator(cache_dir)) {
|
||||
for (const auto & repo : fs::directory_iterator(cache_path)) {
|
||||
if (!repo.is_directory()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -32,4 +32,7 @@ std::string finalize_file(const hf_file & file);
|
||||
// Remove the entire cached directory for a repo, returns true if removed
|
||||
bool remove_cached_repo(const std::string & repo_id);
|
||||
|
||||
// Returns the HuggingFace hub cache path
|
||||
std::string get_cache_path();
|
||||
|
||||
} // namespace hf_cache
|
||||
|
||||
+11
-1
@@ -437,7 +437,8 @@ private:
|
||||
}
|
||||
|
||||
statement_ptr parse_filter_expression() {
|
||||
auto operand = parse_call_member_expression();
|
||||
// Filters/tests bind outside unary so -n|abs is (-n)|abs, not -(n|abs).
|
||||
auto operand = parse_unary_expression();
|
||||
while (is(token::pipe)) {
|
||||
size_t start_pos = current;
|
||||
++current; // consume pipe
|
||||
@@ -448,6 +449,15 @@ private:
|
||||
return operand;
|
||||
}
|
||||
|
||||
statement_ptr parse_unary_expression() {
|
||||
if (is(token::unary_operator)) {
|
||||
size_t start_pos = current;
|
||||
auto op = next();
|
||||
return mk_stmt<unary_expression>(start_pos, op, parse_unary_expression());
|
||||
}
|
||||
return parse_call_member_expression();
|
||||
}
|
||||
|
||||
statement_ptr parse_call_member_expression() {
|
||||
// Handle member expressions recursively
|
||||
auto member = parse_member_expression(parse_primary_expression());
|
||||
|
||||
+30
-34
@@ -51,7 +51,7 @@ static void ensure_key_type_allowed(const value & val) {
|
||||
}
|
||||
|
||||
// execute with error handling
|
||||
value statement::execute(context & ctx) {
|
||||
value statement::execute(context & ctx) const {
|
||||
try {
|
||||
return execute_impl(ctx);
|
||||
} catch (const continue_statement::signal & /* ex */) {
|
||||
@@ -80,7 +80,7 @@ value statement::execute(context & ctx) {
|
||||
}
|
||||
}
|
||||
|
||||
value identifier::execute_impl(context & ctx) {
|
||||
value identifier::execute_impl(context & ctx) const {
|
||||
auto it = ctx.get_val(val);
|
||||
auto builtins = global_builtins();
|
||||
if (!it->is_undefined()) {
|
||||
@@ -98,7 +98,7 @@ value identifier::execute_impl(context & ctx) {
|
||||
}
|
||||
}
|
||||
|
||||
value object_literal::execute_impl(context & ctx) {
|
||||
value object_literal::execute_impl(context & ctx) const {
|
||||
auto obj = mk_val<value_object>();
|
||||
for (const auto & pair : val) {
|
||||
value key = pair.first->execute(ctx);
|
||||
@@ -109,7 +109,7 @@ value object_literal::execute_impl(context & ctx) {
|
||||
return obj;
|
||||
}
|
||||
|
||||
value binary_expression::execute_impl(context & ctx) {
|
||||
value binary_expression::execute_impl(context & ctx) const {
|
||||
value left_val = left->execute(ctx);
|
||||
|
||||
// Logical operators
|
||||
@@ -317,9 +317,7 @@ static value try_builtin_func(context & ctx, const std::string & name, value & i
|
||||
throw std::runtime_error("Unknown (built-in) filter '" + name + "' for type " + input->type());
|
||||
}
|
||||
|
||||
value filter_expression::execute_impl(context & ctx) {
|
||||
value input = operand ? operand->execute(ctx) : val;
|
||||
|
||||
static value apply_filter(context & ctx, const statement_ptr & filter, value input) {
|
||||
JJ_DEBUG("Applying filter to %s", input->type().c_str());
|
||||
|
||||
auto set_filter_alias = [](auto & filter_id) {
|
||||
@@ -375,22 +373,21 @@ value filter_expression::execute_impl(context & ctx) {
|
||||
}
|
||||
}
|
||||
|
||||
value filter_statement::execute_impl(context & ctx) {
|
||||
value filter_expression::execute_impl(context & ctx) const {
|
||||
return apply_filter(ctx, filter, operand->execute(ctx));
|
||||
}
|
||||
|
||||
value filter_statement::execute_impl(context & ctx) const {
|
||||
// eval body as string, then apply filter
|
||||
auto body_val = exec_statements(body, ctx);
|
||||
value_string parts = mk_val<value_string>();
|
||||
gather_string_parts_recursive(body_val, parts);
|
||||
|
||||
JJ_DEBUG("FilterStatement: applying filter to body string of length %zu", parts->val_str.length());
|
||||
filter_expression filter_expr(std::move(parts), std::move(filter));
|
||||
value out = filter_expr.execute(ctx);
|
||||
|
||||
// this node can be reused later, make sure filter is preserved
|
||||
this->filter = std::move(filter_expr.filter);
|
||||
return out;
|
||||
return apply_filter(ctx, filter, parts);
|
||||
}
|
||||
|
||||
value test_expression::execute_impl(context & ctx) {
|
||||
value test_expression::execute_impl(context & ctx) const {
|
||||
// NOTE: "value is something" translates to function call "test_is_something(value)"
|
||||
const auto & builtins = global_builtins();
|
||||
|
||||
@@ -439,7 +436,7 @@ value test_expression::execute_impl(context & ctx) {
|
||||
}
|
||||
}
|
||||
|
||||
value unary_expression::execute_impl(context & ctx) {
|
||||
value unary_expression::execute_impl(context & ctx) const {
|
||||
value operand_val = argument->execute(ctx);
|
||||
JJ_DEBUG("Executing unary expression with operator '%s'", op.value.c_str());
|
||||
|
||||
@@ -453,12 +450,17 @@ value unary_expression::execute_impl(context & ctx) {
|
||||
} else {
|
||||
throw std::runtime_error("Unary - operator requires numeric operand");
|
||||
}
|
||||
} else if (op.value == "+") {
|
||||
if (is_val<value_int>(operand_val) || is_val<value_float>(operand_val)) {
|
||||
return operand_val;
|
||||
}
|
||||
throw std::runtime_error("Unary + operator requires numeric operand");
|
||||
}
|
||||
|
||||
throw std::runtime_error("Unknown unary operator '" + op.value + "'");
|
||||
}
|
||||
|
||||
value if_statement::execute_impl(context & ctx) {
|
||||
value if_statement::execute_impl(context & ctx) const {
|
||||
value test_val = test->execute(ctx);
|
||||
|
||||
auto out = mk_val<value_array>();
|
||||
@@ -479,20 +481,14 @@ value if_statement::execute_impl(context & ctx) {
|
||||
return str;
|
||||
}
|
||||
|
||||
value for_statement::execute_impl(context & ctx) {
|
||||
value for_statement::execute_impl(context & ctx) const {
|
||||
context scope(ctx); // new scope for loop variables
|
||||
|
||||
jinja::select_expression * select_expr = cast_stmt<select_expression>(iterable);
|
||||
const jinja::select_expression * select_expr = cast_stmt<select_expression>(iterable);
|
||||
statement_ptr test_expr_nullptr;
|
||||
|
||||
statement_ptr & iter_expr = [&]() -> statement_ptr & {
|
||||
auto tmp = cast_stmt<select_expression>(iterable);
|
||||
return tmp ? tmp->lhs : iterable;
|
||||
}();
|
||||
statement_ptr & test_expr = [&]() -> statement_ptr & {
|
||||
auto tmp = cast_stmt<select_expression>(iterable);
|
||||
return tmp ? tmp->test : test_expr_nullptr;
|
||||
}();
|
||||
const statement_ptr & iter_expr = select_expr ? select_expr->lhs : iterable;
|
||||
const statement_ptr & test_expr = select_expr ? select_expr->test : test_expr_nullptr;
|
||||
|
||||
JJ_DEBUG("Executing for statement, iterable type: %s", iter_expr->type().c_str());
|
||||
|
||||
@@ -645,7 +641,7 @@ value for_statement::execute_impl(context & ctx) {
|
||||
return str;
|
||||
}
|
||||
|
||||
value set_statement::execute_impl(context & ctx) {
|
||||
value set_statement::execute_impl(context & ctx) const {
|
||||
auto rhs = val ? val->execute(ctx) : exec_statements(body, ctx);
|
||||
|
||||
if (is_stmt<identifier>(assignee)) {
|
||||
@@ -744,7 +740,7 @@ static inline void bind_parameters(const std::string & name, const statements &
|
||||
}
|
||||
}
|
||||
|
||||
value macro_statement::execute_impl(context & ctx) {
|
||||
value macro_statement::execute_impl(context & ctx) const {
|
||||
if (!is_stmt<identifier>(this->name)) {
|
||||
throw std::runtime_error("Macro name must be an identifier");
|
||||
}
|
||||
@@ -767,7 +763,7 @@ value macro_statement::execute_impl(context & ctx) {
|
||||
return mk_val<value_undefined>();
|
||||
}
|
||||
|
||||
value call_statement::execute_impl(context & ctx) {
|
||||
value call_statement::execute_impl(context & ctx) const {
|
||||
auto call_expr = cast_stmt<call_expression>(this->call);
|
||||
if (!call_expr) {
|
||||
throw std::runtime_error("Call statement requires a valid call expression");
|
||||
@@ -807,7 +803,7 @@ value call_statement::execute_impl(context & ctx) {
|
||||
return callee_func->invoke(args);
|
||||
}
|
||||
|
||||
value member_expression::execute_impl(context & ctx) {
|
||||
value member_expression::execute_impl(context & ctx) const {
|
||||
value object = this->object->execute(ctx);
|
||||
|
||||
value property;
|
||||
@@ -940,7 +936,7 @@ value member_expression::execute_impl(context & ctx) {
|
||||
return val;
|
||||
}
|
||||
|
||||
value call_expression::execute_impl(context & ctx) {
|
||||
value call_expression::execute_impl(context & ctx) const {
|
||||
// gather arguments
|
||||
func_args args(ctx);
|
||||
for (auto & arg_stmt : this->args) {
|
||||
@@ -958,7 +954,7 @@ value call_expression::execute_impl(context & ctx) {
|
||||
return callee_func->invoke(args);
|
||||
}
|
||||
|
||||
value keyword_argument_expression::execute_impl(context & ctx) {
|
||||
value keyword_argument_expression::execute_impl(context & ctx) const {
|
||||
if (!is_stmt<identifier>(key)) {
|
||||
throw std::runtime_error("Keyword argument key must be identifiers");
|
||||
}
|
||||
@@ -982,7 +978,7 @@ std::string runtime::debug_dump_program(const program & prog, const std::string
|
||||
return std::string(lvl * 2, ' ');
|
||||
};
|
||||
|
||||
ctx.visitor = [&](bool is_leaf, statement * node, std::vector<visitor_pair> children) {
|
||||
ctx.visitor = [&](bool is_leaf, const statement * node, std::vector<visitor_pair> children) {
|
||||
oss << indent(lvl) << node->type() << ":\n";
|
||||
lvl++;
|
||||
if (is_leaf) {
|
||||
|
||||
+55
-62
@@ -48,9 +48,9 @@ const T * cast_stmt(const statement_ptr & ptr) {
|
||||
void enable_debug(bool enable);
|
||||
|
||||
// for visiting AST nodes
|
||||
// function signature: void(bool is_leaf, statement * node, pair of <label, children>)
|
||||
using visitor_pair = std::pair<std::string, std::vector<statement *>>;
|
||||
using visitor_fn = std::function<void(bool, statement *, std::vector<visitor_pair>)>;
|
||||
// function signature: void(bool is_leaf, const statement * node, pair of <label, children>)
|
||||
using visitor_pair = std::pair<std::string, std::vector<const statement *>>;
|
||||
using visitor_fn = std::function<void(bool, const statement *, std::vector<visitor_pair>)>;
|
||||
|
||||
struct context {
|
||||
std::shared_ptr<std::string> src; // for debugging; use shared_ptr to avoid copying on scope creation
|
||||
@@ -107,8 +107,8 @@ private:
|
||||
};
|
||||
|
||||
// utils for visiting AST nodes
|
||||
static std::vector<statement *> stmts_to_ptr(const statements & stmts) {
|
||||
std::vector<statement *> children;
|
||||
static std::vector<const statement *> stmts_to_ptr(const statements & stmts) {
|
||||
std::vector<const statement *> children;
|
||||
for (const auto & stmt : stmts) {
|
||||
children.push_back(stmt.get());
|
||||
}
|
||||
@@ -117,17 +117,18 @@ static std::vector<statement *> stmts_to_ptr(const statements & stmts) {
|
||||
|
||||
/**
|
||||
* Base class for all nodes in the AST.
|
||||
* The AST is shared between threads, so visit and execute must be const.
|
||||
*/
|
||||
struct statement {
|
||||
size_t pos; // position in source, for debugging
|
||||
virtual ~statement() = default;
|
||||
virtual std::string type() const { return "Statement"; }
|
||||
virtual void visit(context & ctx) { ctx.visitor(true, this, {}); }
|
||||
virtual void visit(context & ctx) const { ctx.visitor(true, this, {}); }
|
||||
|
||||
// execute_impl must be overridden by derived classes
|
||||
virtual value execute_impl(context &) { throw_exec_error(); }
|
||||
virtual value execute_impl(context &) const { throw_exec_error(); }
|
||||
// execute is the public method to execute a statement with error handling
|
||||
value execute(context &);
|
||||
value execute(context &) const;
|
||||
|
||||
private:
|
||||
[[noreturn]] void throw_exec_error() const {
|
||||
@@ -166,7 +167,7 @@ struct program : public statement {
|
||||
program() = default;
|
||||
explicit program(statements && body) : body(std::move(body)) {}
|
||||
std::string type() const override { return "Program"; }
|
||||
[[noreturn]] value execute_impl(context &) override {
|
||||
[[noreturn]] value execute_impl(context &) const override {
|
||||
throw std::runtime_error("Cannot execute program directly, use jinja::runtime instead");
|
||||
}
|
||||
};
|
||||
@@ -182,8 +183,8 @@ struct if_statement : public statement {
|
||||
}
|
||||
|
||||
std::string type() const override { return "If"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"test", {test.get()}},
|
||||
{"body", stmts_to_ptr(body)},
|
||||
@@ -213,8 +214,8 @@ struct for_statement : public statement {
|
||||
}
|
||||
|
||||
std::string type() const override { return "For"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"loopvar", {loopvar.get()}},
|
||||
{"iterable", {iterable.get()}},
|
||||
@@ -233,7 +234,7 @@ struct break_statement : public statement {
|
||||
}
|
||||
};
|
||||
|
||||
[[noreturn]] value execute_impl(context &) override {
|
||||
[[noreturn]] value execute_impl(context &) const override {
|
||||
throw break_statement::signal();
|
||||
}
|
||||
};
|
||||
@@ -247,7 +248,7 @@ struct continue_statement : public statement {
|
||||
}
|
||||
};
|
||||
|
||||
[[noreturn]] value execute_impl(context &) override {
|
||||
[[noreturn]] value execute_impl(context &) const override {
|
||||
throw continue_statement::signal();
|
||||
}
|
||||
};
|
||||
@@ -255,7 +256,7 @@ struct continue_statement : public statement {
|
||||
// do nothing
|
||||
struct noop_statement : public statement {
|
||||
std::string type() const override { return "Noop"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_undefined>();
|
||||
}
|
||||
};
|
||||
@@ -272,8 +273,8 @@ struct set_statement : public statement {
|
||||
}
|
||||
|
||||
std::string type() const override { return "Set"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"assignee", {assignee.get()}},
|
||||
{"value", {val.get()}},
|
||||
@@ -294,8 +295,8 @@ struct macro_statement : public statement {
|
||||
}
|
||||
|
||||
std::string type() const override { return "Macro"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"name", {name.get()}},
|
||||
{"args", stmts_to_ptr(args)},
|
||||
@@ -308,7 +309,7 @@ struct comment_statement : public statement {
|
||||
std::string val;
|
||||
explicit comment_statement(const std::string & v) : val(v) {}
|
||||
std::string type() const override { return "Comment"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_undefined>();
|
||||
}
|
||||
};
|
||||
@@ -318,7 +319,7 @@ struct comment_statement : public statement {
|
||||
// Represents an omitted expression in a computed member, e.g. `a[]`.
|
||||
struct blank_expression : public expression {
|
||||
std::string type() const override { return "BlankExpression"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_undefined>();
|
||||
}
|
||||
};
|
||||
@@ -334,8 +335,8 @@ struct member_expression : public expression {
|
||||
chk_type<expression>(this->property);
|
||||
}
|
||||
std::string type() const override { return "MemberExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"object", {object.get()}},
|
||||
{"property", {property.get()}}
|
||||
@@ -353,8 +354,8 @@ struct call_expression : public expression {
|
||||
for (const auto& arg : this->args) chk_type<expression>(arg);
|
||||
}
|
||||
std::string type() const override { return "CallExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"callee", {callee.get()}},
|
||||
{"args", stmts_to_ptr(args)}
|
||||
@@ -369,7 +370,7 @@ struct identifier : public expression {
|
||||
std::string val;
|
||||
explicit identifier(const std::string & val) : val(val) {}
|
||||
std::string type() const override { return "Identifier"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
value execute_impl(context & ctx) const override;
|
||||
};
|
||||
|
||||
// Literals
|
||||
@@ -378,7 +379,7 @@ struct integer_literal : public expression {
|
||||
int64_t val;
|
||||
explicit integer_literal(int64_t val) : val(val) {}
|
||||
std::string type() const override { return "IntegerLiteral"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_int>(val);
|
||||
}
|
||||
};
|
||||
@@ -387,7 +388,7 @@ struct float_literal : public expression {
|
||||
double val;
|
||||
explicit float_literal(double val) : val(val) {}
|
||||
std::string type() const override { return "FloatLiteral"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_float>(val);
|
||||
}
|
||||
};
|
||||
@@ -396,7 +397,7 @@ struct string_literal : public expression {
|
||||
std::string val;
|
||||
explicit string_literal(const std::string & val) : val(val) {}
|
||||
std::string type() const override { return "StringLiteral"; }
|
||||
value execute_impl(context &) override {
|
||||
value execute_impl(context &) const override {
|
||||
return mk_val<value_string>(val);
|
||||
}
|
||||
};
|
||||
@@ -407,7 +408,7 @@ struct array_literal : public expression {
|
||||
for (const auto& item : this->val) chk_type<expression>(item);
|
||||
}
|
||||
std::string type() const override { return "ArrayLiteral"; }
|
||||
value execute_impl(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override {
|
||||
auto arr = mk_val<value_array>();
|
||||
for (const auto & item_stmt : val) {
|
||||
arr->push_back(item_stmt->execute(ctx));
|
||||
@@ -422,7 +423,7 @@ struct tuple_literal : public expression {
|
||||
for (const auto& item : this->val) chk_type<expression>(item);
|
||||
}
|
||||
std::string type() const override { return "TupleLiteral"; }
|
||||
value execute_impl(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override {
|
||||
auto arr = mk_val<value_array>();
|
||||
for (const auto & item_stmt : val) {
|
||||
arr->push_back(item_stmt->execute(ctx));
|
||||
@@ -441,7 +442,7 @@ struct object_literal : public expression {
|
||||
}
|
||||
}
|
||||
std::string type() const override { return "ObjectLiteral"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
value execute_impl(context & ctx) const override;
|
||||
};
|
||||
|
||||
// Complex Expressions
|
||||
@@ -462,8 +463,8 @@ struct binary_expression : public expression {
|
||||
chk_type<expression>(this->right);
|
||||
}
|
||||
std::string type() const override { return "BinaryExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"left", {left.get()}},
|
||||
{"right", {right.get()}}
|
||||
@@ -476,10 +477,7 @@ struct binary_expression : public expression {
|
||||
* Operator precedence: https://github.com/pallets/jinja/issues/379#issuecomment-168076202
|
||||
*/
|
||||
struct filter_expression : public expression {
|
||||
// either an expression or a value is allowed
|
||||
statement_ptr operand;
|
||||
value_string val; // will be set by filter_statement
|
||||
|
||||
statement_ptr filter;
|
||||
|
||||
filter_expression(statement_ptr && operand, statement_ptr && filter)
|
||||
@@ -488,14 +486,9 @@ struct filter_expression : public expression {
|
||||
chk_type<identifier, call_expression>(this->filter);
|
||||
}
|
||||
|
||||
filter_expression(value_string && val, statement_ptr && filter)
|
||||
: val(std::move(val)), filter(std::move(filter)) {
|
||||
chk_type<identifier, call_expression>(this->filter);
|
||||
}
|
||||
|
||||
std::string type() const override { return "FilterExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"operand", {operand.get()}},
|
||||
{"filter", {filter.get()}}
|
||||
@@ -512,8 +505,8 @@ struct filter_statement : public statement {
|
||||
chk_type<identifier, call_expression>(this->filter);
|
||||
}
|
||||
std::string type() const override { return "FilterStatement"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"filter", {filter.get()}},
|
||||
{"body", stmts_to_ptr(body)}
|
||||
@@ -537,14 +530,14 @@ struct select_expression : public expression {
|
||||
chk_type<expression>(this->test);
|
||||
}
|
||||
std::string type() const override { return "SelectExpression"; }
|
||||
value execute_impl(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override {
|
||||
auto predicate = test->execute_impl(ctx);
|
||||
if (!predicate->as_bool()) {
|
||||
return mk_val<value_undefined>();
|
||||
}
|
||||
return lhs->execute_impl(ctx);
|
||||
}
|
||||
void visit(context & ctx) override {
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"lhs", {lhs.get()}},
|
||||
{"test", {test.get()}}
|
||||
@@ -567,8 +560,8 @@ struct test_expression : public expression {
|
||||
chk_type<identifier, call_expression>(this->test);
|
||||
}
|
||||
std::string type() const override { return "TestExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"operand", {operand.get()}},
|
||||
{"test", {test.get()}}
|
||||
@@ -588,8 +581,8 @@ struct unary_expression : public expression {
|
||||
chk_type<expression>(this->argument);
|
||||
}
|
||||
std::string type() const override { return "UnaryExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"argument", {argument.get()}}
|
||||
});
|
||||
@@ -608,10 +601,10 @@ struct slice_expression : public expression {
|
||||
chk_type<expression>(this->step_expr);
|
||||
}
|
||||
std::string type() const override { return "SliceExpression"; }
|
||||
[[noreturn]] value execute_impl(context &) override {
|
||||
[[noreturn]] value execute_impl(context &) const override {
|
||||
throw std::runtime_error("must be handled by MemberExpression");
|
||||
}
|
||||
void visit(context & ctx) override {
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"start_expr", {start_expr.get()}},
|
||||
{"stop_expr", {stop_expr.get()}},
|
||||
@@ -630,8 +623,8 @@ struct keyword_argument_expression : public expression {
|
||||
chk_type<expression>(this->val);
|
||||
}
|
||||
std::string type() const override { return "KeywordArgumentExpression"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"key", {key.get()}},
|
||||
{"val", {val.get()}}
|
||||
@@ -645,7 +638,7 @@ struct spread_expression : public expression {
|
||||
chk_type<expression>(this->argument);
|
||||
}
|
||||
std::string type() const override { return "SpreadExpression"; }
|
||||
void visit(context & ctx) override {
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"argument", {argument.get()}}
|
||||
});
|
||||
@@ -663,8 +656,8 @@ struct call_statement : public statement {
|
||||
for (const auto & arg : this->caller_args) chk_type<expression>(arg);
|
||||
}
|
||||
std::string type() const override { return "CallStatement"; }
|
||||
value execute_impl(context & ctx) override;
|
||||
void visit(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override;
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"call", {call.get()}},
|
||||
{"caller_args", stmts_to_ptr(caller_args)},
|
||||
@@ -685,7 +678,7 @@ struct ternary_expression : public expression {
|
||||
chk_type<expression>(this->false_expr);
|
||||
}
|
||||
std::string type() const override { return "Ternary"; }
|
||||
value execute_impl(context & ctx) override {
|
||||
value execute_impl(context & ctx) const override {
|
||||
value cond_val = condition->execute(ctx);
|
||||
if (cond_val->as_bool()) {
|
||||
return true_expr->execute(ctx);
|
||||
@@ -693,7 +686,7 @@ struct ternary_expression : public expression {
|
||||
return false_expr->execute(ctx);
|
||||
}
|
||||
}
|
||||
void visit(context & ctx) override {
|
||||
void visit(context & ctx) const override {
|
||||
ctx.visitor(false, this, {
|
||||
{"condition", {condition.get()}},
|
||||
{"true_expr", {true_expr.get()}},
|
||||
|
||||
@@ -321,7 +321,8 @@ static size_t gbnf_escape_length(const std::string & pattern, size_t pos) {
|
||||
case 'x': n_hex = 2; break;
|
||||
case 'u': n_hex = 4; break;
|
||||
case 'U': n_hex = 8; break;
|
||||
case 't': case 'r': case 'n': case '\\': case '"': case '[': case ']':
|
||||
// keep in sync with parse_char() in src/llama-grammar.cpp
|
||||
case 't': case 'r': case 'n': case '\\': case '"': case '[': case ']': case '-':
|
||||
return 2;
|
||||
default:
|
||||
return 0;
|
||||
|
||||
@@ -82,6 +82,9 @@ struct common_json_value {
|
||||
// note: a nested pair {"a", "b"} does not build, use common_json::array({"a", "b"}) for an array
|
||||
common_json_value(std::initializer_list<common_json_item> items);
|
||||
|
||||
template <typename T, typename std::enable_if<std::is_enum<T>::value, int>::type = 0>
|
||||
common_json_value(T val) : common_json_value((typename std::underlying_type<T>::type) val) {}
|
||||
|
||||
template <typename T, typename std::enable_if<std::is_integral<T>::value && !std::is_same<T, bool>::value, int>::type = 0>
|
||||
common_json_value(T val) : type(std::is_signed<T>::value ? VAL_INT : VAL_UINT) {
|
||||
if (std::is_signed<T>::value) {
|
||||
@@ -111,6 +114,7 @@ struct common_json_item {
|
||||
// the types common_json_value holds on its own
|
||||
// anything else reaches its common_json ctor and recurses forever
|
||||
template <typename T> struct common_json_is_value : std::integral_constant<bool,
|
||||
std::is_enum<T>::value ||
|
||||
std::is_arithmetic<T>::value ||
|
||||
std::is_same<T, std::nullptr_t>::value ||
|
||||
std::is_same<T, std::string>::value ||
|
||||
|
||||
@@ -104,6 +104,12 @@ common_chat_params common_chat_params_init_deepseek_v3_2(const common_chat_templ
|
||||
const std::string GEN_PROMPT = "<|Assistant|>";
|
||||
const std::string TC_SEPARATOR = "\n\n";
|
||||
|
||||
// lets the server find user turns in the prompt and place context checkpoints there
|
||||
data.message_delimiters = {
|
||||
{ COMMON_CHAT_ROLE_ASSISTANT, GEN_PROMPT },
|
||||
{ COMMON_CHAT_ROLE_USER, "<|User|>" },
|
||||
};
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(
|
||||
tmpl, inputs, adjusted_messages, std::nullopt, additional_context);
|
||||
data.generation_prompt = common_chat_template_generation_prompt_impl(
|
||||
|
||||
@@ -272,6 +272,10 @@ common_chat_params common_chat_params_init_gemma4(const common_chat_template &
|
||||
/* max = */ inputs.parallel_tool_calls ? -1 : 1
|
||||
));
|
||||
|
||||
if (inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED) {
|
||||
return start + thought + tool_call;
|
||||
}
|
||||
|
||||
auto scan_to_toolcall = p.rule("scan-to-toolcall", p.until("<|tool_call>"));
|
||||
auto content = p.rule("content", p.content(p.until_one_of({"<|channel>", "<channel|>", "<|tool_call>"})));
|
||||
auto message = p.rule("message", thought + content);
|
||||
|
||||
@@ -0,0 +1,194 @@
|
||||
#include "parsers.h"
|
||||
|
||||
// Ling 3.0 / Bailing V3 - <role>X</role> sections with tagged tool calls:
|
||||
// assistant := [<think> ... </think>] [content] {<tool_call>name
|
||||
// <arg_key>k</arg_key>\n<arg_value>v</arg_value> ...</tool_call>}
|
||||
// The generation prompt ends with "<role>ASSISTANT</role>\n<think>", so the model
|
||||
// never emits the opening think tag, and a tool call can arrive before any
|
||||
// </think>. Reasoning therefore terminates at the think close tag or at a tool
|
||||
// call start, like the Qwen3-Coder and Kimi K3 parsers. With thinking off the
|
||||
// template pre-closes the think block instead, and the model emits bare content.
|
||||
common_chat_params common_chat_params_init_ling3(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
common_chat_params data;
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);
|
||||
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
data.supports_thinking = true;
|
||||
|
||||
const std::string ROLE = "<role>ASSISTANT</role>";
|
||||
const std::string THINK_START = "<think>";
|
||||
const std::string THINK_END = "</think>";
|
||||
const std::string CALL_START = "<tool_call>";
|
||||
const std::string CALL_END = "</tool_call>";
|
||||
const std::string ARG_KEY = "<arg_key>";
|
||||
const std::string ARG_KEY_END = "</arg_key>";
|
||||
const std::string ARG_VAL = "<arg_value>";
|
||||
const std::string ROLE_END = "<|role_end|>";
|
||||
const std::string ARG_VAL_END = "</arg_value>";
|
||||
|
||||
data.preserved_tokens = {
|
||||
THINK_START, THINK_END, CALL_START, CALL_END,
|
||||
ARG_KEY, ARG_KEY_END, ARG_VAL, ARG_VAL_END, ROLE_END,
|
||||
};
|
||||
|
||||
data.thinking_start_tag = THINK_START;
|
||||
// Support both </think> and <tool_call> as reasoning end sequences: a call
|
||||
// can be emitted before the think block is closed.
|
||||
data.thinking_end_tags = { THINK_END, CALL_START };
|
||||
|
||||
data.message_delimiters = {
|
||||
{ COMMON_CHAT_ROLE_ASSISTANT, "<role>ASSISTANT</role>" },
|
||||
{ COMMON_CHAT_ROLE_USER, "<role>HUMAN</role>" },
|
||||
{ COMMON_CHAT_ROLE_TOOL, "<role>OBSERVATION</role>" },
|
||||
{ COMMON_CHAT_ROLE_SYSTEM, "<role>SYSTEM</role>" },
|
||||
};
|
||||
|
||||
// the model may spell the end-of-turn control token out as text tokens,
|
||||
// which does not stop generation; a literal stop string catches it either
|
||||
// way (as the Laguna patch does for its </assistant> token)
|
||||
data.additional_stops = { ROLE_END };
|
||||
|
||||
if (inputs.has_continuation()) {
|
||||
const auto & msg = inputs.continue_msg;
|
||||
|
||||
data.generation_prompt = ROLE + "\n" + THINK_START + msg.reasoning_content;
|
||||
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
|
||||
data.generation_prompt += THINK_END + msg.render_content();
|
||||
}
|
||||
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
// The generation prompt pre-opens the think block when thinking is on, so
|
||||
// the opening tag is optional here and reasoning runs until </think> or a
|
||||
// tool call start; with thinking off the template pre-closes the block and
|
||||
// everything the model emits is content.
|
||||
bool think_open = false;
|
||||
if (inputs.has_continuation()) {
|
||||
think_open = inputs.continue_final_message != COMMON_CHAT_CONTINUATION_CONTENT;
|
||||
} else {
|
||||
auto last_open = data.generation_prompt.rfind(THINK_START);
|
||||
auto last_close = data.generation_prompt.rfind(THINK_END);
|
||||
think_open = last_open != std::string::npos &&
|
||||
(last_close == std::string::npos || last_open > last_close);
|
||||
}
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto end = p.end();
|
||||
|
||||
// the effective parse input is generation_prompt + model output, so the
|
||||
// assistant opener is optionally consumed here
|
||||
auto opener = p.optional(p.literal(ROLE) + p.optional(p.space()));
|
||||
|
||||
// the generation prompt pre-opens the think block, so the opening tag
|
||||
// is optional; a missing close tag does not swallow a tool call
|
||||
auto body_end = think_open ? p.until_one_of({ THINK_END, CALL_START }) : p.until_one_of({ THINK_END });
|
||||
auto think_body = extract_reasoning ? p.reasoning(body_end) : p.content(body_end);
|
||||
|
||||
auto reasoning = p.optional(p.optional(p.literal(THINK_START)) + think_body +
|
||||
p.optional(p.literal(THINK_END)));
|
||||
|
||||
// content between the think block and the first tool call, plus any
|
||||
// trailing text after the last tool call, are plain content
|
||||
auto content = p.optional(p.content(p.until_one_of({ CALL_START })));
|
||||
|
||||
// a trailing end-of-turn token is consumed instead of leaking into content
|
||||
auto tail = p.optional(p.content(p.until(ROLE_END))) + p.optional(p.literal(ROLE_END));
|
||||
|
||||
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
return opener + reasoning + tail + end;
|
||||
}
|
||||
|
||||
auto tool_choices = p.choice();
|
||||
auto arg_close = p.tool_arg_close(p.literal(ARG_VAL_END));
|
||||
auto arg_string = p.rule("ling3-arg-string",
|
||||
p.tool_arg_string_value(p.until(ARG_VAL_END)) + arg_close);
|
||||
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
|
||||
std::vector<common_peg_parser> required_args;
|
||||
std::vector<common_peg_parser> optional_args;
|
||||
|
||||
// each argument may be preceded by whitespace: the model emits
|
||||
// newlines between arguments, the template history does not
|
||||
foreach_parameter(function, [&](const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
auto rule_name = "ling3-arg-" + name + "-" + param.name;
|
||||
|
||||
auto types = param.schema->value_types();
|
||||
|
||||
// string arguments are raw text up to the closing tag, other
|
||||
// types parse as JSON per their schema; each alternative
|
||||
// consumes the closing tag itself so a JSON prefix can not
|
||||
// commit the choice before the tag matches
|
||||
auto arg_value = p.eps();
|
||||
if (!types.has(common_chat_schema::TYPE_STRING)) {
|
||||
arg_value = p.tool_arg_json_value(p.schema(p.json(), rule_name + "-schema", doc, *param.schema)) + arg_close;
|
||||
} else if (types.is_only(common_chat_schema::TYPE_STRING)) {
|
||||
arg_value = arg_string;
|
||||
} else {
|
||||
// the parser tries the JSON alternative first to type the value
|
||||
arg_value = p.gbnf(p.atomic(p.tool_arg_json_value(p.schema(p.json(), rule_name + "-schema", doc, *param.schema)) + arg_close) | arg_string,
|
||||
"ling3-arg-string");
|
||||
}
|
||||
|
||||
auto arg = p.rule(rule_name,
|
||||
p.optional(p.space()) +
|
||||
p.tool_arg(p.tool_arg_open(p.literal(ARG_KEY) + p.tool_arg_name(p.literal(param.name)) +
|
||||
p.literal(ARG_KEY_END)) +
|
||||
p.optional(p.space()) + p.literal(ARG_VAL) +
|
||||
arg_value));
|
||||
|
||||
(param.required ? required_args : optional_args).push_back(arg);
|
||||
});
|
||||
|
||||
// required arguments in any order (as Qwen3-Coder does), then
|
||||
// optional ones in any order and number
|
||||
auto args = p.permute("ling3-" + name + "-args", required_args);
|
||||
if (!optional_args.empty()) {
|
||||
args = args + p.zero_or_more(p.choice(optional_args));
|
||||
}
|
||||
|
||||
auto call = p.tool(p.tool_open(p.literal(CALL_START) + p.tool_name(p.literal(name)) +
|
||||
p.optional(p.space())) +
|
||||
p.tool_args(args) +
|
||||
p.tool_close(p.optional(p.space()) + p.literal(CALL_END)));
|
||||
|
||||
tool_choices |= p.rule("ling3-tool-" + name, call);
|
||||
});
|
||||
|
||||
auto calls = inputs.parallel_tool_calls ?
|
||||
tool_choices + p.zero_or_more(p.space() + tool_choices) :
|
||||
tool_choices;
|
||||
|
||||
auto tools_section = p.trigger_rule("ling3-tool-call", calls + p.space() +
|
||||
p.optional(p.content(p.until(ROLE_END))) + p.optional(p.literal(ROLE_END)));
|
||||
|
||||
auto tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED ? tools_section :
|
||||
p.optional(tools_section);
|
||||
|
||||
return opener + reasoning + content + tools + tail + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
{ COMMON_GRAMMAR_TRIGGER_TYPE_WORD, CALL_START },
|
||||
};
|
||||
}
|
||||
|
||||
return data;
|
||||
}
|
||||
@@ -130,7 +130,7 @@ common_chat_params common_chat_params_init_muse_glimmer(const common_chat_templa
|
||||
});
|
||||
data.grammar_triggers = {
|
||||
{ COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN,
|
||||
"<\\|start\\|>assistant( to=(?!self<\\|message\\|>)(?!user<\\|message\\|>)[^<]*?<\\|message\\|>)" },
|
||||
"(?:^|<\\|start\\|>assistant)( to=(?!self<\\|message\\|>)(?!user<\\|message\\|>)[^<]*?<\\|message\\|>)" },
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -63,6 +63,8 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
|
||||
common_chat_params common_chat_params_init_kimi_k3(const common_chat_template & tmpl, const autoparser::generation_params & inputs);
|
||||
|
||||
common_chat_params common_chat_params_init_ling3(const common_chat_template & tmpl, const autoparser::generation_params & inputs);
|
||||
|
||||
// tool_list_tokens preserves the LFM2 system tool-list markers; LFM2.5 renders without them
|
||||
common_chat_params common_chat_params_init_lfm2(const common_chat_template & tmpl, const autoparser::generation_params & inputs, bool tool_list_tokens);
|
||||
|
||||
|
||||
@@ -23,8 +23,9 @@ common_chat_params common_chat_params_init_qwen3_coder(const common_chat_templat
|
||||
if (supports_reasoning) {
|
||||
data.thinking_start_tag = "<think>";
|
||||
// Support both </think> and <tool_call> as reasoning end sequences.
|
||||
// The newline variant comes first so it is included in the forced message
|
||||
// <function= is omitted, as it is a workaround for Qwen3-Coder which is not a thinking model
|
||||
data.thinking_end_tags = { "</think>", "<tool_call>" };
|
||||
data.thinking_end_tags = { "\n</think>", "</think>", "<tool_call>" };
|
||||
data.preserved_tokens.insert(data.preserved_tokens.end(), { "<think>", "</think>" });
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ set(LLAMA_CHAT_PARSERS_SOURCES
|
||||
${CMAKE_CURRENT_LIST_DIR}/gpt-oss.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/kimi-k2.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/kimi-k3.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/ling3.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/lfm2.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/minicpm5.cpp
|
||||
${CMAKE_CURRENT_LIST_DIR}/minimax-m3.cpp
|
||||
|
||||
+50
-24
@@ -166,6 +166,25 @@ common_peg_ast_id common_peg_ast_arena::find_by_rule(const common_peg_ast_node &
|
||||
return COMMON_PEG_INVALID_AST_ID;
|
||||
}
|
||||
|
||||
std::string common_peg_ast_node::sanitized_text() const {
|
||||
if (invalid_utf8.empty()) {
|
||||
return std::string(text);
|
||||
}
|
||||
|
||||
std::string out;
|
||||
out.reserve(text.size() + 2 * invalid_utf8.size());
|
||||
|
||||
size_t seg_start = start;
|
||||
for (const auto & invalid : invalid_utf8) {
|
||||
out.append(text.data() + (seg_start - start), invalid.pos - seg_start);
|
||||
out.append("\xEF\xBF\xBD");
|
||||
seg_start = invalid.pos + invalid.len;
|
||||
}
|
||||
out.append(text.data() + (seg_start - start), end - seg_start);
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
void common_peg_ast_arena::visit(common_peg_ast_id id, const common_peg_ast_visitor & visitor) const {
|
||||
if (id == COMMON_PEG_INVALID_AST_ID) {
|
||||
return;
|
||||
@@ -282,6 +301,7 @@ struct parser_executor {
|
||||
|
||||
auto pos = start_pos;
|
||||
std::vector<common_peg_ast_id> nodes;
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8;
|
||||
|
||||
for (size_t i = 0; i < p.children.size(); i++) {
|
||||
const auto & child_id = p.children[i];
|
||||
@@ -306,13 +326,14 @@ struct parser_executor {
|
||||
if (!result.nodes.empty()) {
|
||||
nodes.insert(nodes.end(), result.nodes.begin(), result.nodes.end());
|
||||
}
|
||||
invalid_utf8.insert(invalid_utf8.end(), result.invalid_utf8.begin(), result.invalid_utf8.end());
|
||||
|
||||
if (result.need_more_input()) {
|
||||
ctx.parse_depth--;
|
||||
if (ctx.is_debug()) {
|
||||
fprintf(stderr, "%sSEQ -> NEED_MORE\n", debug_indent().c_str());
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, result.end, std::move(nodes));
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, result.end, std::move(nodes), std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
pos = result.end;
|
||||
@@ -322,7 +343,7 @@ struct parser_executor {
|
||||
if (ctx.is_debug()) {
|
||||
fprintf(stderr, "%sSEQ -> SUCCESS at %zu->%zu\n", debug_indent().c_str(), start_pos, pos);
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, std::move(nodes));
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, std::move(nodes), std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
common_peg_parse_result operator()(const common_peg_choice_parser & p) {
|
||||
@@ -370,6 +391,7 @@ struct parser_executor {
|
||||
auto pos = start_pos;
|
||||
int match_count = 0;
|
||||
std::vector<common_peg_ast_id> nodes;
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8;
|
||||
|
||||
// Try to match up to max_count times (or unlimited if max_count is -1)
|
||||
while (p.max_count == -1 || match_count < p.max_count) {
|
||||
@@ -400,6 +422,7 @@ struct parser_executor {
|
||||
if (!result.nodes.empty()) {
|
||||
nodes.insert(nodes.end(), result.nodes.begin(), result.nodes.end());
|
||||
}
|
||||
invalid_utf8.insert(invalid_utf8.end(), result.invalid_utf8.begin(), result.invalid_utf8.end());
|
||||
|
||||
pos = result.end;
|
||||
match_count++;
|
||||
@@ -410,13 +433,14 @@ struct parser_executor {
|
||||
if (!result.nodes.empty()) {
|
||||
nodes.insert(nodes.end(), result.nodes.begin(), result.nodes.end());
|
||||
}
|
||||
invalid_utf8.insert(invalid_utf8.end(), result.invalid_utf8.begin(), result.invalid_utf8.end());
|
||||
|
||||
ctx.parse_depth--;
|
||||
if (ctx.is_debug()) {
|
||||
fprintf(stderr, "%sREPEAT -> NEED_MORE (count=%d, nodes=%zu)\n", debug_indent().c_str(),
|
||||
match_count, nodes.size());
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, result.end, std::move(nodes));
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, result.end, std::move(nodes), std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
// Child failed - stop trying
|
||||
@@ -434,7 +458,7 @@ struct parser_executor {
|
||||
fprintf(stderr, "%sREPEAT -> NEED_MORE (not enough matches: %d < %d)\n", debug_indent().c_str(),
|
||||
match_count, p.min_count);
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, pos, std::move(nodes));
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, pos, std::move(nodes), std::move(invalid_utf8));
|
||||
}
|
||||
if (ctx.is_debug()) {
|
||||
fprintf(stderr, "%sREPEAT -> FAIL (not enough matches: %d < %d)\n", debug_indent().c_str(), match_count,
|
||||
@@ -448,7 +472,7 @@ struct parser_executor {
|
||||
fprintf(stderr, "%sREPEAT -> SUCCESS (count=%d, nodes=%zu)\n", debug_indent().c_str(), match_count,
|
||||
nodes.size());
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, std::move(nodes));
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, std::move(nodes), std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
common_peg_parse_result operator()(const common_peg_and_parser & p) {
|
||||
@@ -664,23 +688,23 @@ struct parser_executor {
|
||||
// Scan input and check for delimiters
|
||||
size_t pos = start_pos;
|
||||
size_t last_valid_pos = start_pos;
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8;
|
||||
|
||||
while (pos < ctx.input.size()) {
|
||||
auto utf8_result = common_parse_utf8_codepoint(ctx.input, pos);
|
||||
|
||||
if (utf8_result.status == utf8_parse_result::INCOMPLETE) {
|
||||
// Incomplete UTF-8 sequence
|
||||
if (!ctx.is_lenient()) {
|
||||
// Input is complete but UTF-8 is incomplete = malformed
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_FAIL, start_pos);
|
||||
}
|
||||
// Return what we have so far (before incomplete sequence)
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, last_valid_pos);
|
||||
if (utf8_result.status == utf8_parse_result::INCOMPLETE && ctx.is_lenient()) {
|
||||
// The rest of the sequence may still arrive, return what we have so far
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, last_valid_pos, {}, std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
if (utf8_result.status == utf8_parse_result::INVALID) {
|
||||
// Malformed UTF-8
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_FAIL, start_pos);
|
||||
if (utf8_result.status != utf8_parse_result::SUCCESS) {
|
||||
// Malformed UTF-8, or a sequence truncated by the end of a complete input.
|
||||
// A delimiter cannot start inside bytes that fail to decode, so consume them and move on
|
||||
invalid_utf8.push_back({pos, utf8_result.bytes_consumed});
|
||||
pos += utf8_result.bytes_consumed;
|
||||
last_valid_pos = pos;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Check if a delimiter starts at this position
|
||||
@@ -688,12 +712,12 @@ struct parser_executor {
|
||||
|
||||
if (match == common_trie::COMPLETE_MATCH) {
|
||||
// Found a complete delimiter, return everything before it
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos);
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, {}, std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
if (match == common_trie::PARTIAL_MATCH) {
|
||||
// Found a partial match extending to end of input, return everything before it
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos);
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, pos, {}, std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
pos += utf8_result.bytes_consumed;
|
||||
@@ -702,9 +726,9 @@ struct parser_executor {
|
||||
|
||||
if (last_valid_pos == ctx.input.size() && ctx.is_lenient()) {
|
||||
// Reached the end of a partial stream, there might still be more input that we need to consume.
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, last_valid_pos);
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT, start_pos, last_valid_pos, {}, std::move(invalid_utf8));
|
||||
}
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, last_valid_pos);
|
||||
return common_peg_parse_result(COMMON_PEG_PARSE_RESULT_SUCCESS, start_pos, last_valid_pos, {}, std::move(invalid_utf8));
|
||||
}
|
||||
|
||||
common_peg_parse_result operator()(const common_peg_schema_parser & p) {
|
||||
@@ -728,10 +752,11 @@ struct parser_executor {
|
||||
result.end,
|
||||
text,
|
||||
std::move(result.nodes),
|
||||
result.need_more_input()
|
||||
result.need_more_input(),
|
||||
result.invalid_utf8
|
||||
);
|
||||
|
||||
return common_peg_parse_result(result.type, result.start, result.end, { node_id });
|
||||
return common_peg_parse_result(result.type, result.start, result.end, { node_id }, std::move(result.invalid_utf8));
|
||||
}
|
||||
|
||||
return result;
|
||||
@@ -757,10 +782,11 @@ struct parser_executor {
|
||||
result.end,
|
||||
text,
|
||||
std::move(result.nodes),
|
||||
result.need_more_input()
|
||||
result.need_more_input(),
|
||||
result.invalid_utf8
|
||||
);
|
||||
|
||||
return common_peg_parse_result(result.type, result.start, result.end, { node_id });
|
||||
return common_peg_parse_result(result.type, result.start, result.end, { node_id }, std::move(result.invalid_utf8));
|
||||
}
|
||||
|
||||
return result;
|
||||
|
||||
+21
-4
@@ -72,6 +72,12 @@ enum common_peg_parse_result_type {
|
||||
|
||||
const char * common_peg_parse_result_type_name(common_peg_parse_result_type type);
|
||||
|
||||
// A run of input bytes that does not decode as UTF-8
|
||||
struct common_peg_invalid_utf8 {
|
||||
size_t pos;
|
||||
size_t len;
|
||||
};
|
||||
|
||||
struct common_peg_ast_node {
|
||||
common_peg_ast_id id;
|
||||
std::string rule;
|
||||
@@ -82,6 +88,12 @@ struct common_peg_ast_node {
|
||||
std::vector<common_peg_ast_id> children;
|
||||
|
||||
bool is_partial = false;
|
||||
|
||||
// Invalid UTF-8 inside the node, in ascending order
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8;
|
||||
|
||||
// Returns the text with every invalid run replaced by U+FFFD
|
||||
std::string sanitized_text() const;
|
||||
};
|
||||
|
||||
struct common_peg_parse_result;
|
||||
@@ -98,10 +110,11 @@ class common_peg_ast_arena {
|
||||
size_t end,
|
||||
std::string_view text,
|
||||
std::vector<common_peg_ast_id> children,
|
||||
bool is_partial = false
|
||||
bool is_partial = false,
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8 = {}
|
||||
) {
|
||||
common_peg_ast_id id = nodes_.size();
|
||||
nodes_.push_back({id, rule, tag, start, end, text, std::move(children), is_partial});
|
||||
nodes_.push_back({id, rule, tag, start, end, text, std::move(children), is_partial, std::move(invalid_utf8)});
|
||||
return id;
|
||||
}
|
||||
|
||||
@@ -127,6 +140,9 @@ struct common_peg_parse_result {
|
||||
|
||||
std::vector<common_peg_ast_id> nodes;
|
||||
|
||||
// Invalid UTF-8 consumed by this result, carried up to the enclosing AST nodes
|
||||
std::vector<common_peg_invalid_utf8> invalid_utf8;
|
||||
|
||||
common_peg_parse_result() = default;
|
||||
|
||||
common_peg_parse_result(common_peg_parse_result_type type, size_t start)
|
||||
@@ -135,8 +151,8 @@ struct common_peg_parse_result {
|
||||
common_peg_parse_result(common_peg_parse_result_type type, size_t start, size_t end)
|
||||
: type(type), start(start), end(end) {}
|
||||
|
||||
common_peg_parse_result(common_peg_parse_result_type type, size_t start, size_t end, std::vector<common_peg_ast_id> nodes)
|
||||
: type(type), start(start), end(end), nodes(std::move(nodes)) {}
|
||||
common_peg_parse_result(common_peg_parse_result_type type, size_t start, size_t end, std::vector<common_peg_ast_id> nodes, std::vector<common_peg_invalid_utf8> invalid_utf8 = {})
|
||||
: type(type), start(start), end(end), nodes(std::move(nodes)), invalid_utf8(std::move(invalid_utf8)) {}
|
||||
|
||||
bool fail() const { return type == COMMON_PEG_PARSE_RESULT_FAIL; }
|
||||
bool need_more_input() const { return type == COMMON_PEG_PARSE_RESULT_NEED_MORE_INPUT; }
|
||||
@@ -430,6 +446,7 @@ class common_peg_parser_builder {
|
||||
common_peg_parser space() { return add(common_peg_space_parser{}); }
|
||||
|
||||
// Matches all characters until a delimiter is found (delimiter not consumed).
|
||||
// Invalid UTF-8 is consumed and recorded on the AST nodes.
|
||||
// S -> (!delim .)*
|
||||
common_peg_parser until(const std::string & delimiter) { return add(common_peg_until_parser{{delimiter}}); }
|
||||
|
||||
|
||||
@@ -228,9 +228,7 @@ struct common_speculative_impl_draft_simple : public common_speculative_impl {
|
||||
common_params_sampling params;
|
||||
params.no_perf = false;
|
||||
params.top_k = 10;
|
||||
params.samplers = {
|
||||
COMMON_SAMPLER_TYPE_TOP_K,
|
||||
};
|
||||
params.samplers.assign(1, COMMON_SAMPLER_TYPE_TOP_K);
|
||||
|
||||
smpl.reset(common_sampler_init(llama_get_model(ctx_dft), params));
|
||||
}
|
||||
|
||||
+19
-14
@@ -26,16 +26,16 @@ utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t off
|
||||
|
||||
// Invalid: continuation byte as first byte
|
||||
if (!(input[offset] & 0x40)) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID);
|
||||
return utf8_parse_result(utf8_parse_result::INVALID, 0, 1);
|
||||
}
|
||||
|
||||
// 2-byte sequence
|
||||
if (!(input[offset] & 0x20)) {
|
||||
if (offset + 1 >= input.size()) {
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE);
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE, 0, 1);
|
||||
}
|
||||
if ((input[offset + 1] & 0xc0) != 0x80) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID);
|
||||
return utf8_parse_result(utf8_parse_result::INVALID, 0, 1);
|
||||
}
|
||||
auto result = ((input[offset] & 0x1f) << 6) | (input[offset + 1] & 0x3f);
|
||||
return utf8_parse_result(utf8_parse_result::SUCCESS, result, 2);
|
||||
@@ -43,11 +43,14 @@ utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t off
|
||||
|
||||
// 3-byte sequence
|
||||
if (!(input[offset] & 0x10)) {
|
||||
if (offset + 2 >= input.size()) {
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE);
|
||||
}
|
||||
if ((input[offset + 1] & 0xc0) != 0x80 || (input[offset + 2] & 0xc0) != 0x80) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID);
|
||||
// Check one byte at a time so a bad byte is reported before a short input
|
||||
for (size_t i = 1; i < 3; i++) {
|
||||
if (offset + i >= input.size()) {
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE, 0, i);
|
||||
}
|
||||
if ((input[offset + i] & 0xc0) != 0x80) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID, 0, i);
|
||||
}
|
||||
}
|
||||
auto result = ((input[offset] & 0x0f) << 12) | ((input[offset + 1] & 0x3f) << 6) | (input[offset + 2] & 0x3f);
|
||||
return utf8_parse_result(utf8_parse_result::SUCCESS, result, 3);
|
||||
@@ -55,18 +58,20 @@ utf8_parse_result common_parse_utf8_codepoint(std::string_view input, size_t off
|
||||
|
||||
// 4-byte sequence
|
||||
if (!(input[offset] & 0x08)) {
|
||||
if (offset + 3 >= input.size()) {
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE);
|
||||
}
|
||||
if ((input[offset + 1] & 0xc0) != 0x80 || (input[offset + 2] & 0xc0) != 0x80 || (input[offset + 3] & 0xc0) != 0x80) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID);
|
||||
for (size_t i = 1; i < 4; i++) {
|
||||
if (offset + i >= input.size()) {
|
||||
return utf8_parse_result(utf8_parse_result::INCOMPLETE, 0, i);
|
||||
}
|
||||
if ((input[offset + i] & 0xc0) != 0x80) {
|
||||
return utf8_parse_result(utf8_parse_result::INVALID, 0, i);
|
||||
}
|
||||
}
|
||||
auto result = ((input[offset] & 0x07) << 18) | ((input[offset + 1] & 0x3f) << 12) | ((input[offset + 2] & 0x3f) << 6) | (input[offset + 3] & 0x3f);
|
||||
return utf8_parse_result(utf8_parse_result::SUCCESS, result, 4);
|
||||
}
|
||||
|
||||
// Invalid first byte
|
||||
return utf8_parse_result(utf8_parse_result::INVALID);
|
||||
return utf8_parse_result(utf8_parse_result::INVALID, 0, 1);
|
||||
}
|
||||
|
||||
bool common_utf8_is_complete(const std::string & s) {
|
||||
|
||||
+1
-1
@@ -9,7 +9,7 @@
|
||||
|
||||
struct utf8_parse_result {
|
||||
uint32_t codepoint; // Decoded codepoint (only valid if status == SUCCESS)
|
||||
size_t bytes_consumed; // How many bytes this codepoint uses (1-4)
|
||||
size_t bytes_consumed; // How many bytes this codepoint uses (1-4), or the length of the valid prefix if status != SUCCESS
|
||||
enum status { SUCCESS, INCOMPLETE, INVALID } status;
|
||||
|
||||
utf8_parse_result(enum status s, uint32_t cp = 0, size_t bytes = 0)
|
||||
|
||||
@@ -28,6 +28,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"BailingMoeForCausalLM": "bailingmoe",
|
||||
"BailingMoeV2ForCausalLM": "bailingmoe",
|
||||
"BailingMoeV3ForCausalLM": "bailingmoe3",
|
||||
"BailingMoeV3VLForConditionalGeneration": "bailingmoe3",
|
||||
"BambaForCausalLM": "granite",
|
||||
"BertForMaskedLM": "bert",
|
||||
"BertForSequenceClassification": "bert",
|
||||
@@ -94,6 +95,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"Gemma3nForCausalLM": "gemma",
|
||||
"Gemma3nForConditionalGeneration": "gemma",
|
||||
"Gemma4AssistantForCausalLM": "gemma",
|
||||
"Gemma4DSparkModel": "gemma",
|
||||
"Gemma4ForConditionalGeneration": "gemma",
|
||||
"Gemma4ForCausalLM": "gemma",
|
||||
"Gemma4UnifiedForConditionalGeneration": "gemma",
|
||||
@@ -123,6 +125,7 @@ TEXT_MODEL_MAP: dict[str, str] = {
|
||||
"HunYuanDenseV1ForCausalLM": "hunyuan",
|
||||
"HunYuanMoEV1ForCausalLM": "hunyuan",
|
||||
"HunYuanVLForConditionalGeneration": "hunyuan",
|
||||
"HrmTextForCausalLM": "hrm_text",
|
||||
"HYV3ForCausalLM": "hunyuan",
|
||||
"HYV4ForCausalLM": "hy_v4",
|
||||
"IQuestCoderForCausalLM": "llama",
|
||||
@@ -300,6 +303,7 @@ MMPROJ_MODEL_MAP: dict[str, str] = {
|
||||
"Gemma4ForConditionalGeneration": "gemma",
|
||||
"Gemma4UnifiedForConditionalGeneration": "gemma",
|
||||
"Glm4vForConditionalGeneration": "qwen3vl",
|
||||
"BailingMoeV3VLForConditionalGeneration": "bailingmoe3",
|
||||
"Glm4vMoeForConditionalGeneration": "qwen3vl",
|
||||
"Glm5vForConditionalGeneration": "kimivl",
|
||||
"GlmOcrForConditionalGeneration": "qwen3vl",
|
||||
|
||||
+112
-2
@@ -9,7 +9,9 @@ import torch
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import ModelBase, TextModel, gguf
|
||||
from .base import ModelBase, MmprojModel, TextModel, gguf
|
||||
|
||||
from .qwen3vl import Qwen3VLVisionModel
|
||||
|
||||
|
||||
@ModelBase.register("BailingMoeV3ForCausalLM")
|
||||
@@ -74,7 +76,7 @@ class BailingMoeV3Model(TextModel):
|
||||
|
||||
self.gguf_writer.add_expert_feed_forward_length(self.hparams["moe_intermediate_size"])
|
||||
self.gguf_writer.add_expert_shared_feed_forward_length(self.hparams["moe_shared_expert_intermediate_size"])
|
||||
self.gguf_writer.add_expert_shared_count(self.hparams["num_shared_experts"])
|
||||
self.gguf_writer.add_expert_shared_count(self.hparams.get("num_shared_experts", 1))
|
||||
self.gguf_writer.add_leading_dense_block_count(self.hparams["first_k_dense_replace"])
|
||||
self.gguf_writer.add_expert_weights_scale(self.hparams["routed_scaling_factor"])
|
||||
self.gguf_writer.add_expert_weights_norm(self.hparams["norm_topk_prob"])
|
||||
@@ -191,3 +193,111 @@ class BailingMoeV3Model(TextModel):
|
||||
experts = [name for layer in self._experts for name in layer]
|
||||
if experts:
|
||||
raise ValueError(f"Unprocessed experts: {experts}")
|
||||
|
||||
|
||||
@ModelBase.register("BailingMoeV3VLForConditionalGeneration")
|
||||
@ModelBase.example("inclusionAI/Ling-3.0-flash-VL")
|
||||
class BailingMoeV3VLModel(BailingMoeV3Model):
|
||||
model_arch = gguf.MODEL_ARCH.BAILINGMOE3
|
||||
|
||||
def index_tensors(self, remote_hf_model_id: str | None = None):
|
||||
# hoist text_config before the shared BailingMoeV3 logic runs:
|
||||
# ModelBase.__init__ calls this with the raw VL config, where the text
|
||||
# dims still live under text_config
|
||||
if "text_config" in self.hparams:
|
||||
self.hparams = {**self.hparams, **self.hparams["text_config"]}
|
||||
return super().index_tensors(remote_hf_model_id=remote_hf_model_id)
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
mrope_section = self.hparams.get("mrope_section")
|
||||
if mrope_section is None:
|
||||
raise ValueError("BailingMoeV3VL requires mrope_section in the config")
|
||||
if sum(mrope_section[:3]) * 2 != self.hparams["qk_rope_head_dim"]:
|
||||
raise ValueError(
|
||||
f"mrope_section {mrope_section[:3]} counts rope pairs and must sum to"
|
||||
f" qk_rope_head_dim / 2 = {self.hparams['qk_rope_head_dim'] // 2}"
|
||||
)
|
||||
# mrope_section is [t, h, w]; pad to the 4-wide sections array
|
||||
self.gguf_writer.add_rope_dimension_sections(list(mrope_section[:3]) + [0])
|
||||
|
||||
@classmethod
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
name, gen = item
|
||||
|
||||
# Skip projector tensors; the vision tower is skipped by TextModel.filter_tensors
|
||||
if name.startswith("linear_proj"):
|
||||
return None
|
||||
|
||||
return super().filter_tensors(item)
|
||||
|
||||
|
||||
@ModelBase.register("BailingMoeV3VLForConditionalGeneration")
|
||||
@ModelBase.example("inclusionAI/Ling-3.0-flash-VL")
|
||||
class BailingMoeV3VLVisionModel(Qwen3VLVisionModel):
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
assert self.hparams_vision is not None
|
||||
|
||||
if self.hparams_vision.get("disable_merger_proj") is not True:
|
||||
raise ValueError("BailingMoeV3VL requires disable_merger_proj=true")
|
||||
|
||||
# out_hidden_size is the vision encoder output (post spatial merge, pre linear_proj)
|
||||
self.image_emb_dim = self.hparams_vision.get("out_hidden_size")
|
||||
if self.image_emb_dim is None:
|
||||
raise ValueError("BailingMoeV3VL vision config requires out_hidden_size")
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
assert self.hparams_vision is not None
|
||||
MmprojModel.set_gguf_parameters(self) # skip Qwen3VLVisionModel parameters
|
||||
self.gguf_writer.add_clip_projector_type(gguf.VisionProjectorType.LING3VL)
|
||||
self.gguf_writer.add_vision_use_gelu(True)
|
||||
|
||||
merge_size = self.hparams_vision.get("spatial_merge_size")
|
||||
if merge_size is not None:
|
||||
self.gguf_writer.add_vision_spatial_merge_size(int(merge_size))
|
||||
|
||||
rms_norm_eps = self.global_config.get("text_config", {}).get("rms_norm_eps", 1e-6)
|
||||
self.gguf_writer.add_vision_attention_layernorm_eps(rms_norm_eps)
|
||||
|
||||
@classmethod
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
name, gen = item
|
||||
|
||||
if name.startswith("lm_head."):
|
||||
return None
|
||||
|
||||
if name.startswith("linear_proj"):
|
||||
# top-level projector MLP: linear_proj.0 -> mm.0, linear_proj.2 -> mm.2
|
||||
parts = name.split(".")
|
||||
if len(parts) != 3:
|
||||
raise ValueError(f"Unexpected linear_proj tensor: {name}")
|
||||
idx, suffix = int(parts[1]), parts[2]
|
||||
name = f"mm.{idx}.{suffix}"
|
||||
# the qwen3vl filter keeps only visual.*; skip it for the renamed projector tensors
|
||||
return MmprojModel.filter_tensors((name, gen))
|
||||
|
||||
if name.startswith("model.visual."):
|
||||
name = name.replace("model.visual.", "visual.", 1)
|
||||
|
||||
if not name.startswith("visual."):
|
||||
return None
|
||||
|
||||
return super().filter_tensors((name, gen))
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
assert self.hparams_vision is not None
|
||||
|
||||
if name.startswith("mm.0.") or name.startswith("mm.2."):
|
||||
# top-level projector MLP (linear_proj.0 / linear_proj.2, renamed by filter_tensors)
|
||||
yield (name, data_torch)
|
||||
return
|
||||
|
||||
if name == "visual.merger.norm.weight" or name == "visual.merger.norm.bias":
|
||||
# the merger is norm-only for Ling: per-patch LayerNorm before the spatial merge
|
||||
new_name = f"mm.input_norm.{name.split('.')[-1]}"
|
||||
yield (new_name, data_torch)
|
||||
return
|
||||
|
||||
# Ling has no patch bias; the Conv3D split below matches the stock qwen3vl path
|
||||
yield from Qwen3VLVisionModel.modify_tensors(self, data_torch, name, bid)
|
||||
|
||||
@@ -776,6 +776,36 @@ class ModelBase:
|
||||
raw = torch.cat((s.unsqueeze(-1), qs.to(torch.uint8)), dim=-1)
|
||||
return raw.reshape(rows, n_blocks * 17).cpu().numpy()
|
||||
|
||||
def _mxfp4_expert_tensor(self, loaders: list[tuple[Callable[[], Tensor], Callable[[], Tensor]]]):
|
||||
"""
|
||||
One stacked [n_expert, rows, cols] MXFP4 tensor, built lazily.
|
||||
|
||||
gguf_writer holds every added tensor until the final write, so building
|
||||
this eagerly (like the DeepSeek-V4 path does) keeps every expert in
|
||||
memory at once. lazy means only the tensor being written is resident.
|
||||
"""
|
||||
# meta shapes, so this does not read any weights
|
||||
rows, packed_cols = loaders[0][0]().shape
|
||||
n_blocks = (packed_cols * 2) // 32
|
||||
byte_shape = (len(loaders), rows, n_blocks * 17)
|
||||
|
||||
def load(fns: list[tuple[Callable[[], Tensor], Callable[[], Tensor]]]) -> np.ndarray:
|
||||
out = np.empty(byte_shape, dtype=np.uint8)
|
||||
for eid, (packed_fn, scale_fn) in enumerate(fns):
|
||||
out[eid] = self.repack_mxfp4_blocks(
|
||||
LazyTorchTensor.to_eager(packed_fn()),
|
||||
LazyTorchTensor.to_eager(scale_fn()),
|
||||
)
|
||||
return out
|
||||
|
||||
# loaders goes through args, not the closure, so that `func` matches
|
||||
# LazyBase's single-argument shape
|
||||
return gguf.LazyNumpyTensor(
|
||||
meta=gguf.LazyNumpyTensor.meta_with_dtype_and_shape(np.uint8, byte_shape),
|
||||
args=(loaders,),
|
||||
func=load,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _nvfp4_pack(weight: Tensor, scale: Tensor) -> tuple[np.ndarray, list[int]]:
|
||||
"""Repack NVFP4 ModelOpt tensors into ggml super-block layout.
|
||||
@@ -1633,6 +1663,9 @@ class TextModel(ModelBase):
|
||||
if chkhsh == "9e454714343b69b99b71795c1d27a68c2a1d15dab111f4d353109f966af29da7":
|
||||
# ref: https://huggingface.co/LiquidAI/LFM2.5-8B-A1B
|
||||
res = "lfm2"
|
||||
if chkhsh == "846deafc5b0fa786186fa4ae6c7b49903cf2f1d1895bdb80b9120d60be135252":
|
||||
# ref: https://huggingface.co/danish-foundation-models/DFM-Mimir
|
||||
res = "gemma4"
|
||||
if chkhsh == "0a766d034107bc736a3f2dc4968fd62e54a3570f1454443e0c5a4cc6bd7941ed":
|
||||
# ref: https://huggingface.co/XHToken/Spark-X2.5-1.7B
|
||||
res = "spark2_5"
|
||||
@@ -1858,6 +1891,9 @@ class TextModel(ModelBase):
|
||||
if chkhsh == "972da7b59cec44d1f0a490a86c96df53859e486e481563e5dddac155013d87ac":
|
||||
# ref: https://huggingface.co/poolside/Laguna-XS.2
|
||||
res = "laguna"
|
||||
if chkhsh == "653660222fb704f61cbf2b618a8ae6502b7f8b20c980f9a5de07ed78e13319cd":
|
||||
# ref: https://huggingface.co/ufakai/ufakzeka-1
|
||||
res = "ufakzeka"
|
||||
|
||||
if res is None:
|
||||
logger.warning("\n")
|
||||
|
||||
@@ -11,6 +11,7 @@ if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import MmprojModel, ModelBase, TextModel, gguf, logger
|
||||
from .qwen import DFlashModel
|
||||
|
||||
|
||||
@ModelBase.register("GemmaForCausalLM")
|
||||
@@ -809,6 +810,105 @@ class Gemma4Model(Gemma3Model):
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
|
||||
@ModelBase.register("Gemma4DSparkModel")
|
||||
class Gemma4DSparkModel(DFlashModel):
|
||||
model_arch = gguf.MODEL_ARCH.DFLASH
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
if not self.hparams.get("attention_k_eq_v", False):
|
||||
raise ValueError("Gemma4 DSpark currently requires attention_k_eq_v")
|
||||
if self.hparams.get("layer_types") != ["full_attention"] * self.block_count:
|
||||
raise ValueError("Gemma4 DSpark currently requires uniform full_attention layer types")
|
||||
if self.hparams.get("hidden_activation", "gelu_pytorch_tanh") != "gelu_pytorch_tanh":
|
||||
raise ValueError("Gemma4 DSpark currently requires hidden_activation=gelu_pytorch_tanh")
|
||||
if self.hparams.get("attention_bias", False) or self.hparams.get("enable_moe_block", False):
|
||||
raise ValueError("Gemma4 DSpark attention bias and MoE are not supported")
|
||||
if (self.hparams.get("draft_vocab_size") or self.hparams["vocab_size"]) != self.hparams["vocab_size"]:
|
||||
raise ValueError("Gemma4 DSpark currently requires a full draft vocabulary")
|
||||
if "model.lm_head.weight" not in self.model_tensors and self.hparams.get("tie_word_embeddings") is not True:
|
||||
raise ValueError("Gemma4 DSpark requires lm_head.weight unless tie_word_embeddings is true")
|
||||
|
||||
self.dflash_config = self.hparams.get("dflash_config", {})
|
||||
markov_type = self.dflash_config.get("markov_head_type", self.hparams.get("markov_head_type", "vanilla"))
|
||||
if markov_type != "vanilla":
|
||||
raise ValueError("Gemma4 DSpark currently requires a vanilla Markov head")
|
||||
|
||||
# Gemma4TextConfig supplies these defaults when rope_parameters is absent.
|
||||
rope = self.hparams.get("rope_parameters") or {
|
||||
"full_attention": {"rope_type": "proportional", "partial_rotary_factor": 0.25, "rope_theta": 1000000.0},
|
||||
}
|
||||
self.rope_parameters = rope.get("full_attention", rope)
|
||||
if self.rope_parameters.get("rope_type") not in ("default", "proportional"):
|
||||
raise ValueError("Gemma4 DSpark requires default or proportional RoPE")
|
||||
|
||||
def set_vocab(self):
|
||||
super().set_vocab()
|
||||
mask_id = self.dflash_config.get("mask_token_id", self.hparams.get("mask_token_id"))
|
||||
if mask_id is None:
|
||||
raise ValueError("Gemma4 DSpark requires mask_token_id")
|
||||
if "mask_token_id" not in self.dflash_config:
|
||||
self.gguf_writer.add_mask_token_id(mask_id)
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
head_dim = int(self.hparams["global_head_dim"])
|
||||
self.gguf_writer.add_head_count_kv(self.hparams["num_global_key_value_heads"])
|
||||
self.gguf_writer.add_key_length(head_dim)
|
||||
self.gguf_writer.add_value_length(head_dim)
|
||||
self.gguf_writer.add_rope_dimension_count(head_dim)
|
||||
self.gguf_writer.add_embedding_scale(self.hparams["hidden_size"] ** 0.5)
|
||||
self.gguf_writer.add_attention_scale(1.0)
|
||||
self.gguf_writer.add_hidden_act("gelu_pytorch_tanh")
|
||||
|
||||
self.gguf_writer.add_sample_from_anchor(self.hparams.get("sample_from_anchor", True))
|
||||
target_layers = self.dflash_config.get("target_layer_ids", self.hparams.get("target_layer_ids"))
|
||||
if not target_layers:
|
||||
raise ValueError("Gemma4 DSpark requires target_layer_ids")
|
||||
self.gguf_writer.add_has_confidence_head(any("confidence_head.proj" in name for name in self.model_tensors))
|
||||
|
||||
if self.hparams.get("final_logit_softcapping"):
|
||||
raise ValueError("Gemma4 DSpark logit softcapping is not supported")
|
||||
# The top-level sliding_window is inert unless the draft enables SWA.
|
||||
if self.dflash_config.get("use_swa", False):
|
||||
window = self.dflash_config["swa_window_size"]
|
||||
if window <= 0:
|
||||
raise ValueError("Gemma4 DSpark swa_window_size must be positive")
|
||||
|
||||
@classmethod
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
name, gen = item
|
||||
if not name.startswith("model."):
|
||||
name = "model." + name
|
||||
if name.endswith(".layer_scalar"):
|
||||
name += ".weight"
|
||||
name = name.replace("model.confidence_proj.", "model.confidence_head.proj.")
|
||||
return super().filter_tensors((name, gen))
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
# The shared DFlash map assigns this name to Qwen's pre-FFN norm.
|
||||
if name.endswith(".post_attention_layernorm.weight"):
|
||||
name = self.format_tensor_name(gguf.MODEL_TENSOR.ATTN_POST_NORM, bid)
|
||||
elif name.endswith(".pre_feedforward_layernorm.weight"):
|
||||
name = self.format_tensor_name(gguf.MODEL_TENSOR.FFN_NORM, bid)
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
|
||||
if self.rope_parameters["rope_type"] == "proportional":
|
||||
# Keep the unrotated dimensions in place, as in the Gemma4 converter.
|
||||
head_dim = int(self.hparams["global_head_dim"])
|
||||
fraction_value = self.rope_parameters.get("partial_rotary_factor", 0.25)
|
||||
if not isinstance(fraction_value, (int, float)):
|
||||
raise ValueError("Gemma4 DSpark partial_rotary_factor must be numeric")
|
||||
fraction = float(fraction_value)
|
||||
n_rot = int(head_dim * fraction / 2)
|
||||
if not 0 < fraction <= 1 or head_dim * fraction != 2 * n_rot:
|
||||
raise ValueError("Gemma4 DSpark rotary dimension count must be positive and even")
|
||||
factors = torch.tensor([1.0] * n_rot + [1e30] * (head_dim // 2 - n_rot), dtype=torch.float32)
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.ROPE_FREQS), factors
|
||||
|
||||
|
||||
@ModelBase.register("Gemma4UnifiedForConditionalGeneration")
|
||||
@ModelBase.example("hf-tiny-v2/tiny-random-Gemma4UnifiedForConditionalGeneration")
|
||||
class Gemma4UnifiedModel(Gemma4Model):
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from typing import Iterable, TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import ModelBase, TextModel, gguf
|
||||
|
||||
|
||||
@ModelBase.register("HrmTextForCausalLM")
|
||||
@ModelBase.example("danish-foundation-models/DFM-Mimir")
|
||||
class HrmTextModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.HRM_TEXT
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
# training-style configs store the per-stack count in num_hidden_layers,
|
||||
# transformers-style configs keep it in num_layers_per_stack
|
||||
self.layers_per_stack = self.hparams.get("num_layers_per_stack") or self.hparams["num_hidden_layers"]
|
||||
self.h_cycles = self.hparams["H_cycles"]
|
||||
self.l_cycles = self.hparams["L_cycles"]
|
||||
|
||||
# block_count is the expanded cache-slot count; the file only holds
|
||||
# 2 * layers_per_stack physical blocks
|
||||
self.block_count = self.layers_per_stack * self.h_cycles * (self.l_cycles + 1)
|
||||
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, 2 * self.layers_per_stack)
|
||||
|
||||
def set_vocab(self):
|
||||
self._set_vocab_gpt2()
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
super().set_gguf_parameters()
|
||||
|
||||
head_dim = self.hparams.get("head_dim") or self.hparams["hidden_size"] // self.hparams["num_attention_heads"]
|
||||
self.gguf_writer.add_rope_dimension_count(head_dim)
|
||||
self.gguf_writer.add_embedding_scale(self.hparams["embedding_scale"])
|
||||
self.gguf_writer.add_hrm_layers_per_stack(self.layers_per_stack)
|
||||
self.gguf_writer.add_hrm_h_cycles(self.h_cycles)
|
||||
self.gguf_writer.add_hrm_l_cycles(self.l_cycles)
|
||||
self.gguf_writer.add_hrm_prefix_lm(bool(self.hparams.get("prefix_lm", False)))
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
if name == "model.embed_tokens.weight":
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.TOKEN_EMBD), data_torch
|
||||
return
|
||||
if name == "lm_head.weight":
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.OUTPUT), data_torch
|
||||
return
|
||||
if name == "model.z_L_init":
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.HRM_Z_L_INIT, suffix=""), data_torch
|
||||
return
|
||||
|
||||
match = re.fullmatch(r"model\.([LH])_module\.layers\.(\d+)\.(.+)", name)
|
||||
if match is None:
|
||||
raise ValueError(f"can not map tensor: {name}")
|
||||
|
||||
stack, layer_s, tensor_name = match.groups()
|
||||
# the L stack occupies blocks [0, layers_per_stack), the H stack follows it
|
||||
layer_idx = int(layer_s) + (self.layers_per_stack if stack == "H" else 0)
|
||||
|
||||
if tensor_name == "attn.gqkv_proj.weight":
|
||||
gate, q, k, v = data_torch.chunk(4, dim=0)
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.ATTN_GATE, layer_idx), gate.contiguous()
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.ATTN_Q, layer_idx), q.contiguous()
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.ATTN_K, layer_idx), k.contiguous()
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.ATTN_V, layer_idx), v.contiguous()
|
||||
elif tensor_name == "mlp.gate_up_proj.weight":
|
||||
gate, up = data_torch.chunk(2, dim=0)
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.FFN_GATE, layer_idx), gate.contiguous()
|
||||
yield self.format_tensor_name(gguf.MODEL_TENSOR.FFN_UP, layer_idx), up.contiguous()
|
||||
else:
|
||||
if tensor_name.startswith("attn."):
|
||||
tensor_name = "self_attn." + tensor_name[len("attn."):]
|
||||
tensor_name = "model.layers.{bid}." + tensor_name
|
||||
yield from super().modify_tensors(data_torch, tensor_name.format(bid=layer_idx), layer_idx)
|
||||
+19
-27
@@ -159,32 +159,14 @@ class HunYuanMoEModel(TextModel):
|
||||
class HunYuanModel(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.HUNYUAN_DENSE
|
||||
|
||||
def _get_eod_token_id(self) -> int | None:
|
||||
"""Get the actual end-of-generation token from config (eod_token_id)."""
|
||||
return self.hparams.get("eod_token_id")
|
||||
|
||||
def _get_eot_token_id(self) -> int | None:
|
||||
"""Get the end-of-turn token from generation_config.json.
|
||||
This is the first entry in eos_token_id when it's a list."""
|
||||
gen_cfg_path = self.dir_model / "generation_config.json"
|
||||
if gen_cfg_path.is_file():
|
||||
with open(gen_cfg_path, encoding="utf-8") as f:
|
||||
gen_cfg = json.load(f)
|
||||
eos = gen_cfg.get("eos_token_id")
|
||||
if isinstance(eos, list) and len(eos) >= 2:
|
||||
return eos[0]
|
||||
return None
|
||||
|
||||
def _fix_special_tokens(self):
|
||||
"""Fix EOS/EOT tokens that are incorrect in upstream configs."""
|
||||
eod_id = self._get_eod_token_id()
|
||||
if eod_id is not None:
|
||||
self.gguf_writer.add_eos_token_id(eod_id)
|
||||
eot_id = self._get_eot_token_id()
|
||||
if eot_id is not None:
|
||||
self.gguf_writer.add_eot_token_id(eot_id)
|
||||
|
||||
def set_vocab(self):
|
||||
# Also called by draft models (e.g. DFlash), with dir_model pointing at
|
||||
# the target model.
|
||||
config = ModelBase.load_hparams(self.dir_model, self.is_mistral_format)
|
||||
config = {**config, **config.get("text_config", {})}
|
||||
self.hparams["pad_token_id"] = config.get("pad_token_id")
|
||||
self.hparams["eod_token_id"] = config.get("eod_token_id")
|
||||
|
||||
if (self.dir_model / "tokenizer.json").is_file():
|
||||
tokens, toktypes, tokpre = self.get_vocab_base()
|
||||
self.gguf_writer.add_tokenizer_model("gpt2")
|
||||
@@ -199,7 +181,6 @@ class HunYuanModel(TextModel):
|
||||
token_types = ('bos', 'eos', 'unk', 'sep', 'cls', 'mask')
|
||||
special_vocab = gguf.SpecialVocab(self.dir_model, load_merges=True, special_token_types=token_types)
|
||||
special_vocab.add_to_gguf(self.gguf_writer)
|
||||
self._fix_special_tokens()
|
||||
else:
|
||||
from transformers import AutoTokenizer
|
||||
tokenizer = AutoTokenizer.from_pretrained(self.dir_model, trust_remote_code=True)
|
||||
@@ -251,7 +232,18 @@ class HunYuanModel(TextModel):
|
||||
# FIX for BOS token: Overwrite incorrect id read from config.json
|
||||
if self.hparams['hidden_size'] == 4096:
|
||||
self.gguf_writer.add_bos_token_id(127958) # only for 7b dense, fix <|bos|> token
|
||||
self._fix_special_tokens()
|
||||
|
||||
# Fix EOS/EOT tokens that are incorrect in upstream configs.
|
||||
eod_id = self.hparams.get("eod_token_id")
|
||||
if eod_id is not None:
|
||||
self.gguf_writer.add_eos_token_id(eod_id)
|
||||
|
||||
gen_cfg = self.dir_model / "generation_config.json"
|
||||
if gen_cfg.is_file():
|
||||
with open(gen_cfg, encoding="utf-8") as f:
|
||||
eos = json.load(f).get("eos_token_id")
|
||||
if isinstance(eos, list) and len(eos) >= 2:
|
||||
self.gguf_writer.add_eot_token_id(eos[0])
|
||||
|
||||
def set_gguf_parameters(self):
|
||||
# Some HunYuanVL variants set num_experts=1 (not real MoE);
|
||||
|
||||
+2
-33
@@ -2,15 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Callable, Iterable, Iterator, TYPE_CHECKING
|
||||
from typing import Iterable, Iterator, TYPE_CHECKING
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import LazyTorchTensor, ModelBase, TextModel, gguf, logger
|
||||
from .base import ModelBase, TextModel, gguf, logger
|
||||
|
||||
from .kimi_linear import KimiLinearModel
|
||||
|
||||
@@ -104,36 +103,6 @@ class KimiK3Model(TextModel):
|
||||
"only the routed experts have a repack path"
|
||||
)
|
||||
|
||||
def _mxfp4_expert_tensor(self, loaders: list[tuple[Callable[[], Tensor], Callable[[], Tensor]]]):
|
||||
"""
|
||||
One stacked [n_expert, rows, cols] MXFP4 tensor, built lazily.
|
||||
|
||||
gguf_writer holds every added tensor until the final write, so building
|
||||
this eagerly (like the DeepSeek-V4 path does) keeps all ~1.38 TB of
|
||||
experts in memory. lazy means only the tensor being written is resident.
|
||||
"""
|
||||
# meta shapes, so this does not read any weights
|
||||
rows, packed_cols = loaders[0][0]().shape
|
||||
n_blocks = (packed_cols * 2) // 32
|
||||
byte_shape = (len(loaders), rows, n_blocks * 17)
|
||||
|
||||
def load(fns: list[tuple[Callable[[], Tensor], Callable[[], Tensor]]]) -> np.ndarray:
|
||||
out = np.empty(byte_shape, dtype=np.uint8)
|
||||
for eid, (packed_fn, scale_fn) in enumerate(fns):
|
||||
out[eid] = self.repack_mxfp4_blocks(
|
||||
LazyTorchTensor.to_eager(packed_fn()),
|
||||
LazyTorchTensor.to_eager(scale_fn()),
|
||||
)
|
||||
return out
|
||||
|
||||
# loaders goes through args, not the closure, so that `func` matches
|
||||
# LazyBase's single-argument shape
|
||||
return gguf.LazyNumpyTensor(
|
||||
meta=gguf.LazyNumpyTensor.meta_with_dtype_and_shape(np.uint8, byte_shape),
|
||||
args=(loaders,),
|
||||
func=load,
|
||||
)
|
||||
|
||||
def _write_mxfp4_experts(self) -> None:
|
||||
n_experts = self.hparams["num_experts"]
|
||||
|
||||
|
||||
+117
-5
@@ -10,13 +10,14 @@ import torch
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import MmprojModel, ModelBase, TextModel, gguf
|
||||
from .base import MmprojModel, ModelBase, TextModel, gguf, logger
|
||||
|
||||
|
||||
@ModelBase.register("MiMoV2FlashForCausalLM", "MiMoV2ForCausalLM")
|
||||
@ModelBase.example("XiaomiMiMo/MiMo-V2.5")
|
||||
class MimoV2Model(TextModel):
|
||||
model_arch = gguf.MODEL_ARCH.MIMO2
|
||||
supports_mtp_export = True
|
||||
|
||||
# MiMo V2-Flash, V2.5 and V2.5-Pro all ship 3 trained MTP layers under model.mtp.layers.{0,1,2}.
|
||||
# The HF config does not expose the count, so it's hardcoded to match the count found in the safetensors.
|
||||
@@ -25,6 +26,8 @@ class MimoV2Model(TextModel):
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
if self.no_mtp:
|
||||
self._n_nextn = 0
|
||||
self.block_count = self.hparams["num_hidden_layers"] + self._n_nextn
|
||||
self.tensor_map = gguf.get_tensor_name_map(self.model_arch, self.block_count)
|
||||
|
||||
@@ -101,7 +104,7 @@ class MimoV2Model(TextModel):
|
||||
qkv_overrides: dict[str, tuple[Callable, Callable, int]] = {}
|
||||
qc = self.hparams.get("quantization_config")
|
||||
if isinstance(qc, dict) and qc.get("quant_method") == "fp8":
|
||||
pat = re.compile(r"^model\.layers\.(\d+)\.self_attn\.qkv_proj\.weight_scale_inv$")
|
||||
pat = re.compile(r"^model\.(mtp\.)?layers\.(\d+)\.self_attn\.qkv_proj\.weight_scale_inv$")
|
||||
for name in list(self.model_tensors.keys()):
|
||||
m = pat.match(name)
|
||||
if not m:
|
||||
@@ -109,10 +112,13 @@ class MimoV2Model(TextModel):
|
||||
weight_name = name.removesuffix("_scale_inv")
|
||||
if weight_name not in self.model_tensors:
|
||||
continue
|
||||
bid = int(m.group(2))
|
||||
if m.group(1) is not None:
|
||||
bid += self.hparams["num_hidden_layers"]
|
||||
qkv_overrides[weight_name] = (
|
||||
self.model_tensors[weight_name],
|
||||
self.model_tensors[name],
|
||||
int(m.group(1)),
|
||||
bid,
|
||||
)
|
||||
|
||||
super().dequant_model()
|
||||
@@ -165,7 +171,86 @@ class MimoV2Model(TextModel):
|
||||
if v_scale is not None:
|
||||
self.gguf_writer.add_attn_value_scale(float(v_scale))
|
||||
|
||||
self.gguf_writer.add_nextn_predict_layers(self._n_nextn)
|
||||
if self._n_nextn > 0:
|
||||
self.gguf_writer.add_nextn_predict_layers(self._n_nextn)
|
||||
|
||||
_MXFP4_EXPERT_RE = re.compile(
|
||||
r"^model\.layers\.(\d+)\.mlp\.experts\.(\d+)\.(gate|up|down)_proj\.weight$"
|
||||
)
|
||||
_MXFP4_PROJ = {
|
||||
"gate": gguf.MODEL_TENSOR.FFN_GATE_EXP,
|
||||
"up": gguf.MODEL_TENSOR.FFN_UP_EXP,
|
||||
"down": gguf.MODEL_TENSOR.FFN_DOWN_EXP,
|
||||
}
|
||||
|
||||
def _is_mxfp4_packed(self) -> bool:
|
||||
quant_config = self.hparams.get("quantization_config") or {}
|
||||
if quant_config.get("store_dtype") != "mxfp4":
|
||||
return False
|
||||
# repack_mxfp4_blocks assumes ggml's 32-element group
|
||||
block_size = quant_config.get("mxfp4_block_size", 32)
|
||||
if block_size != 32:
|
||||
raise NotImplementedError(
|
||||
f"MXFP4 block size {block_size} is not ggml's QK_MXFP4 (32)")
|
||||
return True
|
||||
|
||||
def _write_mxfp4_experts(self) -> None:
|
||||
n_experts = self.hparams["n_routed_experts"]
|
||||
|
||||
# the FP8 half uses `weight_scale_inv` and is left to dequant_model
|
||||
stray = [n for n in self.model_tensors
|
||||
if n.endswith(".weight_scale") and not self._MXFP4_EXPERT_RE.match(n.removesuffix("_scale"))]
|
||||
if stray:
|
||||
raise NotImplementedError(
|
||||
f"{len(stray)} MXFP4 tensor(s) outside the routed experts, e.g. {stray[0]!r}; "
|
||||
"only the routed experts have a repack path"
|
||||
)
|
||||
|
||||
# (bid, proj) -> {expert id: (weight name, scale name)}
|
||||
groups: dict[tuple[int, str], dict[int, tuple[str, str]]] = {}
|
||||
for name in self.model_tensors:
|
||||
m = self._MXFP4_EXPERT_RE.match(name)
|
||||
if m is None:
|
||||
continue
|
||||
bid, eid, proj = int(m.group(1)), int(m.group(2)), m.group(3)
|
||||
scale_name = name + "_scale"
|
||||
if scale_name not in self.model_tensors:
|
||||
raise KeyError(f"missing {scale_name} for {name}")
|
||||
groups.setdefault((bid, proj), {})[eid] = (name, scale_name)
|
||||
|
||||
consumed: list[str] = []
|
||||
for (bid, proj), experts in sorted(groups.items()):
|
||||
missing = [e for e in range(n_experts) if e not in experts]
|
||||
if missing or len(experts) != n_experts:
|
||||
raise KeyError(
|
||||
f"layer {bid} {proj}_proj: {len(experts)} of {n_experts} experts present"
|
||||
+ (f", first missing is {missing[0]}" if missing else "")
|
||||
)
|
||||
|
||||
loaders = []
|
||||
for eid in range(n_experts):
|
||||
weight_name, scale_name = experts[eid]
|
||||
loaders.append((self.model_tensors[weight_name], self.model_tensors[scale_name]))
|
||||
consumed += [weight_name, scale_name]
|
||||
|
||||
data = self._mxfp4_expert_tensor(loaders)
|
||||
new_name = self.format_tensor_name(self._MXFP4_PROJ[proj], bid)
|
||||
shape = gguf.quant_shape_from_byte_shape(data.shape, gguf.GGMLQuantizationType.MXFP4)
|
||||
logger.info(
|
||||
f"{new_name}: repacked {n_experts} experts to MXFP4, "
|
||||
f"shape = {{{', '.join(str(n) for n in reversed(shape))}}}"
|
||||
)
|
||||
self.gguf_writer.add_tensor(new_name, data, raw_dtype=gguf.GGMLQuantizationType.MXFP4)
|
||||
|
||||
for name in consumed:
|
||||
del self.model_tensors[name]
|
||||
|
||||
def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
|
||||
# not a generator on purpose: base.py chains this with get_tensors(), so the
|
||||
# tensors used here must be removed from model_tensors before that starts
|
||||
if self._is_mxfp4_packed():
|
||||
self._write_mxfp4_experts()
|
||||
return ()
|
||||
|
||||
_experts: list[dict[str, Tensor]] | None = None
|
||||
|
||||
@@ -173,11 +258,32 @@ class MimoV2Model(TextModel):
|
||||
def filter_tensors(cls, item: tuple[str, Callable[[], Tensor]]) -> tuple[str, Callable[[], Tensor]] | None:
|
||||
name, gen = item
|
||||
|
||||
is_mtp = name.startswith("model.mtp.layers.")
|
||||
if is_mtp and cls.no_mtp:
|
||||
return None
|
||||
if cls.mtp_only and not is_mtp and name not in (
|
||||
"model.embed_tokens.weight", "model.norm.weight", "lm_head.weight",
|
||||
):
|
||||
return None
|
||||
|
||||
if "attention_sink" in name and not name.endswith(".weight"):
|
||||
name += ".weight"
|
||||
|
||||
return super().filter_tensors((name, gen))
|
||||
|
||||
def prepare_metadata(self, vocab_only: bool):
|
||||
from_dir = self.fname_out.is_dir()
|
||||
super().prepare_metadata(vocab_only=vocab_only)
|
||||
|
||||
if not self.mtp_only or not from_dir:
|
||||
return
|
||||
|
||||
output_type: str = self.ftype.name.partition("_")[2]
|
||||
fname_default: str = gguf.naming_convention(
|
||||
self.metadata.name, self.metadata.basename, self.metadata.finetune,
|
||||
self.metadata.version, size_label=None, output_type=output_type, model_type=None)
|
||||
self.fname_out = self.fname_out.parent / f"mtp-{fname_default}.gguf"
|
||||
|
||||
def modify_tensors(self, data_torch, name, bid):
|
||||
# Remap MTP/NextN tensors to additional layer slots so the standard tensor map handles them.
|
||||
# HF: model.mtp.layers.{i}.foo -> model.layers.{n_layer_text + i}.foo
|
||||
@@ -192,7 +298,7 @@ class MimoV2Model(TextModel):
|
||||
bid = new_bid
|
||||
|
||||
# process the experts separately
|
||||
if name.find("mlp.experts") != -1:
|
||||
if ".mlp.experts." in name and name.endswith(".weight"):
|
||||
n_experts = self.hparams["n_routed_experts"]
|
||||
assert bid is not None
|
||||
|
||||
@@ -229,6 +335,10 @@ class MimoV2Model(TextModel):
|
||||
if len(experts) > 0:
|
||||
raise ValueError(f"Unprocessed experts: {experts}")
|
||||
|
||||
if self._is_mxfp4_packed():
|
||||
self._is_mxfp4 = True
|
||||
self.ftype = gguf.LlamaFileType.MOSTLY_MXFP4_MOE
|
||||
|
||||
|
||||
@ModelBase.register("MiMoV2ForCausalLM")
|
||||
@ModelBase.example("XiaomiMiMo/MiMo-V2.5")
|
||||
@@ -382,6 +492,8 @@ class MiMoV2VisionAudioModel(MmprojModel):
|
||||
"_codebook.inited",
|
||||
)
|
||||
for name, tensor in state_dict.items():
|
||||
if name.startswith("decoder."):
|
||||
continue
|
||||
if name.endswith(skip_suffixes):
|
||||
continue
|
||||
if m := codebook_re.match(name):
|
||||
|
||||
+6
-18
@@ -10,7 +10,7 @@ import torch
|
||||
if TYPE_CHECKING:
|
||||
from torch import Tensor
|
||||
|
||||
from .base import LazyTorchTensor, ModelBase, TextModel, gguf, logger
|
||||
from .base import LazyTorchTensor, ModelBase, ModelType, TextModel, get_model_architecture, gguf, logger
|
||||
|
||||
|
||||
@ModelBase.register("QWenLMHeadModel")
|
||||
@@ -666,7 +666,7 @@ class DFlashModel(Qwen3Model):
|
||||
from . import get_model_class
|
||||
with open(self.target_model_dir / "config.json", "r", encoding="utf-8") as f:
|
||||
target_hparams = json.load(f)
|
||||
target_arch = target_hparams["architectures"][0]
|
||||
target_arch = get_model_architecture(target_hparams, ModelType.TEXT)
|
||||
target_cls = get_model_class(target_arch)
|
||||
|
||||
if target_cls is not type(self):
|
||||
@@ -711,7 +711,7 @@ class DFlashModel(Qwen3Model):
|
||||
if embedding_scale is not None:
|
||||
self.gguf_writer.add_embedding_scale(float(embedding_scale))
|
||||
|
||||
target_layer_ids = dflash_config.get("target_layer_ids", [])
|
||||
target_layer_ids = dflash_config.get("target_layer_ids", self.hparams.get("target_layer_ids", []))
|
||||
if target_layer_ids:
|
||||
extract_layer_ids = [i + 1 for i in target_layer_ids]
|
||||
self.gguf_writer.add_target_layers(extract_layer_ids)
|
||||
@@ -719,8 +719,9 @@ class DFlashModel(Qwen3Model):
|
||||
use_sliding_window = self.hparams.get("use_sliding_window", False) or dflash_config.get("use_swa", False)
|
||||
sliding_window = dflash_config.get("swa_window_size") or self.hparams.get("sliding_window")
|
||||
layer_types = self.hparams.get("layer_types")
|
||||
if use_sliding_window and sliding_window and layer_types:
|
||||
is_swa = [lt == "sliding_attention" for lt in layer_types]
|
||||
if use_sliding_window and sliding_window:
|
||||
is_swa = ([True] * self.block_count if dflash_config.get("use_swa", False)
|
||||
else [lt == "sliding_attention" for lt in layer_types or []])
|
||||
self.gguf_writer.add_sliding_window(sliding_window)
|
||||
self.gguf_writer.add_sliding_window_pattern(is_swa)
|
||||
|
||||
@@ -840,13 +841,6 @@ class DSparkModel(DFlashModel):
|
||||
return None
|
||||
return super().filter_tensors(item)
|
||||
|
||||
_ROPE_PERMUTE_SUFFIXES = (
|
||||
"self_attn.q_proj.weight",
|
||||
"self_attn.k_proj.weight",
|
||||
"self_attn.q_norm.weight",
|
||||
"self_attn.k_norm.weight",
|
||||
)
|
||||
|
||||
def modify_tensors(self, data_torch: Tensor, name: str, bid: int | None) -> Iterable[tuple[str, Tensor]]:
|
||||
if name == "model.d2t":
|
||||
self._d2t = data_torch
|
||||
@@ -855,12 +849,6 @@ class DSparkModel(DFlashModel):
|
||||
if self._n_vocab_draft == self.hparams["vocab_size"] and name.endswith("lm_head.weight"):
|
||||
return
|
||||
|
||||
# interleaved-rope checkpoints (rope_is_neox_style = false) -> NeoX layout: per head, even dims first then odd
|
||||
if not self.hparams.get("rope_is_neox_style", True) and name.endswith(self._ROPE_PERMUTE_SUFFIXES):
|
||||
head_dim = self.hparams["head_dim"]
|
||||
shape = data_torch.shape
|
||||
data_torch = data_torch.reshape(-1, head_dim // 2, 2, *shape[1:]).transpose(1, 2).reshape(shape)
|
||||
|
||||
yield from super().modify_tensors(data_torch, name, bid)
|
||||
|
||||
def prepare_tensors(self):
|
||||
|
||||
@@ -163,6 +163,7 @@ models = [
|
||||
{"name": "granite-embed-multi-311m", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/ibm-granite/granite-embedding-311m-multilingual-r2", },
|
||||
{"name": "mellum2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/JetBrains/Mellum2-12B-A2.5B-Base"},
|
||||
{"name": "laguna", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/poolside/Laguna-XS.2", },
|
||||
{"name": "ufakzeka", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/ufakai/ufakzeka-1", },
|
||||
]
|
||||
|
||||
# some models are known to be broken upstream, so we will skip them as exceptions
|
||||
@@ -191,6 +192,10 @@ pre_computed_hashes = [
|
||||
{"name": "gpt-2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/evilfreelancer/ruGPT3XL", "chkhsh": "0fe1cf6eda062318a1af7270f3331a85c539a01778ff948e24388e949c5282f4"},
|
||||
# lfm2 variants
|
||||
{"name": "lfm2", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LiquidAI/LFM2.5-8B-A1B", "chkhsh": "9e454714343b69b99b71795c1d27a68c2a1d15dab111f4d353109f966af29da7"},
|
||||
# hrm-text (DFM Mimir) is SPM-style BPE: normalizer maps ' ' -> '▁', merges
|
||||
# over the whole text (fix_mistral_regex inserts a tekken regex that is a
|
||||
# no-op here); the gemma4 pre (escape ws, split on newlines only) matches it.
|
||||
{"name": "gemma4", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/danish-foundation-models/DFM-Mimir", "chkhsh": "846deafc5b0fa786186fa4ae6c7b49903cf2f1d1895bdb80b9120d60be135252"},
|
||||
{"name": "spark2_5", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/XHToken/Spark-X2.5-1.7B", "chkhsh": "0a766d034107bc736a3f2dc4968fd62e54a3570f1454443e0c5a4cc6bd7941ed"},
|
||||
]
|
||||
|
||||
|
||||
+14
-12
@@ -12,6 +12,8 @@ The OpenVINO backend is implemented in `ggml/src/ggml-openvino` and provides a t
|
||||
- Compiles and caches the model for the target device.
|
||||
- Binds GGML tensor memory to OpenVINO inference tensors and runs inference.
|
||||
|
||||
For guidance on contributing to the OpenVINO backend, see the [OpenVINO Backend Contributing Guide](https://github.com/ravi9/llamacpp-ov-dev-guide/blob/main/contributing-llamacpp-ov.md).
|
||||
|
||||
## Contents
|
||||
|
||||
- [Supported Devices](#supported-devices)
|
||||
@@ -96,7 +98,7 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
- **SL** = Stateless (`GGML_OPENVINO_STATEFUL_EXECUTION=0`)
|
||||
- **SF** = Stateful (`GGML_OPENVINO_STATEFUL_EXECUTION=1`)
|
||||
- Note: The NPU operates in stateless mode only.
|
||||
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.35.0.
|
||||
- **Validation system:** Intel® Core™ Ultra 5 238V (Lunar Lake) | 32 GB RAM | Ubuntu 24.04 | Intel Graphics Compiler 2.41.5 | Intel OpenCL GPU Driver 26.31.39395.13-0 | Intel NPU Driver 1.38.0.
|
||||
- See [Known Limitations](#known-limitations) for context on observed failures.
|
||||
|
||||
| Model | CPU (SL / SF) | GPU (SL / SF) | NPU (SL) |
|
||||
@@ -117,9 +119,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
| [lmstudio-community/Qwen3.5-9B-Q4_K_M](https://huggingface.co/lmstudio-community/Qwen3.5-9B-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| | | | |
|
||||
| [unsloth/gemma-3-4b-it-Q4_K_M](https://huggingface.co/unsloth/gemma-3-4b-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✗ |
|
||||
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ |
|
||||
| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✗ | ✓ / ✗ | ✓ |
|
||||
| [bartowski/google_gemma-4-E2B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E2B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| [bartowski/google_gemma-4-E4B-it-Q4_K_M](https://huggingface.co/bartowski/google_gemma-4-E4B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/gemma-4-12B-it-Q4_K_M](https://huggingface.co/bartowski/gemma-4-12B-it-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| | | | |
|
||||
| [bartowski/Phi-3-mini-4k-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3-mini-4k-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/Phi-3.5-mini-instruct-Q4_K_M](https://huggingface.co/bartowski/Phi-3.5-mini-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
@@ -132,9 +134,9 @@ Although, the validated models below were tested with `llama-cli` using the `Q4_
|
||||
| [bartowski/DeepSeek-R1-Distill-Llama-8B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Llama-8B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [bartowski/DeepSeek-R1-Distill-Qwen-7B-Q4_K_M](https://huggingface.co/bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| | | | |
|
||||
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✗ / ✗ | ✓ |
|
||||
| [ibm-granite/granite-4.0-350m-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-350m-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [ibm-granite/granite-4.0-micro-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-micro-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✗ / ✗ | ✗ |
|
||||
| [ibm-granite/granite-4.0-1b-Q4_K_M](https://huggingface.co/ibm-granite/granite-4.0-1b-GGUF) | ✓ / ✓ | ✓ / ✓ | ✗ |
|
||||
| [ibm-research/granite-3.2-8b-instruct-Q4_K_M](https://huggingface.co/ibm-research/granite-3.2-8b-instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
| | | | |
|
||||
| [HuggingFaceTB/smollm2-1.7b-instruct-q4_k_m](https://huggingface.co/HuggingFaceTB/SmolLM2-1.7B-Instruct-GGUF) | ✓ / ✓ | ✓ / ✓ | ✓ |
|
||||
@@ -242,8 +244,8 @@ chmod +x build-llamacpp-ov.sh
|
||||
# ============================================
|
||||
set -euo pipefail
|
||||
|
||||
OPENVINO_VERSION_MAJOR="2026.3.1"
|
||||
OPENVINO_VERSION_FULL="2026.3.1.22476.56d9685302d"
|
||||
OPENVINO_VERSION_MAJOR="2026.4"
|
||||
OPENVINO_VERSION_FULL="2026.4.0.22959.99c81491cc3"
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
OPENVINO_INSTALL_DIR="/opt/intel/openvino_${OPENVINO_VERSION_MAJOR}"
|
||||
@@ -340,7 +342,7 @@ echo " ./build/ReleaseOV/bin/llama-cli -m model.gguf"
|
||||
```
|
||||
|
||||
> [!NOTE]
|
||||
> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
|
||||
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release.
|
||||
|
||||
</details>
|
||||
|
||||
@@ -370,8 +372,8 @@ REM ============================================
|
||||
REM llama.cpp OpenVINO Build Script (Ninja)
|
||||
REM ============================================
|
||||
|
||||
set "OPENVINO_VERSION_MAJOR=2026.3.1"
|
||||
set "OPENVINO_VERSION_FULL=2026.3.1.22476.56d9685302d"
|
||||
set "OPENVINO_VERSION_MAJOR=2026.4"
|
||||
set "OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3"
|
||||
|
||||
set "SCRIPT_DIR=%~dp0"
|
||||
set "VCPKG_DIR=C:\vcpkg"
|
||||
@@ -550,7 +552,7 @@ endlocal
|
||||
```
|
||||
|
||||
> [!NOTE]
|
||||
> The script pins OpenVINO `2026.3.1` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
|
||||
> The script pins OpenVINO `2026.4` via the `OPENVINO_VERSION_MAJOR` / `OPENVINO_VERSION_FULL` variables at the top — edit them to track a different release. From any new shell, source the matching `setupvars` script via the junction — `call "C:\Intel\openvino\setupvars.bat"` from `cmd`, or `& "C:\Intel\openvino\setupvars.ps1"` from PowerShell. If `winget` cannot register Visual Studio Build Tools on first run, install them once manually and re-run the script from an elevated **Developer Command Prompt for VS 2022**.
|
||||
|
||||
</details>
|
||||
|
||||
|
||||
@@ -327,6 +327,9 @@ on 4 physical NPUs, or `--devices 'HTP0[0-1:0],HTP1[0-1:1]'` on 2 physical NPUs
|
||||
- `GGML_HEXAGON_HOSTBUF=1` (default: 0, disabled)
|
||||
Enables allocating host buffers for debugging. By default, host buffers are disabled.
|
||||
|
||||
- `GGML_HEXAGON_DMA64=0` (default: enabled on v81+)
|
||||
Disables 64-bit DMA for model weights. Set to `1` to enable it explicitly on a supported architecture.
|
||||
|
||||
- `GGML_HEXAGON_VERBOSE=1`
|
||||
Enables verbose logging of Ops from the backend. Example output:
|
||||
|
||||
|
||||
@@ -146,6 +146,28 @@ Writing high-performance operators for Hexagon requires following specific guide
|
||||
python3 scripts/snapdragon/ggml-hexagon-align-macros.py --fix ggml/src/ggml-hexagon/htp/
|
||||
```
|
||||
|
||||
### Binary Inspection and Spill Analysis
|
||||
|
||||
Use [`scripts/snapdragon/ggml-hexagon-inspect.py`](../../../scripts/snapdragon/ggml-hexagon-inspect.py) to audit Hexagon binaries for register
|
||||
spills, unexpected float promotions, or disassembly:
|
||||
|
||||
- Always verify that compute kernels have zero in-loop vector spills (`--spills --strict`) and no float promotions (`--promotions`).
|
||||
- Avoid excessive loop unrolling (`#pragma unroll`), which increases register pressure and causes spills.
|
||||
|
||||
```bash
|
||||
# Check for vector and scalar register spills
|
||||
python3 scripts/snapdragon/ggml-hexagon-inspect.py --spills --strict --func "^compute_"
|
||||
|
||||
# Check for float promotions
|
||||
python3 scripts/snapdragon/ggml-hexagon-inspect.py --promotions --func "^compute_"
|
||||
|
||||
# Disassemble with annotated loops and spill markers
|
||||
python3 scripts/snapdragon/ggml-hexagon-inspect.py --disasm compute_same_shape_div_f32
|
||||
|
||||
# Resolve crash addresses to function symbols and lines
|
||||
python3 scripts/snapdragon/ggml-hexagon-inspect.py --addr2line 0x51a30 0x5ba54
|
||||
```
|
||||
|
||||
## Multi-Device Partitioning (mdev)
|
||||
|
||||
Multi-device (mdev) mode enables row-level tensor parallel execution across multiple physical NPU cores or virtual NPU
|
||||
|
||||
+7
-7
@@ -29,7 +29,7 @@ Legend:
|
||||
| CONT | ❌ | 🟡 | ✅ | ✅ | 🟡 | 🟡 | ✅ | 🟡 | ✅ | ✅ | 🟡 | ❌ | ❌ |
|
||||
| CONV_2D | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | 🟡 | ✅ | ❌ | ❌ |
|
||||
| CONV_2D_DW | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| CONV_3D | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| CONV_3D | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| CONV_TRANSPOSE_1D | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| CONV_TRANSPOSE_2D | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| COS | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
@@ -60,7 +60,7 @@ Legend:
|
||||
| GELU_ERF | ❌ | ✅ | ✅ | 🟡 | 🟡 | ❌ | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| GELU_QUICK | ❌ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| GET_ROWS | ❌ | 🟡 | ✅ | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | ❌ | ❌ |
|
||||
| GET_ROWS_BACK | ❌ | ❌ | 🟡 | 🟡 | ❌ | ❌ | ❌ | ❌ | ❌ | 🟡 | ❌ | ❌ | ❌ |
|
||||
| GET_ROWS_BACK | ❌ | ❌ | 🟡 | 🟡 | ❌ | ❌ | ❌ | ❌ | 🟡 | 🟡 | ❌ | ❌ | ❌ |
|
||||
| GROUP_NORM | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| HARDSIGMOID | ❌ | ✅ | ✅ | 🟡 | 🟡 | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| HARDSWISH | ❌ | ✅ | ✅ | 🟡 | 🟡 | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
@@ -70,11 +70,11 @@ Legend:
|
||||
| LEAKY_RELU | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | 🟡 | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| LIGHTNING_INDEXER | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ |
|
||||
| LOG | ❌ | ✅ | ✅ | ✅ | ❌ | 🟡 | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| MEAN | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| MEAN | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | 🟡 | ✅ | ❌ | ❌ | ❌ |
|
||||
| MUL | ❌ | ✅ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| MUL_MAT | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 |
|
||||
| MUL_MAT_HADAMARD | ❌ | ❌ | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| MUL_MAT_ID | ❌ | 🟡 | ✅ | ✅ | 🟡 | 🟡 | 🟡 | 🟡 | ✅ | ✅ | 🟡 | 🟡 | ❌ |
|
||||
| MUL_MAT_ID | ❌ | 🟡 | ✅ | ✅ | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | 🟡 | ❌ |
|
||||
| NEG | ❌ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| NORM | ❌ | ✅ | ✅ | ✅ | 🟡 | 🟡 | ✅ | ✅ | ✅ | ✅ | 🟡 | ❌ | ❌ |
|
||||
| OPT_STEP_ADAMW | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
@@ -98,7 +98,7 @@ Legend:
|
||||
| RWKV_WKV7 | ❌ | ❌ | ✅ | ✅ | ✅ | ❌ | ✅ | ❌ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| SCALE | ❌ | 🟡 | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| SET | ❌ | ❌ | ✅ | ✅ | ❌ | ❌ | ✅ | ❌ | 🟡 | ✅ | ✅ | ❌ | ❌ |
|
||||
| SET_ROWS | ❌ | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | 🟡 | ❌ | ❌ |
|
||||
| SET_ROWS | ❌ | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | 🟡 | ❌ | ❌ |
|
||||
| SGN | ❌ | ✅ | ✅ | 🟡 | 🟡 | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| SIGMOID | ❌ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| SILU | ❌ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
@@ -117,11 +117,11 @@ Legend:
|
||||
| SUM | ❌ | 🟡 | ✅ | 🟡 | ❌ | ❌ | 🟡 | ❌ | 🟡 | 🟡 | 🟡 | ❌ | ❌ |
|
||||
| SUM_ROWS | ❌ | ✅ | ✅ | 🟡 | ❌ | 🟡 | ✅ | 🟡 | 🟡 | ✅ | ✅ | ❌ | ❌ |
|
||||
| SWIGLU | ❌ | ✅ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| SWIGLU_CLAMP | ❌ | ❌ | ❌ | ❌ | ❌ | 🟡 | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ | ❌ |
|
||||
| SWIGLU_CLAMP | ❌ | ❌ | ❌ | ❌ | ❌ | 🟡 | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ❌ |
|
||||
| SWIGLU_OAI | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| TANH | ❌ | ✅ | ✅ | 🟡 | 🟡 | 🟡 | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| TIMESTEP_EMBEDDING | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ❌ | ❌ | ❌ |
|
||||
| TOP_K | ❌ | ❌ | ✅ | ❌ | ❌ | ❌ | ✅ | ❌ | 🟡 | 🟡 | ✅ | ❌ | ❌ |
|
||||
| TOP_K | ❌ | ❌ | ✅ | ❌ | ❌ | 🟡 | ✅ | ❌ | ✅ | 🟡 | ✅ | ❌ | ❌ |
|
||||
| TRI | ❌ | ❌ | ✅ | ✅ | ✅ | ✅ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| TRUNC | ❌ | ❌ | ✅ | 🟡 | 🟡 | ❌ | ✅ | ❌ | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
| UPSCALE | ❌ | 🟡 | ✅ | ✅ | ❌ | ❌ | ✅ | 🟡 | ✅ | ✅ | ✅ | ❌ | ❌ |
|
||||
|
||||
+276
-258
@@ -4588,264 +4588,282 @@
|
||||
"CUDA0","CONV_2D_DW","ne_input=[17,34,9,1],ne_kernel=[3,3,1,9],stride=1,padding=0,dilation=1,cwhn=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_2D_DW","ne_input=[32,8,64,1],ne_kernel=[3,3,1,64],stride=2,padding=1,dilation=1,cwhn=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_2D_DW","ne_input=[32,8,64,1],ne_kernel=[3,3,1,64],stride=2,padding=1,dilation=1,cwhn=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=4,ID=8,IH=8,IW=8,OC=8,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=4,ID=8,IH=8,IW=8,OC=8,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16","support","0","no","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=4,ID=8,IH=8,IW=8,OC=8,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=5,IH=11,IW=9,OC=65,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=5,ID=7,IH=9,IW=13,OC=17,KD=2,KH=3,KW=4,s0=2,s1=1,s2=3,p0=3,p1=2,p2=2,d0=2,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=3,IC=16,ID=3,IH=7,IW=9,OC=33,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=7,IH=5,IW=9,OC=33,KD=3,KH=1,KW=1,s0=1,s1=1,s2=2,p0=0,p1=0,p2=2,d0=1,d1=1,d2=2,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=1,IH=2,IW=1,OC=7,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=3,p1=4,p2=2,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=5,IH=7,IW=9,OC=17,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=1,KH=1,KW=0,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=1,KH=0,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=0,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f32,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=1,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=1,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=0,p1=0,p2=0,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=3,KW=3,s0=2,s1=2,s2=2,p0=1,p1=1,p2=1,d0=2,d1=2,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=18,IH=22,IW=20,OC=4,KD=3,KH=1,KW=5,s0=2,s1=1,s2=1,p0=2,p1=0,p2=1,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=4,ID=8,IH=8,IW=8,OC=8,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=5,IH=11,IW=9,OC=65,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=5,ID=7,IH=9,IW=13,OC=17,KD=2,KH=3,KW=4,s0=2,s1=1,s2=3,p0=3,p1=2,p2=2,d0=2,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=3,IC=16,ID=3,IH=7,IW=9,OC=33,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=7,IH=5,IW=9,OC=33,KD=3,KH=1,KW=1,s0=1,s1=1,s2=2,p0=0,p1=0,p2=2,d0=1,d1=1,d2=2,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=3,ID=1,IH=2,IW=1,OC=7,KD=1,KH=1,KW=1,s0=1,s1=1,s2=1,p0=3,p1=4,p2=2,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=2,IC=8,ID=5,IH=7,IW=9,OC=17,KD=3,KH=3,KW=3,s0=1,s1=1,s2=1,p0=1,p1=1,p2=1,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=1,KH=1,KW=0,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=1,KH=0,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_3D","N=1,IC=1,ID=2,IH=2,IW=2,OC=1,KD=0,KH=1,KW=1,s0=1,s1=1,s2=1,p0=0,p1=0,p2=0,d0=1,d1=1,d2=1,type_kernel=f16,kernel_offset=0","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_TRANSPOSE_1D","ne_input=[1,1,1,1],ne_kernel=[1,1,1,1],s0=1,p0=0,d0=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_TRANSPOSE_1D","ne_input=[1,1,1,1],ne_kernel=[1,1,1,1],s0=2,p0=0,d0=1","support","1","yes","CUDA"
|
||||
"CUDA0","CONV_TRANSPOSE_1D","ne_input=[1,1,1,1],ne_kernel=[1,1,1,1],s0=3,p0=0,d0=1","support","1","yes","CUDA"
|
||||
|
||||
|
Can't render this file because it is too large.
|
+298
-292
@@ -13936,247 +13936,247 @@
|
||||
"HTP0","ARGSORT","type=f32,ne=[2049,2,1,3],order=1","support","1","yes","HTP"
|
||||
"HTP0","ARGSORT","type=f32,ne=[2,8,8192,1],order=1","support","1","yes","HTP"
|
||||
"HTP0","ARGSORT","type=f32,ne=[2048,512,1,1],order=1","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[12,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[13,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[13,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[12,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[13,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[13,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[15,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[19,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[27,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[43,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[64,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[75,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[128,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[139,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[256,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[267,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[512,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[523,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1035,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2059,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4107,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8203,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16395,1,2,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32768,1,1,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[32779,1,2,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65536,1,1,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[65547,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131072,1,1,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[131083,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=100,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=100,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=500,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=500,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=1023,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262144,1,1,1],k=9999,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[262155,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[524288,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[524299,1,2,1],k=1,ties=0","support","0","no","HTP"
|
||||
@@ -14196,99 +14196,105 @@
|
||||
"HTP0","TOP_K","type=f32,ne=[524299,1,2,1],k=1023,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[524288,1,1,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[524299,1,2,1],k=9999,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,8,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,8,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,16,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,16,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=4,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=4,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,8,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,8,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,16,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,16,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=8,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=8,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,8,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,8,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,16,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,16,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=16,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=16,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,1,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,1,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,1,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,1,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,8,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,8,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,8,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,8,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[202048,16,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[151936,16,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=32,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=1,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=2,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=3,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=7,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=15,ties=0","support","0","no","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,16,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8193,16,1,1],k=32,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=1,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=2,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=3,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=7,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16,10,10,10],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[60,10,10,10],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1023,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1025,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[16384,1,1,1],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2047,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2049,2,1,3],k=15,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[1024,1,1,1],k=1024,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[2048,2,1,1],k=1024,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[4096,1,1,1],k=2048,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[8192,2,1,1],k=2051,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[33024,1,1,1],k=2051,ties=0","support","1","yes","HTP"
|
||||
"HTP0","TOP_K","type=f32,ne=[33024,4,1,1],k=2051,ties=0","support","1","yes","HTP"
|
||||
"HTP0","UPSCALE","type=f32,ne=[512,512,3,2],scale_factor=2,mode=nearest,transpose=0","support","0","no","HTP"
|
||||
"HTP0","UPSCALE","type=f32,ne=[512,512,3,2],scale_factor=2,mode=nearest,transpose=1","support","0","no","HTP"
|
||||
"HTP0","UPSCALE","type=f32,ne=[2,5,7,11],ne_tgt=[5,7,11,13],mode=nearest","support","0","no","HTP"
|
||||
|
||||
|
Can't render this file because it is too large.
|
+9424
-8717
File diff suppressed because it is too large
Load Diff
@@ -1,5 +1,7 @@
|
||||
set(TARGET llama-convert-llama2c-to-ggml)
|
||||
add_executable(${TARGET} convert-llama2c-to-ggml.cpp)
|
||||
install(TARGETS ${TARGET} RUNTIME)
|
||||
target_link_libraries(${TARGET} PRIVATE llama-common llama ${CMAKE_THREAD_LIBS_INIT})
|
||||
target_compile_features(${TARGET} PRIVATE cxx_std_17)
|
||||
if (GGML_CPU)
|
||||
set(TARGET llama-convert-llama2c-to-ggml)
|
||||
add_executable(${TARGET} convert-llama2c-to-ggml.cpp)
|
||||
install(TARGETS ${TARGET} RUNTIME)
|
||||
target_link_libraries(${TARGET} PRIVATE llama-common llama ${CMAKE_THREAD_LIBS_INIT})
|
||||
target_compile_features(${TARGET} PRIVATE cxx_std_17)
|
||||
endif()
|
||||
|
||||
@@ -68,6 +68,9 @@ causal-run-converted-model:
|
||||
@CONVERTED_MODEL="$(CONVERTED_MODEL)" ./scripts/causal/run-converted-model.sh
|
||||
|
||||
causal-verify-logits: causal-run-original-model causal-run-converted-model
|
||||
$(MAKE) causal-compare-logits
|
||||
|
||||
causal-compare-logits:
|
||||
@MODEL_PATH="$(MODEL_PATH)" ./scripts/causal/compare-logits.py
|
||||
@MODEL_PATH="$(MODEL_PATH)" ./scripts/utils/check-nmse.py -m ${MODEL_PATH}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
#include <clocale>
|
||||
#include <cmath>
|
||||
#include <cstdio>
|
||||
#include <random>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
#include <ctime>
|
||||
@@ -156,7 +157,7 @@ static std::vector<std::string> split_string(const std::string& input, char deli
|
||||
int main(int argc, char ** argv) {
|
||||
std::setlocale(LC_NUMERIC, "C");
|
||||
|
||||
srand(1234);
|
||||
std::mt19937 rng(1234);
|
||||
|
||||
common_params params;
|
||||
|
||||
@@ -321,7 +322,7 @@ int main(int argc, char ** argv) {
|
||||
client.t_start_prompt = ggml_time_us();
|
||||
client.t_start_gen = 0;
|
||||
|
||||
client.input = k_prompts[rand() % k_prompts.size()];
|
||||
client.input = k_prompts[rng() % k_prompts.size()];
|
||||
client.response = "";
|
||||
|
||||
// construct the prompt:
|
||||
@@ -334,10 +335,10 @@ int main(int argc, char ** argv) {
|
||||
client.prompt += k_system;
|
||||
}
|
||||
|
||||
const int n_junk_cur = rand() % n_junk;
|
||||
const int n_junk_cur = rng() % n_junk;
|
||||
|
||||
for (int i = 0; i < n_junk_cur; ++i) {
|
||||
const int r = rand() % k_questions.size();
|
||||
const int r = rng() % k_questions.size();
|
||||
client.prompt += "User:\n" + k_questions[r] + "\nAssistant:\n " + k_answers[r] + "\n";
|
||||
}
|
||||
client.prompt += "User:\n" + client.input + "\nAssistant:\n";
|
||||
|
||||
@@ -5,6 +5,9 @@ set(TARGET llama-simple-cmake-pkg)
|
||||
|
||||
find_package(Llama REQUIRED)
|
||||
|
||||
# Check that repeated package discovery does not redefine imported targets.
|
||||
find_package(Llama REQUIRED)
|
||||
|
||||
add_executable(${TARGET} ${CMAKE_CURRENT_LIST_DIR}/../simple/simple.cpp)
|
||||
install(TARGETS ${TARGET} RUNTIME)
|
||||
target_link_libraries(${TARGET} PRIVATE llama ggml::all ${CMAKE_THREAD_LIBS_INIT})
|
||||
|
||||
+2
-2
@@ -4,8 +4,8 @@ project("ggml" C CXX ASM)
|
||||
|
||||
### GGML Version
|
||||
set(GGML_VERSION_MAJOR 0)
|
||||
set(GGML_VERSION_MINOR 24)
|
||||
set(GGML_VERSION_PATCH 0)
|
||||
set(GGML_VERSION_MINOR 25)
|
||||
set(GGML_VERSION_PATCH 3)
|
||||
set(GGML_VERSION_BASE "${GGML_VERSION_MAJOR}.${GGML_VERSION_MINOR}.${GGML_VERSION_PATCH}")
|
||||
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake/")
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#define RPC_PROTO_MAJOR_VERSION 6
|
||||
#define RPC_PROTO_MAJOR_VERSION 7
|
||||
#define RPC_PROTO_MINOR_VERSION 0
|
||||
#define RPC_PROTO_PATCH_VERSION 0
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ GGML_BACKEND_API bool ggml_backend_is_sycl(ggml_backend_t backend);
|
||||
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_buffer_type(int device);
|
||||
|
||||
// split tensor buffer that splits matrices by rows across multiple devices
|
||||
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(const float * tensor_split);
|
||||
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_split_buffer_type(int main_device, const float * tensor_split);
|
||||
|
||||
// Tensor parallelism (--split-mode tensor): comm_init/free/allreduce_tensor
|
||||
// trio queried by the meta-backend via ggml_backend_reg_get_proc_address.
|
||||
@@ -36,6 +36,8 @@ GGML_BACKEND_API void ggml_backend_sycl_comm_free(void * comm_ctx);
|
||||
GGML_BACKEND_API bool ggml_backend_sycl_comm_allreduce_tensor(void * comm_ctx, struct ggml_tensor ** tensors);
|
||||
|
||||
// pinned host buffer for use with the CPU backend for faster copies between CPU and GPU
|
||||
// pins on device 0 - a copy between another device and this memory can fail,
|
||||
// use ggml_backend_dev_host_buffer_type to pin on the device that does the copy
|
||||
GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_sycl_host_buffer_type(void);
|
||||
|
||||
GGML_BACKEND_API void ggml_backend_sycl_print_sycl_devices(void);
|
||||
@@ -43,7 +45,7 @@ GGML_BACKEND_API void ggml_backend_sycl_get_gpu_list(int *id_list, int max_len);
|
||||
GGML_BACKEND_API void ggml_backend_sycl_get_device_description(int device,
|
||||
char *description,
|
||||
size_t description_size);
|
||||
GGML_BACKEND_API int ggml_backend_sycl_get_device_count();
|
||||
GGML_BACKEND_API int ggml_backend_sycl_get_device_count(void);
|
||||
GGML_BACKEND_API void ggml_backend_sycl_get_device_memory(int device, size_t *free, size_t *total);
|
||||
|
||||
// SYCL doesn't support registering host memory, keep here for reference
|
||||
|
||||
@@ -2703,11 +2703,21 @@ extern "C" {
|
||||
struct ggml_tensor * x,
|
||||
struct ggml_tensor * weights);
|
||||
|
||||
// hc_pre with a per-element gate (Qwen3.8-Flash-Next): gate [n_embd, hc, n_tokens]
|
||||
// result[i, t] = scale*sum_h x[i, h, t]*sigmoid(gate[i, h, t])
|
||||
//
|
||||
GGML_API struct ggml_tensor * ggml_dsv4_hc_pre_gated(
|
||||
struct ggml_context * ctx,
|
||||
struct ggml_tensor * x,
|
||||
struct ggml_tensor * gate,
|
||||
float scale);
|
||||
|
||||
// hc_post: x [n_embd, n_tokens], residual [n_embd, hc, n_tokens],
|
||||
// post [hc, n_tokens], comb [dst_hc, src_hc, n_tokens]
|
||||
// -> [n_embd, hc, n_tokens]
|
||||
// result[i, dst, t] = x[i, t]*post[dst, t]
|
||||
// + sum_src residual[i, src, t]*comb[dst, src, t]
|
||||
// comb == NULL uses the identity: result[i, dst, t] = x[i, t]*post[dst, t] + residual[i, dst, t]
|
||||
//
|
||||
GGML_API struct ggml_tensor * ggml_dsv4_hc_post(
|
||||
struct ggml_context * ctx,
|
||||
|
||||
@@ -865,7 +865,12 @@ static struct ggml_backend_meta_split_state ggml_backend_meta_get_split_state(
|
||||
ggml_backend_meta_split_state split_state;
|
||||
switch (tensor->op) {
|
||||
case GGML_OP_NONE: {
|
||||
split_state = {GGML_BACKEND_SPLIT_AXIS_MIRRORED, {0}, {1}, 1};
|
||||
if (tensor->view_src != nullptr) {
|
||||
// full-tensor view created with ggml_view_tensor, transparent for the split state
|
||||
split_state = ggml_backend_meta_get_split_state(stc, tensor->view_src, assume_sync);
|
||||
} else {
|
||||
split_state = {GGML_BACKEND_SPLIT_AXIS_MIRRORED, {0}, {1}, 1};
|
||||
}
|
||||
} break;
|
||||
case GGML_OP_DUP: {
|
||||
split_state = handle_generic(src_ss, /*scalar_only =*/ true);
|
||||
@@ -1200,11 +1205,6 @@ static enum ggml_status ggml_backend_meta_buffer_init_tensor_impl(ggml_backend_m
|
||||
ggml_context * simple_ctx = stc.ctxs[j].get();
|
||||
ggml_backend_buffer_t simple_buf = buf_ctx->bufs[j].get();
|
||||
|
||||
if ((simple_buf != nullptr) && ggml_backend_buffer_is_multi_buffer(simple_buf)) {
|
||||
// see https://github.com/ggml-org/llama.cpp/issues/22197
|
||||
GGML_ABORT("multi buffers are not supported by the meta backend");
|
||||
}
|
||||
|
||||
if (split_dim >= 0 && split_dim < GGML_MAX_DIMS) {
|
||||
// TODO: the following assert fails for llama-parallel even though the results are correct:
|
||||
// GGML_ASSERT(ggml_is_contiguously_allocated(tensor));
|
||||
@@ -1252,16 +1252,27 @@ static enum ggml_status ggml_backend_meta_buffer_init_tensor_impl(ggml_backend_m
|
||||
}
|
||||
}
|
||||
}
|
||||
// TODO: revisit once the graph allocator has been refactored, see https://github.com/ggml-org/llama.cpp/pull/25051#issuecomment-4842873396
|
||||
ggml_backend_buffer_t init_buf = simple_buf;
|
||||
if (t_ij->view_src != nullptr) {
|
||||
t_ij->data = (char *) t_ij->view_src->data + t_ij->view_offs;
|
||||
// views inherit the source slice's concrete sub-buffer (issue 22197)
|
||||
if (tensor->view_src != nullptr && ggml_backend_buffer_is_meta(tensor->view_src->buffer)
|
||||
&& t_ij->view_src->buffer != nullptr) {
|
||||
t_ij->buffer = t_ij->view_src->buffer;
|
||||
init_buf = t_ij->view_src->buffer;
|
||||
}
|
||||
} else if (simple_buf != nullptr) {
|
||||
if (ggml_backend_buffer_is_multi_buffer(simple_buf)) {
|
||||
GGML_ABORT("multi buffers are not supported by the meta backend");
|
||||
}
|
||||
t_ij->data = (char *) ggml_backend_buffer_get_base(simple_buf)
|
||||
+ size_t(tensor->data) - size_t(ggml_backend_buffer_get_base(tensor->buffer));
|
||||
}
|
||||
|
||||
if (simple_buf) {
|
||||
if (init_buf) {
|
||||
// the backend that owns the buffer will set .extra
|
||||
ggml_backend_buffer_init_tensor(simple_buf, t_ij);
|
||||
ggml_backend_buffer_init_tensor(init_buf, t_ij);
|
||||
} else {
|
||||
t_ij->extra = tensor->extra;
|
||||
}
|
||||
@@ -2283,6 +2294,14 @@ static enum ggml_status ggml_backend_meta_graph_compute(ggml_backend_t backend,
|
||||
cgraph_ij->uid = ggml_graph_next_uid();
|
||||
}
|
||||
}
|
||||
|
||||
// Aux graph contents are rewritten on every compute but are identical across calls while the subgraphs are reused,
|
||||
// so they can get stable uids on rebuild. Only safe without a comm backend, where the fallback usage is deterministic.
|
||||
if (backend_ctx->comm_ctx == nullptr) {
|
||||
for (ggml_cgraph * cgraph_aux : backend_ctx->cgraphs_aux) {
|
||||
cgraph_aux->uid = ggml_graph_next_uid();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
size_t iga = 0; // i graph aux
|
||||
|
||||
@@ -1630,7 +1630,10 @@ static bool ggml_backend_sched_alloc_splits(ggml_backend_sched_t sched) {
|
||||
ggml_backend_synchronize(sched->backends[i]);
|
||||
}
|
||||
|
||||
ggml_gallocr_reserve_n(sched->galloc, &sched->graph, sched->node_backend_ids, sched->leaf_backend_ids);
|
||||
if (!ggml_gallocr_reserve_n(sched->galloc, &sched->graph, sched->node_backend_ids, sched->leaf_backend_ids)) {
|
||||
GGML_LOG_ERROR("%s: failed to reserve graph buffers\n", __func__);
|
||||
return false;
|
||||
}
|
||||
if (!ggml_gallocr_alloc_graph(sched->galloc, &sched->graph)) {
|
||||
GGML_LOG_ERROR("%s: failed to allocate graph\n", __func__);
|
||||
return false;
|
||||
|
||||
@@ -39,6 +39,8 @@
|
||||
#define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0
|
||||
@@ -55,6 +57,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0
|
||||
@@ -87,6 +91,8 @@
|
||||
// repack.cpp
|
||||
#define ggml_quantize_mat_q8_0_4x4_generic ggml_quantize_mat_q8_0_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_K_8x4_q8_K_generic ggml_gemv_q4_K_8x4_q8_K
|
||||
@@ -98,6 +104,8 @@
|
||||
#define ggml_gemv_mxfp4_4x4_q8_0_generic ggml_gemv_mxfp4_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_K_8x4_q8_K_generic ggml_gemm_q4_K_8x4_q8_K
|
||||
@@ -124,6 +132,8 @@
|
||||
#define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0
|
||||
@@ -140,6 +150,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0
|
||||
@@ -171,6 +183,8 @@
|
||||
#define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0
|
||||
@@ -187,6 +201,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0
|
||||
@@ -213,6 +229,8 @@
|
||||
#define ggml_quantize_mat_q8_K_4x1_generic ggml_quantize_mat_q8_K_4x1
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q2_K_8x8_q8_K_generic ggml_gemv_q2_K_8x8_q8_K
|
||||
@@ -228,6 +246,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q2_K_8x8_q8_K_generic ggml_gemm_q2_K_8x8_q8_K
|
||||
@@ -262,6 +282,8 @@
|
||||
#define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0
|
||||
#define ggml_gemv_q2_K_8x8_q8_K_generic ggml_gemv_q2_K_8x8_q8_K
|
||||
@@ -277,6 +299,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0
|
||||
#define ggml_gemm_q2_K_8x8_q8_K_generic ggml_gemm_q2_K_8x8_q8_K
|
||||
@@ -314,6 +338,8 @@
|
||||
#define ggml_quantize_mat_q8_0_4x8_generic ggml_quantize_mat_q8_0_4x8
|
||||
#define ggml_quantize_mat_q8_K_4x4_generic ggml_quantize_mat_q8_K_4x4
|
||||
#define ggml_quantize_mat_q8_K_4x8_generic ggml_quantize_mat_q8_K_4x8
|
||||
#define ggml_gemv_q1_0_4x4_q8_0_generic ggml_gemv_q1_0_4x4_q8_0
|
||||
#define ggml_gemv_q1_0_4x8_q8_0_generic ggml_gemv_q1_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_4x4_q8_0_generic ggml_gemv_q4_0_4x4_q8_0
|
||||
#define ggml_gemv_q4_0_4x8_q8_0_generic ggml_gemv_q4_0_4x8_q8_0
|
||||
#define ggml_gemv_q4_0_8x8_q8_0_generic ggml_gemv_q4_0_8x8_q8_0
|
||||
@@ -330,6 +356,8 @@
|
||||
#define ggml_gemv_mxfp4_8x8_q8_0_generic ggml_gemv_mxfp4_8x8_q8_0
|
||||
#define ggml_gemv_q8_0_4x4_q8_0_generic ggml_gemv_q8_0_4x4_q8_0
|
||||
#define ggml_gemv_q8_0_4x8_q8_0_generic ggml_gemv_q8_0_4x8_q8_0
|
||||
#define ggml_gemm_q1_0_4x4_q8_0_generic ggml_gemm_q1_0_4x4_q8_0
|
||||
#define ggml_gemm_q1_0_4x8_q8_0_generic ggml_gemm_q1_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_4x4_q8_0_generic ggml_gemm_q4_0_4x4_q8_0
|
||||
#define ggml_gemm_q4_0_4x8_q8_0_generic ggml_gemm_q4_0_4x8_q8_0
|
||||
#define ggml_gemm_q4_0_8x8_q8_0_generic ggml_gemm_q4_0_8x8_q8_0
|
||||
|
||||
@@ -48,6 +48,24 @@ static inline void decode_q_Kx8_6bit_scales(const uint8_t * scales_in, int16x8_t
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__aarch64__) && defined(__ARM_NEON) && (defined(__ARM_FEATURE_DOTPROD) || defined(__ARM_FEATURE_MATMUL_INT8))
|
||||
#define B1(c,s,n) 0x ## n ## c , 0x ## n ## s
|
||||
#define B2(c,s,n) B1(c,s,n ## c), B1(c,s,n ## s)
|
||||
#define B3(c,s,n) B2(c,s,n ## c), B2(c,s,n ## s)
|
||||
#define B4(c,s,n) B3(c,s,n ## c), B3(c,s,n ## s)
|
||||
#define B5(c,s,n) B4(c,s,n ## c), B4(c,s,n ## s)
|
||||
#define B6(c,s,n) B5(c,s,n ## c), B5(c,s,n ## s)
|
||||
#define B7(c,s,n) B6(c,s,n ## c), B6(c,s,n ## s)
|
||||
#define B8(c,s ) B7(c,s, c), B7(c,s, s)
|
||||
|
||||
static const uint64_t table_q1_signs[256] = { B8(ff, 01) };
|
||||
|
||||
static inline int8x16_t ggml_q1_0_unpack_pair(uint8_t bits0, uint8_t bits1) {
|
||||
return vreinterpretq_s8_u8(vcombine_u8(vcreate_u8(table_q1_signs[bits0]),
|
||||
vcreate_u8(table_q1_signs[bits1])));
|
||||
}
|
||||
#endif
|
||||
|
||||
void ggml_quantize_mat_q8_0_4x4(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k) {
|
||||
assert(QK8_0 == 32);
|
||||
assert(k % QK8_0 == 0);
|
||||
@@ -1823,6 +1841,132 @@ void ggml_gemv_q8_0_4x8_q8_0(int n,
|
||||
ggml_gemv_q8_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
void ggml_gemv_q1_0_4x4_q8_0(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(nb);
|
||||
UNUSED(ncols_interleaved);
|
||||
|
||||
#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
|
||||
for (int c = 0; c < nc; c += ncols_interleaved) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (c / ncols_interleaved) * nb;
|
||||
const block_q8_0 * a_ptr = (const block_q8_0 *) vy;
|
||||
float32x4_t acc = vdupq_n_f32(0);
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d));
|
||||
float32x4_t accb = vdupq_n_f32(0);
|
||||
|
||||
for (int k = 0; k < 4; k++) {
|
||||
const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * 4 + k;
|
||||
const float ad = GGML_CPU_FP16_TO_FP32(a_blk->d);
|
||||
int32x4_t ret = vdupq_n_s32(0);
|
||||
|
||||
for (int tile = 0; tile < 8; tile += 4) {
|
||||
const int8x16_t signs0 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 0) + 0],
|
||||
b_ptr[l].qs[k * 16 + 2 * (tile + 0) + 1]);
|
||||
const int8x16_t signs1 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 1) + 0],
|
||||
b_ptr[l].qs[k * 16 + 2 * (tile + 1) + 1]);
|
||||
const int8x16_t signs2 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 2) + 0],
|
||||
b_ptr[l].qs[k * 16 + 2 * (tile + 2) + 1]);
|
||||
const int8x16_t signs3 = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * (tile + 3) + 0],
|
||||
b_ptr[l].qs[k * 16 + 2 * (tile + 3) + 1]);
|
||||
const int8x16_t q_tiles = vld1q_s8(a_blk->qs + tile * 4);
|
||||
|
||||
ret = vdotq_laneq_s32(ret, signs0, q_tiles, 0);
|
||||
ret = vdotq_laneq_s32(ret, signs1, q_tiles, 1);
|
||||
ret = vdotq_laneq_s32(ret, signs2, q_tiles, 2);
|
||||
ret = vdotq_laneq_s32(ret, signs3, q_tiles, 3);
|
||||
}
|
||||
|
||||
accb = vfmaq_n_f32(accb, vcvtq_f32_s32(ret), ad);
|
||||
}
|
||||
acc = vfmaq_f32(acc, accb, b_d);
|
||||
}
|
||||
vst1q_f32(s, acc);
|
||||
s += ncols_interleaved;
|
||||
}
|
||||
return;
|
||||
#endif
|
||||
ggml_gemv_q1_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
void ggml_gemv_q1_0_4x8_q8_0(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(nb);
|
||||
UNUSED(ncols_interleaved);
|
||||
|
||||
#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
|
||||
for (int c = 0; c < nc; c += ncols_interleaved) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (c / ncols_interleaved) * nb;
|
||||
const block_q8_0 * a_ptr = (const block_q8_0 *) vy;
|
||||
float32x4_t acc = vdupq_n_f32(0);
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d));
|
||||
float32x4_t accb = vdupq_n_f32(0);
|
||||
|
||||
for (int k = 0; k < 4; ++k) {
|
||||
const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * 4 + k;
|
||||
const uint8_t * GGML_RESTRICT b_qs = (const uint8_t *) b_ptr[l].qs + k * 16;
|
||||
const float ad = GGML_CPU_FP16_TO_FP32(a_blk->d);
|
||||
|
||||
int8x8x4_t a_chunks = vld1_s8_x4(a_blk->qs);
|
||||
int8x16_t a0 = vcombine_s8(a_chunks.val[0], a_chunks.val[0]);
|
||||
int8x16_t a1 = vcombine_s8(a_chunks.val[1], a_chunks.val[1]);
|
||||
int8x16_t a2 = vcombine_s8(a_chunks.val[2], a_chunks.val[2]);
|
||||
int8x16_t a3 = vcombine_s8(a_chunks.val[3], a_chunks.val[3]);
|
||||
|
||||
int32x4_t ret0 = vdupq_n_s32(0);
|
||||
int32x4_t ret1 = vdupq_n_s32(0);
|
||||
|
||||
ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[0], b_qs[1]), a0);
|
||||
ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[2], b_qs[3]), a0);
|
||||
ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[4], b_qs[5]), a1);
|
||||
ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[6], b_qs[7]), a1);
|
||||
ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[8], b_qs[9]), a2);
|
||||
ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[10], b_qs[11]), a2);
|
||||
ret0 = vdotq_s32(ret0, ggml_q1_0_unpack_pair(b_qs[12], b_qs[13]), a3);
|
||||
ret1 = vdotq_s32(ret1, ggml_q1_0_unpack_pair(b_qs[14], b_qs[15]), a3);
|
||||
|
||||
accb = vfmaq_n_f32(accb, vcvtq_f32_s32(vpaddq_s32(ret0, ret1)), ad);
|
||||
}
|
||||
|
||||
acc = vfmaq_f32(acc, accb, b_d);
|
||||
}
|
||||
|
||||
vst1q_f32(s, acc);
|
||||
s += ncols_interleaved;
|
||||
}
|
||||
return;
|
||||
#endif
|
||||
|
||||
ggml_gemv_q1_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
void ggml_gemm_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) {
|
||||
const int qk = QK8_0;
|
||||
const int nb = n / qk;
|
||||
@@ -5154,3 +5298,168 @@ void ggml_gemm_q8_0_4x8_q8_0(int n,
|
||||
#endif // defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
ggml_gemm_q8_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
void ggml_gemm_q1_0_4x4_q8_0(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nr % 4 == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(nb);
|
||||
UNUSED(ncols_interleaved);
|
||||
|
||||
#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_DOTPROD)
|
||||
for (int y = 0; y < nr / 4; y++) {
|
||||
const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb);
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
float32x4_t sumf[4];
|
||||
for (int m = 0; m < 4; m++) {
|
||||
sumf[m] = vdupq_n_f32(0);
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d));
|
||||
float32x4_t blockf_0 = vdupq_n_f32(0);
|
||||
float32x4_t blockf_1 = vdupq_n_f32(0);
|
||||
float32x4_t blockf_2 = vdupq_n_f32(0);
|
||||
float32x4_t blockf_3 = vdupq_n_f32(0);
|
||||
|
||||
for (int k = 0; k < 4; ++k) {
|
||||
const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k;
|
||||
float32x4_t a_d = vcvt_f32_f16(vld1_f16((const float16_t *) a_blk->d));
|
||||
|
||||
int32x4_t sumi_0 = vdupq_n_s32(0);
|
||||
int32x4_t sumi_1 = vdupq_n_s32(0);
|
||||
int32x4_t sumi_2 = vdupq_n_s32(0);
|
||||
int32x4_t sumi_3 = vdupq_n_s32(0);
|
||||
|
||||
for (int tile = 0; tile < 8; ++tile) {
|
||||
const int8x16_t signs = ggml_q1_0_unpack_pair(b_ptr[l].qs[k * 16 + 2 * tile + 0],
|
||||
b_ptr[l].qs[k * 16 + 2 * tile + 1]);
|
||||
const int8x16_t a_tile = vld1q_s8(a_blk->qs + tile * 16);
|
||||
|
||||
sumi_0 = vdotq_laneq_s32(sumi_0, signs, a_tile, 0);
|
||||
sumi_1 = vdotq_laneq_s32(sumi_1, signs, a_tile, 1);
|
||||
sumi_2 = vdotq_laneq_s32(sumi_2, signs, a_tile, 2);
|
||||
sumi_3 = vdotq_laneq_s32(sumi_3, signs, a_tile, 3);
|
||||
}
|
||||
|
||||
blockf_0 = vfmaq_laneq_f32(blockf_0, vcvtq_f32_s32(sumi_0), a_d, 0);
|
||||
blockf_1 = vfmaq_laneq_f32(blockf_1, vcvtq_f32_s32(sumi_1), a_d, 1);
|
||||
blockf_2 = vfmaq_laneq_f32(blockf_2, vcvtq_f32_s32(sumi_2), a_d, 2);
|
||||
blockf_3 = vfmaq_laneq_f32(blockf_3, vcvtq_f32_s32(sumi_3), a_d, 3);
|
||||
}
|
||||
|
||||
sumf[0] = vfmaq_f32(sumf[0], blockf_0, b_d);
|
||||
sumf[1] = vfmaq_f32(sumf[1], blockf_1, b_d);
|
||||
sumf[2] = vfmaq_f32(sumf[2], blockf_2, b_d);
|
||||
sumf[3] = vfmaq_f32(sumf[3], blockf_3, b_d);
|
||||
}
|
||||
|
||||
for (int m = 0; m < 4; m++) {
|
||||
vst1q_f32(s + (y * 4 + m) * bs + x * 4, sumf[m]);
|
||||
}
|
||||
}
|
||||
}
|
||||
return;
|
||||
#endif
|
||||
ggml_gemm_q1_0_4x4_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
void ggml_gemm_q1_0_4x8_q8_0(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nr % 4 == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(nb);
|
||||
UNUSED(ncols_interleaved);
|
||||
|
||||
#if defined(__aarch64__) && defined(__ARM_NEON) && defined(__ARM_FEATURE_MATMUL_INT8)
|
||||
for (int y = 0; y < nr / 4; y++) {
|
||||
const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb);
|
||||
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
float32x4_t sumf[4];
|
||||
for (int m = 0; m < 4; ++m) {
|
||||
sumf[m] = vdupq_n_f32(0);
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float32x4_t b_d = vcvt_f32_f16(vld1_f16((const float16_t *) b_ptr[l].d));
|
||||
float32x4_t blockf[4];
|
||||
for (int m = 0; m < 4; ++m) {
|
||||
blockf[m] = vdupq_n_f32(0);
|
||||
}
|
||||
|
||||
for (int k = 0; k < 4; ++k) {
|
||||
const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k;
|
||||
const uint8_t * GGML_RESTRICT b_qs = (const uint8_t *) b_ptr[l].qs + k * 16;
|
||||
|
||||
int32x4_t acc[4];
|
||||
for (int i = 0; i < 4; ++i) {
|
||||
acc[i] = vdupq_n_s32(0);
|
||||
}
|
||||
|
||||
for (int chunk = 0; chunk < 4; ++chunk) {
|
||||
const int8x16_t a01 = vld1q_s8(a_blk->qs + chunk * 32);
|
||||
const int8x16_t a23 = vld1q_s8(a_blk->qs + chunk * 32 + 16);
|
||||
const int8x16_t b01 = ggml_q1_0_unpack_pair(b_qs[chunk * 4 + 0], b_qs[chunk * 4 + 1]);
|
||||
const int8x16_t b23 = ggml_q1_0_unpack_pair(b_qs[chunk * 4 + 2], b_qs[chunk * 4 + 3]);
|
||||
|
||||
acc[0] = vmmlaq_s32(acc[0], a01, b01);
|
||||
acc[1] = vmmlaq_s32(acc[1], a01, b23);
|
||||
acc[2] = vmmlaq_s32(acc[2], a23, b01);
|
||||
acc[3] = vmmlaq_s32(acc[3], a23, b23);
|
||||
}
|
||||
|
||||
const int32x4_t row0 = vcombine_s32(vget_low_s32(acc[0]), vget_low_s32(acc[1]));
|
||||
const int32x4_t row1 = vcombine_s32(vget_high_s32(acc[0]), vget_high_s32(acc[1]));
|
||||
const int32x4_t row2 = vcombine_s32(vget_low_s32(acc[2]), vget_low_s32(acc[3]));
|
||||
const int32x4_t row3 = vcombine_s32(vget_high_s32(acc[2]), vget_high_s32(acc[3]));
|
||||
const float32x4_t a_d = vcvt_f32_f16(vld1_f16((const float16_t *) a_blk->d));
|
||||
|
||||
blockf[0] = vfmaq_laneq_f32(blockf[0], vcvtq_f32_s32(row0), a_d, 0);
|
||||
blockf[1] = vfmaq_laneq_f32(blockf[1], vcvtq_f32_s32(row1), a_d, 1);
|
||||
blockf[2] = vfmaq_laneq_f32(blockf[2], vcvtq_f32_s32(row2), a_d, 2);
|
||||
blockf[3] = vfmaq_laneq_f32(blockf[3], vcvtq_f32_s32(row3), a_d, 3);
|
||||
}
|
||||
|
||||
sumf[0] = vfmaq_f32(sumf[0], blockf[0], b_d);
|
||||
sumf[1] = vfmaq_f32(sumf[1], blockf[1], b_d);
|
||||
sumf[2] = vfmaq_f32(sumf[2], blockf[2], b_d);
|
||||
sumf[3] = vfmaq_f32(sumf[3], blockf[3], b_d);
|
||||
}
|
||||
|
||||
for (int m = 0; m < 4; ++m) {
|
||||
vst1q_f32(s + (y * 4 + m) * bs + x * 4, sumf[m]);
|
||||
}
|
||||
}
|
||||
}
|
||||
return;
|
||||
#endif
|
||||
|
||||
ggml_gemm_q1_0_4x8_q8_0_generic(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
@@ -1329,7 +1329,9 @@ UseGgmlGemm1:;
|
||||
const size_t nbw3 = nbw2*ne12;
|
||||
|
||||
assert(params->wsize >= ne13*nbw3);
|
||||
GGML_ASSERT(src1->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16);
|
||||
// the F16 path below writes plain floats into wdata, so it needs an F32 vec_dot_type
|
||||
GGML_ASSERT(src1->type == GGML_TYPE_F32 || vec_dot_type == GGML_TYPE_F32);
|
||||
|
||||
#if 0
|
||||
for (int64_t i13 = 0; i13 < ne13; ++i13) {
|
||||
@@ -1348,9 +1350,15 @@ UseGgmlGemm1:;
|
||||
size_t bs = ggml_blck_size(vec_dot_type);
|
||||
int64_t ne10_block_start = (ith * ne10/bs) / nth;
|
||||
int64_t ne10_block_end = ((ith + 1) * ne10/bs) / nth;
|
||||
from_float((float *)((char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10),
|
||||
(void *) (wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0),
|
||||
(ne10_block_end - ne10_block_start) * bs);
|
||||
const char * src1_block = (const char *) src1->data + i13*nb13 + i12*nb12 + i11*nb11 + ne10_block_start*bs*nb10;
|
||||
char * dst_block = wdata + i13*nbw3 + i12*nbw2 + i11*nbw1 + ne10_block_start*nbw0;
|
||||
const int64_t n_block = (ne10_block_end - ne10_block_start) * bs;
|
||||
|
||||
if (src1->type == GGML_TYPE_F32) {
|
||||
from_float((const float *) src1_block, dst_block, n_block);
|
||||
} else {
|
||||
ggml_cpu_fp16_to_fp32((const ggml_fp16_t *) src1_block, (float *) dst_block, n_block);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -451,6 +451,10 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st
|
||||
op->type != GGML_TYPE_IQ1_S &&
|
||||
op->type != GGML_TYPE_IQ1_M; // missing type_traits.from_float
|
||||
case GGML_OP_MUL_MAT:
|
||||
if (ggml_get_op_params_i32(op, 1) == GGML_HINT_SRC0_IS_HADAMARD &&
|
||||
src0->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32) {
|
||||
return src1->type == GGML_TYPE_F32 || src1->type == GGML_TYPE_F16;
|
||||
}
|
||||
return src1->type == GGML_TYPE_F32 || src1->type == ggml_get_type_traits_cpu(src0->type)->vec_dot_type;
|
||||
case GGML_OP_SOFT_MAX_BACK: {
|
||||
if (op->src[0]->type != GGML_TYPE_F32 || op->src[1]->type != GGML_TYPE_F32) {
|
||||
|
||||
+62
-20
@@ -11259,10 +11259,19 @@ static void ggml_compute_forward_dsv4_hc_pre_f32(
|
||||
const int64_t hc = x->ne[1];
|
||||
const int64_t n_tokens = x->ne[2];
|
||||
|
||||
const float scale = ggml_get_op_params_f32(dst, 0);
|
||||
const bool gated = ggml_get_op_params_i32(dst, 1) != 0;
|
||||
|
||||
GGML_ASSERT(dst->ne[0] == n_embd);
|
||||
GGML_ASSERT(dst->ne[1] == n_tokens);
|
||||
GGML_ASSERT(weights->ne[0] == hc);
|
||||
GGML_ASSERT(weights->ne[1] == n_tokens);
|
||||
if (gated) {
|
||||
GGML_ASSERT(weights->ne[0] == n_embd);
|
||||
GGML_ASSERT(weights->ne[1] == hc);
|
||||
GGML_ASSERT(weights->ne[2] == n_tokens);
|
||||
} else {
|
||||
GGML_ASSERT(weights->ne[0] == hc);
|
||||
GGML_ASSERT(weights->ne[1] == n_tokens);
|
||||
}
|
||||
|
||||
GGML_TENSOR_LOCALS(size_t, nbx, x, nb);
|
||||
GGML_TENSOR_LOCALS(size_t, nbw, weights, nb);
|
||||
@@ -11282,12 +11291,18 @@ static void ggml_compute_forward_dsv4_hc_pre_f32(
|
||||
|
||||
float sum = 0.0f;
|
||||
for (int64_t ih = 0; ih < hc; ++ih) {
|
||||
const float xv = *(const float *) ((const char *) x->data + i0*nbx0 + ih*nbx1 + it*nbx2);
|
||||
const float wv = *(const float *) ((const char *) weights->data + ih*nbw0 + it*nbw1);
|
||||
const float xv = *(const float *) ((const char *) x->data + i0*nbx0 + ih*nbx1 + it*nbx2);
|
||||
float wv;
|
||||
if (gated) {
|
||||
const float gv = *(const float *) ((const char *) weights->data + i0*nbw0 + ih*nbw1 + it*nbw2);
|
||||
wv = 1.0f / (1.0f + expf(-gv));
|
||||
} else {
|
||||
wv = *(const float *) ((const char *) weights->data + ih*nbw0 + it*nbw1);
|
||||
}
|
||||
sum += xv * wv;
|
||||
}
|
||||
|
||||
*(float *) ((char *) dst->data + i0*nbd0 + it*nbd1) = sum;
|
||||
*(float *) ((char *) dst->data + i0*nbd0 + it*nbd1) = scale * sum;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -11321,7 +11336,6 @@ static void ggml_compute_forward_dsv4_hc_post_f32(
|
||||
GGML_ASSERT(x->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(residual->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(post->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(comb->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(dst->type == GGML_TYPE_F32);
|
||||
|
||||
const int64_t n_embd = x->ne[0];
|
||||
@@ -11335,14 +11349,24 @@ static void ggml_compute_forward_dsv4_hc_post_f32(
|
||||
GGML_ASSERT(residual->ne[2] == n_tokens);
|
||||
GGML_ASSERT(post->ne[0] == hc);
|
||||
GGML_ASSERT(post->ne[1] == n_tokens);
|
||||
GGML_ASSERT(comb->ne[0] == hc);
|
||||
GGML_ASSERT(comb->ne[1] == hc);
|
||||
GGML_ASSERT(comb->ne[2] == n_tokens);
|
||||
|
||||
// comb == NULL: identity mixing, each stream keeps its own residual
|
||||
size_t nbc0 = 0;
|
||||
size_t nbc1 = 0;
|
||||
size_t nbc2 = 0;
|
||||
if (comb) {
|
||||
GGML_ASSERT(comb->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(comb->ne[0] == hc);
|
||||
GGML_ASSERT(comb->ne[1] == hc);
|
||||
GGML_ASSERT(comb->ne[2] == n_tokens);
|
||||
nbc0 = comb->nb[0];
|
||||
nbc1 = comb->nb[1];
|
||||
nbc2 = comb->nb[2];
|
||||
}
|
||||
|
||||
GGML_TENSOR_LOCALS(size_t, nbx, x, nb);
|
||||
GGML_TENSOR_LOCALS(size_t, nbr, residual, nb);
|
||||
GGML_TENSOR_LOCALS(size_t, nbp, post, nb);
|
||||
GGML_TENSOR_LOCALS(size_t, nbc, comb, nb);
|
||||
GGML_TENSOR_LOCALS(size_t, nbd, dst, nb);
|
||||
|
||||
const int ith = params->ith;
|
||||
@@ -11362,10 +11386,14 @@ static void ggml_compute_forward_dsv4_hc_post_f32(
|
||||
const float pv = *(const float *) ((const char *) post->data + idst*nbp0 + it*nbp1);
|
||||
|
||||
float sum = xv * pv;
|
||||
for (int64_t isrc = 0; isrc < hc; ++isrc) {
|
||||
const float rv = *(const float *) ((const char *) residual->data + i0*nbr0 + isrc*nbr1 + it*nbr2);
|
||||
const float cv = *(const float *) ((const char *) comb->data + idst*nbc0 + isrc*nbc1 + it*nbc2);
|
||||
sum += rv * cv;
|
||||
if (comb) {
|
||||
for (int64_t isrc = 0; isrc < hc; ++isrc) {
|
||||
const float rv = *(const float *) ((const char *) residual->data + i0*nbr0 + isrc*nbr1 + it*nbr2);
|
||||
const float cv = *(const float *) ((const char *) comb->data + idst*nbc0 + isrc*nbc1 + it*nbc2);
|
||||
sum += rv * cv;
|
||||
}
|
||||
} else {
|
||||
sum += *(const float *) ((const char *) residual->data + i0*nbr0 + idst*nbr1 + it*nbr2);
|
||||
}
|
||||
|
||||
*(float *) ((char *) dst->data + i0*nbd0 + idst*nbd1 + it*nbd2) = sum;
|
||||
@@ -11987,11 +12015,20 @@ void ggml_compute_forward_opt_step_sgd(const ggml_compute_params * params, ggml_
|
||||
}
|
||||
}
|
||||
|
||||
static void ggml_compute_forward_fwht_f32(const ggml_compute_params * params, ggml_tensor * dst) {
|
||||
static inline float ggml_fwht_load(const float value) {
|
||||
return value;
|
||||
}
|
||||
|
||||
static inline float ggml_fwht_load(const ggml_fp16_t value) {
|
||||
return ggml_fp16_to_fp32(value);
|
||||
}
|
||||
|
||||
template<typename src_t>
|
||||
static void ggml_compute_forward_fwht_impl(const ggml_compute_params * params, ggml_tensor * dst) {
|
||||
const ggml_tensor * src0 = dst->src[0];
|
||||
const ggml_tensor * src1 = dst->src[1];
|
||||
|
||||
GGML_ASSERT(src1->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(src1->type == (std::is_same_v<src_t, float> ? GGML_TYPE_F32 : GGML_TYPE_F16));
|
||||
GGML_ASSERT(dst->type == GGML_TYPE_F32);
|
||||
|
||||
GGML_TENSOR_BINARY_OP_LOCALS
|
||||
@@ -12018,11 +12055,11 @@ static void ggml_compute_forward_fwht_f32(const ggml_compute_params * params, gg
|
||||
const int64_t i12 = (r - i13 * ne11 * ne12) / ne11;
|
||||
const int64_t i11 = r - i13 * ne11 * ne12 - i12 * ne11;
|
||||
|
||||
const float * src_row = (const float *) ((const char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13);
|
||||
const src_t * src_row = (const src_t *) ((const char *) src1->data + i11 * nb11 + i12 * nb12 + i13 * nb13);
|
||||
float * dst_row = (float *) ((char *) dst->data + i11 * nb1 + i12 * nb2 + i13 * nb3);
|
||||
|
||||
for (int64_t j = 0; j < n; j++) {
|
||||
dst_row[j] = src_row[j] * scale;
|
||||
dst_row[j] = ggml_fwht_load(src_row[j]) * scale;
|
||||
}
|
||||
|
||||
// Scalar passes
|
||||
@@ -12069,12 +12106,17 @@ void ggml_compute_forward_fwht(const ggml_compute_params * params, ggml_tensor *
|
||||
switch (src1->type) {
|
||||
case GGML_TYPE_F32:
|
||||
{
|
||||
ggml_compute_forward_fwht_f32(params, dst);
|
||||
ggml_compute_forward_fwht_impl<float>(params, dst);
|
||||
}
|
||||
break;
|
||||
case GGML_TYPE_F16:
|
||||
{
|
||||
ggml_compute_forward_fwht_impl<ggml_fp16_t>(params, dst);
|
||||
}
|
||||
break;
|
||||
default:
|
||||
{
|
||||
GGML_ABORT("fatal error - fwht is F32 only");
|
||||
GGML_ABORT("fatal error - fwht supports F32 and F16 input");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1365,6 +1365,133 @@ void ggml_gemv_q8_0_4x8_q8_0_generic(int n,
|
||||
}
|
||||
}
|
||||
|
||||
void ggml_gemv_q1_0_4x4_q8_0_generic(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(nr == 1);
|
||||
assert(n % qk == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(bs);
|
||||
UNUSED(nr);
|
||||
|
||||
float sumf[4];
|
||||
|
||||
const block_q8_0 * a_ptr = (const block_q8_0 *) vy;
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
sumf[j] = 0.0;
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float d0[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]),
|
||||
};
|
||||
|
||||
for (int k = 0; k < QK1_0 / QK8_0; ++k) {
|
||||
const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * (QK1_0 / QK8_0) + k;
|
||||
const float d1 = GGML_CPU_FP16_TO_FP32(a_blk->d);
|
||||
const float scale[4] = { d0[0] * d1, d0[1] * d1, d0[2] * d1, d0[3] * d1 };
|
||||
|
||||
for (int tile = 0; tile < QK8_0 / 4; ++tile) {
|
||||
const uint8_t bits_lo = b_ptr[l].qs[k * 16 + 2 * tile + 0];
|
||||
const uint8_t bits_hi = b_ptr[l].qs[k * 16 + 2 * tile + 1];
|
||||
|
||||
for (int p = 0; p < 4; ++p) {
|
||||
const float q = (float) a_blk->qs[tile * 4 + p];
|
||||
|
||||
sumf[0] += ((bits_lo & (1u << p)) ? scale[0] : -scale[0]) * q;
|
||||
sumf[1] += ((bits_lo & (1u << (4 + p))) ? scale[1] : -scale[1]) * q;
|
||||
sumf[2] += ((bits_hi & (1u << p)) ? scale[2] : -scale[2]) * q;
|
||||
sumf[3] += ((bits_hi & (1u << (4 + p))) ? scale[3] : -scale[3]) * q;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
s[x * ncols_interleaved + j] = sumf[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ggml_gemv_q1_0_4x8_q8_0_generic(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
const int blocklen = 8;
|
||||
|
||||
assert(nr == 1);
|
||||
assert(n % qk == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
UNUSED(bs);
|
||||
UNUSED(nr);
|
||||
|
||||
float sumf[4];
|
||||
|
||||
const block_q8_0 * a_ptr = (const block_q8_0 *) vy;
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
sumf[j] = 0.0f;
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float d0[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]),
|
||||
};
|
||||
|
||||
for (int k = 0; k < qk / blocklen; ++k) {
|
||||
const block_q8_0 * GGML_RESTRICT a_blk = a_ptr + l * (qk / QK8_0) + k / (QK8_0 / blocklen);
|
||||
const float d1 = GGML_CPU_FP16_TO_FP32(a_blk->d);
|
||||
const float scale[4] = { d0[0] * d1, d0[1] * d1, d0[2] * d1, d0[3] * d1 };
|
||||
const uint8_t bits0 = b_ptr[l].qs[k * ncols_interleaved + 0];
|
||||
const uint8_t bits1 = b_ptr[l].qs[k * ncols_interleaved + 1];
|
||||
const uint8_t bits2 = b_ptr[l].qs[k * ncols_interleaved + 2];
|
||||
const uint8_t bits3 = b_ptr[l].qs[k * ncols_interleaved + 3];
|
||||
const int q_offset = (k % (QK8_0 / blocklen)) * blocklen;
|
||||
|
||||
for (int p = 0; p < blocklen; ++p) {
|
||||
const float q = (float) a_blk->qs[q_offset + p];
|
||||
|
||||
sumf[0] += ((bits0 & (1u << p)) ? scale[0] : -scale[0]) * q;
|
||||
sumf[1] += ((bits1 & (1u << p)) ? scale[1] : -scale[1]) * q;
|
||||
sumf[2] += ((bits2 & (1u << p)) ? scale[2] : -scale[2]) * q;
|
||||
sumf[3] += ((bits3 & (1u << p)) ? scale[3] : -scale[3]) * q;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
s[x * ncols_interleaved + j] = sumf[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Only enable these for RISC-V.
|
||||
#if defined __riscv_zvfh
|
||||
void ggml_gemv_q4_0_16x1_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) {
|
||||
@@ -2383,6 +2510,176 @@ void ggml_gemm_q8_0_4x8_q8_0_generic(int n,
|
||||
}
|
||||
}
|
||||
|
||||
void ggml_gemm_q1_0_4x4_q8_0_generic(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nr % 4 == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
float sumf[4][4];
|
||||
|
||||
for (int y = 0; y < nr / 4; y++) {
|
||||
const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb);
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
for (int m = 0; m < 4; m++) {
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
sumf[m][j] = 0.0;
|
||||
}
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float d0[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]),
|
||||
};
|
||||
|
||||
for (int k = 0; k < QK1_0 / QK8_0; ++k) {
|
||||
const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k;
|
||||
const float a_d[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[3]),
|
||||
};
|
||||
|
||||
for (int tile = 0; tile < QK8_0 / 4; ++tile) {
|
||||
const uint8_t bits_lo = b_ptr[l].qs[k * 16 + 2 * tile + 0];
|
||||
const uint8_t bits_hi = b_ptr[l].qs[k * 16 + 2 * tile + 1];
|
||||
const int tile_offset = tile * 16;
|
||||
|
||||
for (int p = 0; p < 4; ++p) {
|
||||
const int8_t q_row[4] = {
|
||||
a_blk->qs[tile_offset + 0 * 4 + p],
|
||||
a_blk->qs[tile_offset + 1 * 4 + p],
|
||||
a_blk->qs[tile_offset + 2 * 4 + p],
|
||||
a_blk->qs[tile_offset + 3 * 4 + p],
|
||||
};
|
||||
const int sign[4] = {
|
||||
(bits_lo & (1u << p)) ? 1 : -1,
|
||||
(bits_lo & (1u << (4 + p))) ? 1 : -1,
|
||||
(bits_hi & (1u << p)) ? 1 : -1,
|
||||
(bits_hi & (1u << (4 + p))) ? 1 : -1,
|
||||
};
|
||||
|
||||
for (int m = 0; m < 4; ++m) {
|
||||
const float row_scale = a_d[m];
|
||||
sumf[m][0] += sign[0] * q_row[m] * d0[0] * row_scale;
|
||||
sumf[m][1] += sign[1] * q_row[m] * d0[1] * row_scale;
|
||||
sumf[m][2] += sign[2] * q_row[m] * d0[2] * row_scale;
|
||||
sumf[m][3] += sign[3] * q_row[m] * d0[3] * row_scale;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int m = 0; m < 4; m++) {
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
s[(y * 4 + m) * bs + x * ncols_interleaved + j] = sumf[m][j];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ggml_gemm_q1_0_4x8_q8_0_generic(int n,
|
||||
float * GGML_RESTRICT s,
|
||||
size_t bs,
|
||||
const void * GGML_RESTRICT vx,
|
||||
const void * GGML_RESTRICT vy,
|
||||
int nr,
|
||||
int nc) {
|
||||
const int qk = QK1_0;
|
||||
const int nb = n / qk;
|
||||
const int ncols_interleaved = 4;
|
||||
const int blocklen = 8;
|
||||
|
||||
assert(n % qk == 0);
|
||||
assert(nr % 4 == 0);
|
||||
assert(nc % ncols_interleaved == 0);
|
||||
|
||||
float sumf[4][4];
|
||||
|
||||
for (int y = 0; y < nr / 4; y++) {
|
||||
const block_q8_0x4 * a_ptr = (const block_q8_0x4 *) vy + (4 * y * nb);
|
||||
for (int x = 0; x < nc / ncols_interleaved; x++) {
|
||||
const block_q1_0x4 * b_ptr = (const block_q1_0x4 *) vx + (x * nb);
|
||||
|
||||
for (int m = 0; m < 4; m++) {
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
sumf[m][j] = 0.0f;
|
||||
}
|
||||
}
|
||||
|
||||
for (int l = 0; l < nb; l++) {
|
||||
const float d0[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(b_ptr[l].d[3]),
|
||||
};
|
||||
|
||||
for (int k = 0; k < qk / blocklen; ++k) {
|
||||
const block_q8_0x4 * GGML_RESTRICT a_blk = a_ptr + 4 * l + k / (QK8_0 / blocklen);
|
||||
const float a_d[4] = {
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[0]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[1]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[2]),
|
||||
GGML_CPU_FP16_TO_FP32(a_blk->d[3]),
|
||||
};
|
||||
const uint8_t bits0 = b_ptr[l].qs[k * ncols_interleaved + 0];
|
||||
const uint8_t bits1 = b_ptr[l].qs[k * ncols_interleaved + 1];
|
||||
const uint8_t bits2 = b_ptr[l].qs[k * ncols_interleaved + 2];
|
||||
const uint8_t bits3 = b_ptr[l].qs[k * ncols_interleaved + 3];
|
||||
const int q_offset = (k % (QK8_0 / blocklen)) * 4 * blocklen;
|
||||
|
||||
for (int p = 0; p < blocklen; ++p) {
|
||||
const int8_t q_row[4] = {
|
||||
a_blk->qs[q_offset + 0 * blocklen + p],
|
||||
a_blk->qs[q_offset + 1 * blocklen + p],
|
||||
a_blk->qs[q_offset + 2 * blocklen + p],
|
||||
a_blk->qs[q_offset + 3 * blocklen + p],
|
||||
};
|
||||
const int sign[4] = {
|
||||
(bits0 & (1u << p)) ? 1 : -1,
|
||||
(bits1 & (1u << p)) ? 1 : -1,
|
||||
(bits2 & (1u << p)) ? 1 : -1,
|
||||
(bits3 & (1u << p)) ? 1 : -1,
|
||||
};
|
||||
|
||||
for (int m = 0; m < 4; ++m) {
|
||||
const float row_scale = a_d[m];
|
||||
sumf[m][0] += sign[0] * q_row[m] * d0[0] * row_scale;
|
||||
sumf[m][1] += sign[1] * q_row[m] * d0[1] * row_scale;
|
||||
sumf[m][2] += sign[2] * q_row[m] * d0[2] * row_scale;
|
||||
sumf[m][3] += sign[3] * q_row[m] * d0[3] * row_scale;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (int m = 0; m < 4; m++) {
|
||||
for (int j = 0; j < ncols_interleaved; j++) {
|
||||
s[(y * 4 + m) * bs + x * ncols_interleaved + j] = sumf[m][j];
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Only enable these for RISC-V.
|
||||
#if defined __riscv_zvfh
|
||||
void ggml_gemm_q4_0_16x1_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc) {
|
||||
@@ -2739,6 +3036,50 @@ static block_q8_0x4 make_block_q8_0x4(block_q8_0 * in, unsigned int blck_size_in
|
||||
return out;
|
||||
}
|
||||
|
||||
static block_q1_0x4 make_block_q1_0x4(block_q1_0 * in, unsigned int blck_size_interleave) {
|
||||
block_q1_0x4 out;
|
||||
|
||||
for (int i = 0; i < 4; i++) {
|
||||
out.d[i] = in[i].d;
|
||||
}
|
||||
|
||||
GGML_ASSERT(blck_size_interleave == 4 || blck_size_interleave == 8);
|
||||
|
||||
if (blck_size_interleave == 4) {
|
||||
for (int k = 0; k < QK1_0 / QK8_0; ++k) {
|
||||
for (int tile = 0; tile < QK8_0 / 4; ++tile) {
|
||||
uint8_t packed_lo = 0;
|
||||
uint8_t packed_hi = 0;
|
||||
|
||||
const int weight_base = k * QK8_0 + tile * 4;
|
||||
for (int pos = 0; pos < 4; ++pos) {
|
||||
const int weight_idx = weight_base + pos;
|
||||
const int byte_idx = weight_idx / 8;
|
||||
const int bit_idx = weight_idx % 8;
|
||||
|
||||
packed_lo |= ((in[0].qs[byte_idx] >> bit_idx) & 1u) << pos;
|
||||
packed_lo |= ((in[1].qs[byte_idx] >> bit_idx) & 1u) << (4 + pos);
|
||||
packed_hi |= ((in[2].qs[byte_idx] >> bit_idx) & 1u) << pos;
|
||||
packed_hi |= ((in[3].qs[byte_idx] >> bit_idx) & 1u) << (4 + pos);
|
||||
}
|
||||
|
||||
out.qs[k * 16 + 2 * tile + 0] = packed_lo;
|
||||
out.qs[k * 16 + 2 * tile + 1] = packed_hi;
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
for (int byte_idx = 0; byte_idx < QK1_0 / 8; ++byte_idx) {
|
||||
out.qs[byte_idx * 4 + 0] = in[0].qs[byte_idx];
|
||||
out.qs[byte_idx * 4 + 1] = in[1].qs[byte_idx];
|
||||
out.qs[byte_idx * 4 + 2] = in[2].qs[byte_idx];
|
||||
out.qs[byte_idx * 4 + 3] = in[3].qs[byte_idx];
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
static block_q4_0x4 make_block_q4_0x4(block_q4_0 * in, int blck_size_interleave) {
|
||||
block_q4_0x4 out;
|
||||
|
||||
@@ -3509,6 +3850,38 @@ static int repack_q8_0_to_q8_0_4_bl(struct ggml_tensor * t,
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int repack_q1_0_to_q1_0_4_bl(struct ggml_tensor * t,
|
||||
int interleave_block,
|
||||
const void * GGML_RESTRICT data,
|
||||
size_t data_size) {
|
||||
GGML_ASSERT(t->type == GGML_TYPE_Q1_0);
|
||||
GGML_ASSERT(interleave_block == 4 || interleave_block == 8);
|
||||
constexpr int nrows_interleaved = 4;
|
||||
|
||||
block_q1_0x4 * dst = (block_q1_0x4 *) t->data;
|
||||
const block_q1_0 * src = (const block_q1_0 *) data;
|
||||
block_q1_0 dst_tmp[4];
|
||||
int nrow = ggml_nrows(t);
|
||||
int nblocks = t->ne[0] / QK1_0;
|
||||
|
||||
GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q1_0));
|
||||
|
||||
if (t->ne[1] % nrows_interleaved != 0) {
|
||||
return -1;
|
||||
}
|
||||
|
||||
for (int b = 0; b < nrow; b += nrows_interleaved) {
|
||||
for (int64_t x = 0; x < nblocks; x++) {
|
||||
for (int i = 0; i < nrows_interleaved; i++) {
|
||||
dst_tmp[i] = src[x + i * nblocks];
|
||||
}
|
||||
*dst++ = make_block_q1_0x4(dst_tmp, interleave_block);
|
||||
}
|
||||
src += nrows_interleaved * nblocks;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static block_q8_0x16 make_block_q8_0x16(block_q8_0 * in, unsigned int blck_size_interleave) {
|
||||
block_q8_0x16 out;
|
||||
|
||||
@@ -3865,6 +4238,14 @@ template <typename BLOC_TYPE, int64_t INTER_SIZE, int64_t NB_COLS>
|
||||
int repack(struct ggml_tensor *, const void *, size_t);
|
||||
|
||||
// TODO: generalise.
|
||||
template <> int repack<block_q1_0, 4, 4>(struct ggml_tensor * t, const void * data, size_t data_size) {
|
||||
return repack_q1_0_to_q1_0_4_bl(t, 4, data, data_size);
|
||||
}
|
||||
|
||||
template <> int repack<block_q1_0, 8, 4>(struct ggml_tensor * t, const void * data, size_t data_size) {
|
||||
return repack_q1_0_to_q1_0_4_bl(t, 8, data, data_size);
|
||||
}
|
||||
|
||||
template <> int repack<block_q4_0, 4, 4>(struct ggml_tensor * t, const void * data, size_t data_size) {
|
||||
return repack_q4_0_to_q4_0_4_bl(t, 4, data, data_size);
|
||||
}
|
||||
@@ -3960,6 +4341,14 @@ template <> int repack<block_q2_K, 1, 16>(struct ggml_tensor * t, const void * d
|
||||
template <typename BLOC_TYPE, int64_t INTER_SIZE, int64_t NB_COLS, ggml_type PARAM_TYPE>
|
||||
void gemv(int, float *, size_t, const void *, const void *, int, int);
|
||||
|
||||
template <> void gemv<block_q1_0, 4, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemv_q1_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
template <> void gemv<block_q1_0, 8, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemv_q1_0_4x8_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
template <> void gemv<block_q4_0, 4, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemv_q4_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
@@ -4057,6 +4446,14 @@ template <> void gemv<block_q2_K, 1, 16, GGML_TYPE_Q8_K>(int n, float * s, size_
|
||||
template <typename BLOC_TYPE, int64_t INTER_SIZE, int64_t NB_COLS, ggml_type PARAM_TYPE>
|
||||
void gemm(int, float *, size_t, const void *, const void *, int, int);
|
||||
|
||||
template <> void gemm<block_q1_0, 4, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemm_q1_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
template <> void gemm<block_q1_0, 8, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemm_q1_0_4x8_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
|
||||
template <> void gemm<block_q4_0, 4, 4, GGML_TYPE_Q8_0>(int n, float * s, size_t bs, const void * vx, const void * vy, int nr, int nc) {
|
||||
ggml_gemm_q4_0_4x4_q8_0(n, s, bs, vx, vy, nr, nc);
|
||||
}
|
||||
@@ -4526,6 +4923,10 @@ template <typename BLOC_TYPE, int64_t INTER_SIZE, int64_t NB_COLS, ggml_type PAR
|
||||
} // namespace ggml::cpu::repack
|
||||
|
||||
static const ggml::cpu::tensor_traits * ggml_repack_get_optimal_repack_type(const struct ggml_tensor * cur) {
|
||||
// instance for Q1_0
|
||||
static const ggml::cpu::repack::tensor_traits<block_q1_0, 4, 4, GGML_TYPE_Q8_0> q1_0_4x4_q8_0;
|
||||
static const ggml::cpu::repack::tensor_traits<block_q1_0, 8, 4, GGML_TYPE_Q8_0> q1_0_4x8_q8_0;
|
||||
|
||||
// instance for Q4
|
||||
static const ggml::cpu::repack::tensor_traits<block_q4_0, 4, 4, GGML_TYPE_Q8_0> q4_0_4x4_q8_0;
|
||||
static const ggml::cpu::repack::tensor_traits<block_q4_0, 8, 4, GGML_TYPE_Q8_0> q4_0_4x8_q8_0;
|
||||
@@ -4723,6 +5124,17 @@ static const ggml::cpu::tensor_traits * ggml_repack_get_optimal_repack_type(cons
|
||||
}
|
||||
#endif
|
||||
}
|
||||
} else if (cur->type == GGML_TYPE_Q1_0) {
|
||||
if (ggml_cpu_has_neon() && ggml_cpu_has_matmul_int8()) {
|
||||
if (cur->ne[1] % 4 == 0) {
|
||||
return &q1_0_4x8_q8_0;
|
||||
}
|
||||
}
|
||||
if (ggml_cpu_has_neon() && ggml_cpu_has_dotprod()) {
|
||||
if (cur->ne[1] % 4 == 0) {
|
||||
return &q1_0_4x4_q8_0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return nullptr;
|
||||
|
||||
@@ -11,6 +11,9 @@
|
||||
ggml_backend_buffer_type_t ggml_backend_cpu_repack_buffer_type(void);
|
||||
|
||||
template <int K> constexpr int QK_0() {
|
||||
if constexpr (K == 1) {
|
||||
return QK1_0;
|
||||
}
|
||||
if constexpr (K == 4) {
|
||||
return QK4_0;
|
||||
}
|
||||
@@ -26,6 +29,7 @@ template <int K, int N> struct block {
|
||||
};
|
||||
|
||||
// control size
|
||||
static_assert(sizeof(block<1, 4>) == 4 * sizeof(ggml_half) + QK1_0 / 2, "wrong block<1,4> size/padding");
|
||||
static_assert(sizeof(block<4, 4>) == 4 * sizeof(ggml_half) + QK8_0 * 2, "wrong block<4,4> size/padding");
|
||||
static_assert(sizeof(block<4, 8>) == 8 * sizeof(ggml_half) + QK8_0 * 4, "wrong block<4,8> size/padding");
|
||||
static_assert(sizeof(block<4, 16>) == 16 * sizeof(ggml_half) + QK8_0 * 8, "wrong block<4,16> size/padding");
|
||||
@@ -33,6 +37,7 @@ static_assert(sizeof(block<8, 4>) == 4 * sizeof(ggml_half) + QK8_0 * 4, "wrong b
|
||||
static_assert(sizeof(block<8, 8>) == 8 * sizeof(ggml_half) + QK8_0 * 8, "wrong block<8,8> size/padding");
|
||||
static_assert(sizeof(block<8, 16>) == 16 * sizeof(ggml_half) + QK8_0 * 16, "wrong block<8,16> size/padding");
|
||||
|
||||
using block_q1_0x4 = block<1, 4>;
|
||||
using block_q4_0x4 = block<4, 4>;
|
||||
using block_q4_0x8 = block<4, 8>;
|
||||
using block_q4_0x16 = block<4, 16>;
|
||||
@@ -141,6 +146,8 @@ void ggml_quantize_mat_q8_0_4x4(const float * GGML_RESTRICT x, void * GGML_RESTR
|
||||
void ggml_quantize_mat_q8_0_4x8(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_quantize_mat_q8_K_4x4(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_quantize_mat_q8_K_4x8(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_gemv_q1_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q1_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
@@ -157,6 +164,8 @@ void ggml_gemv_mxfp4_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const v
|
||||
void ggml_gemv_mxfp4_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q8_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q8_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q1_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q1_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_4x4_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_4x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_8x8_q8_0(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
@@ -193,6 +202,8 @@ void ggml_quantize_mat_q8_0_4x4_generic(const float * GGML_RESTRICT x, void * GG
|
||||
void ggml_quantize_mat_q8_0_4x8_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_quantize_mat_q8_K_4x4_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_quantize_mat_q8_K_4x8_generic(const float * GGML_RESTRICT x, void * GGML_RESTRICT vy, int64_t k);
|
||||
void ggml_gemv_q1_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q1_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q4_0_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
@@ -209,6 +220,8 @@ void ggml_gemv_mxfp4_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs,
|
||||
void ggml_gemv_mxfp4_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q8_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemv_q8_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q1_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q1_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_4x4_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_4x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
void ggml_gemm_q4_0_8x8_q8_0_generic(int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT vx, const void * GGML_RESTRICT vy, int nr, int nc);
|
||||
|
||||
@@ -639,7 +639,7 @@ static void permute_transpose_impl(const ggml_tensor * src0,
|
||||
}
|
||||
} else if (n_src_stride == sizeof(int16_t)) {
|
||||
for (int64_t bi = ith; bi < batch; bi += nth) {
|
||||
rvv_transposed_s32_mn_to_nm((int8_t *) ((char *) dst->data + bi * batch_stride), n_dst_stride,
|
||||
rvv_transposed_s16_mn_to_nm((int8_t *) ((char *) dst->data + bi * batch_stride), n_dst_stride,
|
||||
(int8_t *) ((char *) src0->data + bi * batch_stride), m_src_stride, m, n);
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#include "allreduce.cuh"
|
||||
|
||||
#if !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA)
|
||||
#if !defined(GGML_USE_MUSA)
|
||||
|
||||
#include "convert.cuh"
|
||||
#include "ggml-impl.h"
|
||||
@@ -11,11 +11,12 @@
|
||||
#include <limits>
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// CUDA AllReduce for tensor-parallel inference across two GPUs.
|
||||
// AllReduce for tensor-parallel inference across two GPUs (CUDA or
|
||||
// ROCm/HIP).
|
||||
//
|
||||
// Provides an in-place sum reduction over matching tensors on two CUDA
|
||||
// devices in the same process. Used by the tensor-split path alongside
|
||||
// NCCL; targets setups without NVLink, where data is exchanged between the
|
||||
// Provides an in-place sum reduction over matching tensors on two GPUs
|
||||
// in the same process. Used by the tensor-split path alongside NCCL;
|
||||
// targets setups without NVLink/xGMI, where data is exchanged between the
|
||||
// GPUs by staging it through pinned host memory over PCIe.
|
||||
//
|
||||
// Two reduction strategies are selected per call by tensor size:
|
||||
@@ -161,11 +162,14 @@ static __global__ void ggml_cuda_ar_kernel(
|
||||
__threadfence_system(); // make our signal visible system-wide
|
||||
|
||||
while (ggml_cuda_ar_signal_get(other_slot) != token) {
|
||||
#if __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA
|
||||
#ifdef GGML_USE_HIP
|
||||
// Equals ~100ns at 2500 MHz (sleeps for n * [1,64] clock cycles)
|
||||
__builtin_amdgcn_s_sleep(4);
|
||||
#elif __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA
|
||||
__nanosleep(100);
|
||||
#else
|
||||
NO_DEVICE_CODE;
|
||||
#endif // __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA
|
||||
#endif // GGML_USE_HIP
|
||||
}
|
||||
}
|
||||
|
||||
@@ -280,7 +284,7 @@ struct ggml_cuda_ar_host_mapping {
|
||||
}
|
||||
rc = cudaHostGetDevicePointer(reinterpret_cast<void **>(&dev), host, 0);
|
||||
if (rc != cudaSuccess) {
|
||||
cudaFreeHost(host);
|
||||
CUDA_CHECK(cudaFreeHost(host));
|
||||
host = nullptr;
|
||||
dev = nullptr;
|
||||
}
|
||||
@@ -289,7 +293,7 @@ struct ggml_cuda_ar_host_mapping {
|
||||
|
||||
void free() {
|
||||
if (host) {
|
||||
cudaFreeHost(host);
|
||||
CUDA_CHECK(cudaFreeHost(host));
|
||||
host = nullptr;
|
||||
dev = nullptr;
|
||||
}
|
||||
@@ -401,7 +405,8 @@ ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init(const int * devices, size_t n
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
// The chunked kernel uses __nanosleep, which is sm70+ (Volta+).
|
||||
// The chunked kernel uses __nanosleep (NVIDIA, sm70+) or
|
||||
// __builtin_amdgcn_s_sleep (AMD).
|
||||
for (size_t i = 0; i < n_devices; ++i) {
|
||||
const int cc = ggml_cuda_info().devices[devices[i]].cc;
|
||||
if (cc < GGML_CUDA_CC_VOLTA) {
|
||||
@@ -543,7 +548,7 @@ void ggml_cuda_ar_pipeline_free(ggml_cuda_ar_pipeline * p) {
|
||||
for (int i = 0; i < p->n_devices; ++i) {
|
||||
if (p->streams[i]) {
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
cudaStreamSynchronize(p->streams[i]);
|
||||
CUDA_CHECK(cudaStreamSynchronize(p->streams[i]));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -552,28 +557,28 @@ void ggml_cuda_ar_pipeline_free(ggml_cuda_ar_pipeline * p) {
|
||||
p->host_large[i].free();
|
||||
if (p->dev_tmp[i]) {
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
cudaFree(p->dev_tmp[i]);
|
||||
CUDA_CHECK(cudaFree(p->dev_tmp[i]));
|
||||
}
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
for (int s = 0; s < GGML_CUDA_AR_POOL_SIZE; ++s) {
|
||||
if (p->ev_pool[i][s].app) { cudaEventDestroy(p->ev_pool[i][s].app); }
|
||||
if (p->ev_pool[i][s].app) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].app)); }
|
||||
for (int c = 0; c < GGML_CUDA_AR_COPY_MAX_CHUNKS; ++c) {
|
||||
if (p->ev_pool[i][s].cpy[c]) { cudaEventDestroy(p->ev_pool[i][s].cpy[c]); }
|
||||
if (p->ev_pool[i][s].cpy[c]) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].cpy[c])); }
|
||||
}
|
||||
if (p->ev_pool[i][s].h2d) { cudaEventDestroy(p->ev_pool[i][s].h2d); }
|
||||
if (p->ev_pool[i][s].ker) { cudaEventDestroy(p->ev_pool[i][s].ker); }
|
||||
if (p->ev_pool[i][s].h2d) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].h2d)); }
|
||||
if (p->ev_pool[i][s].ker) { CUDA_CHECK(cudaEventDestroy(p->ev_pool[i][s].ker)); }
|
||||
}
|
||||
if (p->host_large_read_done[i]) {
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
cudaEventDestroy(p->host_large_read_done[i]);
|
||||
CUDA_CHECK(cudaEventDestroy(p->host_large_read_done[i]));
|
||||
}
|
||||
if (p->dev_tmp_kernel_done[i]) {
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
cudaEventDestroy(p->dev_tmp_kernel_done[i]);
|
||||
CUDA_CHECK(cudaEventDestroy(p->dev_tmp_kernel_done[i]));
|
||||
}
|
||||
if (p->streams[i]) {
|
||||
ggml_cuda_set_device(p->devices[i]);
|
||||
cudaStreamDestroy(p->streams[i]);
|
||||
CUDA_CHECK(cudaStreamDestroy(p->streams[i]));
|
||||
}
|
||||
}
|
||||
p->arrival.free();
|
||||
@@ -952,13 +957,14 @@ bool ggml_cuda_ar_allreduce(
|
||||
return ok;
|
||||
}
|
||||
|
||||
#else // defined(GGML_USE_HIP) || defined(GGML_USE_MUSA)
|
||||
#else // defined(GGML_USE_MUSA)
|
||||
|
||||
// HIP and MUSA lack the host-mapped pinned-memory APIs (cudaHostAllocPortable
|
||||
// / cudaHostAllocMapped / cudaHostGetDevicePointer) and __nanosleep that this
|
||||
// implementation relies on, so the internal AllReduce is a CUDA-only feature.
|
||||
// The dispatcher in ggml-cuda.cu treats a nullptr pipeline as "init failed"
|
||||
// and silently falls back to the meta backend's generic AllReduce.
|
||||
// MUSA lacks the host-mapped pinned-memory APIs (cudaHostAllocPortable
|
||||
// / cudaHostAllocMapped / cudaHostGetDevicePointer) and a device-side
|
||||
// sleep intrinsic that this implementation relies on, so the internal
|
||||
// AllReduce is unavailable there. The dispatcher in ggml-cuda.cu treats
|
||||
// a nullptr pipeline as "init failed" and silently falls back to the meta
|
||||
// backend's generic AllReduce.
|
||||
ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init(const int *, size_t) {
|
||||
return nullptr;
|
||||
}
|
||||
@@ -968,4 +974,4 @@ bool ggml_cuda_ar_allreduce(ggml_cuda_ar_pipeline *, ggml_backend_t *, ggml_tens
|
||||
return false;
|
||||
}
|
||||
|
||||
#endif // !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA)
|
||||
#endif // !defined(GGML_USE_MUSA)
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
struct ggml_cuda_ar_pipeline;
|
||||
|
||||
// Allocate a pipeline for n_devices GPUs.
|
||||
// devices[] holds the CUDA device IDs in rank order.
|
||||
// devices[] holds the GPU device IDs in rank order.
|
||||
// Returns nullptr on allocation failure.
|
||||
ggml_cuda_ar_pipeline * ggml_cuda_ar_pipeline_init(
|
||||
const int * devices, size_t n_devices);
|
||||
|
||||
@@ -51,9 +51,12 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
|
||||
cudaStream_t stream) {
|
||||
ggml_cuda_pool_alloc<int> temp_indices_alloc(pool, ncols * nrows);
|
||||
ggml_cuda_pool_alloc<float> temp_keys_alloc(pool, ncols * nrows);
|
||||
// Device*Sort algorithms currently do not allow for in-place sorting/aliasing of input/outputs
|
||||
ggml_cuda_pool_alloc<float> temp_keys_out_alloc(pool, ncols * nrows);
|
||||
|
||||
int * temp_indices = temp_indices_alloc.get();
|
||||
float * temp_keys = temp_keys_alloc.get();
|
||||
float * temp_keys_out = temp_keys_out_alloc.get();
|
||||
|
||||
static const int block_size = 256;
|
||||
const dim3 grid_size((ncols + block_size - 1) / block_size, nrows);
|
||||
@@ -85,18 +88,18 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
|
||||
|
||||
if (order == GGML_SORT_ORDER_ASC) {
|
||||
if (nrows == 1) {
|
||||
CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys, // keys (in-place)
|
||||
CUDA_CHECK(DeviceRadixSort::SortPairs(nullptr, temp_storage_bytes, temp_keys, temp_keys_out, // keys in, keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols, 0, sizeof(float) * 8, stream));
|
||||
} else if (is_capturing) {
|
||||
CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(
|
||||
nullptr, temp_storage_bytes, temp_keys, temp_keys, // keys (in-place)
|
||||
nullptr, temp_storage_bytes, temp_keys, temp_keys_out, // keys in, keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols * nrows, nrows, // num items, num segments
|
||||
offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
|
||||
} else {
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairs(nullptr, temp_storage_bytes, temp_keys,
|
||||
temp_keys, // keys (in-place)
|
||||
temp_keys_out, // keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols * nrows, nrows, // num items, num segments
|
||||
offset_iterator, offset_iterator + 1, stream));
|
||||
@@ -104,15 +107,15 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
|
||||
} else {
|
||||
if (nrows == 1) {
|
||||
CUDA_CHECK(DeviceRadixSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys,
|
||||
temp_keys, // keys (in-place)
|
||||
temp_keys_out, // keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols, 0, sizeof(float) * 8, stream));
|
||||
} else if (is_capturing) {
|
||||
CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending(
|
||||
nullptr, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows,
|
||||
nullptr, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
|
||||
offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
|
||||
} else {
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys,
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(nullptr, temp_storage_bytes, temp_keys, temp_keys_out,
|
||||
temp_indices, dst, ncols * nrows, nrows,
|
||||
offset_iterator, offset_iterator + 1, stream));
|
||||
}
|
||||
@@ -124,31 +127,31 @@ void argsort_f32_i32_cuda_cub(ggml_cuda_pool & pool,
|
||||
if (order == GGML_SORT_ORDER_ASC) {
|
||||
if (nrows == 1) {
|
||||
CUDA_CHECK(DeviceRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys,
|
||||
temp_keys, // keys (in-place)
|
||||
temp_keys_out, // keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols, 0, sizeof(float) * 8, stream));
|
||||
} else if (is_capturing) {
|
||||
CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys,
|
||||
CUDA_CHECK(DeviceSegmentedRadixSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out,
|
||||
temp_indices, dst, ncols * nrows, nrows, offset_iterator,
|
||||
offset_iterator + 1, 0, sizeof(float) * 8, stream));
|
||||
} else {
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys,
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairs(d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out,
|
||||
temp_indices, dst, ncols * nrows, nrows, offset_iterator,
|
||||
offset_iterator + 1, stream));
|
||||
}
|
||||
} else {
|
||||
if (nrows == 1) {
|
||||
CUDA_CHECK(DeviceRadixSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys,
|
||||
temp_keys, // keys (in-place)
|
||||
temp_keys_out, // keys out
|
||||
temp_indices, dst, // values (indices)
|
||||
ncols, 0, sizeof(float) * 8, stream));
|
||||
} else if (is_capturing) {
|
||||
CUDA_CHECK(DeviceSegmentedRadixSort::SortPairsDescending(
|
||||
d_temp_storage, temp_storage_bytes, temp_keys, temp_keys, temp_indices, dst, ncols * nrows, nrows,
|
||||
d_temp_storage, temp_storage_bytes, temp_keys, temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
|
||||
offset_iterator, offset_iterator + 1, 0, sizeof(float) * 8, stream));
|
||||
} else {
|
||||
CUDA_CHECK(DeviceSegmentedSort::SortPairsDescending(d_temp_storage, temp_storage_bytes, temp_keys,
|
||||
temp_keys, temp_indices, dst, ncols * nrows, nrows,
|
||||
temp_keys_out, temp_indices, dst, ncols * nrows, nrows,
|
||||
offset_iterator, offset_iterator + 1, stream));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
#include "conv2d-dw.cuh"
|
||||
#include "convert.cuh"
|
||||
|
||||
struct conv_params {
|
||||
int in_w, in_h;
|
||||
@@ -79,7 +80,7 @@ struct cwhn_layout {
|
||||
};
|
||||
|
||||
template <typename T, typename Layout>
|
||||
__global__ void conv2d_dw_kernel(const T * __restrict__ input, const T * __restrict__ kernel, T * __restrict__ output,
|
||||
__global__ void conv2d_dw_kernel(const float * __restrict__ input, const T * __restrict__ kernel, float * __restrict__ output,
|
||||
const int in_w, const int in_h, const int out_w, const int out_h,
|
||||
const int kernel_w, const int kernel_h, const int stride_x, const int stride_y,
|
||||
const int padding_x, const int padding_y, const int dilation_x, const int dilation_y,
|
||||
@@ -97,7 +98,7 @@ __global__ void conv2d_dw_kernel(const T * __restrict__ input, const T * __restr
|
||||
int batch_idx, channel_idx, out_y_idx, out_x_idx;
|
||||
Layout::unpack_indices(global_idx, params, batch_idx, channel_idx, out_y_idx, out_x_idx);
|
||||
|
||||
T accumulator = 0;
|
||||
float accumulator = 0.0f;
|
||||
kernel_bounds bounds = calculate_kernel_bounds(out_x_idx, out_y_idx, params);
|
||||
|
||||
for (int kern_y = bounds.y_min; kern_y < bounds.y_max; ++kern_y) {
|
||||
@@ -106,10 +107,10 @@ __global__ void conv2d_dw_kernel(const T * __restrict__ input, const T * __restr
|
||||
for (int kern_x = bounds.x_min; kern_x < bounds.x_max; ++kern_x) {
|
||||
int in_x_idx = calculate_input_coord(out_x_idx, kern_x, params.stride_x, params.dilation_x, params.padding_x);
|
||||
|
||||
const T input_val = input[Layout::input_index(batch_idx, channel_idx, in_y_idx, in_x_idx, params)];
|
||||
const T kernel_val = kernel[Layout::kernel_index(channel_idx, kern_y, kern_x, params)];
|
||||
const float input_val = input[Layout::input_index(batch_idx, channel_idx, in_y_idx, in_x_idx, params)];
|
||||
const T kernel_val = kernel[Layout::kernel_index(channel_idx, kern_y, kern_x, params)];
|
||||
|
||||
accumulator += input_val * kernel_val;
|
||||
accumulator += input_val * ggml_cuda_cast<float>(kernel_val);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -120,8 +121,9 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst)
|
||||
const ggml_tensor * kernel = dst->src[0];
|
||||
const ggml_tensor * input = dst->src[1];
|
||||
|
||||
GGML_ASSERT(kernel->type == GGML_TYPE_F32 && input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32);
|
||||
const float * w_d = (const float *) kernel->data;
|
||||
GGML_ASSERT(kernel->type == GGML_TYPE_F16 || kernel->type == GGML_TYPE_F32);
|
||||
GGML_ASSERT(input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32);
|
||||
const void * w_d = kernel->data;
|
||||
const float * x_d = (const float *) input->data;
|
||||
float * y_d = (float *) dst->data;
|
||||
|
||||
@@ -148,13 +150,25 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst)
|
||||
const int blocks = (total + CUDA_CONV2D_DW_BLOCK_SIZE - 1) / CUDA_CONV2D_DW_BLOCK_SIZE;
|
||||
|
||||
if (ggml_is_contiguous(input)) {
|
||||
conv2d_dw_kernel<float, whcn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, padding_x, padding_y,
|
||||
dilation_x, dilation_y, channels, batches);
|
||||
if (kernel->type == GGML_TYPE_F16) {
|
||||
conv2d_dw_kernel<half, whcn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
|
||||
padding_x, padding_y, dilation_x, dilation_y, channels, batches);
|
||||
} else {
|
||||
conv2d_dw_kernel<float, whcn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
|
||||
padding_x, padding_y, dilation_x, dilation_y, channels, batches);
|
||||
}
|
||||
} else if (ggml_is_contiguous_channels(input)) {
|
||||
conv2d_dw_kernel<float, cwhn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, padding_x, padding_y,
|
||||
dilation_x, dilation_y, channels, batches);
|
||||
if (kernel->type == GGML_TYPE_F16) {
|
||||
conv2d_dw_kernel<half, cwhn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
|
||||
padding_x, padding_y, dilation_x, dilation_y, channels, batches);
|
||||
} else {
|
||||
conv2d_dw_kernel<float, cwhn_layout><<<blocks, CUDA_CONV2D_DW_BLOCK_SIZE, 0, st>>>(
|
||||
x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
|
||||
padding_x, padding_y, dilation_x, dilation_y, channels, batches);
|
||||
}
|
||||
} else {
|
||||
GGML_ABORT("Unsupported memory layout for conv_2d_dw");
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user