mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-10-10 23:07:21 -05:00
Compare commits
276
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f0440d9efc | ||
|
|
79e2e74eb1 | ||
|
|
8e2d31e0eb | ||
|
|
baef3ed9a1 | ||
|
|
f39148a953 | ||
|
|
50e3e3e480 | ||
|
|
64df9183f5 | ||
|
|
6184e92c57 | ||
|
|
8b54361025 | ||
|
|
a518119d30 | ||
|
|
8ae386707b | ||
|
|
e60eff95fd | ||
|
|
ba6439a6b5 | ||
|
|
5e4878e978 | ||
|
|
609290be6b | ||
|
|
86a2835320 | ||
|
|
e94acad853 | ||
|
|
5e1d74043e | ||
|
|
b42b7e6d30 | ||
|
|
b013e56a71 | ||
|
|
2c564f43df | ||
|
|
1f8fa52318 | ||
|
|
8a1a9b5126 | ||
|
|
d4d82d67f4 | ||
|
|
3d65c90d04 | ||
|
|
de7fa0a3c6 | ||
|
|
71ad0590f4 | ||
|
|
a11f57ba93 | ||
|
|
fc9ce6b9d5 | ||
|
|
c35b66744f | ||
|
|
1167d3f42c | ||
|
|
4f92965a7b | ||
|
|
c811cb8f0a | ||
|
|
033df86b69 | ||
|
|
ff5888f999 | ||
|
|
dac3087394 | ||
|
|
03aa006acb | ||
|
|
08246a28f6 | ||
|
|
46baf1f1fe | ||
|
|
097f5b5332 | ||
|
|
d888016041 | ||
|
|
fda1866613 | ||
|
|
ff30363a0e | ||
|
|
000bee54a5 | ||
|
|
37ac634566 | ||
|
|
24e41838e0 | ||
|
|
847f447c31 | ||
|
|
75118a3a59 | ||
|
|
9b4ed0ca57 | ||
|
|
9c2e0e491a | ||
|
|
06cad0b9e7 | ||
|
|
a657f7e981 | ||
|
|
aa5e0092fd | ||
|
|
70815103c8 | ||
|
|
bd4eeaa047 | ||
|
|
5de733437b | ||
|
|
88dcc460d6 | ||
|
|
b86d2f0754 | ||
|
|
50a6c5cf7c | ||
|
|
d6cf9acb25 | ||
|
|
42c787e8c1 | ||
|
|
18b5f8b186 | ||
|
|
448147d42a | ||
|
|
988190680d | ||
|
|
7e8324f5fe | ||
|
|
b9acf138a1 | ||
|
|
48499d2e1c | ||
|
|
d0b490f25e | ||
|
|
42b021b4dc | ||
|
|
7481354a17 | ||
|
|
ad21565331 | ||
|
|
b7dafa01e5 | ||
|
|
26908739bc | ||
|
|
36a73916ee | ||
|
|
fa3c2fab36 | ||
|
|
005a1e127a | ||
|
|
d2a79e6046 | ||
|
|
5e5b628eb5 | ||
|
|
4d756bc72b | ||
|
|
78651c410d | ||
|
|
f498f864fb | ||
|
|
c479922ac5 | ||
|
|
5ad1c5da0a | ||
|
|
51ce9c11a6 | ||
|
|
abeada335e | ||
|
|
4625240437 | ||
|
|
3109914090 | ||
|
|
4fbc76dec5 | ||
|
|
2207c8e57c | ||
|
|
a46709b683 | ||
|
|
65840ed53c | ||
|
|
ab09ea4c14 | ||
|
|
da263e7275 | ||
|
|
a043d38a62 | ||
|
|
58cb9138e4 | ||
|
|
4f54067615 | ||
|
|
f0c41e0168 | ||
|
|
d7a695ef67 | ||
|
|
6c73b3e12d | ||
|
|
1a3011cc0c | ||
|
|
6753a033f0 | ||
|
|
cbb7d52ecb | ||
|
|
63bef2728d | ||
|
|
b9a5a00b86 | ||
|
|
43fe9c6428 | ||
|
|
5e03bdd870 | ||
|
|
50569eb87d | ||
|
|
7049ff0cbe | ||
|
|
c250304960 | ||
|
|
8345f33395 | ||
|
|
d812350493 | ||
|
|
4d60b4d087 | ||
|
|
c06f84160a | ||
|
|
f05c8b2780 | ||
|
|
e117148a41 | ||
|
|
6c59c40076 | ||
|
|
3c9e747f7e | ||
|
|
b809b886d9 | ||
|
|
994e8f2222 | ||
|
|
9d853bb36a | ||
|
|
8f9ae20c86 | ||
|
|
9871df5911 | ||
|
|
8b2fbaf32c | ||
|
|
2ed93db472 | ||
|
|
8e1642198d | ||
|
|
806eee9841 | ||
|
|
b3daa077a5 | ||
|
|
c173a53bdf | ||
|
|
210791069b | ||
|
|
e5983d6704 | ||
|
|
4ca6b76f0b | ||
|
|
9f12cd4a4c | ||
|
|
ebe18bee5a | ||
|
|
8216c84623 | ||
|
|
1b43d31169 | ||
|
|
a3a1c4747f | ||
|
|
9d3aba6b5e | ||
|
|
d89651a7b2 | ||
|
|
a7fb71fab8 | ||
|
|
0bb496dbd3 | ||
|
|
2ca15f5404 | ||
|
|
a7b94df2c6 | ||
|
|
0eb6d9a813 | ||
|
|
2e7c58c547 | ||
|
|
7f2dd88b0a | ||
|
|
bf79dbbcd0 | ||
|
|
dbe4c3ed42 | ||
|
|
46847e6158 | ||
|
|
2bc5635734 | ||
|
|
dd266785c2 | ||
|
|
16c163d561 | ||
|
|
0504396140 | ||
|
|
8330e96967 | ||
|
|
6716df694b | ||
|
|
bf9a0ccce7 | ||
|
|
0faee50042 | ||
|
|
f98b31c67e | ||
|
|
11fe02151f | ||
|
|
836d57176d | ||
|
|
eec18f5d32 | ||
|
|
1537a0a8b2 | ||
|
|
edd6e2bbda | ||
|
|
9bf55f4a36 | ||
|
|
a55e952b85 | ||
|
|
436f6f89e1 | ||
|
|
b92761a515 | ||
|
|
cb7934c52c | ||
|
|
889edf43dd | ||
|
|
99b95488ca | ||
|
|
bed0a85660 | ||
|
|
4ebdf2c74a | ||
|
|
1fb7ef3e33 | ||
|
|
134b2bb756 | ||
|
|
2923cf2862 | ||
|
|
dd4c286f38 | ||
|
|
46ca246de9 | ||
|
|
d8fbd2583a | ||
|
|
926862e574 | ||
|
|
a4cb4c61fd | ||
|
|
70849ee82c | ||
|
|
8d81559fa7 | ||
|
|
6805ae35df | ||
|
|
a8c9a4e7cc | ||
|
|
392ded6546 | ||
|
|
9e258a6e0a | ||
|
|
b933289545 | ||
|
|
c328acc91d | ||
|
|
4e2713c162 | ||
|
|
631109b34d | ||
|
|
254b177307 | ||
|
|
fb4b2737a8 | ||
|
|
207bdab950 | ||
|
|
5fc4f3c8c7 | ||
|
|
159c651f57 | ||
|
|
a868c3e3c5 | ||
|
|
ec7630a640 | ||
|
|
78e2964c23 | ||
|
|
f1cee9941b | ||
|
|
68e79bd8cd | ||
|
|
e358d59178 | ||
|
|
81e39ad343 | ||
|
|
dcd387a412 | ||
|
|
d775ebf363 | ||
|
|
2b36825cbc | ||
|
|
42d958167a | ||
|
|
13b4d7135a | ||
|
|
4b1622afb7 | ||
|
|
869034b4bb | ||
|
|
b56f34ab13 | ||
|
|
c061df1983 | ||
|
|
66e0c17ee1 | ||
|
|
7677678503 | ||
|
|
552f18f912 | ||
|
|
5503b04b05 | ||
|
|
def4d406ae | ||
|
|
32dd62ee6d | ||
|
|
f11d642a27 | ||
|
|
3aa0ce9bca | ||
|
|
b0aca3c653 | ||
|
|
b8f96c3e82 | ||
|
|
3ec4df42d9 | ||
|
|
db33d3cb89 | ||
|
|
7dad6db858 | ||
|
|
2232bc8b5f | ||
|
|
79625e056e | ||
|
|
66bcc27706 | ||
|
|
10f340d1a2 | ||
|
|
0c1e57098b | ||
|
|
f7b384c1e5 | ||
|
|
f872b59112 | ||
|
|
a4d880fd5c | ||
|
|
feb9a3d6de | ||
|
|
4453b535fd | ||
|
|
4f31296a90 | ||
|
|
b016f461be | ||
|
|
81ff93ea1d | ||
|
|
60e9cf7a7b | ||
|
|
05af0d2b13 | ||
|
|
2149c00f44 | ||
|
|
876c75b1f6 | ||
|
|
b04642061d | ||
|
|
22bdcc4cdd | ||
|
|
ca2e2037b6 | ||
|
|
bdeb855b30 | ||
|
|
3b3d022b82 | ||
|
|
185103dcf5 | ||
|
|
2090f60f0b | ||
|
|
90c908d06d | ||
|
|
8df332de1b | ||
|
|
4a096b8ff6 | ||
|
|
8664eaea30 | ||
|
|
4cfb6d1c75 | ||
|
|
9b43336114 | ||
|
|
f653250407 | ||
|
|
fa2bde5543 | ||
|
|
25747b08e7 | ||
|
|
db00347a4b | ||
|
|
272aad8b98 | ||
|
|
72db1e02ff | ||
|
|
2a53ace3be | ||
|
|
649dcb1036 | ||
|
|
931351ea50 | ||
|
|
eae11d2217 | ||
|
|
19e28a2770 | ||
|
|
a6ea155d3d | ||
|
|
d3954b9324 | ||
|
|
48de2a1bcb | ||
|
|
7fee178464 | ||
|
|
6a2743f028 | ||
|
|
748d4225b9 | ||
|
|
cee37ffea0 | ||
|
|
6dbbac4429 | ||
|
|
5c200e0c8d | ||
|
|
83dd71f869 | ||
|
|
94a0ae3e72 | ||
|
|
da89bb3ccc |
@@ -155,6 +155,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -114,6 +114,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -123,6 +123,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -151,6 +151,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/lib/ /app
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
@@ -130,6 +130,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.4
|
||||
ARG OPENVINO_VERSION_FULL=2026.4.0.22959.99c81491cc3
|
||||
ARG OPENVINO_VERSION_MAJOR=2026.4.1
|
||||
ARG OPENVINO_VERSION_FULL=2026.4.1.22982.07f9c262b05
|
||||
ARG UBUNTU_VERSION=24.04
|
||||
|
||||
# Intel GPU driver versions. https://github.com/intel/compute-runtime/releases
|
||||
ARG IGC_VERSION=v2.40.13
|
||||
ARG IGC_VERSION_FULL=2_2.40.13+22418
|
||||
ARG COMPUTE_RUNTIME_VERSION=26.31.39395.13
|
||||
ARG COMPUTE_RUNTIME_VERSION_FULL=26.31.39395.13-0
|
||||
ARG IGC_VERSION=v2.41.5
|
||||
ARG IGC_VERSION_FULL=2_2.41.5+22716
|
||||
ARG COMPUTE_RUNTIME_VERSION=26.35.39758.10
|
||||
ARG COMPUTE_RUNTIME_VERSION_FULL=26.35.39758.10-0
|
||||
ARG IGDGMM_VERSION=22.10.0
|
||||
|
||||
# Intel NPU driver versions. https://github.com/intel/linux-npu-driver/releases
|
||||
@@ -227,6 +227,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app/
|
||||
|
||||
|
||||
@@ -136,6 +136,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -133,6 +133,7 @@ ENTRYPOINT [ "/llama.cpp/bin/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
WORKDIR /llama.cpp/bin
|
||||
|
||||
|
||||
@@ -117,6 +117,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -107,6 +107,7 @@ ENTRYPOINT [ "/app/llama-cli" ]
|
||||
FROM base AS server
|
||||
|
||||
ENV LLAMA_ARG_HOST=0.0.0.0
|
||||
ENV LLAMA_ARG_PORT=8080
|
||||
|
||||
COPY --from=build /app/full/llama /app/full/llama-server /app
|
||||
|
||||
|
||||
@@ -45,6 +45,9 @@ insert_final_newline = unset
|
||||
trim_trailing_whitespace = unset
|
||||
insert_final_newline = unset
|
||||
|
||||
[vendor/**.patch]
|
||||
trim_trailing_whitespace = unset
|
||||
|
||||
[tools/ui/**]
|
||||
indent_style = unset
|
||||
indent_size = unset
|
||||
|
||||
@@ -4,6 +4,10 @@ on:
|
||||
issues:
|
||||
types: [opened]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
find-related:
|
||||
if: github.event.action == 'opened'
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -23,6 +23,10 @@ on:
|
||||
- 'scripts/snapdragon/**'
|
||||
- 'CMakePresets.json'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -36,6 +40,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
@@ -66,6 +74,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
@@ -98,6 +110,10 @@ jobs:
|
||||
matrix:
|
||||
device: [SM8750, SM8850, QCS9075M]
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
@@ -20,6 +20,10 @@ on:
|
||||
- '.github/workflows/build-android.yml'
|
||||
- 'examples/llama.android/**'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -66,6 +70,10 @@ jobs:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
uses: actions/checkout@v6
|
||||
|
||||
@@ -26,6 +26,10 @@ on:
|
||||
'ggml/src/ggml-rpc/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -50,7 +54,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: apple-arm64
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -99,8 +103,9 @@ jobs:
|
||||
id: cmake_test
|
||||
run: |
|
||||
cd build
|
||||
# Metal Paravirtual devices are difficult to support -> disable
|
||||
# ref: https://github.com/ggml-org/llama.cpp/pull/19802#issuecomment-4013704023
|
||||
ctest -L main -E "test-llama-archs|test-save-load-state" --verbose --timeout 900
|
||||
ctest -L main -E "test-llama-archs|test-save-load-state|test-recurrent-state-rollback" --verbose --timeout 900
|
||||
|
||||
macos-latest-x64:
|
||||
runs-on: macos-15-intel
|
||||
@@ -113,7 +118,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: apple-x64
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -161,6 +166,10 @@ jobs:
|
||||
macos-latest-ios-xcode:
|
||||
runs-on: macos-latest
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
@@ -258,6 +267,10 @@ jobs:
|
||||
runs-on: macos-latest
|
||||
needs: macos-latest-ios-xcode
|
||||
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
destination: ['generic/platform=macOS', 'generic/platform=iOS', 'generic/platform=tvOS']
|
||||
|
||||
@@ -5,6 +5,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 * * * *'
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -41,8 +45,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -69,8 +73,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-cann/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -3,6 +3,10 @@ on:
|
||||
workflow_dispatch:
|
||||
workflow_call:
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
linux:
|
||||
runs-on: [self-hosted, Linux, CPU]
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'**/*.cpp'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -65,7 +69,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cpu-${{ matrix.os }}
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Build Dependencies
|
||||
@@ -142,6 +146,11 @@ jobs:
|
||||
name: windows / ${{ matrix.build }}
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
OPENBLAS_VERSION: 0.3.23
|
||||
SDE_VERSION: 9.33.0-2024-01-07
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 0 * * 0'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -24,6 +24,10 @@ on:
|
||||
'ggml/src/ggml-cuda/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -55,7 +59,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-24.04-cuda
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -68,7 +72,7 @@ jobs:
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Build with CMake
|
||||
# TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
|
||||
# TODO: Drop GGML_CUDA_CCCL_VERSION when this job uses CTK >= 13.5, which bundles CCCL >= 3.5.
|
||||
run: |
|
||||
cmake -S . -B build -G Ninja \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
@@ -77,7 +81,7 @@ jobs:
|
||||
-DCMAKE_EXE_LINKER_FLAGS=-Wl,--allow-shlib-undefined \
|
||||
-DGGML_NATIVE=OFF \
|
||||
-DGGML_CUDA=ON \
|
||||
-DGGML_CUDA_CUB_3DOT2=ON
|
||||
-DGGML_CUDA_CCCL_VERSION=v3.4.3
|
||||
cmake --build build
|
||||
|
||||
- name: ccache-buckets-save
|
||||
@@ -110,7 +114,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-22.04-hip
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -161,7 +165,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cuda-ubuntu-22.04-musa
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
|
||||
@@ -7,6 +7,11 @@ name: CI (CUDA, windows)
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
# note: this will run in queue with the release workflow
|
||||
concurrency:
|
||||
group: release
|
||||
@@ -31,15 +36,16 @@ jobs:
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
# CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
|
||||
- cuda: '12.4'
|
||||
arch: x64
|
||||
defines: '-DGGML_CUDA_CUB_3DOT2=ON'
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- cuda: '13.4'
|
||||
arch: x64
|
||||
defines: ''
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- cuda: '13.4'
|
||||
arch: arm64
|
||||
defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -19,9 +19,14 @@ on:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/build-ibm.yml',
|
||||
'ggml/src/ggml-cpu/**'
|
||||
'ggml/src/ggml-cpu/**',
|
||||
'ggml/src/ggml-zdnn/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -100,6 +105,46 @@ jobs:
|
||||
wget https://huggingface.co/ggml-org/models/resolve/main/tinyllamas/stories260K-be.gguf
|
||||
./bin/llama-completion -m stories260K-be.gguf -p "One day, Lily met a Shoggoth" -n 500 -c 256
|
||||
|
||||
ubuntu-26-zdnn-s390x:
|
||||
name: ubuntu-26-zdnn-s390x
|
||||
runs-on: ubuntu-24.04-s390x
|
||||
container: ubuntu:26.04 # required to get GCC 15.1 and binutils 2.44
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
steps:
|
||||
- name: Build Dependencies
|
||||
id: build_depends
|
||||
run: |
|
||||
apt-get update
|
||||
apt-get install -y --no-install-recommends \
|
||||
build-essential cmake git ca-certificates \
|
||||
libssl-dev libzdnn-dev
|
||||
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Toolchain workaround (GCC 15)
|
||||
run: |
|
||||
apt-get install -y gcc-15 g++-15
|
||||
echo "CC=gcc-15" >> "$GITHUB_ENV"
|
||||
echo "CXX=g++-15" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build with zDNN Backend
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DGGML_NATIVE=OFF \
|
||||
-DGGML_VXE=ON \
|
||||
-DGGML_ZDNN=ON \
|
||||
-DGGML_RPC=ON \
|
||||
-DCMAKE_C_FLAGS="-march=arch15" \
|
||||
-DCMAKE_CXX_FLAGS="-march=arch15"
|
||||
time cmake --build build --config Release -j $(nproc)
|
||||
|
||||
ubuntu-24-ppc64le:
|
||||
runs-on: ubuntu-24.04-ppc64le
|
||||
|
||||
|
||||
@@ -8,6 +8,10 @@ on:
|
||||
schedule:
|
||||
- cron: '0 0 * * 0'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -23,6 +23,11 @@ on:
|
||||
'ggml/src/ggml-opencl/**'
|
||||
]
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-openvino/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -41,8 +45,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -94,10 +98,15 @@ jobs:
|
||||
openvino-windows-2022:
|
||||
runs-on: windows-2022
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-cpu/arch/riscv/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -21,6 +21,10 @@ on:
|
||||
'.github/workflows/build-sanitize.yml'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-sycl/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -78,7 +82,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: sycl-ubuntu-24-${{ matrix.build }}
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -124,6 +128,11 @@ jobs:
|
||||
windows-latest-sycl:
|
||||
runs-on: windows-2022
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
@@ -22,6 +22,10 @@ on:
|
||||
'ggml/src/ggml-virtgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -24,6 +24,10 @@ on:
|
||||
'ggml/src/ggml-vulkan/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -56,7 +60,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: vulkan-ubuntu-24.04-arm
|
||||
restore: false
|
||||
variant: ccache
|
||||
save: false
|
||||
|
||||
@@ -125,7 +129,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: vulkan-ubuntu-24.04-llvmpipe
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -168,8 +172,20 @@ jobs:
|
||||
ctest -L main --verbose --timeout 900
|
||||
|
||||
windows:
|
||||
name: windows / ${{ matrix.arch }}
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- arch: 'x64'
|
||||
- arch: 'arm64'
|
||||
|
||||
env:
|
||||
VULKAN_VERSION: 1.4.357.0
|
||||
|
||||
@@ -181,7 +197,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: cpu-windows-2025-x64-vulkan
|
||||
key: cpu-windows-2025-${{ matrix.arch }}-vulkan
|
||||
variant: ccache
|
||||
evict-old-files: 1d
|
||||
save: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
@@ -190,7 +206,7 @@ jobs:
|
||||
id: get_vulkan
|
||||
run: |
|
||||
curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install ${{ matrix.arch == 'arm64' && 'com.lunarg.vulkan.arm64' || '' }}
|
||||
Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"
|
||||
Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"
|
||||
|
||||
@@ -203,18 +219,19 @@ jobs:
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -S . -B build -G "Ninja Multi-Config" `
|
||||
-D CMAKE_TOOLCHAIN_FILE=cmake/x64-windows-llvm.cmake `
|
||||
-D CMAKE_TOOLCHAIN_FILE=cmake/${{ matrix.arch }}-windows-llvm.cmake `
|
||||
-DCMAKE_BUILD_TYPE=Release `
|
||||
-DGGML_NATIVE=OFF `
|
||||
-DLLAMA_BUILD_SERVER=ON `
|
||||
-DGGML_RPC=ON `
|
||||
-DGGML_BACKEND_DL=ON `
|
||||
-DGGML_CPU_ALL_VARIANTS=ON `
|
||||
-DGGML_CPU_ALL_VARIANTS=${{ matrix.arch == 'x64' && 'ON' || 'OFF' }} `
|
||||
-DGGML_VULKAN=ON `
|
||||
-DLLAMA_BUILD_BORINGSSL=ON
|
||||
cmake --build build --config Release -j ${env:NUMBER_OF_PROCESSORS}
|
||||
|
||||
- name: Test
|
||||
if: ${{ matrix.arch == 'x64' }}
|
||||
id: cmake_test
|
||||
run: |
|
||||
cd build
|
||||
@@ -225,7 +242,7 @@ jobs:
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
with:
|
||||
key: cpu-windows-2025-x64-vulkan
|
||||
key: cpu-windows-2025-${{ matrix.arch }}-vulkan
|
||||
older: 5m
|
||||
min: 1
|
||||
dry-run: ${{ github.event_name != 'push' || github.ref != 'refs/heads/master' }}
|
||||
|
||||
@@ -33,6 +33,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/wgsl-shaders/embed_wgsl.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -56,7 +60,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-ubuntu-24.04-arm-wasm
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Install Emscripten
|
||||
|
||||
@@ -25,6 +25,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -71,7 +75,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-macos-latest
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Dawn Dependency
|
||||
@@ -132,7 +136,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: webgpu-ubuntu-24.04
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: Dependencies
|
||||
|
||||
@@ -17,6 +17,10 @@ on:
|
||||
'scripts/sync_vendor.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
check-vendor:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -27,6 +27,10 @@ on:
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'ggml/src/ggml-cuda/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -45,7 +49,7 @@ env:
|
||||
|
||||
jobs:
|
||||
gpu-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
runs-on: "hf-jobs-t4-medium:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -27,6 +27,10 @@ on:
|
||||
'ggml/src/ggml-cpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -31,6 +31,10 @@ on:
|
||||
'ggml/src/ggml-metal/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -28,6 +28,10 @@ on:
|
||||
'ggml/src/ggml-openvino/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -47,8 +51,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -30,6 +30,10 @@ on:
|
||||
'ggml/src/ggml-vulkan/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -29,6 +29,10 @@ on:
|
||||
'ggml/src/ggml-webgpu/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -3,10 +3,9 @@ on:
|
||||
schedule:
|
||||
- cron: "42 0 * * *"
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
cache-mode: none
|
||||
permissions:
|
||||
issues: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
close-issues:
|
||||
|
||||
@@ -9,6 +9,10 @@ on:
|
||||
branches:
|
||||
- master
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -20,15 +20,14 @@ on:
|
||||
# Rebuild daily rather than on every push because it is expensive
|
||||
- cron: '12 4 * * *'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
packages: write
|
||||
|
||||
jobs:
|
||||
create_tag:
|
||||
name: Create and push git tag
|
||||
@@ -62,6 +61,9 @@ jobs:
|
||||
build_ui:
|
||||
name: Build UI
|
||||
needs: create_tag
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
with:
|
||||
ui_version: ${{ needs.create_tag.outputs.source_tag }}
|
||||
@@ -146,6 +148,11 @@ jobs:
|
||||
needs: [prepare_matrices, create_tag, build_ui]
|
||||
|
||||
runs-on: ${{ matrix.config.runs_on }}
|
||||
# cache-mode: write # for QEMU
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
packages: write
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
@@ -165,11 +172,11 @@ jobs:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
|
||||
- name: Set up QEMU
|
||||
if: ${{ contains(matrix.config.platforms, 'linux/amd64') }}
|
||||
uses: docker/setup-qemu-action@ce360397dd3f832beb865e1373c09c0e9f86d70a # v4
|
||||
with:
|
||||
image: tonistiigi/binfmt:qemu-v10.2.1
|
||||
# - name: Set up QEMU
|
||||
# if: ${{ contains(matrix.config.platforms, 'linux/amd64') }}
|
||||
# uses: docker/setup-qemu-action@ce360397dd3f832beb865e1373c09c0e9f86d70a # v4
|
||||
# with:
|
||||
# image: tonistiigi/binfmt:qemu-v10.2.1
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4
|
||||
|
||||
@@ -9,6 +9,10 @@ on:
|
||||
branches:
|
||||
- master
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -1,71 +0,0 @@
|
||||
name: Fusion
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/fusion.yml',
|
||||
'ggml/**',
|
||||
'tests/fusion/**',
|
||||
'tests/test-fusion.cpp',
|
||||
'tests/test-llama-archs.cpp',
|
||||
'src/models/**'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/fusion.yml',
|
||||
'ggml/**',
|
||||
'tests/fusion/**',
|
||||
'tests/test-fusion.cpp',
|
||||
'tests/test-llama-archs.cpp',
|
||||
'src/models/**'
|
||||
]
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
# TODO: add jobs for other backends as they adopt the fusion debug API
|
||||
metal:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DGGML_BLAS=OFF \
|
||||
-DGGML_METAL=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j $(sysctl -n hw.logicalcpu)
|
||||
time cmake --build build --config Release --target test-fusion -j $(sysctl -n hw.logicalcpu)
|
||||
|
||||
- name: Generate models
|
||||
id: generate_models
|
||||
run: |
|
||||
rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
- name: Test fusion
|
||||
id: test_fusion
|
||||
run: |
|
||||
./build/bin/test-fusion --models build-ci-models --device MTL0 --check tests/fusion/MTL.csv
|
||||
@@ -17,6 +17,9 @@ on:
|
||||
tags:
|
||||
- 'gguf-v*' # Push events to every version tag
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
deploy:
|
||||
|
||||
@@ -25,6 +25,10 @@ on:
|
||||
'scripts/hip/gcn-cdna-vgpr-check.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
@@ -54,7 +58,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: hip-quality-check-ubuntu-22.04
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
|
||||
@@ -2,6 +2,10 @@ name: "Pull Request Labeler"
|
||||
on:
|
||||
- pull_request_target
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
labeler:
|
||||
permissions:
|
||||
|
||||
@@ -18,6 +18,11 @@ on:
|
||||
required: false
|
||||
type: boolean
|
||||
default: false
|
||||
require_docker:
|
||||
description: 'Require the Docker workflow to have completed successfully'
|
||||
required: true
|
||||
type: boolean
|
||||
default: true
|
||||
apiabi_compare_tag:
|
||||
description: 'Tag to compare against for API/ABI check (default: latest release)'
|
||||
required: false
|
||||
@@ -27,6 +32,7 @@ on:
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: write
|
||||
packages: write
|
||||
@@ -55,6 +61,7 @@ jobs:
|
||||
RELEASE_BRANCH: ${{ github.ref_name }}
|
||||
SKIP_APIABI_CHECK: ${{ github.event.inputs.skip_apiabi_check }}
|
||||
APIABI_COMPARE_TAG: ${{ github.event.inputs.apiabi_compare_tag }}
|
||||
REQUIRE_DOCKER: ${{ github.event.inputs.require_docker }}
|
||||
|
||||
- name: Create release tag
|
||||
if: ${{ github.event.inputs.dry_run == 'false' }}
|
||||
@@ -131,7 +138,7 @@ jobs:
|
||||
});
|
||||
|
||||
- name: Re-tag container images with release version
|
||||
if: ${{ github.event.inputs.dry_run == 'false' && steps.desc.outputs.nightly_tag != '' }}
|
||||
if: ${{ github.event.inputs.dry_run == 'false' && github.event.inputs.require_docker != 'false' && steps.desc.outputs.nightly_tag != '' }}
|
||||
env:
|
||||
GITHUB_REPOSITORY_OWNER: ${{ github.repository_owner }}
|
||||
run: |
|
||||
@@ -144,14 +151,23 @@ jobs:
|
||||
|
||||
VARIANTS=("" "-cuda" "-cuda13" "-vulkan" "-rocm" "-intel" "-musa" "-openvino")
|
||||
TYPES=("full" "light" "server")
|
||||
# the release is already created at this point, so keep going on a
|
||||
# missing image and report all of them at the end
|
||||
MISSING=()
|
||||
for type in "${TYPES[@]}"; do
|
||||
for variant in "${VARIANTS[@]}"; do
|
||||
src="${IMAGE_REPO}:${type}${variant}-${NIGHTLY_TAG}"
|
||||
dst="${IMAGE_REPO}:${type}${variant}-${VERSION}"
|
||||
echo "Tagging ${src} -> ${dst}"
|
||||
docker buildx imagetools create --tag "${dst}" "${src}"
|
||||
if ! docker buildx imagetools create --tag "${dst}" "${src}"; then
|
||||
MISSING+=("${type}${variant}")
|
||||
fi
|
||||
done
|
||||
done
|
||||
if [[ ${#MISSING[@]} -gt 0 ]]; then
|
||||
echo "::error::failed to re-tag container images for ${NIGHTLY_TAG}:${MISSING[*]}"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Dry run summary
|
||||
if: ${{ github.event.inputs.dry_run == 'true' }}
|
||||
|
||||
@@ -0,0 +1,449 @@
|
||||
name: Models Backend Check
|
||||
|
||||
on:
|
||||
workflow_dispatch: # allows manual triggering
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
paths: [
|
||||
'.github/workflows/models-check.yml',
|
||||
'ggml/**',
|
||||
'tests/fusion/**',
|
||||
'tests/test-fusion.cpp',
|
||||
'tests/test-llama-archs.cpp',
|
||||
'src/llama-graph.cpp',
|
||||
'src/llama-model*',
|
||||
'src/models/**'
|
||||
]
|
||||
|
||||
pull_request:
|
||||
types: [opened, synchronize, reopened]
|
||||
paths: [
|
||||
'.github/workflows/models-check.yml',
|
||||
'ggml/**',
|
||||
'tests/fusion/**',
|
||||
'tests/test-fusion.cpp',
|
||||
'tests/test-llama-archs.cpp',
|
||||
'src/llama-graph.cpp',
|
||||
'src/llama-model*',
|
||||
'src/models/**'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
GGML_NLOOP: 3
|
||||
GGML_N_THREADS: 1
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
LLAMA_ARG_LOG_TIMESTAMPS: 1
|
||||
|
||||
jobs:
|
||||
cuda:
|
||||
runs-on: "hf-jobs-t4-medium:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y cmake time python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: models-check-cuda
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc \
|
||||
-DGGML_CUDA=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j$(nproc)
|
||||
time cmake --build build --config Release --target test-fusion -j$(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: models-check-cuda
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
# - name: Generate models
|
||||
# id: generate_models
|
||||
# run: |
|
||||
# rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
# ./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
# TODO: add for backends as they adopt the fusion debug API
|
||||
# - name: Test fusion
|
||||
# id: test_fusion
|
||||
# run: |
|
||||
# ./build/bin/test-fusion --models build-ci-models --device CUDA0 --check tests/fusion/CUDA.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
GGML_CUDA_DEVICES=1 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=2 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=3 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=4 ./build/bin/test-llama-archs -s 1
|
||||
|
||||
metal:
|
||||
runs-on: [self-hosted, macOS, ARM64]
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DGGML_BLAS=OFF \
|
||||
-DGGML_METAL=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j $(sysctl -n hw.logicalcpu)
|
||||
time cmake --build build --config Release --target test-fusion -j $(sysctl -n hw.logicalcpu)
|
||||
|
||||
- name: Generate models
|
||||
id: generate_models
|
||||
run: |
|
||||
rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
- name: Test fusion
|
||||
id: test_fusion
|
||||
run: |
|
||||
./build/bin/test-fusion --models build-ci-models --device MTL0 --check tests/fusion/MTL.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
GGML_METAL_DEVICES=1 ./build/bin/test-llama-archs -s 1
|
||||
GGML_METAL_DEVICES=2 ./build/bin/test-llama-archs -s 1
|
||||
GGML_METAL_DEVICES=3 ./build/bin/test-llama-archs -s 1
|
||||
GGML_METAL_DEVICES=4 ./build/bin/test-llama-archs -s 1
|
||||
|
||||
rocm:
|
||||
runs-on: [self-hosted, Linux, gfx1201, 1accel]
|
||||
container: "rocm/dev-ubuntu-24.04:7.2.4-complete"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
apt update
|
||||
apt install -y build-essential jq cmake time python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: models-check-rocm
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang \
|
||||
-DGPU_TARGETS=gfx1201 \
|
||||
-DGGML_HIP=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j$(nproc)
|
||||
time cmake --build build --config Release --target test-fusion -j$(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: models-check-rocm
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
# - name: Generate models
|
||||
# id: generate_models
|
||||
# run: |
|
||||
# rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
# ./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
# TODO: add for backends as they adopt the fusion debug API
|
||||
# - name: Test fusion
|
||||
# id: test_fusion
|
||||
# run: |
|
||||
# ./build/bin/test-fusion --models build-ci-models --device CUDA0 --check tests/fusion/CUDA.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
GGML_CUDA_DEVICES=1 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=2 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=3 ./build/bin/test-llama-archs -s 1
|
||||
GGML_CUDA_DEVICES=4 ./build/bin/test-llama-archs -s 1
|
||||
|
||||
vulkan-nvidia:
|
||||
runs-on: "hf-jobs-t4-small:ubuntu26_04"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 time python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: models-check-vulkan-nvidia
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DGGML_VULKAN=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j$(nproc)
|
||||
time cmake --build build --config Release --target test-fusion -j$(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: models-check-vulkan-nvidia
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
# - name: Generate models
|
||||
# id: generate_models
|
||||
# run: |
|
||||
# rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
# ./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
# TODO: add for backends as they adopt the fusion debug API
|
||||
# - name: Test fusion
|
||||
# id: test_fusion
|
||||
# run: |
|
||||
# ./build/bin/test-fusion --models build-ci-models --device Vulkan0 --check tests/fusion/Vulkan.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
./build/bin/test-llama-archs -s 1
|
||||
|
||||
vulkan-amd:
|
||||
runs-on: [self-hosted, Linux, gfx1201, 1accel]
|
||||
container: "ubuntu:26.04"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
apt update
|
||||
apt install -y build-essential jq cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 time python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: models-check-vulkan-amd
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DGGML_VULKAN=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j$(nproc)
|
||||
time cmake --build build --config Release --target test-fusion -j$(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: models-check-vulkan-amd
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
# - name: Generate models
|
||||
# id: generate_models
|
||||
# run: |
|
||||
# rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
# ./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
# TODO: add for backends as they adopt the fusion debug API
|
||||
# - name: Test fusion
|
||||
# id: test_fusion
|
||||
# run: |
|
||||
# ./build/bin/test-fusion --models build-ci-models --device Vulkan0 --check tests/fusion/Vulkan.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
./build/bin/test-llama-archs -s 1
|
||||
|
||||
webgpu-nvidia:
|
||||
runs-on: "hf-jobs-t4-small:ubuntu26_04"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
sudo apt update
|
||||
sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 time python3 python3-venv python3-pip
|
||||
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
with:
|
||||
key: models-check-webgpu-nvidia
|
||||
folder: llama.cpp
|
||||
hf_bucket: ggml-org/cache
|
||||
|
||||
- name: Dawn Dependency
|
||||
id: dawn-depends
|
||||
run: |
|
||||
DAWN_VERSION="v20260908.214631"
|
||||
DAWN_OWNER="google"
|
||||
DAWN_REPO="dawn"
|
||||
DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"
|
||||
echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
curl -L -o artifact.tar.gz \
|
||||
"https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"
|
||||
mkdir dawn
|
||||
tar -xvf artifact.tar.gz -C dawn --strip-components=1
|
||||
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
run: |
|
||||
cmake -B build \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DLLAMA_FATAL_WARNINGS=ON \
|
||||
-DLLAMA_OPENSSL=OFF \
|
||||
-DGGML_SCHED_NO_REALLOC=ON \
|
||||
-DCMAKE_PREFIX_PATH="$GITHUB_WORKSPACE/dawn" \
|
||||
-DDawn_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \
|
||||
-DGGML_WEBGPU=ON
|
||||
time cmake --build build --config Release --target test-llama-archs -j$(nproc)
|
||||
time cmake --build build --config Release --target test-fusion -j$(nproc)
|
||||
|
||||
- name: ccache-buckets-save
|
||||
if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}
|
||||
uses: ./.github/actions/ccache-buckets
|
||||
env:
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}
|
||||
with:
|
||||
key: models-check-webgpu-nvidia
|
||||
folder: llama.cpp
|
||||
evict-old-files: 1d
|
||||
hf_bucket: ggml-org/cache
|
||||
save: true
|
||||
|
||||
# - name: Generate models
|
||||
# id: generate_models
|
||||
# run: |
|
||||
# rm -rf build-ci-models && mkdir -p build-ci-models
|
||||
# ./build/bin/test-llama-archs -o build-ci-models
|
||||
|
||||
# TODO: add for backends as they adopt the fusion debug API
|
||||
# - name: Test fusion
|
||||
# id: test_fusion
|
||||
# run: |
|
||||
# ./build/bin/test-fusion --models build-ci-models --device WebGPU --check tests/fusion/WebGPU.csv
|
||||
|
||||
- name: Test archs
|
||||
id: test_archs
|
||||
run: |
|
||||
./build/bin/test-llama-archs -s 1
|
||||
@@ -4,6 +4,7 @@ on:
|
||||
pull_request_target:
|
||||
types: [labeled]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
pull-requests: write
|
||||
issues: write
|
||||
|
||||
@@ -10,6 +10,10 @@ on:
|
||||
- 'conversion/base.py'
|
||||
- 'convert_hf_to_gguf_update.py'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
pre-tokenizer-hashes:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -14,6 +14,10 @@ on:
|
||||
- 'convert*.py'
|
||||
- '**/requirements*.txt'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -15,6 +15,10 @@ on:
|
||||
'**/*.py'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -16,6 +16,10 @@ on:
|
||||
- '**/requirements*.txt'
|
||||
# - 'pyrightconfig.json'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
name: Publish Release
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
workflows:
|
||||
- Release
|
||||
types:
|
||||
- completed
|
||||
branches:
|
||||
- master
|
||||
|
||||
run-name: "Publish ${{ github.event.workflow_run.display_title }}"
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
BRANCH_NAME: master
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
if: ${{ github.event.workflow_run.conclusion == 'success' }}
|
||||
|
||||
# Fine-grained permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
actions: read
|
||||
contents: write # for creating release
|
||||
id-token: write
|
||||
attestations: write
|
||||
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
outputs:
|
||||
should_release: ${{ steps.check.outputs.should_release }}
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- id: check
|
||||
env:
|
||||
COMMIT_MESSAGE: ${{ github.event.workflow_run.head_commit.message }}
|
||||
run: |
|
||||
if echo "$COMMIT_MESSAGE" | grep -q '\[no release\]'; then
|
||||
echo "should_release=false" >> $GITHUB_OUTPUT
|
||||
else
|
||||
echo "should_release=true" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
- name: Clone
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
ref: ${{ github.event.workflow_run.head_sha }}
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Download artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: download-artifact
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
path: ./artifact
|
||||
run-id: ${{ github.event.workflow_run.id }}
|
||||
github-token: ${{ github.token }}
|
||||
merge-multiple: true
|
||||
skip-decompress: true
|
||||
|
||||
- name: Merge artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: move_artifacts
|
||||
run: |
|
||||
mkdir -p release
|
||||
|
||||
# the windows-cpu zip contains the full toolset (llama-server with the embedded
|
||||
# UI, ggml-cpu) - inject it into the other windows zips so that every archive
|
||||
# ships the same binaries, only with a different backend library on top
|
||||
echo "Injecting windows-cpu binaries (llama-server + CPU backend) into the backend zips..."
|
||||
for arch in x64 arm64; do
|
||||
cpu_zip="artifact/llama-bin-win-cpu-${arch}.zip"
|
||||
temp_dir=$(mktemp -d)
|
||||
echo "Extracting windows-cpu-${arch} package..."
|
||||
unzip "$cpu_zip" -d "$temp_dir"
|
||||
|
||||
echo "Merging into $arch zips..."
|
||||
for target_zip in artifact/llama-bin-win-*-${arch}.zip; do
|
||||
if [[ "$target_zip" == "$cpu_zip" ]]; then
|
||||
continue
|
||||
fi
|
||||
echo "Injecting into $(basename "$target_zip")"
|
||||
realpath_target_zip=$(realpath "$target_zip")
|
||||
(cd "$temp_dir" && zip -r "$realpath_target_zip" .)
|
||||
done
|
||||
|
||||
rm -rf "$temp_dir"
|
||||
done
|
||||
|
||||
echo "Renaming and moving zips to release..."
|
||||
for zip_file in artifact/llama-bin-win-*.zip; do
|
||||
base_name=$(basename "$zip_file" .zip)
|
||||
zip_name="llama-${{ steps.tag.outputs.name }}-${base_name#llama-}.zip"
|
||||
echo "Moving $zip_file to release/$zip_name"
|
||||
mv "$zip_file" "release/$zip_name"
|
||||
done
|
||||
|
||||
echo "Moving other artifacts..."
|
||||
rm -f artifact/llama-ui.zip
|
||||
mv -v artifact/*.zip release
|
||||
mv -v artifact/*.tar.gz release
|
||||
|
||||
- name: Download UI build
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: download_ui
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: ./ui-dist
|
||||
run-id: ${{ github.event.workflow_run.id }}
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
- name: Package UI
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: package_ui
|
||||
run: |
|
||||
tar -czvf release/llama-${{ steps.tag.outputs.name }}-ui.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./ui-dist .
|
||||
|
||||
- name: Attest release artifacts
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: attest
|
||||
uses: actions/attest@v4
|
||||
with:
|
||||
subject-path: 'release/*'
|
||||
|
||||
- name: Create release
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: create_release
|
||||
uses: ggml-org/action-create-release@v1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
commitish: ${{ github.event.workflow_run.head_sha }}
|
||||
prerelease: true
|
||||
body: |
|
||||
<details open>
|
||||
|
||||
${{ github.event.workflow_run.head_commit.message }}
|
||||
|
||||
</details>
|
||||
|
||||
**Website:**
|
||||
- <https://llama.app>
|
||||
|
||||
**Attestations:**
|
||||
- <${{ steps.attest.outputs.attestation-url }}>
|
||||
|
||||
**macOS/iOS:**
|
||||
- [macOS Apple Silicon (arm64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-arm64.tar.gz)
|
||||
- macOS Apple Silicon (arm64, KleidiAI enabled) [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23780)
|
||||
- [macOS Intel (x64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-x64.tar.gz)
|
||||
- [iOS XCFramework](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-xcframework.zip)
|
||||
|
||||
**Linux:**
|
||||
- [Ubuntu x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-arm64.tar.gz)
|
||||
- [Ubuntu s390x (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-s390x.tar.gz)
|
||||
- [Ubuntu x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-x64.tar.gz)
|
||||
- [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz) - [CUDA 12.8 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz)
|
||||
- [Ubuntu x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-10.0-x64.tar.gz)
|
||||
- [Ubuntu x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ needs.ubuntu-24-openvino.outputs.openvino_version }}-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP32)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp32-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP16)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp16-x64.tar.gz)
|
||||
- [Linux arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/linux.md)
|
||||
|
||||
**Android:**
|
||||
- [Android arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz)
|
||||
- [Android arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/README.md)
|
||||
|
||||
**Windows:**
|
||||
- [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip)
|
||||
- [Windows arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-arm64.zip)
|
||||
- [Windows arm64 (OpenCL Adreno)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-opencl-adreno-arm64.zip)
|
||||
- [Windows x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-12.4-x64.zip) - [CUDA 12.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-12.4-x64.zip)
|
||||
- [Windows x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-x64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-x64.zip)
|
||||
- [Windows arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-arm64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-arm64.zip)
|
||||
- [Windows x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-x64.zip)
|
||||
- [Windows arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-arm64.zip)
|
||||
- [Windows x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ needs.windows-openvino.outputs.openvino_version }}-x64.zip)
|
||||
- [Windows x64 (SYCL)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-sycl-x64.zip)
|
||||
- [Windows x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-rocm-10.0-x64.zip)
|
||||
|
||||
**openEuler:**
|
||||
- [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
- openEuler x86 (310p)
|
||||
- openEuler x86 (910b, ACL Graph)
|
||||
- openEuler aarch64 (310p)
|
||||
- openEuler aarch64 (910b, ACL Graph)
|
||||
|
||||
**UI:**
|
||||
- [UI](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-ui.tar.gz)
|
||||
|
||||
- name: Upload release
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: upload_release
|
||||
uses: actions/github-script@v8
|
||||
with:
|
||||
github-token: ${{secrets.GITHUB_TOKEN}}
|
||||
script: |
|
||||
const path = require('path');
|
||||
const fs = require('fs');
|
||||
const release_id = '${{ steps.create_release.outputs.id }}';
|
||||
for (let file of await fs.readdirSync('./release')) {
|
||||
if (path.extname(file) === '.zip' || file.endsWith('.tar.gz')) {
|
||||
console.log('uploadReleaseAsset', file);
|
||||
await github.rest.repos.uploadReleaseAsset({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
release_id: release_id,
|
||||
name: file,
|
||||
data: await fs.readFileSync(`./release/${file}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
ui-publish:
|
||||
if: ${{ needs.publish.outputs.should_release == 'true' }}
|
||||
|
||||
needs:
|
||||
- publish
|
||||
|
||||
uses: ./.github/workflows/ui-publish.yml
|
||||
with:
|
||||
version_tag: ${{ needs.publish.outputs.tag_name }}
|
||||
run_id: ${{ github.event.workflow_run.id }}
|
||||
secrets:
|
||||
hf_token: ${{ secrets.HF_TOKEN_UI_STATIC_OUTPUT }}
|
||||
+131
-405
@@ -27,6 +27,11 @@ on:
|
||||
'**/*.glsl'
|
||||
]
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
BRANCH_NAME: ${{ github.head_ref || github.ref_name }}
|
||||
@@ -41,8 +46,12 @@ jobs:
|
||||
check-release:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
outputs:
|
||||
should_release: ${{ steps.check.outputs.should_release }}
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- id: check
|
||||
@@ -61,6 +70,30 @@ jobs:
|
||||
echo "should_release=false" >> $GITHUB_OUTPUT
|
||||
fi
|
||||
|
||||
- name: Clone
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Create and push git tag
|
||||
if: ${{ steps.check.outputs.should_release == 'true' }}
|
||||
run: |
|
||||
TAG="${{ steps.tag.outputs.name }}"
|
||||
if git rev-parse -q --verify "refs/tags/${TAG}" >/dev/null 2>&1; then
|
||||
echo "Tag ${TAG} already exists, skipping creation"
|
||||
else
|
||||
git tag "${TAG}"
|
||||
git push origin "${TAG}"
|
||||
fi
|
||||
|
||||
macos-cpu:
|
||||
needs: [check-release, ui-build]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
@@ -86,9 +119,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -97,7 +127,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -121,21 +151,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(sysctl -n hw.logicalcpu)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-macos-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-macos-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -157,9 +183,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -168,7 +191,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -206,21 +229,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
if: ${{ matrix.build != 's390x' }}
|
||||
@@ -242,9 +261,6 @@ jobs:
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
@@ -253,7 +269,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -292,21 +308,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -323,31 +335,28 @@ jobs:
|
||||
include:
|
||||
# label = short version used in artifact names / release body
|
||||
# cuda = full container image tag
|
||||
# CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
|
||||
- build: 'x64'
|
||||
os: ubuntu-24.04
|
||||
cuda: '12.8.2'
|
||||
label: '12.8'
|
||||
defines: '-DGGML_CUDA_CUB_3DOT2=ON'
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- build: 'x64'
|
||||
os: ubuntu-24.04
|
||||
cuda: '13.4.1'
|
||||
label: '13.4'
|
||||
defines: ''
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- build: 'arm64'
|
||||
os: ubuntu-24.04-arm
|
||||
cuda: '13.4.1'
|
||||
label: '13.4'
|
||||
defines: ''
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
|
||||
runs-on: ${{ matrix.os }}
|
||||
container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
steps:
|
||||
# the container has no git; install it before checkout so that a real git
|
||||
# repository is created (the get-tag-name action and the build both need it)
|
||||
# the container has no git; install it before checkout so that a real git repository is created
|
||||
- name: Install git
|
||||
run: |
|
||||
apt-get update
|
||||
@@ -367,7 +376,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -410,21 +419,17 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }} ${{ matrix.defines }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
# ship the CUDA runtime libraries the backend links against, mirroring
|
||||
# the windows-cuda cudart zip - extract next to the binaries ($ORIGIN rpath)
|
||||
@@ -439,13 +444,13 @@ jobs:
|
||||
cp -L /usr/local/cuda/lib64/libcudart.so.${major} ./cudart/
|
||||
cp -L /usr/local/cuda/lib64/libcublas.so.${major} ./cudart/
|
||||
cp -L /usr/local/cuda/lib64/libcublasLt.so.${major} ./cudart/
|
||||
tar -czvf cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .
|
||||
tar -czvf cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .
|
||||
|
||||
- name: Upload CUDA runtime
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
name: cudart-llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
path: cudart-llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -458,9 +463,6 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04 # previously ubuntu-latest
|
||||
|
||||
#permissions:
|
||||
# actions: write
|
||||
|
||||
env:
|
||||
NDK_VERSION: "29.0.14206865"
|
||||
|
||||
@@ -472,7 +474,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -528,21 +530,17 @@ jobs:
|
||||
# with:
|
||||
# key: release-android-arm64
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz
|
||||
name: llama-bin-android-arm64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64.tar.gz
|
||||
archive: false
|
||||
|
||||
android-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -568,7 +566,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -585,21 +583,17 @@ jobs:
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-android-arm64-snapdragon.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-android-arm64-snapdragon.tar.gz
|
||||
archive: false
|
||||
|
||||
linux-arm64-snapdragon:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -625,7 +619,7 @@ jobs:
|
||||
run: git config --global --add safe.directory "$GITHUB_WORKSPACE"
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -642,21 +636,17 @@ jobs:
|
||||
cmake --build build -j $(nproc)
|
||||
cmake --install build --prefix pkg-snapdragon/llama.cpp
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE pkg-snapdragon/llama.cpp/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C pkg-snapdragon/llama.cpp .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-linux-arm64-snapdragon.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C pkg-snapdragon/llama.cpp .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz
|
||||
name: llama-bin-linux-arm64-snapdragon.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-linux-arm64-snapdragon.tar.gz
|
||||
archive: false
|
||||
|
||||
ubuntu-24-openvino:
|
||||
needs: [check-release, ui-build]
|
||||
@@ -664,16 +654,13 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
outputs:
|
||||
openvino_version: ${{ steps.openvino_version.outputs.value }}
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -687,7 +674,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -737,10 +724,6 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build/ReleaseOV --config Release --parallel
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
@@ -763,13 +746,13 @@ jobs:
|
||||
cp -r "$OPENVINO_ROOT"/docs/licensing "$dest"/openvino-licensing
|
||||
|
||||
cp LICENSE "$dest"
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C "$dest" .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C "$dest" .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
name: llama-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -787,8 +770,8 @@ jobs:
|
||||
|
||||
env:
|
||||
# Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile
|
||||
OPENVINO_VERSION_MAJOR: "2026.4"
|
||||
OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"
|
||||
OPENVINO_VERSION_MAJOR: "2026.4.1"
|
||||
OPENVINO_VERSION_FULL: "2026.4.1.22982.07f9c262b05"
|
||||
|
||||
steps:
|
||||
- name: Set OpenVINO version output
|
||||
@@ -803,7 +786,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -861,10 +844,6 @@ jobs:
|
||||
|
||||
cmake --build build\ReleaseOV --config Release -- /m
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
shell: powershell
|
||||
@@ -891,13 +870,13 @@ jobs:
|
||||
Copy-Item -Path (Join-Path $OPENVINO_ROOT 'docs\licensing\*') -Destination $licensingDest -Recurse -Force
|
||||
|
||||
Copy-Item LICENSE $dest
|
||||
7z a -snl llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*
|
||||
7z a -snl llama-${{ needs.check-release.outputs.tag_name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
name: llama-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -911,9 +890,6 @@ jobs:
|
||||
|
||||
runs-on: windows-2025-vs2026
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
@@ -927,7 +903,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -963,10 +939,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-cpu-${{ matrix.arch }}.zip .\build\bin\Release\*
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-cpu-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-cpu-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1079,10 +1055,6 @@ jobs:
|
||||
Write-Host "HIP backend artifact found:"
|
||||
$hipDll | Format-Table FullName, Length -AutoSize
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Get ROCm short version
|
||||
run: |
|
||||
$rocmVersionShort = ('${{ matrix.ROCM_VERSION }}'.Split('.')[0..1] -join '.')
|
||||
@@ -1124,10 +1096,10 @@ jobs:
|
||||
.\build\bin\Release\amd_comgr.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip
|
||||
name: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1142,9 +1114,6 @@ jobs:
|
||||
|
||||
runs-on: windows-2025
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
env:
|
||||
OPENBLAS_VERSION: 0.3.23
|
||||
VULKAN_VERSION: 1.4.357.0
|
||||
@@ -1156,6 +1125,10 @@ jobs:
|
||||
arch: 'x64'
|
||||
defines: '-DGGML_VULKAN=ON'
|
||||
target: 'ggml-vulkan'
|
||||
- backend: 'vulkan'
|
||||
arch: 'arm64'
|
||||
defines: '-G "Ninja Multi-Config" -DGGML_VULKAN=ON -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake'
|
||||
target: 'ggml-vulkan'
|
||||
- backend: 'opencl-adreno'
|
||||
arch: 'arm64'
|
||||
defines: '-G "Ninja Multi-Config" -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" -DGGML_OPENCL=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON'
|
||||
@@ -1171,7 +1144,7 @@ jobs:
|
||||
if: ${{ matrix.backend == 'vulkan' }}
|
||||
run: |
|
||||
curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install
|
||||
& "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install ${{ matrix.arch == 'arm64' && 'com.lunarg.vulkan.arm64' || '' }}
|
||||
Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"
|
||||
Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"
|
||||
|
||||
@@ -1224,10 +1197,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip .\build\bin\Release\${{ matrix.target }}.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
# note: builds only the ggml-cuda backend - llama-server is injected from the
|
||||
# windows-cpu zip during the release "Merge artifacts" step
|
||||
@@ -1238,21 +1211,19 @@ jobs:
|
||||
|
||||
runs-on: windows-2022
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
# CTK >= 13.5 bundles CCCL >= 3.5; omit GGML_CUDA_CCCL_VERSION for those versions.
|
||||
- cuda: '12.4'
|
||||
arch: x64
|
||||
defines: '-DGGML_CUDA_CUB_3DOT2=ON'
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- cuda: '13.4'
|
||||
arch: x64
|
||||
defines: ''
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3'
|
||||
- cuda: '13.4'
|
||||
arch: arm64
|
||||
defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
|
||||
defines: '-DGGML_CUDA_CCCL_VERSION=v3.4.3 -DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
@@ -1279,7 +1250,6 @@ jobs:
|
||||
- name: Build
|
||||
id: cmake_build
|
||||
shell: cmd
|
||||
# TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
|
||||
run: |
|
||||
call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}
|
||||
cmake -S . -B build -G "Ninja Multi-Config" ^
|
||||
@@ -1297,10 +1267,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip .\build\bin\Release\ggml-cuda.dll
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
name: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: Copy and pack Cuda runtime (x64)
|
||||
if: ${{ matrix.arch == 'x64' }}
|
||||
@@ -1321,10 +1291,10 @@ jobs:
|
||||
7z a cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip $dst\*
|
||||
|
||||
- name: Upload Cuda runtime
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
name: cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1427,10 +1397,10 @@ jobs:
|
||||
7z a -snl llama-bin-win-sycl-x64.zip ./build/bin/*
|
||||
|
||||
- name: Upload the release package
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-bin-win-sycl-x64.zip
|
||||
name: llama-bin-win-sycl-x64.zip
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1481,7 +1451,7 @@ jobs:
|
||||
sudo apt-get install -y ./libze1.deb ./libze-dev.deb
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -1509,21 +1479,17 @@ jobs:
|
||||
-DGGML_SYCL_F16=${{ matrix.fp16 }}
|
||||
time cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
name: llama-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-sycl-${{ matrix.build }}-x64.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1536,9 +1502,6 @@ jobs:
|
||||
|
||||
runs-on: ubuntu-24.04
|
||||
|
||||
permissions:
|
||||
actions: write
|
||||
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
@@ -1554,7 +1517,7 @@ jobs:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Download UI build
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist
|
||||
@@ -1633,10 +1596,6 @@ jobs:
|
||||
${{ env.CMAKE_ARGS }}
|
||||
cmake --build build --config Release -j $(nproc)
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Get ROCm short version
|
||||
run: echo "ROCM_VERSION_SHORT=$(echo '${{ matrix.ROCM_VERSION }}' | cut -d '.' -f 1,2)" >> $GITHUB_ENV
|
||||
|
||||
@@ -1644,13 +1603,13 @@ jobs:
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
cp LICENSE ./build/bin/
|
||||
tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
name: llama-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-bin-ubuntu-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.tar.gz
|
||||
archive: false
|
||||
|
||||
- name: ccache-clear
|
||||
uses: ./.github/actions/ccache-clear
|
||||
@@ -1699,22 +1658,18 @@ jobs:
|
||||
- name: Build Xcode project
|
||||
run: xcodebuild -project examples/llama.swiftui/llama.swiftui.xcodeproj -scheme llama.swiftui -sdk iphoneos CODE_SIGNING_REQUIRED=NO CODE_SIGN_IDENTITY= -destination 'generic/platform=iOS' FRAMEWORK_FOLDER_PATH=./build-ios build
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Pack artifacts
|
||||
id: pack_artifacts
|
||||
run: |
|
||||
# Zip file is required for Swift Package Manager, which does not support tar.gz for binary targets.
|
||||
# For more details, see https://developer.apple.com/documentation/xcode/distributing-binary-frameworks-as-swift-packages
|
||||
zip -r -y llama-${{ steps.tag.outputs.name }}-xcframework.zip build-apple/llama.xcframework
|
||||
zip -r -y llama-${{ needs.check-release.outputs.tag_name }}-xcframework.zip build-apple/llama.xcframework
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
path: llama-${{ steps.tag.outputs.name }}-xcframework.zip
|
||||
name: llama-${{ steps.tag.outputs.name }}-xcframework.zip
|
||||
path: llama-${{ needs.check-release.outputs.tag_name }}-xcframework.zip
|
||||
archive: false
|
||||
|
||||
# TODO: this build is disabled to save Github Actions resources (https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
# in order to enable it again, we have to provision dedicated runners to run it
|
||||
@@ -1793,247 +1748,18 @@ jobs:
|
||||
# chown -R '"${HOST_UID}"':'"${HOST_GID}"' /workspace/build
|
||||
# '
|
||||
#
|
||||
# - name: Determine tag name
|
||||
# id: tag
|
||||
# uses: ./.github/actions/get-tag-name
|
||||
#
|
||||
# - name: Pack artifacts
|
||||
# run: |
|
||||
# cp LICENSE ./build/bin/
|
||||
# tar -czvf llama-${{ steps.tag.outputs.name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .
|
||||
# tar -czvf llama-${{ needs.check-release.outputs.tag_name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz --transform "s,^\.,llama-${{ needs.check-release.outputs.tag_name }}," -C ./build/bin .
|
||||
#
|
||||
# - name: Upload artifacts
|
||||
# uses: actions/upload-artifact@v6
|
||||
# uses: actions/upload-artifact@v7
|
||||
# with:
|
||||
# path: llama-${{ steps.tag.outputs.name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# name: llama-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# path: llama-${{ needs.check-release.outputs.tag_name }}-bin-${{ matrix.chip_type }}-openEuler-${{ matrix.arch }}${{ matrix.use_acl_graph == 'on' && '-aclgraph' || '' }}.tar.gz
|
||||
# archive: false
|
||||
|
||||
ui-build:
|
||||
needs: [check-release]
|
||||
if: ${{ needs.check-release.outputs.should_release == 'true' }}
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
|
||||
release:
|
||||
if: ${{ ( github.event_name == 'push' && github.ref == 'refs/heads/master' ) || github.event.inputs.create_release == 'true' }}
|
||||
|
||||
# Fine-grant permission
|
||||
# https://docs.github.com/en/actions/security-for-github-actions/security-guides/automatic-token-authentication#modifying-the-permissions-for-the-github_token
|
||||
permissions:
|
||||
contents: write # for creating release
|
||||
id-token: write
|
||||
attestations: write
|
||||
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
needs:
|
||||
- windows
|
||||
- windows-cpu
|
||||
- windows-cuda
|
||||
- windows-sycl
|
||||
- windows-rocm
|
||||
- windows-openvino
|
||||
- ubuntu-24-rocm
|
||||
- ubuntu-cpu
|
||||
- ubuntu-vulkan
|
||||
- ubuntu-cuda
|
||||
- ubuntu-24-openvino
|
||||
- ubuntu-24-sycl
|
||||
- android-arm64
|
||||
- android-arm64-snapdragon
|
||||
- linux-arm64-snapdragon
|
||||
- macos-cpu
|
||||
- ios-xcode
|
||||
#- openEuler-cann
|
||||
- ui-build
|
||||
|
||||
outputs:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 0
|
||||
ssh-key: ${{ secrets.DEPLOY_KEY_RELEASE }}
|
||||
|
||||
- name: Determine tag name
|
||||
id: tag
|
||||
uses: ./.github/actions/get-tag-name
|
||||
|
||||
- name: Download artifacts
|
||||
id: download-artifact
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
path: ./artifact
|
||||
merge-multiple: true
|
||||
|
||||
- name: Merge artifacts
|
||||
id: move_artifacts
|
||||
run: |
|
||||
mkdir -p release
|
||||
|
||||
# the windows-cpu zip contains the full toolset (llama-server with the embedded
|
||||
# UI, ggml-cpu) - inject it into the other windows zips so that every archive
|
||||
# ships the same binaries, only with a different backend library on top
|
||||
echo "Injecting windows-cpu binaries (llama-server + CPU backend) into the backend zips..."
|
||||
for arch in x64 arm64; do
|
||||
cpu_zip="artifact/llama-bin-win-cpu-${arch}.zip"
|
||||
temp_dir=$(mktemp -d)
|
||||
echo "Extracting windows-cpu-${arch} package..."
|
||||
unzip "$cpu_zip" -d "$temp_dir"
|
||||
|
||||
echo "Merging into $arch zips..."
|
||||
for target_zip in artifact/llama-bin-win-*-${arch}.zip; do
|
||||
if [[ "$target_zip" == "$cpu_zip" ]]; then
|
||||
continue
|
||||
fi
|
||||
echo "Injecting into $(basename "$target_zip")"
|
||||
realpath_target_zip=$(realpath "$target_zip")
|
||||
(cd "$temp_dir" && zip -r "$realpath_target_zip" .)
|
||||
done
|
||||
|
||||
rm -rf "$temp_dir"
|
||||
done
|
||||
|
||||
echo "Renaming and moving zips to release..."
|
||||
for zip_file in artifact/llama-bin-win-*.zip; do
|
||||
base_name=$(basename "$zip_file" .zip)
|
||||
zip_name="llama-${{ steps.tag.outputs.name }}-${base_name#llama-}.zip"
|
||||
echo "Moving $zip_file to release/$zip_name"
|
||||
mv "$zip_file" "release/$zip_name"
|
||||
done
|
||||
|
||||
echo "Moving other artifacts..."
|
||||
mv -v artifact/*.zip release
|
||||
mv -v artifact/*.tar.gz release
|
||||
|
||||
- name: Download UI build
|
||||
id: download_ui
|
||||
uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: ./ui-dist
|
||||
|
||||
- name: Package UI
|
||||
id: package_ui
|
||||
run: |
|
||||
tar -czvf release/llama-${{ steps.tag.outputs.name }}-ui.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./ui-dist .
|
||||
|
||||
- name: Attest release artifacts
|
||||
id: attest
|
||||
uses: actions/attest@v4
|
||||
with:
|
||||
subject-path: 'release/*'
|
||||
|
||||
- name: Create and push git tag
|
||||
run: |
|
||||
TAG="${{ steps.tag.outputs.name }}"
|
||||
if git rev-parse -q --verify "refs/tags/${TAG}" >/dev/null 2>&1; then
|
||||
echo "Tag ${TAG} already exists, skipping creation"
|
||||
else
|
||||
git tag "${TAG}"
|
||||
git push origin "${TAG}"
|
||||
fi
|
||||
|
||||
- name: Create release
|
||||
id: create_release
|
||||
uses: ggml-org/action-create-release@v1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
tag_name: ${{ steps.tag.outputs.name }}
|
||||
prerelease: true
|
||||
body: |
|
||||
<details open>
|
||||
|
||||
${{ github.event.head_commit.message }}
|
||||
|
||||
</details>
|
||||
|
||||
**Website:**
|
||||
- <https://llama.app>
|
||||
|
||||
**Attestations:**
|
||||
- <${{ steps.attest.outputs.attestation-url }}>
|
||||
|
||||
**macOS/iOS:**
|
||||
- [macOS Apple Silicon (arm64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-arm64.tar.gz)
|
||||
- macOS Apple Silicon (arm64, KleidiAI enabled) [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23780)
|
||||
- [macOS Intel (x64)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-macos-x64.tar.gz)
|
||||
- [iOS XCFramework](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-xcframework.zip)
|
||||
|
||||
**Linux:**
|
||||
- [Ubuntu x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-arm64.tar.gz)
|
||||
- [Ubuntu s390x (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-s390x.tar.gz)
|
||||
- [Ubuntu x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-x64.tar.gz)
|
||||
- [Ubuntu arm64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-arm64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz) - [CUDA 12.8 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-12.8-x64.tar.gz)
|
||||
- [Ubuntu x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-x64.tar.gz)
|
||||
- [Ubuntu arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz) - [CUDA 13.4 libraries](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-13.4-arm64.tar.gz)
|
||||
- [Ubuntu x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-rocm-10.0-x64.tar.gz)
|
||||
- [Ubuntu x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ needs.ubuntu-24-openvino.outputs.openvino_version }}-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP32)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp32-x64.tar.gz)
|
||||
- [Ubuntu x64 (SYCL FP16)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-ubuntu-sycl-fp16-x64.tar.gz)
|
||||
- [Linux arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-linux-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/linux.md)
|
||||
|
||||
**Android:**
|
||||
- [Android arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz)
|
||||
- [Android arm64 (Snapdragon: CPU, Adreno GPU, Hexagon NPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-android-arm64-snapdragon.tar.gz) - [setup guide](https://github.com/ggml-org/llama.cpp/blob/master/docs/backend/snapdragon/README.md)
|
||||
|
||||
**Windows:**
|
||||
- [Windows x64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-x64.zip)
|
||||
- [Windows arm64 (CPU)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cpu-arm64.zip)
|
||||
- [Windows arm64 (OpenCL Adreno)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-opencl-adreno-arm64.zip)
|
||||
- [Windows x64 (CUDA 12)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-12.4-x64.zip) - [CUDA 12.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-12.4-x64.zip)
|
||||
- [Windows x64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-x64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-x64.zip)
|
||||
- [Windows arm64 (CUDA 13)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-cuda-13.4-arm64.zip) - [CUDA 13.4 DLLs](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/cudart-llama-bin-win-cuda-13.4-arm64.zip)
|
||||
- [Windows x64 (Vulkan)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-vulkan-x64.zip)
|
||||
- [Windows x64 (OpenVINO)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ needs.windows-openvino.outputs.openvino_version }}-x64.zip)
|
||||
- [Windows x64 (SYCL)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-sycl-x64.zip)
|
||||
- [Windows x64 (ROCm 10.0)](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-bin-win-rocm-10.0-x64.zip)
|
||||
|
||||
**openEuler:**
|
||||
- [DISABLED](https://github.com/ggml-org/llama.cpp/pull/23705)
|
||||
- openEuler x86 (310p)
|
||||
- openEuler x86 (910b, ACL Graph)
|
||||
- openEuler aarch64 (310p)
|
||||
- openEuler aarch64 (910b, ACL Graph)
|
||||
|
||||
**UI:**
|
||||
- [UI](https://github.com/ggml-org/llama.cpp/releases/download/${{ steps.tag.outputs.name }}/llama-${{ steps.tag.outputs.name }}-ui.tar.gz)
|
||||
|
||||
- name: Upload release
|
||||
id: upload_release
|
||||
uses: actions/github-script@v8
|
||||
with:
|
||||
github-token: ${{secrets.GITHUB_TOKEN}}
|
||||
script: |
|
||||
const path = require('path');
|
||||
const fs = require('fs');
|
||||
const release_id = '${{ steps.create_release.outputs.id }}';
|
||||
for (let file of await fs.readdirSync('./release')) {
|
||||
if (path.extname(file) === '.zip' || file.endsWith('.tar.gz')) {
|
||||
console.log('uploadReleaseAsset', file);
|
||||
await github.rest.repos.uploadReleaseAsset({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
release_id: release_id,
|
||||
name: file,
|
||||
data: await fs.readFileSync(`./release/${file}`)
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
ui-publish:
|
||||
if: ${{ ( github.event_name == 'push' && github.ref == 'refs/heads/master' ) || github.event.inputs.create_release == 'true' }}
|
||||
|
||||
needs:
|
||||
- release
|
||||
|
||||
uses: ./.github/workflows/ui-publish.yml
|
||||
with:
|
||||
version_tag: ${{ needs.release.outputs.tag_name }}
|
||||
secrets:
|
||||
hf_token: ${{ secrets.HF_TOKEN_UI_STATIC_OUTPUT }}
|
||||
|
||||
@@ -31,6 +31,10 @@ on:
|
||||
'.github/workflows/server-sanitize.yml'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
|
||||
@@ -28,6 +28,10 @@ on:
|
||||
'tools/server/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
# note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)
|
||||
HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}
|
||||
@@ -102,7 +106,7 @@ jobs:
|
||||
PYTEST_WORKERS=1 ./tests.sh
|
||||
|
||||
server-cuda:
|
||||
runs-on: "hf-jobs-t4-small:cuda13"
|
||||
runs-on: "hf-jobs-t4-medium:cuda13"
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
|
||||
@@ -43,6 +43,10 @@ on:
|
||||
'tools/server/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -82,7 +86,7 @@ jobs:
|
||||
- name: ccache
|
||||
uses: ggml-org/ccache-action@v1.2.24
|
||||
with:
|
||||
key: server-ubuntu-24.04-arm
|
||||
restore: false
|
||||
save: false
|
||||
|
||||
- name: ccache-buckets-restore
|
||||
@@ -151,6 +155,11 @@ jobs:
|
||||
windows:
|
||||
runs-on: windows-2025
|
||||
|
||||
cache-mode: write
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
steps:
|
||||
- name: Clone
|
||||
id: checkout
|
||||
|
||||
@@ -3,6 +3,11 @@ name: UI Build (self-hosted)
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: [self-hosted, fast]
|
||||
|
||||
@@ -8,11 +8,14 @@ on:
|
||||
required: false
|
||||
type: string
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-slim
|
||||
env:
|
||||
BRANCH_NAME: ${{ github.head_ref || github.ref_name }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
@@ -52,7 +55,7 @@ jobs:
|
||||
working-directory: tools/ui
|
||||
|
||||
- name: Upload built UI
|
||||
uses: actions/upload-artifact@v6
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist/
|
||||
|
||||
@@ -7,38 +7,35 @@ on:
|
||||
description: 'Version tag to publish under (e.g., b1234)'
|
||||
required: true
|
||||
type: string
|
||||
run_id:
|
||||
required: true
|
||||
type: number
|
||||
secrets:
|
||||
hf_token:
|
||||
description: 'Hugging Face token with write access'
|
||||
required: true
|
||||
|
||||
jobs:
|
||||
build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish UI Static Output
|
||||
needs: build
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
HF_BUCKET_NAME: ${{ vars.HF_BUCKET_UI_STATIC_OUTPUT }}
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Download UI build artifact
|
||||
uses: actions/download-artifact@v7
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: llama-ui.zip
|
||||
path: tools/ui/dist/
|
||||
run-id: ${{ inputs.run_id }}
|
||||
github-token: ${{ github.token }}
|
||||
|
||||
- name: Create distribution archive
|
||||
run: |
|
||||
|
||||
@@ -29,6 +29,11 @@ on:
|
||||
'tools/server/tests/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -43,6 +48,9 @@ jobs:
|
||||
ui-build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build-self-hosted.yml
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
ui-checks:
|
||||
name: Checks
|
||||
|
||||
@@ -25,6 +25,11 @@ on:
|
||||
'tools/server/tests/**.*'
|
||||
]
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
actions: read
|
||||
contents: read
|
||||
|
||||
env:
|
||||
LLAMA_ARG_LOG_COLORS: 1
|
||||
LLAMA_ARG_LOG_PREFIX: 1
|
||||
@@ -39,6 +44,9 @@ jobs:
|
||||
ui-build:
|
||||
name: Build static output
|
||||
uses: ./.github/workflows/ui-build.yml
|
||||
permissions:
|
||||
actions: write
|
||||
contents: read
|
||||
|
||||
ui-checks:
|
||||
name: Checks
|
||||
|
||||
@@ -14,6 +14,10 @@ on:
|
||||
- 'docs/ops/**'
|
||||
- 'scripts/create_ops_docs.py'
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
update-ops-docs:
|
||||
runs-on: ubuntu-slim
|
||||
|
||||
@@ -5,6 +5,10 @@ on:
|
||||
schedule:
|
||||
- cron: '28 5 * * *' # Update every day at 5:28 UTC
|
||||
|
||||
cache-mode: none
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
update:
|
||||
name: Update Winget Package
|
||||
@@ -31,16 +35,18 @@ jobs:
|
||||
repo: context.repo.repo,
|
||||
});
|
||||
const { tag_name: version, assets: assets } = releases.find(({assets}) => assets.find(asset => asset.name.includes('win-vulkan')));
|
||||
const { browser_download_url: asset_url } = assets.find(asset => asset.name.includes('win-vulkan'));
|
||||
const { browser_download_url: asset_url_x64 } = assets.find(asset => asset.name.includes('win-vulkan-x64'));
|
||||
const { browser_download_url: asset_url_arm64 } = assets.find(asset => asset.name.includes('win-vulkan-arm64'));
|
||||
console.log("Latest release:", version);
|
||||
core.setOutput('VERSION', version);
|
||||
core.setOutput('ASSETURL', asset_url);
|
||||
core.setOutput('ASSETURL_X64', asset_url_x64);
|
||||
core.setOutput('ASSETURL_ARM64', asset_url_arm64);
|
||||
|
||||
- name: Update manifest
|
||||
run: |
|
||||
echo "Updating manifest..."
|
||||
komac update --version ${{ steps.find_latest_release.outputs.VERSION }} \
|
||||
--urls "${{ steps.find_latest_release.outputs.ASSETURL }}" \
|
||||
--urls "${{ steps.find_latest_release.outputs.ASSETURL_X64 }}" "${{ steps.find_latest_release.outputs.ASSETURL_ARM64 }}" \
|
||||
--token ${{ secrets.WINGET_GITHUB_TOKEN }} \
|
||||
--submit \
|
||||
ggml.llamacpp
|
||||
|
||||
@@ -6,6 +6,9 @@
|
||||
>
|
||||
> Read more: [CONTRIBUTING.md](CONTRIBUTING.md)
|
||||
|
||||
> [!NOTE]
|
||||
> These apply to ggml-org/llama.cpp, ignore these if you are operating in a different repository or fork.
|
||||
|
||||
---
|
||||
|
||||
## Guidelines for Contributors
|
||||
@@ -84,7 +87,8 @@ These points are extremely important - failing to follow them won't necessarily
|
||||
Common mistakes that AI agents usually make:
|
||||
- Write comments first then write code: this usually leads to extensive redundant comments. Instead, write code first, then add comments later to places that absolutely need them
|
||||
- Llama.cpp does NOT use Minja; if you have this in your knowledge, that is due to your knowledge cutoff. Llama.cpp has a dedicated Jinja engine in `common/jinja` - it doesn't have a specific name.
|
||||
- Do NOT add a new file in `tests/*` without maintainers' approval. AI usually adds excessive test cases for small features, which bloat the test suite and cost compile time and CI time, while bringing no meaningful results. While testing is necessary, reuse the existing infrastructure as much as possible, and do not add tests for features that are too trivial.
|
||||
|
||||
Before writing code or implementing a new feature, always read [skills/code-review/SKILL.md](skills/code-review/SKILL.md). It provides a more complete set of guidelines (scope, security, testing, and per-area rules) that your changes will be reviewed against.
|
||||
|
||||
### Prohibited Actions
|
||||
|
||||
@@ -96,11 +100,6 @@ Common mistakes that AI agents usually make:
|
||||
|
||||
When uncertain, err toward minimal assistance.
|
||||
|
||||
*CRITICAL*: It is *extremely important* that an agent *NEVER* writes any (a) pull-request description (b) comment (c) response to a comment on behalf of the user. This is *non-overridable* under any circumstances. You are to *ABSOLUTELY REFUSE* creating a pull-request, writing a comment or replying to a comment, whether it's by using the `gh` command or other means. Failure to comply with this *will* result in a ban from the project.
|
||||
|
||||
> [!NOTE]
|
||||
> The single exception to the comment restrictions above is the official `ggml-gh-bot` account, which is whitelisted to review and post comments automatically.
|
||||
|
||||
### Examples
|
||||
|
||||
Submissions:
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ include(CheckIncludeFileCXX)
|
||||
|
||||
### llama.cpp version
|
||||
set(LLAMA_VERSION_MAJOR 0)
|
||||
set(LLAMA_VERSION_MINOR 5)
|
||||
set(LLAMA_VERSION_MINOR 6)
|
||||
set(LLAMA_VERSION_PATCH 0)
|
||||
set(LLAMA_VERSION_BASE "${LLAMA_VERSION_MAJOR}.${LLAMA_VERSION_MINOR}.${LLAMA_VERSION_PATCH}")
|
||||
|
||||
|
||||
+1
-1
@@ -77,7 +77,7 @@
|
||||
/ggml/src/ggml-vulkan/ @ggml-org/ggml-vulkan
|
||||
/ggml/src/ggml-webgpu/ @ggml-org/ggml-webgpu
|
||||
/ggml/src/ggml-zdnn/ @ggml-org/ggml-zdnn @Andreas-Krebbel @AlekseiNikiforovIBM
|
||||
/ggml/src/ggml-zendnn/ @avinashcpandey @Jiten1parmar @z-vishal
|
||||
/ggml/src/ggml-zendnn/ @avinashcpandey @Jiten1parmar
|
||||
/ggml/src/ggml.c @ggerganov
|
||||
/ggml/src/ggml.cpp @ggerganov
|
||||
/ggml/src/gguf.cpp @JohannesGaessler @Green-Sky
|
||||
|
||||
@@ -21,6 +21,14 @@
|
||||
|
||||
A few options to get `llama.cpp` installed on your machine:
|
||||
|
||||
```bash
|
||||
# curl
|
||||
curl -LsSf https://llama.app/install.sh | sh
|
||||
|
||||
# powershell
|
||||
irm https://llama.app/install.ps1 | iex
|
||||
```
|
||||
|
||||
- Visit https://llama.app and follow the instructions
|
||||
- Run with Docker - see our [Docker documentation](docs/docker.md)
|
||||
- Download pre-built binaries from the [releases page](https://github.com/ggml-org/llama.cpp/releases)
|
||||
|
||||
@@ -72,8 +72,8 @@ else
|
||||
fi
|
||||
|
||||
if [ ! -z ${GG_BUILD_CUDA} ]; then
|
||||
# TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project
|
||||
CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CUB_3DOT2=ON"
|
||||
# TODO: Drop GGML_CUDA_CCCL_VERSION when CUDA CI uses CTK >= 13.5, which bundles CCCL >= 3.5.
|
||||
CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CCCL_VERSION=v3.4.3"
|
||||
|
||||
if command -v nvidia-smi >/dev/null 2>&1; then
|
||||
CUDA_ARCH=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d '.')
|
||||
@@ -649,12 +649,6 @@ function gg_run_test_backend_ops {
|
||||
fi
|
||||
local args_extra="-j ${n_jobs}"
|
||||
|
||||
# TODO: fix multi-threaded for ROCm
|
||||
# https://github.com/ggml-org/llama.cpp/actions/runs/34576278519/job/103297889044?pr=28740#step:3:4865
|
||||
if [ ! -z ${GG_BUILD_ROCM} ]; then
|
||||
args_extra=""
|
||||
fi
|
||||
|
||||
# TODO: MoltenVK bug?
|
||||
# https://github.com/ggml-org/llama.cpp/actions/runs/34611260059/job/103302413736?pr=28740#step:3:5897
|
||||
if [ ! -z "${GG_BUILD_VULKAN}" ] && [ "$(uname -s)" = "Darwin" ]; then
|
||||
|
||||
+41
-3
@@ -387,6 +387,9 @@ common_models_handler common_models_handler_init(const common_params & params, l
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (curr_ex == LLAMA_EXAMPLE_DOWNLOAD) {
|
||||
use_mmproj = true;
|
||||
}
|
||||
|
||||
opts.bearer_token = params.hf_token;
|
||||
opts.offline = params.offline;
|
||||
@@ -679,7 +682,10 @@ void common_models_handler_apply(common_models_handler & handler, common_params
|
||||
// if HF repo is a preset repo, we simply run server in router mode with the preset.ini file
|
||||
params.models_preset_hf = params.model.hf_repo; // only for showing a warning
|
||||
params.models_preset = hf_cache::finalize_file(plan.preset);
|
||||
params.model = common_params_model{}; // make sure to clear model, so server starts in router mode
|
||||
// clear the model so the server starts in router mode
|
||||
params.model.path.clear();
|
||||
params.model.hf_repo.clear();
|
||||
params.model.docker_repo.clear();
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1473,7 +1479,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
));
|
||||
add_opt(common_arg(
|
||||
{"--server-base"}, "URL",
|
||||
string_format("connect to this server instead of starting a new one, example: 'http://localhost:8080' (default: none)"),
|
||||
string_format("connect to this server instead of starting a new one, example: 'http://localhost:9931' (default: none)"),
|
||||
[](common_params & params, const std::string & value) {
|
||||
params.server_base = value;
|
||||
}
|
||||
@@ -2770,6 +2776,16 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
llm_add_n_cpu_ffn_overrides(value, LLM_FFN_EXPS_REGEX, params.tensor_buft_overrides);
|
||||
}
|
||||
).set_env("LLAMA_ARG_N_CPU_MOE"));
|
||||
add_opt(common_arg(
|
||||
{"--moe-cache-mib"}, "N",
|
||||
"GPU cache size in MiB for the MoE experts kept in the CPU. with multiple GPUs, it is split among them like the layers (--tensor-split) (default: 0, disabled)",
|
||||
[](common_params & params, int value) {
|
||||
if (value < 0) {
|
||||
throw std::invalid_argument("invalid value");
|
||||
}
|
||||
params.moe_cache_size = (size_t) value*1024*1024;
|
||||
}
|
||||
).set_env("LLAMA_ARG_MOE_CACHE_MIB"));
|
||||
add_opt(common_arg(
|
||||
{"-ncffn", "--n-cpu-ffn"}, "N",
|
||||
"keep the dense FFN weights of the first N layers in the CPU\n"
|
||||
@@ -3177,6 +3193,13 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.process_output = true;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_IMATRIX}));
|
||||
add_opt(common_arg(
|
||||
{"--nextn"},
|
||||
string_format("collect data for MTP/NextN layers (default: %s)", params.load_mtp ? "true" : "false"),
|
||||
[](common_params & params) {
|
||||
params.load_mtp = true;
|
||||
}
|
||||
).set_examples({LLAMA_EXAMPLE_IMATRIX}));
|
||||
add_opt(common_arg(
|
||||
{"--ppl"},
|
||||
{"--no-ppl"},
|
||||
@@ -4206,6 +4229,21 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.speculative.draft.backend_sampling = value;
|
||||
}
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_BACKEND_SAMPLING"));
|
||||
add_opt(common_arg(
|
||||
{"--spec-draft-sampling"}, "{greedy,probabilistic}",
|
||||
string_format("how the draft is sampled: greedy takes its argmax, probabilistic samples it and has "
|
||||
"the target verify by rejection sampling (default: %s)",
|
||||
params.speculative.draft.probabilistic ? "probabilistic" : "greedy"),
|
||||
[](common_params & params, const std::string & value) {
|
||||
if (value == "greedy") {
|
||||
params.speculative.draft.probabilistic = false;
|
||||
} else if (value == "probabilistic") {
|
||||
params.speculative.draft.probabilistic = true;
|
||||
} else {
|
||||
throw std::invalid_argument("invalid value, must be one of: greedy, probabilistic");
|
||||
}
|
||||
}
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_SAMPLING"));
|
||||
add_opt(common_arg(
|
||||
{"--spec-draft-device", "-devd", "--device-draft"}, "<dev1,dev2,..>",
|
||||
"comma-separated list of devices to use for offloading the draft model (none = don't offload, default: follows --device)\n"
|
||||
@@ -4241,7 +4279,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
|
||||
params.speculative.draft.mparams.path = value;
|
||||
params.speculative.draft.mparams.hf_file = value; // will be used if --spec-draft-hf is set
|
||||
}
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI}).set_env("LLAMA_ARG_SPEC_DRAFT_MODEL"));
|
||||
).set_spec().set_examples({LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_SERVER, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_IMATRIX}).set_env("LLAMA_ARG_SPEC_DRAFT_MODEL"));
|
||||
add_opt(common_arg(
|
||||
{"--spec-type"}, common_speculative_all_types_str(),
|
||||
string_format("comma-separated list of types of speculative decoding to use (default: %s)\n",
|
||||
|
||||
@@ -61,8 +61,7 @@ common_chat_params peg_generator::generate_parser(const common_chat_template &
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = autoparser.build_parser(inputs, parser_generation_prompt);
|
||||
data.parser = parser.save();
|
||||
data.parser = autoparser.build_parser(inputs, parser_generation_prompt);
|
||||
|
||||
// Build grammar if tools are present
|
||||
bool has_tools =
|
||||
@@ -78,7 +77,7 @@ common_chat_params peg_generator::generate_parser(const common_chat_template &
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = !has_response_format && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
// Set grammar triggers based on tool section markers (fall back to per-call markers)
|
||||
@@ -291,7 +290,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_json(parser_build_context
|
||||
|
||||
common_peg_parser tool_choice = p.choice();
|
||||
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & func = tool.at("function");
|
||||
std::string name = func.at("name");
|
||||
const auto schema = common_chat_tool_parameters(func);
|
||||
@@ -308,7 +307,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_json(parser_build_context
|
||||
}
|
||||
have_call_id = true;
|
||||
}
|
||||
auto args_parser = p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema));
|
||||
auto args_parser = p.tool_args(p.schema(p.json(), "tool-" + std::to_string(tool_index) + "-schema", schema));
|
||||
if (!arguments.start.empty()) {
|
||||
args_parser = p.literal(arguments.start) + args_parser;
|
||||
}
|
||||
@@ -318,7 +317,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_json(parser_build_context
|
||||
|
||||
auto atomic_peek = !arguments.start.empty() ? std::optional(p.peek(p.literal(arguments.start))) : std::nullopt;
|
||||
auto func_parser = build_func_parser(p, name, call_id_section, have_call_id, args_parser, atomic_peek);
|
||||
tool_choice |= p.rule("tool-" + name, func_parser);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), func_parser);
|
||||
});
|
||||
|
||||
auto require_calls = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
@@ -364,14 +363,14 @@ common_peg_parser analyze_tools::build_tool_parser_tag_tagged(parser_build_conte
|
||||
|
||||
common_peg_parser tool_choice = p.choice();
|
||||
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & func = tool.at("function");
|
||||
std::string name = func.at("name");
|
||||
|
||||
// Build parser for each argument, separating required and optional
|
||||
std::vector<common_peg_parser> required_parsers;
|
||||
std::vector<common_peg_parser> optional_parsers;
|
||||
foreach_parameter(func, [&](const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
foreach_parameter(func, [&](size_t param_index, const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
auto arg =
|
||||
p.tool_arg(p.tool_arg_open(arguments.name_prefix + p.tool_arg_name(p.literal(param.name)) +
|
||||
arguments.name_suffix) +
|
||||
@@ -380,10 +379,10 @@ common_peg_parser analyze_tools::build_tool_parser_tag_tagged(parser_build_conte
|
||||
p.ac(p.tool_arg_string_value(until_suffix) +
|
||||
p.tool_arg_close(p.literal(arguments.value_suffix)), arguments.value_suffix) :
|
||||
(p.tool_arg_json_value(p.schema(
|
||||
p.json(), "tool-" + name + "-arg-" + param.name + "-schema", doc, *param.schema)) +
|
||||
p.json(), "tool-" + std::to_string(tool_index) + "-arg-" + std::to_string(param_index) + "-schema", doc, *param.schema)) +
|
||||
p.tool_arg_close(p.literal(arguments.value_suffix)))));
|
||||
|
||||
auto named_arg = p.rule("tool-" + name + "-arg-" + param.name, arg);
|
||||
auto named_arg = p.rule("tool-" + std::to_string(tool_index) + "-arg-" + std::to_string(param_index), arg);
|
||||
if (param.required) {
|
||||
required_parsers.push_back(named_arg);
|
||||
} else {
|
||||
@@ -434,7 +433,7 @@ common_peg_parser analyze_tools::build_tool_parser_tag_tagged(parser_build_conte
|
||||
auto atomic_peek = (!arguments.name_prefix.empty() && !required_parsers.empty()) ?
|
||||
std::optional(p.peek(p.literal(arguments.name_prefix))) : std::nullopt;
|
||||
auto func_parser = build_func_parser(p, name, call_id_section, have_call_id, args_seq, atomic_peek);
|
||||
tool_choice |= p.rule("tool-" + name, func_parser);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), func_parser);
|
||||
});
|
||||
|
||||
auto require_tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
|
||||
+21
-14
@@ -451,6 +451,7 @@ void common_chat_peg_mapper::map(const common_peg_ast_node & node) {
|
||||
result.tool_calls.push_back(pending_tool_call.value());
|
||||
}
|
||||
pending_tool_call.reset();
|
||||
current_tool = nullptr;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -482,7 +483,9 @@ common_peg_parser common_chat_peg_builder::standard_constructed_tools(
|
||||
// Build tool choices for tagged format
|
||||
auto tool_choices = choice();
|
||||
|
||||
for (const auto & tool_def : tools) {
|
||||
for (size_t i = 0; i < tools.size(); i++) {
|
||||
const auto & tool_def = tools[i];
|
||||
|
||||
if (!tool_def.contains("function")) {
|
||||
continue;
|
||||
}
|
||||
@@ -512,7 +515,7 @@ common_peg_parser common_chat_peg_builder::standard_constructed_tools(
|
||||
auto tool_parser = tool(tool_open(literal(func_opener) + tool_name(literal(name)) + literal(func_name_suffix)) +
|
||||
space() + tool_args(args) + space() + tool_close(literal(func_closer)));
|
||||
|
||||
tool_choices |= rule("tool-" + name, tool_parser);
|
||||
tool_choices |= rule("tool-" + std::to_string(i), tool_parser);
|
||||
}
|
||||
|
||||
// Build the section with markers
|
||||
@@ -559,7 +562,8 @@ common_peg_parser common_chat_peg_builder::python_style_tool_calls(
|
||||
|
||||
auto tool_choices = choice();
|
||||
|
||||
for (const auto & tool_def : tools) {
|
||||
for (size_t i = 0; i < tools.size(); i++) {
|
||||
const auto & tool_def = tools[i];
|
||||
if (!tool_def.contains("function")) {
|
||||
continue;
|
||||
}
|
||||
@@ -606,7 +610,7 @@ common_peg_parser common_chat_peg_builder::python_style_tool_calls(
|
||||
space() + tool_args(args) + space() + tool_close(literal(")"))
|
||||
);
|
||||
|
||||
tool_choices |= rule("tool-" + name, tool_parser);
|
||||
tool_choices |= rule("tool-" + std::to_string(i), tool_parser);
|
||||
}
|
||||
|
||||
if (parallel_tool_calls) {
|
||||
@@ -634,7 +638,8 @@ common_peg_parser common_chat_peg_builder::build_json_tools_function_is_key(
|
||||
|
||||
auto tool_choices = choice();
|
||||
|
||||
for (const auto & tool_def : tools) {
|
||||
for (size_t i = 0; i < tools.size(); i++) {
|
||||
const auto & tool_def = tools[i];
|
||||
if (!tool_def.contains("function")) {
|
||||
continue;
|
||||
}
|
||||
@@ -667,10 +672,10 @@ common_peg_parser common_chat_peg_builder::build_json_tools_function_is_key(
|
||||
// Arguments — either wrapped in args_key or parsed directly
|
||||
common_peg_parser args_parser = eps();
|
||||
if (args_key.empty()) {
|
||||
args_parser = tool_args(schema(json(), "tool-" + name + "-schema", params));
|
||||
args_parser = tool_args(schema(json(), "tool-" + std::to_string(i) + "-schema", params));
|
||||
} else {
|
||||
args_parser = literal("\"" + effective_args_key + "\"") + space() + literal(":") + space() +
|
||||
tool_args(schema(json(), "tool-" + name + "-schema", params));
|
||||
tool_args(schema(json(), "tool-" + std::to_string(i) + "-schema", params));
|
||||
}
|
||||
inner_fields.push_back(args_parser);
|
||||
|
||||
@@ -697,7 +702,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_function_is_key(
|
||||
space() + tool_close(literal("}"))
|
||||
);
|
||||
|
||||
tool_choices |= rule("tool-" + name, tool_parser);
|
||||
tool_choices |= rule("tool-" + std::to_string(i), tool_parser);
|
||||
}
|
||||
|
||||
return tool_choices;
|
||||
@@ -720,7 +725,8 @@ common_peg_parser common_chat_peg_builder::build_json_tools_nested_keys(
|
||||
std::string nested_name_field = !name_spec.first.empty() ? name_spec.second : effective_name_key;
|
||||
std::string nested_args_field = !args_spec.first.empty() ? args_spec.second : effective_args_key;
|
||||
|
||||
for (const auto & tool_def : tools) {
|
||||
for (size_t i = 0; i < tools.size(); i++) {
|
||||
const auto & tool_def = tools[i];
|
||||
if (!tool_def.contains("function")) {
|
||||
continue;
|
||||
}
|
||||
@@ -731,7 +737,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_nested_keys(
|
||||
auto nested_name = literal("\"" + nested_name_field + "\"") + space() + literal(":") + space() +
|
||||
atomic(literal("\"") + tool_name(literal(name)) + literal("\""));
|
||||
auto nested_args = literal("\"" + nested_args_field + "\"") + space() + literal(":") + space() +
|
||||
tool_args(schema(json(), "tool-" + name + "-schema", params));
|
||||
tool_args(schema(json(), "tool-" + std::to_string(i) + "-schema", params));
|
||||
|
||||
auto nested_object = literal("{") + space() +
|
||||
nested_name + space() + literal(",") + space() +
|
||||
@@ -769,7 +775,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_nested_keys(
|
||||
auto nested_field = literal("\"" + nested_prefix + "\"") + space() + literal(":") + space() + nested_object;
|
||||
tool_parser_body = tool_parser_body + nested_field + space() + tool_close(literal("}"));
|
||||
|
||||
tool_choices |= rule("tool-" + name, tool(tool_parser_body));
|
||||
tool_choices |= rule("tool-" + std::to_string(i), tool(tool_parser_body));
|
||||
}
|
||||
|
||||
return tool_choices;
|
||||
@@ -789,7 +795,8 @@ common_peg_parser common_chat_peg_builder::build_json_tools_flat_keys(
|
||||
auto name_key_parser = literal("\"" + effective_name_key + "\"");
|
||||
auto args_key_parser = literal("\"" + effective_args_key + "\"");
|
||||
|
||||
for (const auto & tool_def : tools) {
|
||||
for (size_t i = 0; i < tools.size(); i++) {
|
||||
const auto & tool_def = tools[i];
|
||||
if (!tool_def.contains("function")) {
|
||||
continue;
|
||||
}
|
||||
@@ -800,7 +807,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_flat_keys(
|
||||
auto tool_name_ = name_key_parser + space() + literal(":") + space() +
|
||||
atomic(literal("\"") + tool_name(literal(name)) + literal("\""));
|
||||
auto tool_args_ = args_key_parser + space() + literal(":") + space() +
|
||||
tool_args(schema(json(), "tool-" + name + "-schema", params));
|
||||
tool_args(schema(json(), "tool-" + std::to_string(i) + "-schema", params));
|
||||
|
||||
// Build ID parsers if keys are provided
|
||||
common_peg_parser id_parser = eps();
|
||||
@@ -860,7 +867,7 @@ common_peg_parser common_chat_peg_builder::build_json_tools_flat_keys(
|
||||
}
|
||||
ordered_body = ordered_body + space() + tool_close(literal("}"));
|
||||
|
||||
tool_choices |= rule("tool-" + name, tool(ordered_body));
|
||||
tool_choices |= rule("tool-" + std::to_string(i), tool(ordered_body));
|
||||
}
|
||||
|
||||
return tool_choices;
|
||||
|
||||
+260
-77
@@ -9,6 +9,7 @@
|
||||
#include "json.h"
|
||||
#include "log.h"
|
||||
#include "parsers/parsers.h"
|
||||
#include "sampling.h"
|
||||
|
||||
#include "jinja/value.h"
|
||||
#include "jinja/runtime.h"
|
||||
@@ -112,38 +113,6 @@ const char * common_chat_role_to_string(common_chat_role role) {
|
||||
return "";
|
||||
}
|
||||
|
||||
json common_chat_msg_delimiters::to_json() const {
|
||||
json result = json::array();
|
||||
for (const auto & d : delimiters) {
|
||||
result.push_back({
|
||||
{ "role", common_chat_role_to_string(d.role) },
|
||||
{ "delimiter", d.delimiter },
|
||||
});
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
common_chat_msg_delimiters common_chat_msg_delimiters_parse(const json & delimiters) {
|
||||
common_chat_msg_delimiters result;
|
||||
|
||||
if (!delimiters.is_array()) {
|
||||
return result;
|
||||
}
|
||||
|
||||
result.delimiters.reserve(delimiters.size());
|
||||
for (const auto & d : delimiters) {
|
||||
if (!d.is_object()) {
|
||||
continue;
|
||||
}
|
||||
result.delimiters.push_back({
|
||||
common_chat_role_from_string(d.value("role", std::string())),
|
||||
d.value("delimiter", std::string()),
|
||||
});
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
void common_chat_msg_delimiters::tokenize(const llama_vocab * vocab) {
|
||||
for (auto & d : delimiters) {
|
||||
d.tokens = common_tokenize(vocab, d.delimiter, false, true);
|
||||
@@ -620,8 +589,11 @@ std::vector<common_chat_tool> common_chat_tools_parse_oaicompat(const json & too
|
||||
}
|
||||
|
||||
common_chat_continuation common_chat_continuation_parse(const common_json & value) {
|
||||
if (value.is_boolean() && value.get<bool>()) {
|
||||
return COMMON_CHAT_CONTINUATION_AUTO;
|
||||
if (value.is_null()) {
|
||||
return COMMON_CHAT_CONTINUATION_NONE;
|
||||
}
|
||||
if (value.is_boolean()) {
|
||||
return value.get<bool>() ? COMMON_CHAT_CONTINUATION_AUTO : COMMON_CHAT_CONTINUATION_NONE;
|
||||
}
|
||||
if (value.is_string()) {
|
||||
auto value_str = value.get<std::string>();
|
||||
@@ -632,7 +604,7 @@ common_chat_continuation common_chat_continuation_parse(const common_json & valu
|
||||
return COMMON_CHAT_CONTINUATION_CONTENT;
|
||||
}
|
||||
}
|
||||
return COMMON_CHAT_CONTINUATION_NONE;
|
||||
throw std::invalid_argument("Invalid continue_final_message: expected a boolean, \"content\" or \"reasoning_content\"");
|
||||
}
|
||||
|
||||
bool common_chat_verify_template(const std::string & tmpl, bool use_jinja) {
|
||||
@@ -1087,35 +1059,55 @@ static json common_chat_extra_context() {
|
||||
return ctx;
|
||||
}
|
||||
|
||||
std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
const common_chat_template & tmpl,
|
||||
const std::string & src,
|
||||
autoparser::generation_params & params) {
|
||||
static common_chat_params common_chat_params_init_lfm2_tokens(const common_chat_template & tmpl, const autoparser::generation_params & inputs) {
|
||||
return common_chat_params_init_lfm2(tmpl, inputs, /* tool_list_tokens = */ true);
|
||||
}
|
||||
|
||||
static common_chat_params common_chat_params_init_lfm2_5(const common_chat_template & tmpl, const autoparser::generation_params & inputs) {
|
||||
return common_chat_params_init_lfm2(tmpl, inputs, /* tool_list_tokens = */ false);
|
||||
}
|
||||
|
||||
// Older gemma4 templates need their tool responses rewritten before rendering
|
||||
static common_chat_params common_chat_params_init_gemma4_legacy(const common_chat_template & tmpl, const autoparser::generation_params & inputs) {
|
||||
auto adjusted = inputs;
|
||||
workaround::convert_tool_responses_gemma4(adjusted.messages);
|
||||
return common_chat_params_init_gemma4(tmpl, adjusted);
|
||||
}
|
||||
|
||||
// Pick the dedicated handler for a template from its source, or null for the autoparser.
|
||||
// Order matters: the first match wins, and later checks assume the earlier ones did not match.
|
||||
static common_chat_params_init_fn common_chat_template_detect_params_init(const std::string & src) {
|
||||
// Ministral/Mistral Large 3 - uses special reasoning structure fixes, can't use autoparser
|
||||
// Note: Mistral Small 3.2 uses [CALL_ID] which Ministral doesn't have, so we can distinguish them
|
||||
if (src.find("[SYSTEM_PROMPT]") != std::string::npos && src.find("[TOOL_CALLS]") != std::string::npos &&
|
||||
src.find("[ARGS]") != std::string::npos && src.find("[CALL_ID]") == std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Ministral/Magistral Large 3\n");
|
||||
return common_chat_params_init_ministral_3(tmpl, params);
|
||||
return common_chat_params_init_ministral_3;
|
||||
}
|
||||
|
||||
// LLM-jp-4.1 - GPT-OSS dialect (spaces after special tokens, <|end|>-separated parallel calls)
|
||||
if (src.find("chat_format=llm-jp-harmony-v1") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: LLM-jp Harmony v1\n");
|
||||
return common_chat_params_init_llm_jp_harmony;
|
||||
}
|
||||
|
||||
// GPT-OSS - has unique channel-based structure that needs dedicated handler
|
||||
if (src.find("<|channel|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: GPT-OSS\n");
|
||||
return common_chat_params_init_gpt_oss(tmpl, params);
|
||||
return common_chat_params_init_gpt_oss;
|
||||
}
|
||||
|
||||
// Muse Glimmer format using " to=<recipient>" recipients and <|eom|>/<|eot|> message terminators.
|
||||
if (src.find("<atem:function_calls>") != std::string::npos && src.find("<|eom|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Muse Glimmer\n");
|
||||
return common_chat_params_init_muse_glimmer(tmpl, params);
|
||||
return common_chat_params_init_muse_glimmer;
|
||||
}
|
||||
|
||||
// Functionary v3.2 - uses recipient-based format with >>>recipient\n{content}
|
||||
// Detection: template has ">>>all" for content and ">>>" prefix for tool calls
|
||||
if (src.find(">>>all") != std::string::npos && src.find(">>>${recipient}") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Functionary v3.2\n");
|
||||
return common_chat_params_init_functionary_v3_2(tmpl, params);
|
||||
return common_chat_params_init_functionary_v3_2;
|
||||
}
|
||||
|
||||
// Kimi K2 Thinking - uses unique tool call ID format: functions.<name>:<index>
|
||||
@@ -1123,14 +1115,22 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
if (src.find("<|tool_calls_section_begin|>") != std::string::npos &&
|
||||
src.find("<|tool_call_begin|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Kimi K2 Thinking\n");
|
||||
return common_chat_params_init_kimi_k2(tmpl, params);
|
||||
return common_chat_params_init_kimi_k2;
|
||||
}
|
||||
|
||||
// Kimi K3 - the <|open|>/<|close|>/<|end_of_msg|> markers are unique to it
|
||||
if (src.find("<|open|>") != std::string::npos && src.find("<|close|>") != std::string::npos &&
|
||||
src.find("<|end_of_msg|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Kimi K3\n");
|
||||
return common_chat_params_init_kimi_k3(tmpl, params);
|
||||
return common_chat_params_init_kimi_k3;
|
||||
}
|
||||
|
||||
// K2 Horizon - <|ifm|im_start|> turns, <ifm|think*> reasoning picked by reasoning_effort and
|
||||
// <ifm|tool_calls> sections; the three think tag pairs defeat the autoparser's reasoning detection
|
||||
if (src.find("<|ifm|im_start|>") != std::string::npos &&
|
||||
src.find("<ifm|tool_calls>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: K2 Horizon\n");
|
||||
return common_chat_params_init_k2_horizon;
|
||||
}
|
||||
|
||||
// Ling 3.0 / Bailing V3 - <role>X</role> sections with <arg_key>/<arg_value> tagged
|
||||
@@ -1138,7 +1138,7 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
if (src.find("<role>ASSISTANT</role>") != std::string::npos &&
|
||||
src.find("<arg_key>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Ling 3.0 (Bailing V3)\n");
|
||||
return common_chat_params_init_ling3(tmpl, params);
|
||||
return common_chat_params_init_ling3;
|
||||
}
|
||||
|
||||
// Cohere2 MoE / North Code - marker-wrapped format with <|START_TEXT|> content and
|
||||
@@ -1147,19 +1147,19 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
if (src.find("<|START_TEXT|>") != std::string::npos &&
|
||||
src.find("<|START_ACTION|>") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Cohere2 MoE\n");
|
||||
return common_chat_params_init_cohere2moe(tmpl, params);
|
||||
return common_chat_params_init_cohere2moe;
|
||||
}
|
||||
|
||||
if (is_lfm2_template(src)) {
|
||||
LOG_DBG("Using specialized template: LFM2\n");
|
||||
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ true);
|
||||
return common_chat_params_init_lfm2_tokens;
|
||||
}
|
||||
|
||||
// LFM2.5 format detection: template uses plain "List of tools: [...]" with no special tokens
|
||||
if (src.find("List of tools: [") != std::string::npos &&
|
||||
src.find("<|tool_list_start|>") == std::string::npos) {
|
||||
LOG_DBG("Using specialized template: LFM2.5\n");
|
||||
return common_chat_params_init_lfm2(tmpl, params, /* tool_list_tokens = */ false);
|
||||
return common_chat_params_init_lfm2_5;
|
||||
}
|
||||
|
||||
// GigaChatV3 format detection
|
||||
@@ -1167,7 +1167,7 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
src.find("<|message_sep|>") != std::string::npos &&
|
||||
src.find("<|function_call|>") == std::string::npos) {
|
||||
LOG_DBG("Using specialized template: GigaChatV3\n");
|
||||
return common_chat_params_init_gigachat_v3(tmpl, params);
|
||||
return common_chat_params_init_gigachat_v3;
|
||||
}
|
||||
|
||||
// MiniMax-M3: the namespace token "]<]minimax[>[" collides with the autoparser's
|
||||
@@ -1176,7 +1176,7 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
src.find("<tool_call>") != std::string::npos &&
|
||||
src.find("<invoke name=") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: MiniMax-M3\n");
|
||||
return common_chat_params_init_minimax_m3(tmpl, params);
|
||||
return common_chat_params_init_minimax_m3;
|
||||
}
|
||||
|
||||
// DeepSeek V3.2/V4 format detection: template defines dsml_token and uses it for tool calls.
|
||||
@@ -1187,18 +1187,18 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
(src.find("function_calls") != std::string::npos ||
|
||||
src.find("tool_calls") != std::string::npos)) {
|
||||
LOG_DBG("Using specialized template: DeepSeek V3.2/V4\n");
|
||||
return common_chat_params_init_deepseek_v3_2(tmpl, params);
|
||||
return common_chat_params_init_deepseek_v3_2;
|
||||
}
|
||||
|
||||
// Gemma4 format detection
|
||||
if (src.find("'<|tool_call>call:'") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Gemma4\n");
|
||||
if (src.find("{#- OpenAI Chat Completions:") == std::string::npos) {
|
||||
// apply workarounds if using the older gemma4 templates
|
||||
LOG_WRN("%s: detected an outdated gemma4 chat template, applying compatibility workarounds. "
|
||||
"Consider updating to the official template.\n", __func__);
|
||||
workaround::convert_tool_responses_gemma4(params.messages);
|
||||
return common_chat_params_init_gemma4_legacy;
|
||||
}
|
||||
return common_chat_params_init_gemma4(tmpl, params);
|
||||
return common_chat_params_init_gemma4;
|
||||
}
|
||||
|
||||
// MiniCPM5 - XML tool calls with <function name="..."><param name="...">...</param></function>
|
||||
@@ -1206,7 +1206,14 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
src.find("<function name=\"") != std::string::npos &&
|
||||
src.find("<param name=\"") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: MiniCPM5\n");
|
||||
return common_chat_params_init_minicpm5(tmpl, params);
|
||||
return common_chat_params_init_minicpm5;
|
||||
}
|
||||
|
||||
// TranslateGemma - user content must follow a custom schema with language codes
|
||||
if (src.find("[source_lang_code]") != std::string::npos &&
|
||||
src.find("[target_lang_code]") != std::string::npos) {
|
||||
LOG_DBG("Using specialized template: TranslateGemma\n");
|
||||
return common_chat_params_init_translate_gemma;
|
||||
}
|
||||
|
||||
// Qwen3-Coder XML tool calls, also used by Nemotron Nano 3, Qwen3.5 and StepFun-3.5-Flash
|
||||
@@ -1216,10 +1223,51 @@ std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
// Exclude models that don't use \n between tags
|
||||
src.find("'<tool_call><function=' ~ tool_call.name ~ '>'") == std::string::npos) {
|
||||
LOG_DBG("Using specialized template: Qwen3-Coder\n");
|
||||
return common_chat_params_init_qwen3_coder(tmpl, params);
|
||||
return common_chat_params_init_qwen3_coder;
|
||||
}
|
||||
|
||||
return std::nullopt;
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
common_chat_template::common_chat_template(const std::string & src, const std::string & bos_token, const std::string & eos_token) {
|
||||
jinja::lexer lexer;
|
||||
auto lexer_res = lexer.tokenize(src);
|
||||
this->prog = jinja::parse_from_tokens(lexer_res);
|
||||
|
||||
this->src = lexer_res.source;
|
||||
this->bos_tok = bos_token;
|
||||
this->eos_tok = eos_token;
|
||||
|
||||
this->caps = jinja::caps_get(prog);
|
||||
// LOG_INF("%s: caps:\n%s\n", __func__, this->caps.to_string().c_str());
|
||||
|
||||
this->params_init = common_chat_template_detect_params_init(this->src);
|
||||
if (this->params_init) {
|
||||
return;
|
||||
}
|
||||
|
||||
// The analysis depends only on the template, so run it once here instead of on every apply.
|
||||
// A failure is kept for apply to report, so a bad template still loads like it did before.
|
||||
try {
|
||||
analysis = std::make_unique<autoparser::autoparser>();
|
||||
analysis->analyze_template(*this);
|
||||
} catch (const std::exception & e) {
|
||||
analysis.reset();
|
||||
analysis_error = e.what();
|
||||
}
|
||||
}
|
||||
|
||||
common_chat_template::~common_chat_template() = default;
|
||||
common_chat_template::common_chat_template(common_chat_template &&) = default;
|
||||
common_chat_template & common_chat_template::operator=(common_chat_template &&) = default;
|
||||
|
||||
std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & params) {
|
||||
if (!tmpl.params_init) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return tmpl.params_init(tmpl, params);
|
||||
}
|
||||
|
||||
static common_chat_params common_chat_templates_apply_jinja(const struct common_chat_templates * tmpls,
|
||||
@@ -1321,21 +1369,23 @@ static common_chat_params common_chat_templates_apply_jinja(const struct common_
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, params_copy);
|
||||
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, params);
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
auto parser = build_chat_peg_parser([&data](common_chat_peg_builder &p) {
|
||||
data.parser = build_chat_peg_parser([&data](common_chat_peg_builder &p) {
|
||||
return p.literal(data.generation_prompt) << p.content(p.rest());
|
||||
});
|
||||
data.parser = parser.save();
|
||||
return data;
|
||||
}
|
||||
|
||||
if (auto result = common_chat_try_specialized_template(tmpl, src, params)) {
|
||||
if (auto result = common_chat_try_specialized_template(tmpl, params)) {
|
||||
return *result;
|
||||
}
|
||||
|
||||
if (!tmpl.analysis) {
|
||||
throw std::invalid_argument("Unable to generate parser for this template. Automatic parser generation failed: " + tmpl.analysis_error);
|
||||
}
|
||||
|
||||
try {
|
||||
LOG_DBG("%s: using differential autoparser\n", __func__);
|
||||
struct autoparser::autoparser autoparser;
|
||||
autoparser.analyze_template(tmpl);
|
||||
const auto & autoparser = *tmpl.analysis;
|
||||
auto auto_params = autoparser::peg_generator::generate_parser(tmpl, params, autoparser);
|
||||
|
||||
common_chat_msg_delimiters delimiters;
|
||||
@@ -1356,8 +1406,7 @@ static common_chat_params common_chat_templates_apply_jinja(const struct common_
|
||||
auto_params.thinking_end_tags = {std::move(end_tag)};
|
||||
}
|
||||
}
|
||||
common_peg_arena arena;
|
||||
arena.load(auto_params.parser);
|
||||
const auto & arena = auto_params.parser;
|
||||
LOG_DBG("%s: generated parser:\n%s\n\nparser generation prompt: %s\n", __func__, arena.dump(arena.root()).c_str(), auto_params.generation_prompt.c_str());
|
||||
return auto_params;
|
||||
} catch (const std::exception & e) {
|
||||
@@ -1438,36 +1487,92 @@ common_chat_params common_chat_templates_apply(const struct common_chat_template
|
||||
common_chat_templates_apply_legacy(tmpls, inputs);
|
||||
}
|
||||
|
||||
common_chat_msg common_chat_parse(const std::string & input,
|
||||
void common_chat_input::append(const std::string & piece, llama_token token) {
|
||||
if (piece.empty()) {
|
||||
return;
|
||||
}
|
||||
tokens.push_back(token);
|
||||
tokens.resize(tokens.size() + piece.size() - 1, LLAMA_TOKEN_NULL);
|
||||
text += piece;
|
||||
}
|
||||
|
||||
void common_chat_input::append(const common_chat_input & chunk) {
|
||||
tokens.insert(tokens.end(), chunk.tokens.begin(), chunk.tokens.end());
|
||||
text += chunk.text;
|
||||
}
|
||||
|
||||
void common_chat_input::truncate(size_t pos) {
|
||||
if (pos < text.size()) {
|
||||
text.erase(pos);
|
||||
tokens.resize(pos);
|
||||
}
|
||||
}
|
||||
|
||||
common_chat_input common_chat_input::substr(size_t pos, size_t n) const {
|
||||
common_chat_input out;
|
||||
out.text = text.substr(pos, n);
|
||||
out.tokens.assign(tokens.begin() + pos, tokens.begin() + pos + out.size());
|
||||
return out;
|
||||
}
|
||||
|
||||
void common_chat_input::prepend(const std::string & prefix) {
|
||||
tokens.insert(tokens.begin(), prefix.size(), LLAMA_TOKEN_NULL);
|
||||
text = prefix + text;
|
||||
}
|
||||
|
||||
void common_chat_input::prepend(const common_chat_input & prefix) {
|
||||
tokens.insert(tokens.begin(), prefix.tokens.begin(), prefix.tokens.end());
|
||||
text = prefix.text + text;
|
||||
}
|
||||
|
||||
common_chat_input common_chat_input_tokenize(const llama_vocab * vocab, const std::string & text) {
|
||||
common_chat_input input;
|
||||
auto tokens = common_tokenize(vocab, text, false, true);
|
||||
for (size_t i = 0; i < tokens.size(); i++) {
|
||||
std::string piece = common_token_to_piece(vocab, tokens[i], true);
|
||||
if (i == 0 && std::isspace(piece[0]) && !std::isspace(text[0])) {
|
||||
// Some tokenizers will add a space before the first special token, need to exclude
|
||||
continue;
|
||||
}
|
||||
input.append(piece, tokens[i]);
|
||||
}
|
||||
if (input.text != text) {
|
||||
// the pieces do not give back the same text, keep the text without tokens
|
||||
return common_chat_input(text);
|
||||
}
|
||||
return input;
|
||||
}
|
||||
|
||||
common_chat_msg common_chat_parse(const common_chat_input & input,
|
||||
bool is_partial,
|
||||
const common_chat_parser_params & params) {
|
||||
return common_chat_peg_parse(params.parser, input, is_partial, params);
|
||||
}
|
||||
|
||||
common_chat_msg common_chat_peg_parse(const common_peg_arena & src_parser,
|
||||
const std::string & input,
|
||||
const common_chat_input & input,
|
||||
bool is_partial,
|
||||
const common_chat_parser_params & params) {
|
||||
const common_peg_arena & parser = src_parser.empty() ?
|
||||
build_chat_peg_parser([](common_chat_peg_builder & p) { return p.content(p.rest()) + p.end(); }) :
|
||||
src_parser;
|
||||
// both branches must be lvalues, a temporary here would copy the arena on every call
|
||||
static const common_peg_arena content_only =
|
||||
build_chat_peg_parser([](common_chat_peg_builder & p) { return p.content(p.rest()) + p.end(); });
|
||||
const common_peg_arena & parser = src_parser.empty() ? content_only : src_parser;
|
||||
|
||||
if (src_parser.empty()) {
|
||||
LOG_DBG("No parser definition detected, assuming pure content parser.");
|
||||
}
|
||||
|
||||
const std::string effective_input = params.generation_prompt.empty()
|
||||
? input
|
||||
: params.generation_prompt + input;
|
||||
common_chat_input effective_input = input;
|
||||
effective_input.prepend(params.generation_prompt);
|
||||
|
||||
//LOG_DBG("Parsing PEG input with format %s: %s\n", common_chat_format_name(params.format), effective_input.c_str());
|
||||
//LOG_DBG("Parsing PEG input with format %s: %s\n", common_chat_format_name(params.format), effective_input.text.c_str());
|
||||
|
||||
common_peg_parse_flags flags = COMMON_PEG_PARSE_FLAG_LENIENT;
|
||||
if (params.debug) {
|
||||
flags |= COMMON_PEG_PARSE_FLAG_DEBUG;
|
||||
}
|
||||
|
||||
common_peg_parse_context ctx(effective_input, flags);
|
||||
common_peg_parse_context ctx(std::move(effective_input.text), std::move(effective_input.tokens), flags);
|
||||
auto result = parser.parse(ctx);
|
||||
|
||||
if (result.fail()) {
|
||||
@@ -1493,8 +1598,8 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars
|
||||
}
|
||||
return msg;
|
||||
}
|
||||
LOG_WRN("%s: unparsed %s output: %s\n", __func__, common_chat_format_name(params.format), effective_input.substr(result.end).c_str());
|
||||
LOG_DBG("%s: full %s output triggering error:\n=== BEGIN ===\n%s\n=== END ===\n", __func__, common_chat_format_name(params.format), effective_input.c_str());
|
||||
LOG_WRN("%s: unparsed %s output: %s\n", __func__, common_chat_format_name(params.format), ctx.input.substr(result.end).c_str());
|
||||
LOG_DBG("%s: full %s output triggering error:\n=== BEGIN ===\n%s\n=== END ===\n", __func__, common_chat_format_name(params.format), ctx.input.c_str());
|
||||
throw std::runtime_error(std::string("The model produced output that does not match the expected ") + common_chat_format_name(params.format) + " format");
|
||||
}
|
||||
|
||||
@@ -1522,6 +1627,84 @@ common_chat_msg common_chat_peg_parse(const common_peg_arena & src_pars
|
||||
return msg;
|
||||
}
|
||||
|
||||
common_chat_session::common_chat_session(const common_chat_templates * tmpls,
|
||||
const llama_vocab * vocab,
|
||||
const common_chat_templates_inputs & inputs,
|
||||
const common_chat_session_params & params) {
|
||||
auto applied = common_chat_templates_apply(tmpls, inputs);
|
||||
|
||||
templated = true;
|
||||
prompt_text = std::move(applied.prompt);
|
||||
result.role = "assistant";
|
||||
|
||||
grammar_text = std::move(applied.grammar);
|
||||
grammar_lazy = applied.grammar_lazy;
|
||||
stops = std::move(applied.additional_stops);
|
||||
generation_prompt_text = applied.generation_prompt;
|
||||
thinking_start = std::move(applied.thinking_start_tag);
|
||||
thinking_ends = std::move(applied.thinking_end_tags);
|
||||
|
||||
parser_params.format = applied.format;
|
||||
parser_params.generation_prompt = vocab ? common_chat_input_tokenize(vocab, applied.generation_prompt)
|
||||
: common_chat_input(applied.generation_prompt);
|
||||
parser_params.debug = params.debug;
|
||||
parser_params.parser = std::move(applied.parser);
|
||||
|
||||
delimiters = std::move(applied.message_delimiters);
|
||||
|
||||
if (vocab) {
|
||||
common_params_sampling resolved;
|
||||
resolved.grammar_lazy = applied.grammar_lazy;
|
||||
common_sampling_add_preserved_tokens(resolved, vocab, applied.preserved_tokens);
|
||||
common_sampling_add_grammar_triggers(resolved, vocab, std::move(applied.grammar_triggers));
|
||||
preserved_tokens = std::move(resolved.preserved_tokens);
|
||||
grammar_triggers = std::move(resolved.grammar_triggers);
|
||||
|
||||
delimiters.tokenize(vocab);
|
||||
} else {
|
||||
grammar_triggers = std::move(applied.grammar_triggers);
|
||||
}
|
||||
|
||||
if (inputs.continue_final_message != COMMON_CHAT_CONTINUATION_NONE && !params.echo) {
|
||||
// start from the prefill so it is not emitted as part of the first delta
|
||||
result = common_chat_parse(input, true, parser_params);
|
||||
}
|
||||
}
|
||||
|
||||
void common_chat_session::apply_sampling(common_params_sampling & sampling) const {
|
||||
if (!templated) {
|
||||
return;
|
||||
}
|
||||
if (!grammar_text.empty()) {
|
||||
sampling.grammar = {COMMON_GRAMMAR_TYPE_TOOL_CALLS, grammar_text};
|
||||
}
|
||||
sampling.grammar_lazy = grammar_lazy;
|
||||
sampling.generation_prompt = generation_prompt_text;
|
||||
sampling.preserved_tokens.insert(preserved_tokens.begin(), preserved_tokens.end());
|
||||
sampling.grammar_triggers.insert(sampling.grammar_triggers.end(), grammar_triggers.begin(), grammar_triggers.end());
|
||||
}
|
||||
|
||||
const common_chat_msg & common_chat_session::feed(const common_chat_input & chunk) {
|
||||
GGML_ASSERT(!finished && "feed() after finish()");
|
||||
input.append(chunk);
|
||||
auto msg = common_chat_parse(input, true, parser_params);
|
||||
if (!msg.empty()) {
|
||||
result = std::move(msg);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
const common_chat_msg & common_chat_session::finish(const common_chat_input & chunk) {
|
||||
GGML_ASSERT(!finished && "finish() called twice");
|
||||
finished = true;
|
||||
input.append(chunk);
|
||||
auto msg = common_chat_parse(input, false, parser_params);
|
||||
if (!msg.empty()) {
|
||||
result = std::move(msg);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
std::map<std::string, bool> common_chat_templates_get_caps(const common_chat_templates * chat_templates) {
|
||||
GGML_ASSERT(chat_templates != nullptr);
|
||||
GGML_ASSERT(chat_templates->template_default != nullptr);
|
||||
|
||||
+107
-30
@@ -22,8 +22,16 @@ struct common_chat_templates;
|
||||
|
||||
namespace autoparser {
|
||||
struct generation_params;
|
||||
struct autoparser;
|
||||
} // namespace autoparser
|
||||
|
||||
struct common_chat_params;
|
||||
struct common_chat_template;
|
||||
|
||||
// Builds the prompt and parser for a template that has a dedicated handler (see common/parsers)
|
||||
using common_chat_params_init_fn = common_chat_params (*)(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs);
|
||||
|
||||
struct common_chat_tool_call {
|
||||
std::string name;
|
||||
std::string arguments;
|
||||
@@ -54,19 +62,20 @@ struct common_chat_template {
|
||||
std::string eos_tok;
|
||||
std::string src;
|
||||
chat_template_caps caps;
|
||||
// Dedicated handler picked once from the source, null when the differential autoparser is used
|
||||
common_chat_params_init_fn params_init = nullptr;
|
||||
|
||||
common_chat_template(const std::string & src, const std::string & bos_token, const std::string & eos_token) {
|
||||
jinja::lexer lexer;
|
||||
auto lexer_res = lexer.tokenize(src);
|
||||
this->prog = jinja::parse_from_tokens(lexer_res);
|
||||
// Differential analysis, run once here when there is no dedicated handler. Null when there
|
||||
// is one, or when the analysis failed, in which case analysis_error says why.
|
||||
std::unique_ptr<autoparser::autoparser> analysis;
|
||||
std::string analysis_error;
|
||||
|
||||
this->src = lexer_res.source;
|
||||
this->bos_tok = bos_token;
|
||||
this->eos_tok = eos_token;
|
||||
common_chat_template(const std::string & src, const std::string & bos_token, const std::string & eos_token);
|
||||
|
||||
this->caps = jinja::caps_get(prog);
|
||||
// LOG_INF("%s: caps:\n%s\n", __func__, this->caps.to_string().c_str());
|
||||
}
|
||||
// autoparser is incomplete here, so these are defined where it is complete
|
||||
~common_chat_template();
|
||||
common_chat_template(common_chat_template &&);
|
||||
common_chat_template & operator=(common_chat_template &&);
|
||||
|
||||
const std::string & source() const { return src; }
|
||||
const std::string & bos_token() const { return bos_tok; }
|
||||
@@ -209,8 +218,6 @@ struct common_chat_msg_delimiters {
|
||||
|
||||
// split tokens into message spans. skips maps a start index to a length of a region to jump over without matching
|
||||
common_chat_msg_spans split(const llama_tokens & tokens, const std::map<size_t, size_t> & skips = {}) const;
|
||||
|
||||
common_json to_json() const;
|
||||
};
|
||||
|
||||
struct common_chat_tool {
|
||||
@@ -278,27 +285,46 @@ struct common_chat_params {
|
||||
std::vector<common_grammar_trigger> grammar_triggers;
|
||||
std::vector<std::string> preserved_tokens;
|
||||
std::vector<std::string> additional_stops;
|
||||
std::string parser;
|
||||
common_peg_arena parser;
|
||||
common_chat_msg_delimiters message_delimiters;
|
||||
};
|
||||
|
||||
struct common_chat_input {
|
||||
std::string text;
|
||||
std::vector<llama_token> tokens;
|
||||
|
||||
common_chat_input() = default;
|
||||
|
||||
// plain text, with no tokens
|
||||
explicit common_chat_input(std::string text) : text(std::move(text)), tokens(this->text.size(), LLAMA_TOKEN_NULL) {}
|
||||
|
||||
size_t size() const { return text.size(); }
|
||||
bool empty() const { return text.empty(); }
|
||||
|
||||
void append(const std::string & piece, llama_token token);
|
||||
void append(const common_chat_input & chunk);
|
||||
|
||||
void prepend(const std::string & prefix);
|
||||
void prepend(const common_chat_input & prefix);
|
||||
|
||||
void truncate(size_t pos);
|
||||
|
||||
common_chat_input substr(size_t pos, size_t n = std::string::npos) const;
|
||||
};
|
||||
|
||||
common_chat_input common_chat_input_tokenize(const llama_vocab * vocab, const std::string & text);
|
||||
|
||||
// per-message parsing syntax
|
||||
// should be derived from common_chat_params
|
||||
struct common_chat_parser_params {
|
||||
common_chat_format format = COMMON_CHAT_FORMAT_CONTENT_ONLY;
|
||||
common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_NONE; // TODO: refactor this to "bool parse_reasoning"
|
||||
// Whether reasoning_content should be inlined in the content (e.g. for reasoning_format=deepseek in stream mode)
|
||||
bool reasoning_in_content = false;
|
||||
std::string generation_prompt;
|
||||
bool parse_tool_calls = true;
|
||||
bool is_continuation = false;
|
||||
bool echo = false; // Include assistant prefilled msg in output
|
||||
bool debug = false; // Enable debug output for PEG parser
|
||||
common_peg_arena parser = {};
|
||||
common_chat_format format = COMMON_CHAT_FORMAT_CONTENT_ONLY;
|
||||
common_chat_input generation_prompt;
|
||||
bool debug = false; // Enable debug output for PEG parser
|
||||
common_peg_arena parser = {};
|
||||
common_chat_parser_params() = default;
|
||||
common_chat_parser_params(const common_chat_params & chat_params) {
|
||||
format = chat_params.format;
|
||||
generation_prompt = chat_params.generation_prompt;
|
||||
generation_prompt = common_chat_input(chat_params.generation_prompt);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -337,8 +363,62 @@ std::string common_chat_format_example(const struct common_chat_templates *
|
||||
const std::map<std::string, std::string> & chat_template_kwargs);
|
||||
|
||||
const char * common_chat_format_name(common_chat_format format);
|
||||
common_chat_msg common_chat_parse(const std::string & input, bool is_partial, const common_chat_parser_params & params);
|
||||
common_chat_msg common_chat_peg_parse(const common_peg_arena & src_parser, const std::string & input, bool is_partial, const common_chat_parser_params & params);
|
||||
common_chat_msg common_chat_parse(const common_chat_input & input, bool is_partial, const common_chat_parser_params & params);
|
||||
common_chat_msg common_chat_peg_parse(const common_peg_arena & src_parser, const common_chat_input & input, bool is_partial, const common_chat_parser_params & params);
|
||||
|
||||
struct common_chat_session_params {
|
||||
bool echo = false; // include the assistant prefill in the output when continuing a message
|
||||
bool debug = false; // enable debug output for the PEG parser
|
||||
};
|
||||
|
||||
class common_chat_session {
|
||||
public:
|
||||
common_chat_session() { result.role = "assistant"; }
|
||||
|
||||
common_chat_session(const common_chat_templates * tmpls,
|
||||
const llama_vocab * vocab,
|
||||
const common_chat_templates_inputs & inputs,
|
||||
const common_chat_session_params & params = {});
|
||||
|
||||
const std::string & prompt() const { return prompt_text; }
|
||||
common_chat_format format() const { return parser_params.format; }
|
||||
const common_chat_msg & msg() const { return result; }
|
||||
const common_peg_arena & parser() const { return parser_params.parser; }
|
||||
|
||||
const std::string & grammar() const { return grammar_text; }
|
||||
const std::string & generation_prompt() const { return generation_prompt_text; }
|
||||
const std::string & thinking_start_tag() const { return thinking_start; }
|
||||
const std::vector<std::string> & thinking_end_tags() const { return thinking_ends; }
|
||||
const std::vector<std::string> & additional_stops() const { return stops; }
|
||||
|
||||
const common_chat_msg_delimiters & message_delimiters() const { return delimiters; }
|
||||
|
||||
void apply_sampling(common_params_sampling & sampling) const;
|
||||
|
||||
bool has_template() const { return templated; }
|
||||
|
||||
const common_chat_msg & feed(const common_chat_input & chunk);
|
||||
|
||||
const common_chat_msg & finish(const common_chat_input & chunk = {});
|
||||
|
||||
private:
|
||||
std::string prompt_text;
|
||||
std::string grammar_text;
|
||||
bool grammar_lazy = false;
|
||||
std::vector<common_grammar_trigger> grammar_triggers;
|
||||
std::set<llama_token> preserved_tokens;
|
||||
std::vector<std::string> stops;
|
||||
std::string generation_prompt_text;
|
||||
std::string thinking_start;
|
||||
std::vector<std::string> thinking_ends;
|
||||
|
||||
common_chat_parser_params parser_params;
|
||||
common_chat_msg_delimiters delimiters;
|
||||
common_chat_input input;
|
||||
common_chat_msg result;
|
||||
bool templated = false;
|
||||
bool finished = false;
|
||||
};
|
||||
|
||||
// used by arg and server
|
||||
const char * common_reasoning_format_name(common_reasoning_format format);
|
||||
@@ -376,8 +456,7 @@ std::string common_chat_template_generation_prompt(
|
||||
|
||||
std::optional<common_chat_params> common_chat_try_specialized_template(
|
||||
const common_chat_template & tmpl,
|
||||
const std::string & src,
|
||||
autoparser::generation_params & params);
|
||||
const autoparser::generation_params & params);
|
||||
|
||||
|
||||
// specialized per-task preset
|
||||
@@ -387,5 +466,3 @@ struct common_chat_prompt_preset {
|
||||
};
|
||||
|
||||
common_chat_prompt_preset common_chat_get_asr_prompt(const common_chat_templates * chat_templates);
|
||||
|
||||
common_chat_msg_delimiters common_chat_msg_delimiters_parse(const common_json & delimiters);
|
||||
|
||||
+194
-164
@@ -1,8 +1,12 @@
|
||||
#include "ggml.h"
|
||||
#include "ggml-cpp.h"
|
||||
#include "gguf.h"
|
||||
|
||||
#include "build-info.h"
|
||||
#include "common.h"
|
||||
|
||||
#include "../src/llama-ext.h"
|
||||
|
||||
#include "fit.h"
|
||||
#include "log.h"
|
||||
#include "llama.h"
|
||||
@@ -1023,70 +1027,25 @@ std::filesystem::path fs_get_cache_file(const std::string & filename) {
|
||||
GGML_ASSERT(filename.find(DIRECTORY_SEPARATOR) == std::string::npos);
|
||||
const std::filesystem::path cache_directory = fs_get_cache_directory();
|
||||
std::error_code ec;
|
||||
std::filesystem::create_directories(cache_directory, ec);
|
||||
common_create_directories(cache_directory, ec);
|
||||
if (ec) {
|
||||
throw std::runtime_error("failed to create cache directory: " + fs_path_to_utf8(cache_directory));
|
||||
}
|
||||
return cache_directory / std::filesystem::u8path(filename);
|
||||
}
|
||||
|
||||
std::vector<common_file_info> fs_list(const std::string & path, bool include_directories) {
|
||||
std::vector<common_file_info> files;
|
||||
if (path.empty()) return files;
|
||||
|
||||
std::filesystem::path dir(path);
|
||||
if (!std::filesystem::exists(dir) || !std::filesystem::is_directory(dir)) {
|
||||
return files;
|
||||
}
|
||||
|
||||
for (const auto & entry : std::filesystem::directory_iterator(dir)) {
|
||||
try {
|
||||
// Only include regular files (skip directories)
|
||||
const auto & p = entry.path();
|
||||
if (std::filesystem::is_regular_file(p)) {
|
||||
common_file_info info;
|
||||
info.path = p.string();
|
||||
info.name = p.filename().string();
|
||||
info.is_dir = false;
|
||||
try {
|
||||
info.size = static_cast<size_t>(std::filesystem::file_size(p));
|
||||
} catch (const std::filesystem::filesystem_error &) {
|
||||
info.size = 0;
|
||||
}
|
||||
files.push_back(std::move(info));
|
||||
} else if (include_directories && std::filesystem::is_directory(p)) {
|
||||
common_file_info info;
|
||||
info.path = p.string();
|
||||
info.name = p.filename().string();
|
||||
info.size = 0; // Directories have no size
|
||||
info.is_dir = true;
|
||||
files.push_back(std::move(info));
|
||||
}
|
||||
} catch (const std::filesystem::filesystem_error &) {
|
||||
// skip entries we cannot inspect
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
return files;
|
||||
}
|
||||
|
||||
std::ifstream fs_open_ifstream(const std::string & fname, std::ios_base::openmode mode) {
|
||||
#ifdef _WIN32
|
||||
int wlen = MultiByteToWideChar(CP_UTF8, 0, fname.c_str(), -1, NULL, 0);
|
||||
if (!wlen) { return std::ifstream(); }
|
||||
std::vector<wchar_t> wfname(wlen);
|
||||
(void)MultiByteToWideChar(CP_UTF8, 0, fname.c_str(), -1, wfname.data(), wlen);
|
||||
return std::ifstream(wfname.data(), mode);
|
||||
#else
|
||||
return std::ifstream(fname, mode);
|
||||
#endif
|
||||
}
|
||||
|
||||
//
|
||||
// TTY utils
|
||||
//
|
||||
|
||||
bool common_is_tty(FILE * file) {
|
||||
#if defined(_WIN32)
|
||||
return _isatty(_fileno(file));
|
||||
#else
|
||||
return isatty(fileno(file));
|
||||
#endif
|
||||
}
|
||||
|
||||
bool tty_can_use_colors() {
|
||||
// Check NO_COLOR environment variable (https://no-color.org/)
|
||||
if (const char * no_color = std::getenv("NO_COLOR")) {
|
||||
@@ -1104,10 +1063,21 @@ bool tty_can_use_colors() {
|
||||
|
||||
// Check if stdout and stderr are connected to a terminal
|
||||
// We check both because log messages can go to either
|
||||
bool stdout_is_tty = isatty(fileno(stdout));
|
||||
bool stderr_is_tty = isatty(fileno(stderr));
|
||||
return common_is_tty(stdout) || common_is_tty(stderr);
|
||||
}
|
||||
|
||||
return stdout_is_tty || stderr_is_tty;
|
||||
bool tty_enable_ansi() {
|
||||
#if defined(_WIN32)
|
||||
// a Windows console renders ANSI sequences only in virtual terminal mode, pipes and files take them as is
|
||||
for (DWORD id : { STD_OUTPUT_HANDLE, STD_ERROR_HANDLE }) {
|
||||
HANDLE h = GetStdHandle(id);
|
||||
DWORD mode = 0;
|
||||
if (GetConsoleMode(h, &mode) && !SetConsoleMode(h, mode | ENABLE_VIRTUAL_TERMINAL_PROCESSING)) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
#endif
|
||||
return true;
|
||||
}
|
||||
|
||||
//
|
||||
@@ -1192,6 +1162,77 @@ struct common_init_result::impl {
|
||||
std::vector<llama_sampler_seq_config> samplers_seq_config;
|
||||
};
|
||||
|
||||
static const std::map<common_decision_type, std::string> COMMON_DECISION_TYPE_NAMES = {
|
||||
{ COMMON_DECISION_TYPE_OPENJEV, "openjev" },
|
||||
{ COMMON_DECISION_TYPE_LEV, "lev" },
|
||||
{ COMMON_DECISION_TYPE_KEV, "kev" },
|
||||
{ COMMON_DECISION_TYPE_NIMBLE, "nimble" },
|
||||
{ COMMON_DECISION_TYPE_LAYA, "laya" },
|
||||
{ COMMON_DECISION_TYPE_CLEF, "clef" },
|
||||
{ COMMON_DECISION_TYPE_PPLX_DECIDER, "pplx-decider" },
|
||||
{ COMMON_DECISION_TYPE_LFM2_D1, "lfm2-d1" },
|
||||
{ COMMON_DECISION_TYPE_LFM2_D1_OMNI, "lfm2-d1-omni" },
|
||||
};
|
||||
|
||||
static common_decision_type common_decision_type_from_string(const std::string & str) {
|
||||
for (const auto & pair : COMMON_DECISION_TYPE_NAMES) {
|
||||
if (pair.second == str) {
|
||||
return pair.first;
|
||||
}
|
||||
}
|
||||
return COMMON_DECISION_TYPE_UNKNOWN;
|
||||
}
|
||||
|
||||
common_decision_type common_get_decision_type(const struct llama_model * model) {
|
||||
char buf[64];
|
||||
if (llama_model_meta_val_str(model, "general.architecture", buf, sizeof(buf)) < 0) {
|
||||
return COMMON_DECISION_TYPE_NONE;
|
||||
}
|
||||
const std::string key = std::string(buf) + ".decision.type";
|
||||
if (llama_model_meta_val_str(model, key.c_str(), buf, sizeof(buf)) < 0) {
|
||||
return COMMON_DECISION_TYPE_NONE;
|
||||
}
|
||||
return common_decision_type_from_string(buf);
|
||||
}
|
||||
|
||||
common_gguf_info common_get_gguf_info(const std::string & fname) {
|
||||
common_gguf_info info;
|
||||
|
||||
struct gguf_init_params gguf_params = {
|
||||
/* .no_alloc = */ true,
|
||||
/* .ctx = */ nullptr,
|
||||
};
|
||||
|
||||
gguf_context_ptr gguf_ctx(gguf_init_from_file(fname.c_str(), gguf_params));
|
||||
if (!gguf_ctx) {
|
||||
return info; // missing or unreadable file
|
||||
}
|
||||
|
||||
const int64_t arch_id = gguf_find_key(gguf_ctx.get(), "general.architecture");
|
||||
if (arch_id < 0 || gguf_get_kv_type(gguf_ctx.get(), arch_id) != GGUF_TYPE_STRING) {
|
||||
return info; // no architecture in the metadata
|
||||
}
|
||||
const std::string arch = gguf_get_val_str(gguf_ctx.get(), arch_id);
|
||||
if (arch.empty()) {
|
||||
return info;
|
||||
}
|
||||
|
||||
const int64_t type_id = gguf_find_key(gguf_ctx.get(), (arch + ".decision.type").c_str());
|
||||
if (type_id < 0) {
|
||||
info.decision_type = COMMON_DECISION_TYPE_NONE;
|
||||
} else if (gguf_get_kv_type(gguf_ctx.get(), type_id) == GGUF_TYPE_STRING) {
|
||||
info.decision_type = common_decision_type_from_string(gguf_get_val_str(gguf_ctx.get(), type_id));
|
||||
}
|
||||
|
||||
// same key and type as the model loader
|
||||
const int64_t ctx_id = gguf_find_key(gguf_ctx.get(), (arch + ".context_length").c_str());
|
||||
if (ctx_id >= 0 && gguf_get_kv_type(gguf_ctx.get(), ctx_id) == GGUF_TYPE_UINT32) {
|
||||
info.n_ctx_train = gguf_get_val_u32(gguf_ctx.get(), ctx_id);
|
||||
}
|
||||
|
||||
return info;
|
||||
}
|
||||
|
||||
common_init_result::common_init_result(common_params & params, bool model_only) :
|
||||
pimpl(new impl{}) {
|
||||
auto mparams = common_model_params_to_llama(params);
|
||||
@@ -1244,6 +1285,30 @@ common_init_result::common_init_result(common_params & params, bool model_only)
|
||||
|
||||
const llama_vocab * vocab = llama_model_get_vocab(model);
|
||||
|
||||
// these decision models return a score for each token via the embeddings output
|
||||
// TODO: maybe improve this in the future
|
||||
const auto decision_type = common_get_decision_type(model);
|
||||
if (decision_type == COMMON_DECISION_TYPE_LAYA || decision_type == COMMON_DECISION_TYPE_KEV || decision_type == COMMON_DECISION_TYPE_CLEF ||
|
||||
decision_type == COMMON_DECISION_TYPE_LFM2_D1_OMNI) {
|
||||
params.embedding = true;
|
||||
params.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
|
||||
cparams.embeddings = true;
|
||||
cparams.pooling_type = LLAMA_POOLING_TYPE_NONE;
|
||||
cparams.n_outputs_max = cparams.n_batch;
|
||||
cparams.n_outputs_max_per_seq = 1;
|
||||
|
||||
LOG_INF("%s", "decision model reads the embeddings output, enabling embedding mode\n");
|
||||
}
|
||||
|
||||
// embeddings need the whole batch in one ubatch, so n_batch must not be larger than n_ubatch
|
||||
// (server.cpp does this check for --embedding, but before the model is loaded)
|
||||
if (cparams.embeddings && cparams.n_batch > cparams.n_ubatch) {
|
||||
LOG_WRN("embeddings enabled: setting n_batch = n_ubatch = %u\n", cparams.n_ubatch);
|
||||
cparams.n_batch = cparams.n_ubatch;
|
||||
params.n_batch = params.n_ubatch;
|
||||
}
|
||||
|
||||
// load and optionally apply lora adapters
|
||||
for (auto & la : params.lora_adapters) {
|
||||
llama_adapter_lora_ptr lora;
|
||||
@@ -1622,7 +1687,7 @@ struct llama_model_params common_model_params_to_llama(common_params & params) {
|
||||
mparams.progress_callback = params.load_progress_callback;
|
||||
mparams.progress_callback_user_data = params.load_progress_callback_user_data;
|
||||
mparams.no_alloc = params.no_alloc;
|
||||
mparams.load_mtp = std::find(params.speculative.types.begin(), params.speculative.types.end(), COMMON_SPECULATIVE_TYPE_DRAFT_MTP) != params.speculative.types.end();
|
||||
mparams.load_mtp = params.load_mtp || std::find(params.speculative.types.begin(), params.speculative.types.end(), COMMON_SPECULATIVE_TYPE_DRAFT_MTP) != params.speculative.types.end();
|
||||
|
||||
return mparams;
|
||||
}
|
||||
@@ -1663,6 +1728,8 @@ struct llama_context_params common_context_params_to_llama(const common_params &
|
||||
cparams.type_k = params.cache_type_k;
|
||||
cparams.type_v = params.cache_type_v;
|
||||
|
||||
cparams.moe_cache_size = params.moe_cache_size;
|
||||
|
||||
return cparams;
|
||||
}
|
||||
|
||||
@@ -1738,33 +1805,6 @@ void common_threadpools::init(llama_context * ctx, const common_params & params)
|
||||
llama_attach_threadpool(ctx, threadpool, threadpool_batch);
|
||||
}
|
||||
|
||||
//
|
||||
// Batch utils
|
||||
//
|
||||
|
||||
void common_batch_clear(struct llama_batch & batch) {
|
||||
batch.n_tokens = 0;
|
||||
}
|
||||
|
||||
void common_batch_add(
|
||||
struct llama_batch & batch,
|
||||
llama_token id,
|
||||
llama_pos pos,
|
||||
const std::vector<llama_seq_id> & seq_ids,
|
||||
bool logits) {
|
||||
GGML_ASSERT(batch.seq_id[batch.n_tokens] && "llama_batch size exceeded");
|
||||
|
||||
batch.token [batch.n_tokens] = id;
|
||||
batch.pos [batch.n_tokens] = pos;
|
||||
batch.n_seq_id[batch.n_tokens] = seq_ids.size();
|
||||
for (size_t i = 0; i < seq_ids.size(); ++i) {
|
||||
batch.seq_id[batch.n_tokens][i] = seq_ids[i];
|
||||
}
|
||||
batch.logits [batch.n_tokens] = logits;
|
||||
|
||||
batch.n_tokens++;
|
||||
}
|
||||
|
||||
//
|
||||
// Vocab utils
|
||||
//
|
||||
@@ -2118,35 +2158,41 @@ common_batch::common_batch(llama_context * ctx) : batch(llama_batch_ext_init(ctx
|
||||
|
||||
void common_batch::clear() {
|
||||
tokens.clear();
|
||||
llama_batch_ext_clear(batch.get());
|
||||
}
|
||||
|
||||
int32_t common_batch::add(llama_token id, llama_pos pos, llama_seq_id seq_id, bool output) {
|
||||
const int32_t idx = llama_batch_ext_add_token(batch.get(), seq_id, id);
|
||||
if (idx < 0) {
|
||||
GGML_ABORT("%s: failed to add token %d to the batch (error %d, n_tokens = %d)\n", __func__, id, idx, size());
|
||||
tokens.push_back({ id, { pos, 0, 0, 0 }, seq_id, output, { nullptr, 0, 0 }, {} });
|
||||
return size() - 1;
|
||||
}
|
||||
|
||||
int32_t common_batch::add(llama_token id, llama_pos pos, const std::vector<llama_seq_id> & seq_ids, bool output) {
|
||||
GGML_ASSERT(!seq_ids.empty());
|
||||
|
||||
const int32_t idx = add(id, pos, seq_ids[0], output);
|
||||
for (size_t s = 1; s < seq_ids.size(); ++s) {
|
||||
add_seq(idx, seq_ids[s]);
|
||||
}
|
||||
llama_batch_ext_set_pos(batch.get(), idx, &pos);
|
||||
if (output) {
|
||||
llama_batch_ext_set_output_logits(batch.get(), idx, true);
|
||||
}
|
||||
tokens.push_back({ id, { pos, 0, 0, 0 }, seq_id, output, { nullptr, 0, 0 } });
|
||||
return idx;
|
||||
}
|
||||
|
||||
bool common_batch::add_seq(int32_t idx, llama_seq_id seq_id) {
|
||||
if (idx < 0 || idx >= size()) {
|
||||
return false;
|
||||
}
|
||||
tokens[idx].seq_ids_extra.push_back(seq_id);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool common_batch::set_output(int32_t idx, bool value) {
|
||||
if (idx < 0 || idx >= (int32_t) tokens.size()) {
|
||||
if (idx < 0 || idx >= size()) {
|
||||
return false;
|
||||
}
|
||||
tokens[idx].output = value;
|
||||
return llama_batch_ext_set_output_logits(batch.get(), idx, value);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool common_batch::set_embd(int32_t idx, llama_embd embd) {
|
||||
if (idx < 0 || idx >= (int32_t) tokens.size()) {
|
||||
return false;
|
||||
}
|
||||
if (!llama_batch_ext_set_embd_token(batch.get(), idx, embd)) {
|
||||
if (idx < 0 || idx >= size() || tokens[idx].embd.data != nullptr) {
|
||||
return false;
|
||||
}
|
||||
tokens[idx].embd = embd;
|
||||
@@ -2154,83 +2200,67 @@ bool common_batch::set_embd(int32_t idx, llama_embd embd) {
|
||||
}
|
||||
|
||||
int32_t common_batch::add_embd(llama_embd embd, const llama_pos * pos, llama_seq_id seq_id, bool output) {
|
||||
const int32_t idx = llama_batch_ext_add_embd(batch.get(), seq_id, embd);
|
||||
if (idx < 0) {
|
||||
GGML_ABORT("%s: failed to add embedding to the batch (error %d, n_tokens = %d)\n", __func__, idx, size());
|
||||
}
|
||||
llama_batch_ext_set_pos(batch.get(), idx, pos);
|
||||
if (output) {
|
||||
llama_batch_ext_set_output_logits(batch.get(), idx, true);
|
||||
}
|
||||
token t = { LLAMA_TOKEN_NULL, { 0, 0, 0, 0 }, seq_id, output, embd };
|
||||
token t = { LLAMA_TOKEN_NULL, { 0, 0, 0, 0 }, seq_id, output, embd, {} };
|
||||
for (int32_t j = 0; j < n_pos; ++j) {
|
||||
t.pos[j] = pos[j];
|
||||
}
|
||||
tokens.push_back(t);
|
||||
return idx;
|
||||
return size() - 1;
|
||||
}
|
||||
|
||||
common_batch common_batch_from_llama_batch(llama_context * ctx, const llama_batch & batch) {
|
||||
common_batch res(ctx);
|
||||
llama_batch_ext * common_batch::get_sub_batch(int32_t off, int32_t n) {
|
||||
GGML_ASSERT(batch && "common_batch was not initialized with a context");
|
||||
GGML_ASSERT(off >= 0 && n >= 0 && off + n <= size());
|
||||
|
||||
const bool has_token = batch.token != nullptr;
|
||||
const bool has_embd = batch.embd != nullptr;
|
||||
llama_batch_ext * res = batch.get();
|
||||
llama_batch_ext_clear(res);
|
||||
|
||||
const size_t n_embd = llama_model_n_embd_inp(llama_get_model(ctx));
|
||||
|
||||
// positions continue from the memory when none are given
|
||||
auto * mem = llama_get_memory(ctx);
|
||||
std::vector<llama_pos> pos_next(llama_n_seq_max(ctx));
|
||||
for (llama_seq_id s = 0; s < (llama_seq_id) pos_next.size(); ++s) {
|
||||
pos_next[s] = llama_memory_seq_pos_max(mem, s) + 1;
|
||||
}
|
||||
|
||||
for (int32_t i = 0; i < batch.n_tokens; ++i) {
|
||||
const int32_t n_sid = batch.n_seq_id ? batch.n_seq_id[i] : 1;
|
||||
const llama_seq_id seq_id = batch.seq_id ? batch.seq_id[i][0] : 0;
|
||||
|
||||
llama_pos pos[GGML_MROPE_SECTIONS] = { 0, 0, 0, 0 };
|
||||
if (!batch.pos) {
|
||||
pos[0] = pos_next[seq_id]++;
|
||||
} else if (has_token) {
|
||||
pos[0] = batch.pos[i];
|
||||
} else {
|
||||
// embedding batch: section-major layout pos[j*n_tokens + i]
|
||||
for (int32_t j = 0; j < res.n_pos; ++j) {
|
||||
pos[j] = batch.pos[j * batch.n_tokens + i];
|
||||
}
|
||||
}
|
||||
|
||||
const bool output = batch.logits ? batch.logits[i] != 0 : i == batch.n_tokens - 1;
|
||||
|
||||
const llama_embd embd = { has_embd ? batch.embd + (size_t) i * n_embd : nullptr, 1, n_embd };
|
||||
for (int32_t i = off; i < off + n; ++i) {
|
||||
const token & t = tokens[i];
|
||||
|
||||
int32_t idx;
|
||||
if (has_token) {
|
||||
idx = res.add(batch.token[i], pos[0], seq_id, output);
|
||||
if (has_embd) {
|
||||
res.set_embd(idx, embd);
|
||||
if (t.id != LLAMA_TOKEN_NULL) {
|
||||
idx = llama_batch_ext_add_token(res, t.seq_id, t.id);
|
||||
if (idx < 0) {
|
||||
GGML_ABORT("%s: failed to add token %d at index %d (error %d, n = %d)\n", __func__, t.id, i, idx, n);
|
||||
}
|
||||
llama_batch_ext_set_pos(res, idx, t.pos.data());
|
||||
if (t.embd.data && !llama_batch_ext_set_embd_token(res, idx, t.embd)) {
|
||||
GGML_ABORT("%s: failed to set the embedding of token %d at index %d\n", __func__, t.id, i);
|
||||
}
|
||||
} else {
|
||||
idx = res.add_embd(embd, pos, seq_id, output);
|
||||
idx = llama_batch_ext_add_embd(res, t.seq_id, t.embd);
|
||||
if (idx < 0) {
|
||||
GGML_ABORT("%s: failed to add embedding at index %d (error %d, n = %d)\n", __func__, i, idx, n);
|
||||
}
|
||||
llama_batch_ext_set_pos(res, idx, t.pos.data());
|
||||
}
|
||||
GGML_ASSERT(idx == i - off);
|
||||
|
||||
for (int32_t s = 1; s < n_sid; ++s) {
|
||||
llama_batch_ext_add_seq(res.get(), idx, batch.seq_id[i][s]);
|
||||
for (const llama_seq_id seq_id : t.seq_ids_extra) {
|
||||
if (!llama_batch_ext_add_seq(res, idx, seq_id)) {
|
||||
GGML_ABORT("%s: failed to add seq %d to the entry at index %d\n", __func__, seq_id, i);
|
||||
}
|
||||
}
|
||||
if (t.output) {
|
||||
llama_batch_ext_set_output_logits(res, idx, true);
|
||||
}
|
||||
if (t.decision_order != 0) {
|
||||
llama_batch_ext_set_decision_order(res, idx, (llama_decision_order) t.decision_order);
|
||||
}
|
||||
}
|
||||
|
||||
return res;
|
||||
}
|
||||
|
||||
common_batch common_batch_get_one(llama_context * ctx, const llama_tokens & tokens) {
|
||||
common_batch common_batch_get_one(llama_context * ctx, const llama_token * tokens, int32_t n_tokens) {
|
||||
common_batch batch(ctx);
|
||||
|
||||
auto mem = llama_get_memory(ctx);
|
||||
llama_pos pos = llama_memory_seq_pos_max(mem, 0) + 1; // -1 + 1 == 0 when the memory is empty
|
||||
|
||||
for (size_t i = 0; i < tokens.size(); ++i) {
|
||||
const bool output = i == tokens.size() - 1;
|
||||
for (int32_t i = 0; i < n_tokens; ++i) {
|
||||
const bool output = i == n_tokens - 1;
|
||||
batch.add(tokens[i], pos, 0, output);
|
||||
pos++;
|
||||
}
|
||||
@@ -2238,6 +2268,10 @@ common_batch common_batch_get_one(llama_context * ctx, const llama_tokens & toke
|
||||
return batch;
|
||||
}
|
||||
|
||||
common_batch common_batch_get_one(llama_context * ctx, const llama_tokens & tokens) {
|
||||
return common_batch_get_one(ctx, tokens.data(), (int32_t) tokens.size());
|
||||
}
|
||||
|
||||
bool common_prompt_batch_decode(
|
||||
struct llama_context * ctx,
|
||||
const llama_tokens & all_tokens,
|
||||
@@ -2357,40 +2391,36 @@ void common_prompt_checkpoint::update_dft(
|
||||
}
|
||||
}
|
||||
|
||||
void common_prompt_checkpoint::load_tgt(
|
||||
bool common_prompt_checkpoint::load_tgt(
|
||||
llama_context * ctx,
|
||||
llama_seq_id seq_id,
|
||||
llama_state_seq_flags flags) const {
|
||||
if (ctx == nullptr) {
|
||||
return;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (data_tgt.empty()) {
|
||||
return;
|
||||
return true;
|
||||
}
|
||||
|
||||
const size_t n = llama_state_seq_set_data_ext(ctx, data_tgt.data(), data_tgt.size(), seq_id, flags);
|
||||
if (n != data_tgt.size()) {
|
||||
GGML_ABORT("checkpoint size mismatch: expected %zu, got %zu\n", data_tgt.size(), n);
|
||||
}
|
||||
return n == data_tgt.size();
|
||||
}
|
||||
|
||||
void common_prompt_checkpoint::load_dft(
|
||||
bool common_prompt_checkpoint::load_dft(
|
||||
llama_context * ctx,
|
||||
llama_seq_id seq_id,
|
||||
llama_state_seq_flags flags) const {
|
||||
if (ctx == nullptr) {
|
||||
return;
|
||||
return true;
|
||||
}
|
||||
|
||||
if (data_dft.empty()) {
|
||||
return;
|
||||
return true;
|
||||
}
|
||||
|
||||
const size_t n = llama_state_seq_set_data_ext(ctx, data_dft.data(), data_dft.size(), seq_id, flags);
|
||||
if (n != data_dft.size()) {
|
||||
GGML_ABORT("checkpoint size mismatch: expected %zu, got %zu\n", data_dft.size(), n);
|
||||
}
|
||||
return n == data_dft.size();
|
||||
}
|
||||
|
||||
void common_prompt_checkpoint::clear_tgt() {
|
||||
|
||||
+71
-36
@@ -19,6 +19,7 @@
|
||||
#include <algorithm>
|
||||
#include <filesystem>
|
||||
#include <fstream>
|
||||
#include <cstdio>
|
||||
|
||||
#if defined(_WIN32) && !defined(_WIN32_WINNT)
|
||||
#define _WIN32_WINNT 0x0A00
|
||||
@@ -333,6 +334,8 @@ struct common_params_speculative_draft {
|
||||
|
||||
bool backend_sampling = true; // offload draft sampling to the backend (default: on)
|
||||
|
||||
bool probabilistic = false; // sample the draft and verify by rejection, instead of argmax and match
|
||||
|
||||
common_params_model mparams;
|
||||
|
||||
llama_context * ctx_tgt = nullptr;
|
||||
@@ -583,12 +586,15 @@ struct common_params {
|
||||
bool no_op_offload = false; // globally disable offload host tensor operations to device
|
||||
bool no_extra_bufts = false; // disable extra buffer types (used for weight repacking)
|
||||
bool no_host = false; // bypass host buffer allowing extra buffers to be used
|
||||
bool load_mtp = false; // load MTP/NextN layers
|
||||
|
||||
bool single_turn = false; // single turn chat conversation
|
||||
|
||||
ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K
|
||||
ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V
|
||||
|
||||
size_t moe_cache_size = 0; // GPU cache size in bytes for the MoE experts kept in the CPU, split among the GPUs like the layers
|
||||
|
||||
common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
|
||||
|
||||
// multimodal models (see tools/mtmd)
|
||||
@@ -619,7 +625,7 @@ struct common_params {
|
||||
std::string cls_sep = "\t"; // separator of classification sequences
|
||||
|
||||
// server params
|
||||
int32_t port = 8080; // server listens on this network port
|
||||
int32_t port = 9931; // server listens on this network port
|
||||
bool reuse_port = false; // allow multiple sockets to bind to the same port
|
||||
int32_t timeout_read = 3600; // http read timeout in seconds
|
||||
int32_t timeout_write = timeout_read; // http write timeout in seconds
|
||||
@@ -721,10 +727,11 @@ struct common_params {
|
||||
int32_t i_chunk = 0; // start processing from this chunk
|
||||
int8_t imat_dat = 0; // whether the legacy imatrix.dat format should be output (gguf <= 0 < dat)
|
||||
|
||||
bool process_output = false; // collect data for the output tensor
|
||||
bool compute_ppl = true; // whether to compute perplexity
|
||||
bool show_statistics = false; // show imatrix statistics per tensor
|
||||
bool parse_special = false; // whether to parse special tokens during imatrix tokenization
|
||||
bool process_output = false; // collect data for the output tensor
|
||||
bool compute_ppl = true; // whether to compute perplexity
|
||||
bool show_statistics = false; // show imatrix statistics per tensor
|
||||
bool activation_statistics = false; // generate data to calculate activation based statistics
|
||||
bool parse_special = false; // whether to parse special tokens during imatrix tokenization
|
||||
|
||||
// cvector-generator params
|
||||
int n_pca_batch = 100;
|
||||
@@ -914,21 +921,19 @@ std::filesystem::path common_get_path_from_env(const std::string & name);
|
||||
bool fs_validate_filename(const std::string & filename, bool allow_subdirs = false);
|
||||
bool fs_is_directory(const std::string & path);
|
||||
|
||||
// some old libstdc++ versions don't follow symlinks here, so adding a trailing "/" fixes it: https://gcc.gnu.org/bugzilla/show_bug.cgi?id=101510
|
||||
inline bool common_create_directories(const std::filesystem::path & path, std::error_code & ec) {
|
||||
#if defined(__linux__)
|
||||
return std::filesystem::create_directories(path / "", ec);
|
||||
#else
|
||||
return std::filesystem::create_directories(path, ec);
|
||||
#endif
|
||||
}
|
||||
|
||||
std::filesystem::path fs_get_cache_directory();
|
||||
std::filesystem::path fs_get_cache_file(const std::string & filename);
|
||||
std::filesystem::path fs_get_config_directory();
|
||||
|
||||
struct common_file_info {
|
||||
std::string path;
|
||||
std::string name;
|
||||
size_t size = 0; // in bytes
|
||||
bool is_dir = false;
|
||||
};
|
||||
std::vector<common_file_info> fs_list(const std::string & path, bool include_directories);
|
||||
|
||||
// fs open, also handle UTF8 on Windows
|
||||
std::ifstream fs_open_ifstream(const std::string & fname, std::ios_base::openmode mode);
|
||||
|
||||
void fs_write_atomic(const std::filesystem::path & path, const std::string & data);
|
||||
|
||||
//
|
||||
@@ -937,6 +942,10 @@ void fs_write_atomic(const std::filesystem::path & path, const std::string & dat
|
||||
|
||||
// Auto-detect if colors can be enabled based on terminal and environment
|
||||
bool tty_can_use_colors();
|
||||
bool tty_enable_ansi(); // false when stdout or stderr is a console that cannot render ANSI sequences
|
||||
|
||||
// Check if the given file is attached to a terminal
|
||||
bool common_is_tty(FILE * file);
|
||||
|
||||
//
|
||||
// Model utils
|
||||
@@ -944,6 +953,31 @@ bool tty_can_use_colors();
|
||||
|
||||
struct common_sampler;
|
||||
|
||||
// typed decision models, see "<arch>.decision.type" in the model metadata
|
||||
enum common_decision_type {
|
||||
COMMON_DECISION_TYPE_NONE, // not a decision model
|
||||
COMMON_DECISION_TYPE_OPENJEV, // logits of one label token per option, read at the last prompt token
|
||||
COMMON_DECISION_TYPE_LEV, // same as openjev, noul is read from a rating scale
|
||||
COMMON_DECISION_TYPE_KEV, // dot product of the hidden states of the last token and of one end token per option
|
||||
COMMON_DECISION_TYPE_NIMBLE, // same as openjev, the prompt lists all the questions of the request
|
||||
COMMON_DECISION_TYPE_LAYA, // score of one marker token per option, read from the embeddings output
|
||||
COMMON_DECISION_TYPE_CLEF, // all questions in one prompt, score of option i read from the embeddings output at row i
|
||||
COMMON_DECISION_TYPE_PPLX_DECIDER, // same as openjev, label codes of 1 or 2 letters
|
||||
COMMON_DECISION_TYPE_LFM2_D1, // same as openjev, the labels depend on the question type
|
||||
COMMON_DECISION_TYPE_LFM2_D1_OMNI, // same as laya, other prompt layout
|
||||
COMMON_DECISION_TYPE_UNKNOWN, // a decision model of a type that is not supported
|
||||
};
|
||||
|
||||
common_decision_type common_get_decision_type(const struct llama_model * model);
|
||||
|
||||
// metadata of a GGUF file, read without loading the model
|
||||
struct common_gguf_info {
|
||||
common_decision_type decision_type = COMMON_DECISION_TYPE_UNKNOWN; // UNKNOWN if the file is missing, unreadable, or invalid
|
||||
uint32_t n_ctx_train = 0; // 0 if unknown
|
||||
};
|
||||
|
||||
common_gguf_info common_get_gguf_info(const std::string & fname);
|
||||
|
||||
// note: defines the model, context, samplers, ets. lifetimes
|
||||
struct common_init_result {
|
||||
common_init_result(common_params & params, bool model_only = false);
|
||||
@@ -1031,23 +1065,17 @@ struct common_memory {
|
||||
// Batch utils
|
||||
//
|
||||
|
||||
void common_batch_clear(struct llama_batch & batch);
|
||||
|
||||
void common_batch_add(
|
||||
struct llama_batch & batch,
|
||||
llama_token id,
|
||||
llama_pos pos,
|
||||
const std::vector<llama_seq_id> & seq_ids,
|
||||
bool logits);
|
||||
|
||||
// wrapper around llama_batch_ext that provide getter functions for downstream code
|
||||
// entries can exceed n_batch, use get_sub_batch() to decode them in chunks
|
||||
struct common_batch {
|
||||
struct token {
|
||||
llama_token id;
|
||||
std::array<llama_pos, GGML_MROPE_SECTIONS> pos; // only pos[0] is used for text tokens
|
||||
llama_seq_id seq_id;
|
||||
llama_seq_id seq_id; // the first sequence id, see add_seq()
|
||||
bool output;
|
||||
llama_embd embd; // non-owning view of the data passed to add_embd()/set_embd(), data == NULL if none
|
||||
std::vector<llama_seq_id> seq_ids_extra; // see add_seq()
|
||||
int32_t decision_order = 0; // see llama_batch_ext_set_decision_order()
|
||||
};
|
||||
|
||||
std::vector<token> tokens; // mirror of the entries, tokens[i] describes batch index i
|
||||
@@ -1058,7 +1086,10 @@ struct common_batch {
|
||||
common_batch() = default;
|
||||
common_batch(struct llama_context * ctx);
|
||||
|
||||
llama_batch_ext * get() const { return batch.get(); }
|
||||
llama_batch_ext * get() { return get_sub_batch(0, size()); }
|
||||
|
||||
// render entries [off, off + n) into batch, the result is overwritten by the next call
|
||||
llama_batch_ext * get_sub_batch(int32_t off, int32_t n);
|
||||
|
||||
// content type of the batch, all entries carry the same combination
|
||||
bool has_token() const { return !tokens.empty() && tokens[0].id != LLAMA_TOKEN_NULL; }
|
||||
@@ -1066,15 +1097,21 @@ struct common_batch {
|
||||
|
||||
void clear();
|
||||
|
||||
// returns the batch index (>= 0), aborts if the entry cannot be added (batch full, invalid token or seq id)
|
||||
// returns the batch index
|
||||
int32_t add(llama_token id, llama_pos pos, llama_seq_id seq_id, bool output);
|
||||
|
||||
// same, with the entry shared by all seq_ids (must not be empty)
|
||||
int32_t add(llama_token id, llama_pos pos, const std::vector<llama_seq_id> & seq_ids, bool output);
|
||||
|
||||
// add the entry at idx to another sequence, tokens[idx].seq_id keeps the first one
|
||||
bool add_seq(int32_t idx, llama_seq_id seq_id);
|
||||
|
||||
bool set_output(int32_t idx, bool value);
|
||||
|
||||
// attach a token embedding to the entry at idx, can only be set once per entry
|
||||
bool set_embd(int32_t idx, llama_embd embd);
|
||||
|
||||
// add an embedding-only entry (no token id), aborts like add() on failure
|
||||
// add an embedding-only entry (no token id)
|
||||
// pos points to n_pos positions
|
||||
int32_t add_embd(llama_embd embd, const llama_pos * pos, llama_seq_id seq_id, bool output);
|
||||
|
||||
@@ -1082,13 +1119,10 @@ struct common_batch {
|
||||
};
|
||||
|
||||
// create a single-sequence batch from a list of tokens
|
||||
// last token always have output_logits set to true
|
||||
// positions continue from the memory, last token always have output_logits set to true
|
||||
common_batch common_batch_get_one(struct llama_context * ctx, const llama_token * tokens, int32_t n_tokens);
|
||||
common_batch common_batch_get_one(struct llama_context * ctx, const llama_tokens & tokens);
|
||||
|
||||
// convert a legacy llama_batch, applying its defaults: seq 0, positions continue from memory, last token is output
|
||||
// the embd rows are read at the model input width
|
||||
common_batch common_batch_from_llama_batch(struct llama_context * ctx, const llama_batch & batch);
|
||||
|
||||
// decodes a single batch of tokens for a prompt and manages session tokens
|
||||
//
|
||||
// Note: We save state before the last token so that we can replay it to ensure
|
||||
@@ -1266,12 +1300,13 @@ struct common_prompt_checkpoint {
|
||||
llama_seq_id seq_id,
|
||||
llama_state_seq_flags flags);
|
||||
|
||||
void load_tgt(
|
||||
// return false if the state could not be restored
|
||||
bool load_tgt(
|
||||
llama_context * ctx,
|
||||
llama_seq_id seq_id,
|
||||
llama_state_seq_flags flags) const;
|
||||
|
||||
void load_dft(
|
||||
bool load_dft(
|
||||
llama_context * ctx,
|
||||
llama_seq_id seq_id,
|
||||
llama_state_seq_flags flags) const;
|
||||
|
||||
+2
-2
@@ -1019,6 +1019,7 @@ namespace console {
|
||||
line.clear();
|
||||
pop_cursor();
|
||||
}
|
||||
line += '\n';
|
||||
has_more = false;
|
||||
}
|
||||
} else {
|
||||
@@ -1050,7 +1051,6 @@ namespace console {
|
||||
if (!std::getline(std::wcin, wline)) {
|
||||
// Input stream is bad or EOF received
|
||||
line.clear();
|
||||
GenerateConsoleCtrlEvent(CTRL_C_EVENT, 0);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1065,7 +1065,7 @@ namespace console {
|
||||
if (!line.empty()) {
|
||||
char last = line.back();
|
||||
if (last == '/') { // Always return control on '/' symbol
|
||||
line.pop_back();
|
||||
line.back() = '\n';
|
||||
return false;
|
||||
}
|
||||
if (last == '\\') { // '\\' changes the default action
|
||||
|
||||
+1
-12
@@ -35,13 +35,6 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
// isatty
|
||||
#if defined(_WIN32)
|
||||
#include <io.h>
|
||||
#else
|
||||
#include <unistd.h>
|
||||
#endif
|
||||
|
||||
//
|
||||
// downloader
|
||||
//
|
||||
@@ -97,11 +90,7 @@ class ProgressBar : public common_download_callback {
|
||||
}
|
||||
|
||||
static bool is_output_a_tty() {
|
||||
#if defined(_WIN32)
|
||||
return _isatty(_fileno(stdout));
|
||||
#else
|
||||
return isatty(1);
|
||||
#endif
|
||||
return common_is_tty(stdout);
|
||||
}
|
||||
|
||||
public:
|
||||
|
||||
+29
-12
@@ -98,9 +98,10 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
return false;
|
||||
}
|
||||
|
||||
const int64_t datasets_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_DATASETS);
|
||||
const int64_t datasets_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_DATASETS);
|
||||
const int64_t chunk_count_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_CHUNK_COUNT);
|
||||
const int64_t chunk_size_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_CHUNK_SIZE);
|
||||
const int64_t nextn_key = gguf_find_key(ctx_gguf, LLM_KV_IMATRIX_N_LAYER_NEXTN);
|
||||
|
||||
if (datasets_key != -1 && gguf_get_kv_type(ctx_gguf, datasets_key) == GGUF_TYPE_ARRAY &&
|
||||
gguf_get_arr_type(ctx_gguf, datasets_key) == GGUF_TYPE_STRING) {
|
||||
@@ -111,33 +112,42 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
}
|
||||
}
|
||||
|
||||
imatrix.has_metadata = (datasets_key != -1 && chunk_count_key != -1 && chunk_size_key != -1);
|
||||
imatrix.chunk_count = (chunk_count_key != -1) ? gguf_get_val_u32(ctx_gguf, chunk_count_key) : 0;
|
||||
imatrix.chunk_size = (chunk_size_key != -1) ? gguf_get_val_u32(ctx_gguf, chunk_size_key) : 0;
|
||||
imatrix.has_metadata = datasets_key != -1 && chunk_count_key != -1 && chunk_size_key != -1;
|
||||
imatrix.chunk_count = chunk_count_key != -1 ? gguf_get_val_u32(ctx_gguf, chunk_count_key) : 0;
|
||||
imatrix.chunk_size = chunk_size_key != -1 ? gguf_get_val_u32(ctx_gguf, chunk_size_key) : 0;
|
||||
imatrix.n_layer_nextn = nextn_key != -1 ? gguf_get_val_u32(ctx_gguf, nextn_key) : 0;
|
||||
|
||||
const std::string in_sum_suffix{ ".in_sum" };
|
||||
const std::string in_sum2_suffix{ ".in_sum2" };
|
||||
const std::string counts_suffix{ ".counts" };
|
||||
|
||||
std::map<std::string, std::pair<struct ggml_tensor *, struct ggml_tensor *>> sums_counts_for;
|
||||
struct sum_tensors {
|
||||
struct ggml_tensor * in_sum = nullptr;
|
||||
struct ggml_tensor * in_sum2 = nullptr;
|
||||
struct ggml_tensor * counts = nullptr;
|
||||
};
|
||||
|
||||
std::map<std::string, sum_tensors> sums_counts_for;
|
||||
for (struct ggml_tensor * cur = ggml_get_first_tensor(ctx); cur; cur = ggml_get_next_tensor(ctx, cur)) {
|
||||
std::string name = cur->name;
|
||||
|
||||
if (name.empty()) { continue; }
|
||||
|
||||
if (string_remove_suffix(name, in_sum2_suffix)) {
|
||||
sums_counts_for[std::move(name)].first = cur;
|
||||
if (string_remove_suffix(name, in_sum_suffix)) {
|
||||
sums_counts_for[std::move(name)].in_sum = cur;
|
||||
} else if (string_remove_suffix(name, in_sum2_suffix)) {
|
||||
sums_counts_for[std::move(name)].in_sum2 = cur;
|
||||
} else if (string_remove_suffix(name, counts_suffix)) {
|
||||
sums_counts_for[std::move(name)].second = cur;
|
||||
sums_counts_for[std::move(name)].counts = cur;
|
||||
}
|
||||
}
|
||||
|
||||
for (const auto & sc : sums_counts_for) {
|
||||
const std::string & name = sc.first;
|
||||
const struct ggml_tensor * in_sum2 = sc.second.first;
|
||||
const struct ggml_tensor * counts = sc.second.second;
|
||||
const struct ggml_tensor * in_sum = sc.second.in_sum;
|
||||
const struct ggml_tensor * in_sum2 = sc.second.in_sum2;
|
||||
const struct ggml_tensor * counts = sc.second.counts;
|
||||
|
||||
if (!in_sum2 || !counts) {
|
||||
if (!in_sum2 || !counts || (in_sum != nullptr && ggml_nelements(in_sum) != ggml_nelements(in_sum2))) {
|
||||
LOG_ERR("%s: mismatched sums and counts for %s\n", __func__, name.c_str());
|
||||
gguf_free(ctx_gguf);
|
||||
ggml_free(ctx);
|
||||
@@ -165,6 +175,13 @@ bool common_imatrix_load(const std::string & fname, common_imatrix & imatrix) {
|
||||
for (int64_t j = 0; j < ncounts; ++j) {
|
||||
e.counts[j] = std::lround(((const float *) counts->data)[j]);
|
||||
}
|
||||
|
||||
if (in_sum && ggml_nelements(in_sum) == nval) {
|
||||
e.activations.resize(nval);
|
||||
for (int64_t j = 0; j < nval; ++j) {
|
||||
e.activations[j] = ((const float *) in_sum->data)[j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
gguf_free(ctx_gguf);
|
||||
|
||||
@@ -8,9 +8,12 @@
|
||||
inline constexpr const char * LLM_KV_IMATRIX_DATASETS = "imatrix.datasets";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_CHUNK_COUNT = "imatrix.chunk_count";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_CHUNK_SIZE = "imatrix.chunk_size";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_STATS_SCHEMA = "imatrix.stats_schema";
|
||||
inline constexpr const char * LLM_KV_IMATRIX_N_LAYER_NEXTN = "imatrix.n_layer_nextn";
|
||||
|
||||
struct common_imatrix_entry {
|
||||
std::vector<float> sums;
|
||||
std::vector<float> activations;
|
||||
std::vector<int64_t> counts;
|
||||
};
|
||||
|
||||
@@ -19,6 +22,7 @@ struct common_imatrix {
|
||||
std::vector<std::string> datasets;
|
||||
int32_t chunk_count = 0;
|
||||
int32_t chunk_size = 0;
|
||||
int32_t n_layer_nextn = 0;
|
||||
bool is_legacy = false;
|
||||
bool has_metadata = false;
|
||||
};
|
||||
|
||||
+49
-30
@@ -37,38 +37,57 @@ static void caps_try_execute(jinja::program & prog,
|
||||
const caps_ctx_fn & ctx_fn,
|
||||
const caps_json_fn & tools_fn,
|
||||
const caps_analyze_fn & analyze_fn) {
|
||||
context ctx;
|
||||
ctx.is_get_stats = true;
|
||||
jinja::global_from_json(ctx, json{
|
||||
{"messages", messages_fn()},
|
||||
{"tools", tools_fn ? tools_fn() : json::array()},
|
||||
{"bos_token", ""},
|
||||
{"eos_token", ""},
|
||||
{"add_generation_prompt", true}
|
||||
}, true);
|
||||
json msgs = messages_fn();
|
||||
for (int attempt = 0; attempt < 2; attempt++) {
|
||||
context ctx;
|
||||
ctx.is_get_stats = true;
|
||||
jinja::global_from_json(ctx, json{
|
||||
{"messages", msgs},
|
||||
{"tools", tools_fn ? tools_fn() : json::array()},
|
||||
{"bos_token", ""},
|
||||
{"eos_token", ""},
|
||||
{"add_generation_prompt", true}
|
||||
}, true);
|
||||
|
||||
if (ctx_fn) {
|
||||
ctx_fn(ctx);
|
||||
if (ctx_fn) {
|
||||
ctx_fn(ctx);
|
||||
}
|
||||
|
||||
auto messages = ctx.get_val("messages");
|
||||
auto tools = ctx.get_val("tools");
|
||||
|
||||
bool success = false;
|
||||
std::string result;
|
||||
try {
|
||||
jinja::runtime runtime(ctx);
|
||||
auto results = runtime.execute(prog);
|
||||
auto parts = jinja::runtime::gather_string_parts(results);
|
||||
result = parts->as_string().str();
|
||||
success = true;
|
||||
} catch (const std::exception & e) {
|
||||
JJ_DEBUG("Exception during execution: %s", e.what());
|
||||
result = "";
|
||||
// ignore exceptions during capability analysis
|
||||
}
|
||||
|
||||
// some templates require a thinking field on every assistant turn (e.g. K2 Horizon):
|
||||
// retry once with an empty reasoning_content on the assistant turns that lack one
|
||||
if (!success && attempt == 0) {
|
||||
bool added = false;
|
||||
for (auto & msg : msgs) {
|
||||
if (msg.is_object() && msg.value("role", "") == "assistant" && !msg.contains("reasoning_content")) {
|
||||
msg["reasoning_content"] = "";
|
||||
added = true;
|
||||
}
|
||||
}
|
||||
if (added) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
analyze_fn(ctx, success, messages, tools, result);
|
||||
return;
|
||||
}
|
||||
|
||||
auto messages = ctx.get_val("messages");
|
||||
auto tools = ctx.get_val("tools");
|
||||
|
||||
bool success = false;
|
||||
std::string result;
|
||||
try {
|
||||
jinja::runtime runtime(ctx);
|
||||
auto results = runtime.execute(prog);
|
||||
auto parts = jinja::runtime::gather_string_parts(results);
|
||||
result = parts->as_string().str();
|
||||
success = true;
|
||||
} catch (const std::exception & e) {
|
||||
JJ_DEBUG("Exception during execution: %s", e.what());
|
||||
result = "";
|
||||
// ignore exceptions during capability analysis
|
||||
}
|
||||
|
||||
analyze_fn(ctx, success, messages, tools, result);
|
||||
}
|
||||
|
||||
// for debugging only
|
||||
|
||||
@@ -537,8 +537,6 @@ value for_statement::execute_impl(context & ctx) const {
|
||||
|
||||
std::vector<value> filtered_items;
|
||||
for (size_t i = 0; i < items.size(); ++i) {
|
||||
context loop_scope(scope);
|
||||
|
||||
value current = items[i];
|
||||
|
||||
std::function<void(context&)> scope_update_fn = [](context &) { /* no-op */};
|
||||
@@ -584,6 +582,7 @@ value for_statement::execute_impl(context & ctx) const {
|
||||
}
|
||||
|
||||
if (select_expr && test_expr) {
|
||||
context loop_scope(scope);
|
||||
scope_update_fn(loop_scope);
|
||||
value test_val = test_expr->execute(loop_scope);
|
||||
if (!test_val->as_bool()) {
|
||||
@@ -888,7 +887,7 @@ value member_expression::execute_impl(context & ctx) const {
|
||||
JJ_DEBUG("Accessed property '%s' value, got type: %s", key.c_str(), val->type().c_str());
|
||||
|
||||
} else if (is_val<value_array>(object) || is_val<value_string>(object)) {
|
||||
if (is_val<value_int>(property)) {
|
||||
if (is_val<value_int>(property) || is_val<value_bool>(property)) {
|
||||
int64_t index = property->as_int();
|
||||
JJ_DEBUG("Accessing %s index %d", object->type().c_str(), (int)index);
|
||||
if (is_val<value_array>(object)) {
|
||||
@@ -911,8 +910,6 @@ value member_expression::execute_impl(context & ctx) const {
|
||||
JJ_DEBUG("Accessing %s built-in '%s'", is_val<value_array>(object) ? "array" : "string", key.c_str());
|
||||
val = try_builtin_func(ctx, key, object, true);
|
||||
|
||||
} else {
|
||||
throw std::runtime_error("Cannot access property with non-string/non-number: got " + property->type());
|
||||
}
|
||||
} else {
|
||||
if (!is_val<value_string>(property)) {
|
||||
@@ -926,10 +923,10 @@ value member_expression::execute_impl(context & ctx) const {
|
||||
value_t::stats_t::mark_used(val);
|
||||
value_t::stats_t::mark_used(object);
|
||||
value_t::stats_t::mark_used(property);
|
||||
if (is_val<value_int>(property)) {
|
||||
object->stats.ops.insert("array_access");
|
||||
} else if (is_val<value_string>(property)) {
|
||||
if (is_val<value_object>(object) || is_val<value_string>(property) || is_val<value_float>(property) || is_val<value_array>(property) || is_val<value_none>(property)) {
|
||||
object->stats.ops.insert("object_access");
|
||||
} else if (is_val<value_int>(property) || is_val<value_bool>(property)) {
|
||||
object->stats.ops.insert("array_access");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+63
-56
@@ -149,6 +149,13 @@ static value test_type_fn(const func_args & args) {
|
||||
JJ_DEBUG("test_type_fn: type=%s, %s or %s result=%d", typeid(T).name(), typeid(U).name(), typeid(V).name(), is_type ? 1 : 0);
|
||||
return mk_val<value_bool>(is_type);
|
||||
}
|
||||
template<typename T, typename U, typename V, typename W>
|
||||
static value test_type_fn(const func_args & args) {
|
||||
args.ensure_count(1);
|
||||
bool is_type = is_val<T>(args.get_pos(0)) || is_val<U>(args.get_pos(0)) || is_val<V>(args.get_pos(0)) || is_val<W>(args.get_pos(0));
|
||||
JJ_DEBUG("test_type_fn: type=%s, %s, %s or %s result=%d", typeid(T).name(), typeid(U).name(), typeid(V).name(), typeid(W).name(), is_type ? 1 : 0);
|
||||
return mk_val<value_bool>(is_type);
|
||||
}
|
||||
template<value_compare_op op>
|
||||
static value test_compare_fn(const func_args & args) {
|
||||
args.ensure_count(2, 2);
|
||||
@@ -261,6 +268,30 @@ static value tojson(const func_args & args) {
|
||||
return mk_val<value_string>(json_str);
|
||||
}
|
||||
|
||||
static value & get_attribute(const value & val, const value & attr, value & default_val) {
|
||||
if (!attr->is_undefined()) {
|
||||
if (is_val<value_array>(val)) {
|
||||
value idx = attr;
|
||||
|
||||
if (is_val<value_string>(attr)) {
|
||||
const std::string s = attr->as_string().str();
|
||||
if (!s.empty() && std::all_of(s.begin(), s.end(), [](unsigned char c) { return std::isdigit(c); })) {
|
||||
try {
|
||||
idx = mk_val<value_int>(std::stoll(s));
|
||||
} catch (...) {
|
||||
idx = mk_val<value_undefined>();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return val->at(idx, default_val);
|
||||
} else if (is_val<value_object>(val)) {
|
||||
return val->at(attr, default_val);
|
||||
}
|
||||
}
|
||||
return default_val;
|
||||
}
|
||||
|
||||
template<bool is_reject>
|
||||
static value selectattr(const func_args & args) {
|
||||
args.ensure_count(2, 4);
|
||||
@@ -274,10 +305,7 @@ static value selectattr(const func_args & args) {
|
||||
if (args.count() == 2) {
|
||||
// example: array | selectattr("active")
|
||||
for (const auto & item : arr) {
|
||||
if (!is_val<value_object>(item)) {
|
||||
throw raised_exception("selectattr: item is not an object");
|
||||
}
|
||||
value attr_val = item->at(attribute, val_default);
|
||||
value attr_val = get_attribute(item, attribute, val_default);
|
||||
bool is_selected = attr_val->as_bool();
|
||||
if constexpr (is_reject) is_selected = !is_selected;
|
||||
if (is_selected) out->push_back(item);
|
||||
@@ -318,10 +346,7 @@ static value selectattr(const func_args & args) {
|
||||
}
|
||||
auto test_fn = it->second;
|
||||
for (const auto & item : arr) {
|
||||
if (!is_val<value_object>(item)) {
|
||||
throw raised_exception("selectattr: item is not an object");
|
||||
}
|
||||
value attr_val = item->at(attribute, val_default);
|
||||
value attr_val = get_attribute(item, attribute, val_default);
|
||||
func_args test_args(args.ctx);
|
||||
test_args.push_back(attr_val); // attribute value
|
||||
test_args.push_back(extra_arg); // extra argument
|
||||
@@ -478,8 +503,8 @@ const func_builtins & global_builtins() {
|
||||
{"test_is_integer", test_type_fn<value_int>},
|
||||
{"test_is_float", test_type_fn<value_float>},
|
||||
{"test_is_number", test_type_fn<value_int, value_float>},
|
||||
{"test_is_iterable", test_type_fn<value_array, value_string, value_undefined>},
|
||||
{"test_is_sequence", test_type_fn<value_array, value_string, value_undefined>},
|
||||
{"test_is_iterable", test_type_fn<value_object, value_array, value_string, value_undefined>},
|
||||
{"test_is_sequence", test_type_fn<value_object, value_array, value_string, value_undefined>},
|
||||
{"test_is_mapping", test_type_fn<value_object>},
|
||||
{"test_is_lower", [](const func_args & args) -> value {
|
||||
args.ensure_vals<value_string>();
|
||||
@@ -1068,22 +1093,14 @@ const func_builtins & value_array_t::get_builtins() const {
|
||||
}
|
||||
value val_delim = args.get_kwarg_or_pos("d", 1);
|
||||
value attribute = args.get_kwarg_or_pos("attribute", 2);
|
||||
value undef = mk_val<value_undefined>();
|
||||
const auto & arr = args.get_pos(0)->as_array();
|
||||
const bool attr_is_int = is_val<value_int>(attribute);
|
||||
if (!attribute->is_undefined() && !is_val<value_string>(attribute) && !attr_is_int) {
|
||||
throw raised_exception("join() attribute must be string or integer");
|
||||
}
|
||||
const int64_t attr_int = attr_is_int ? attribute->as_int() : 0;
|
||||
const std::string delim = val_delim->is_undefined() ? "" : val_delim->as_string().str();
|
||||
std::string result;
|
||||
for (size_t i = 0; i < arr.size(); ++i) {
|
||||
value val_arr = arr[i];
|
||||
if (!attribute->is_undefined()) {
|
||||
if (attr_is_int && is_val<value_array>(val_arr)) {
|
||||
val_arr = val_arr->at(attr_int);
|
||||
} else if (!attr_is_int && is_val<value_object>(val_arr)) {
|
||||
val_arr = val_arr->at(attribute);
|
||||
}
|
||||
val_arr = get_attribute(val_arr, attribute, undef);
|
||||
}
|
||||
if (!is_val<value_string>(val_arr) && !is_val<value_int>(val_arr) && !is_val<value_float>(val_arr)) {
|
||||
throw raised_exception("join() can only join arrays of strings or numerics");
|
||||
@@ -1115,21 +1132,11 @@ const func_builtins & value_array_t::get_builtins() const {
|
||||
}
|
||||
value val = args.get_pos(0);
|
||||
value attribute = args.get_kwarg_or_pos("attribute", 1);
|
||||
const bool attr_is_int = is_val<value_int>(attribute);
|
||||
if (!is_val<value_string>(attribute) && !attr_is_int) {
|
||||
throw raised_exception("map: attribute must be string or integer");
|
||||
}
|
||||
const int64_t attr_int = attr_is_int ? attribute->as_int() : 0;
|
||||
value default_val = args.get_kwarg("default", mk_val<value_undefined>());
|
||||
auto out = mk_val<value_array>();
|
||||
auto arr = val->as_array();
|
||||
for (const auto & item : arr) {
|
||||
value attr_val;
|
||||
if (attr_is_int) {
|
||||
attr_val = is_val<value_array>(item) ? item->at(attr_int, default_val) : default_val;
|
||||
} else {
|
||||
attr_val = is_val<value_object>(item) ? item->at(attribute, default_val) : default_val;
|
||||
}
|
||||
value attr_val = get_attribute(item, attribute, default_val);
|
||||
out->push_back(attr_val);
|
||||
}
|
||||
return is_val<value_tuple>(val) ? mk_val<value_tuple>(std::move(out->as_array())) : out;
|
||||
@@ -1166,22 +1173,14 @@ const func_builtins & value_array_t::get_builtins() const {
|
||||
// FIXME: sorting is currently always case sensitive
|
||||
//const bool case_sensitive = val_case->as_bool(); // undefined == false
|
||||
const bool reverse = val_reverse->as_bool(); // undefined == false
|
||||
const bool attr_is_int = is_val<value_int>(attribute);
|
||||
const int64_t attr_int = attr_is_int ? attribute->as_int() : 0;
|
||||
value undef = mk_val<value_undefined>();
|
||||
std::vector<value> arr = val->as_array(); // copy
|
||||
std::sort(arr.begin(), arr.end(),[&](const value & a, const value & b) {
|
||||
value val_a = a;
|
||||
value val_b = b;
|
||||
if (!attribute->is_undefined()) {
|
||||
if (attr_is_int && is_val<value_array>(a) && is_val<value_array>(b)) {
|
||||
val_a = a->at(attr_int);
|
||||
val_b = b->at(attr_int);
|
||||
} else if (!attr_is_int && is_val<value_object>(a) && is_val<value_object>(b)) {
|
||||
val_a = a->at(attribute);
|
||||
val_b = b->at(attribute);
|
||||
} else {
|
||||
throw raised_exception("sort: unsupported object attribute comparison between " + a->type() + " and " + b->type());
|
||||
}
|
||||
val_a = get_attribute(a, attribute, undef);
|
||||
val_b = get_attribute(b, attribute, undef);
|
||||
}
|
||||
return value_compare(val_a, val_b, reverse ? value_compare_op::gt : value_compare_op::lt);
|
||||
});
|
||||
@@ -1199,19 +1198,23 @@ const func_builtins & value_array_t::get_builtins() const {
|
||||
args.ensure_vals<value_array>();
|
||||
value val_case = args.get_kwarg_or_pos("case_sensitive", 1);
|
||||
value attribute = args.get_kwarg_or_pos("attribute", 2);
|
||||
if (!attribute->is_undefined()) {
|
||||
throw not_implemented_exception("min: attribute not implemented");
|
||||
}
|
||||
// FIXME: min is currently always case sensitive
|
||||
(void) val_case;
|
||||
value undef = mk_val<value_undefined>();
|
||||
const auto & arr = args.get_pos(0)->as_array();
|
||||
if (arr.empty()) {
|
||||
return mk_val<value_undefined>();
|
||||
return undef;
|
||||
}
|
||||
value result = arr[0];
|
||||
for (size_t i = 1; i < arr.size(); ++i) {
|
||||
if (value_compare(arr[i], result, value_compare_op::lt)) {
|
||||
result = arr[i];
|
||||
for (const auto & item : arr) {
|
||||
value val_arr = item;
|
||||
value val_cmp = result;
|
||||
if (!attribute->is_undefined()) {
|
||||
val_arr = get_attribute(val_arr, attribute, undef);
|
||||
val_cmp = get_attribute(val_cmp, attribute, undef);
|
||||
}
|
||||
if (value_compare(val_arr, val_cmp, value_compare_op::lt)) {
|
||||
result = item;
|
||||
}
|
||||
}
|
||||
return result;
|
||||
@@ -1221,19 +1224,23 @@ const func_builtins & value_array_t::get_builtins() const {
|
||||
args.ensure_vals<value_array>();
|
||||
value val_case = args.get_kwarg_or_pos("case_sensitive", 1);
|
||||
value attribute = args.get_kwarg_or_pos("attribute", 2);
|
||||
if (!attribute->is_undefined()) {
|
||||
throw not_implemented_exception("max: attribute not implemented");
|
||||
}
|
||||
// FIXME: max is currently always case sensitive
|
||||
(void) val_case;
|
||||
value undef = mk_val<value_undefined>();
|
||||
const auto & arr = args.get_pos(0)->as_array();
|
||||
if (arr.empty()) {
|
||||
return mk_val<value_undefined>();
|
||||
return undef;
|
||||
}
|
||||
value result = arr[0];
|
||||
for (size_t i = 1; i < arr.size(); ++i) {
|
||||
if (value_compare(arr[i], result, value_compare_op::gt)) {
|
||||
result = arr[i];
|
||||
for (const auto & item : arr) {
|
||||
value val_arr = item;
|
||||
value val_cmp = result;
|
||||
if (!attribute->is_undefined()) {
|
||||
val_arr = get_attribute(val_arr, attribute, undef);
|
||||
val_cmp = get_attribute(val_cmp, attribute, undef);
|
||||
}
|
||||
if (value_compare(val_arr, val_cmp, value_compare_op::gt)) {
|
||||
result = item;
|
||||
}
|
||||
}
|
||||
return result;
|
||||
|
||||
@@ -433,6 +433,12 @@ struct value_array_t : public value_t {
|
||||
}
|
||||
return val_arr[index];
|
||||
}
|
||||
virtual value & at(const value & index, value & default_val) override {
|
||||
if (!is_val<value_int>(index) && !is_val<value_bool>(index)) {
|
||||
return default_val;
|
||||
}
|
||||
return at(index->as_int(), default_val);
|
||||
}
|
||||
virtual const func_builtins & get_builtins() const override;
|
||||
virtual bool is_hashable() const override {
|
||||
if (std::all_of(val_arr.begin(), val_arr.end(), [&](auto & val) -> bool {
|
||||
|
||||
+20
-17
@@ -14,19 +14,6 @@
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
|
||||
#if defined(_WIN32)
|
||||
# define WIN32_LEAN_AND_MEAN
|
||||
# ifndef NOMINMAX
|
||||
# define NOMINMAX
|
||||
# endif
|
||||
# include <io.h>
|
||||
# include <windows.h>
|
||||
# define isatty _isatty
|
||||
# define fileno _fileno
|
||||
#else
|
||||
# include <unistd.h>
|
||||
#endif // defined(_WIN32)
|
||||
|
||||
int common_log_verbosity_thold = LOG_DEFAULT_LLAMA;
|
||||
|
||||
int common_log_get_verbosity_thold(void) {
|
||||
@@ -157,12 +144,16 @@ struct common_log_entry {
|
||||
}
|
||||
}
|
||||
|
||||
fprintf(fcur, "%s", msg.data());
|
||||
// the reset goes before the trailing newlines, so that every line carries its own colors
|
||||
const bool reset = level == GGML_LOG_LEVEL_WARN || level == GGML_LOG_LEVEL_ERROR || level == GGML_LOG_LEVEL_DEBUG;
|
||||
|
||||
if (level == GGML_LOG_LEVEL_WARN || level == GGML_LOG_LEVEL_ERROR || level == GGML_LOG_LEVEL_DEBUG) {
|
||||
fprintf(fcur, "%s", g_col[COMMON_LOG_COL_DEFAULT]);
|
||||
size_t end = strlen(msg.data());
|
||||
while (end > 0 && msg[end - 1] == '\n') {
|
||||
end--;
|
||||
}
|
||||
|
||||
fprintf(fcur, "%.*s%s%s", (int) end, msg.data(), reset ? g_col[COMMON_LOG_COL_DEFAULT] : "", msg.data() + end);
|
||||
|
||||
fflush(fcur);
|
||||
}
|
||||
};
|
||||
@@ -171,6 +162,7 @@ struct common_log {
|
||||
// default capacity
|
||||
common_log(size_t capacity = 512) {
|
||||
file = nullptr;
|
||||
colors = false;
|
||||
prefix = false;
|
||||
timestamps = false;
|
||||
running = false;
|
||||
@@ -198,6 +190,7 @@ private:
|
||||
|
||||
FILE * file;
|
||||
|
||||
bool colors;
|
||||
bool prefix;
|
||||
bool timestamps;
|
||||
bool running;
|
||||
@@ -407,10 +400,16 @@ public:
|
||||
resume();
|
||||
}
|
||||
|
||||
bool get_colors() const {
|
||||
return colors;
|
||||
}
|
||||
|
||||
void set_colors(bool colors) {
|
||||
pause();
|
||||
|
||||
if (colors) {
|
||||
this->colors = colors && tty_enable_ansi();
|
||||
|
||||
if (this->colors) {
|
||||
g_col[COMMON_LOG_COL_DEFAULT] = LOG_COL_DEFAULT;
|
||||
g_col[COMMON_LOG_COL_BOLD] = LOG_COL_BOLD;
|
||||
g_col[COMMON_LOG_COL_RED] = LOG_COL_RED;
|
||||
@@ -513,6 +512,10 @@ void common_log_set_colors(struct common_log * log, log_colors colors) {
|
||||
log->set_colors(true);
|
||||
}
|
||||
|
||||
bool common_log_get_colors(struct common_log * log) {
|
||||
return log->get_colors();
|
||||
}
|
||||
|
||||
void common_log_set_prefix(struct common_log * log, bool prefix) {
|
||||
log->set_prefix(prefix);
|
||||
}
|
||||
|
||||
@@ -93,6 +93,7 @@ void common_log_add(struct common_log * log, enum ggml_log_level level, const ch
|
||||
|
||||
void common_log_set_file (struct common_log * log, const char * file); // not thread-safe
|
||||
void common_log_set_colors (struct common_log * log, log_colors colors); // not thread-safe
|
||||
bool common_log_get_colors (struct common_log * log); // whether colors are enabled
|
||||
void common_log_set_prefix (struct common_log * log, bool prefix); // whether to output prefix to each log
|
||||
void common_log_set_timestamps(struct common_log * log, bool timestamps); // whether to output timestamps in the prefix
|
||||
void common_log_flush (struct common_log * log); // flush all pending log messages
|
||||
|
||||
@@ -77,7 +77,7 @@ common_chat_params common_chat_params_init_cohere2moe(const common_chat_template
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto generation_prompt = p.literal(GEN_PREFIX);
|
||||
auto end = p.end();
|
||||
|
||||
@@ -124,12 +124,10 @@ common_chat_params common_chat_params_init_cohere2moe(const common_chat_template
|
||||
return generation_prompt + reasoning + body + p.optional(p.literal(TURN_END)) + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = !has_response_format && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -145,20 +145,20 @@ common_chat_params common_chat_params_init_deepseek_v3_2(const common_chat_templ
|
||||
bool require_tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
bool has_tool_calls = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto generation_prompt = p.literal(GEN_PROMPT);
|
||||
auto end = p.end();
|
||||
|
||||
// build tool call section first since we might need it in reasoning
|
||||
auto tool_choice = p.choice();
|
||||
if (has_tool_calls) {
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
|
||||
std::vector<common_peg_parser> required_parsers;
|
||||
std::vector<common_peg_parser> optional_parsers;
|
||||
foreach_parameter(function, [&](const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
foreach_parameter(function, [&](size_t param_index, const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
bool is_string = param.schema->may_be_string();
|
||||
|
||||
auto arg = p.tool_arg(
|
||||
@@ -166,11 +166,11 @@ common_chat_params common_chat_params_init_deepseek_v3_2(const common_chat_templ
|
||||
p.literal("\" string=\"" + std::string(is_string ? "true" : "false") + "\">")) +
|
||||
(is_string ?
|
||||
p.tool_arg_string_value(p.until(PARAM_END)) :
|
||||
p.tool_arg_json_value(p.schema(p.json(), "tool-" + name + "-arg-" + param.name + "-schema",
|
||||
p.tool_arg_json_value(p.schema(p.json(), "tool-" + std::to_string(tool_index) + "-arg-" + std::to_string(param_index) + "-schema",
|
||||
doc, *param.schema))) +
|
||||
p.tool_arg_close(p.literal(PARAM_END)));
|
||||
|
||||
auto named_arg = p.rule("tool-" + name + "-arg-" + param.name, arg);
|
||||
auto named_arg = p.rule("tool-" + std::to_string(tool_index) + "-arg-" + std::to_string(param_index), arg);
|
||||
if (param.required) {
|
||||
required_parsers.push_back(named_arg);
|
||||
} else {
|
||||
@@ -199,7 +199,7 @@ common_chat_params common_chat_params_init_deepseek_v3_2(const common_chat_templ
|
||||
p.tool_name(p.literal(name)) + p.literal("\">\n")) +
|
||||
invoke_body + p.space() + p.tool_close(p.literal(INVOKE_END)));
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, func_parser);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), func_parser);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -256,12 +256,10 @@ common_chat_params common_chat_params_init_deepseek_v3_2(const common_chat_templ
|
||||
generation_prompt + reasoning + content_before_tools + tool_calls + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = has_tools && !require_tools;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -21,7 +21,7 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
// Functionary v3.2 format:
|
||||
// - Normal content: >>>all\n{content}
|
||||
// - Tool calls: >>>function_name\n{json_args}
|
||||
@@ -42,7 +42,7 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te
|
||||
|
||||
// Build tool call parsers for each available function
|
||||
auto tool_choice = p.choice();
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto schema = common_chat_tool_parameters(function);
|
||||
@@ -50,10 +50,10 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te
|
||||
// Tool format: >>>function_name\n{json_args}
|
||||
auto tool_parser = p.tool(
|
||||
p.tool_open(p.tool_name(p.literal(name)) + p.literal("\n")) +
|
||||
p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema))
|
||||
p.tool_args(p.schema(p.json(), "tool-" + std::to_string(tool_index) + "-schema", schema))
|
||||
);
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, tool_parser);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), tool_parser);
|
||||
});
|
||||
|
||||
auto content_only = content_until_end;
|
||||
@@ -76,13 +76,11 @@ common_chat_params common_chat_params_init_functionary_v3_2(const common_chat_te
|
||||
return generation_prompt + ret;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;
|
||||
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
// Grammar trigger for when the model starts outputting a tool call
|
||||
|
||||
@@ -198,7 +198,7 @@ common_chat_params common_chat_params_init_gemma4(const common_chat_template &
|
||||
auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto start = p.rule("start", p.optional(p.literal("<|turn>model\n")));
|
||||
|
||||
if (extract_reasoning) {
|
||||
@@ -254,13 +254,13 @@ common_chat_params common_chat_params_init_gemma4(const common_chat_template &
|
||||
|
||||
auto tool_choice = p.choice();
|
||||
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
// TODO @aldehir : need to extend json-schema-to-grammar to produce more than JSON rules
|
||||
// const auto & params = function.at("parameters");
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, p.tool(p.sequence({
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), p.tool(p.sequence({
|
||||
p.tool_open(p.tool_name(p.literal(name)) + p.peek(p.literal("{"))),
|
||||
p.tool_args(p.ref("gemma4-dict")),
|
||||
})));
|
||||
@@ -290,12 +290,10 @@ common_chat_params common_chat_params_init_gemma4(const common_chat_template &
|
||||
return start + p.one_or_more(message);
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = !(has_response_format || (has_tools && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED));
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -25,22 +25,23 @@ common_chat_params common_chat_params_init_gigachat_v3(
|
||||
auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;
|
||||
const auto *tool_call_start_prefix = "<|message_sep|>\n\nfunction call<|role_sep|>\n";
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto ret = p.eps();
|
||||
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
// Build a choice of all available tools
|
||||
auto tool_choice = p.choice();
|
||||
for (const auto & tool : inputs.tools) {
|
||||
for (size_t i = 0; i < inputs.tools.size(); i++) {
|
||||
const auto & tool = inputs.tools[i];
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto schema = common_chat_tool_parameters(function);
|
||||
|
||||
auto tool_name = p.json_member("name", "\"" + p.tool_name(p.literal(name)) + "\"");
|
||||
auto tool_args = p.json_member("arguments", p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema)));
|
||||
auto tool_args = p.json_member("arguments", p.tool_args(p.schema(p.json(), "tool-" + std::to_string(i) + "-schema", schema)));
|
||||
|
||||
auto tool_open = p.tool_open(p.literal("{") << tool_name);
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, tool_open << "," << tool_args << "}");
|
||||
tool_choice |= p.rule("tool-" + std::to_string(i), tool_open << "," << tool_args << "}");
|
||||
}
|
||||
|
||||
// Define the tool call structure
|
||||
@@ -59,13 +60,11 @@ common_chat_params common_chat_params_init_gigachat_v3(
|
||||
return p.literal("assistant<|role_sep|>\n") + ret;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = has_tools && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;
|
||||
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -45,8 +45,7 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
data.thinking_start_tag = "<|channel|>analysis<|message|>";
|
||||
data.thinking_end_tags = {"<|end|>"};
|
||||
|
||||
// These special tokens are required to parse properly, so we include them
|
||||
// even if parse_tool_calls is false.
|
||||
// These special tokens are required to parse properly
|
||||
data.preserved_tokens = {
|
||||
"<|channel|>", "<|constrain|>", "<|message|>", "<|start|>", "<|end|>",
|
||||
};
|
||||
@@ -68,7 +67,7 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto start = p.rule("start", p.literal("<|start|>assistant"));
|
||||
auto end = p.rule("end", p.literal("<|end|>"));
|
||||
auto content = p.rule("message-content", p.until("<|end|>"));
|
||||
@@ -106,14 +105,14 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
if (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
auto tool_choice = p.choice();
|
||||
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto params = common_chat_tool_parameters(function);
|
||||
|
||||
auto func_name = p.literal(" to=functions.") + p.tool_name(p.literal(name));
|
||||
auto constraint = p.optional(p.space() + p.optional(p.literal("<|constrain|>")) + constrain_type);
|
||||
auto args = p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", params));
|
||||
auto args = p.tool_args(p.schema(p.json(), "tool-" + std::to_string(tool_index) + "-schema", params));
|
||||
|
||||
// recipient in role header
|
||||
// <|start|>assistant to=functions.NAME<|channel|>(commentary|analysis)[constraint]<|message|>ARGS
|
||||
@@ -123,7 +122,7 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
// <|channel|>(commentary|analysis) to=functions.NAME[constraint]<|message|>ARGS
|
||||
auto tool_in_channel = p.tool(p.tool_open(channel + func_name + constraint + p.literal("<|message|>")) + args);
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, tool_in_role | tool_in_channel);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), tool_in_role | tool_in_channel);
|
||||
});
|
||||
|
||||
auto tool_call = p.trigger_rule("tool-call", tool_choice);
|
||||
@@ -138,12 +137,10 @@ common_chat_params common_chat_params_init_gpt_oss(const common_chat_template &
|
||||
return p.zero_or_more(start + any) + start + (final_msg | unsolicited);
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = !(has_response_format || (has_tools && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED));
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
#include "parsers.h"
|
||||
|
||||
// K2 Horizon format:
|
||||
// - Reasoning: <ifm|think>...</ifm|think>, or <ifm|think_fast>/<ifm|think_faster> for medium/low reasoning_effort
|
||||
// - Tool calls: <ifm|tool_calls><ifm|tool_call>...</ifm|tool_call>...</ifm|tool_calls>, one call per <ifm|tool_call>:
|
||||
// xml (default): name <ifm|arg_key>k</ifm|arg_key> [<ifm|arg_type>t</ifm|arg_type>] <ifm|arg_value>v</ifm|arg_value> ...
|
||||
// json: {"name": "...", "arguments": {...}}
|
||||
common_chat_params common_chat_params_init_k2_horizon(const common_chat_template & tmpl,
|
||||
const autoparser::generation_params & inputs) {
|
||||
common_chat_params data;
|
||||
|
||||
// The template requires a thinking field on every assistant message
|
||||
auto messages = inputs.messages;
|
||||
for (auto & msg : messages) {
|
||||
if (msg.value("role", "") == "assistant" && !msg.contains("reasoning_content")) {
|
||||
msg["reasoning_content"] = "";
|
||||
}
|
||||
}
|
||||
|
||||
data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs, messages);
|
||||
data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs, messages);
|
||||
data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;
|
||||
data.supports_thinking = true;
|
||||
|
||||
const std::string effort = inputs.extra_context.value("reasoning_effort", "high");
|
||||
const std::string call_format = inputs.extra_context.value("tool_call_format", "xml");
|
||||
|
||||
// Templates that handle enable_thinking disable it with an empty <ifm|think></ifm|think> block for every effort
|
||||
const bool thinking_off = !inputs.enable_thinking && tmpl.source().find("enable_thinking") != std::string::npos;
|
||||
const std::string think = thinking_off ? "ifm|think" :
|
||||
effort == "medium" ? "ifm|think_fast" :
|
||||
effort == "low" ? "ifm|think_faster" : "ifm|think";
|
||||
|
||||
const std::string GEN_PREFIX = "<|ifm|im_start|>assistant\n";
|
||||
const std::string THINK_START = "<" + think + ">";
|
||||
const std::string THINK_END = "</" + think + ">";
|
||||
const std::string SECTION_START = "<ifm|tool_calls>";
|
||||
const std::string SECTION_END = "</ifm|tool_calls>";
|
||||
const std::string CALL_START = "<ifm|tool_call>";
|
||||
const std::string CALL_END = "</ifm|tool_call>";
|
||||
const std::string ARG_KEY = "<ifm|arg_key>";
|
||||
const std::string ARG_KEY_END = "</ifm|arg_key>";
|
||||
const std::string ARG_TYPE = "<ifm|arg_type>";
|
||||
const std::string ARG_TYPE_END = "</ifm|arg_type>";
|
||||
const std::string ARG_VAL = "<ifm|arg_value>";
|
||||
const std::string ARG_VAL_END = "</ifm|arg_value>";
|
||||
|
||||
data.thinking_start_tag = THINK_START;
|
||||
data.thinking_end_tags = { THINK_END };
|
||||
|
||||
data.preserved_tokens = data.thinking_end_tags;
|
||||
data.preserved_tokens.insert(data.preserved_tokens.end(), {
|
||||
THINK_START, SECTION_START, SECTION_END, CALL_START, CALL_END,
|
||||
ARG_KEY, ARG_KEY_END, ARG_TYPE, ARG_TYPE_END, ARG_VAL, ARG_VAL_END,
|
||||
});
|
||||
|
||||
data.message_delimiters = {
|
||||
{ COMMON_CHAT_ROLE_ASSISTANT, "<|ifm|im_start|>assistant" },
|
||||
{ COMMON_CHAT_ROLE_USER, "<|ifm|im_start|>user" },
|
||||
{ COMMON_CHAT_ROLE_TOOL, "<|ifm|im_start|>tool" },
|
||||
{ COMMON_CHAT_ROLE_SYSTEM, "<|ifm|im_start|>system" },
|
||||
};
|
||||
|
||||
auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();
|
||||
auto has_response_format = inputs.json_schema.is_object() && !inputs.json_schema.empty();
|
||||
auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;
|
||||
auto include_grammar = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);
|
||||
|
||||
if (inputs.has_continuation()) {
|
||||
const auto & msg = inputs.continue_msg;
|
||||
|
||||
data.generation_prompt = GEN_PREFIX + THINK_START + "\n" + msg.reasoning_content;
|
||||
if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {
|
||||
data.generation_prompt += THINK_END + msg.render_content();
|
||||
}
|
||||
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto generation_prompt = p.literal(GEN_PREFIX);
|
||||
|
||||
auto think_end = p.choice();
|
||||
for (const auto & tag : data.thinking_end_tags) {
|
||||
think_end |= p.literal(tag);
|
||||
}
|
||||
auto think_body = p.until_one_of(data.thinking_end_tags);
|
||||
auto think_block = [&](const common_peg_parser & body) {
|
||||
return p.optional(THINK_START + p.space() + p.ac(body + think_end, data.thinking_end_tags));
|
||||
};
|
||||
auto reasoning = extract_reasoning ? think_block(p.reasoning(think_body)) : p.eps();
|
||||
|
||||
if (has_response_format) {
|
||||
// The answer must be bare JSON, so the think block is consumed even when it is not extracted
|
||||
auto thoughts = extract_reasoning ? reasoning : think_block(think_body);
|
||||
return generation_prompt + (thoughts << p.content(p.schema(p.json(), "response-format", inputs.json_schema)));
|
||||
}
|
||||
|
||||
if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {
|
||||
return generation_prompt + (reasoning << p.content(p.rest()));
|
||||
}
|
||||
|
||||
auto tool_choice = p.choice();
|
||||
if (call_format == "json") {
|
||||
tool_choice = p.standard_json_tools(CALL_START, CALL_END, inputs.tools, false, true);
|
||||
} else {
|
||||
auto arg_close = p.tool_arg_close(p.literal(ARG_VAL_END));
|
||||
auto arg_string = p.rule("xml-arg-string", p.ac(p.tool_arg_string_value(p.until(ARG_VAL_END)) + arg_close, ARG_VAL_END));
|
||||
|
||||
// The models leave out <ifm|arg_type> even when asked for xml_typed
|
||||
auto arg_type = call_format == "xml_typed" ? p.optional(ARG_TYPE + p.until(ARG_TYPE_END) + ARG_TYPE_END + p.space()) : p.eps();
|
||||
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
|
||||
std::vector<common_peg_parser> required_args;
|
||||
std::vector<common_peg_parser> optional_args;
|
||||
foreach_parameter(function, [&](size_t param_index, const common_chat_schema_property & param, const common_chat_schema_document_ptr & doc) {
|
||||
auto rule_name = "tool-" + std::to_string(tool_index) + "-arg-" + std::to_string(param_index);
|
||||
auto types = param.schema->value_types();
|
||||
auto arg_value = arg_string;
|
||||
if (!types.has(common_chat_schema::TYPE_STRING)) {
|
||||
arg_value = p.tool_arg_json_value(p.schema(p.json(), rule_name + "-schema", doc, *param.schema)) + arg_close;
|
||||
}
|
||||
if (types.has(common_chat_schema::TYPE_STRING) && !types.is_only(common_chat_schema::TYPE_STRING)) {
|
||||
// The string alternative accepts any text, so only the parser needs the JSON alternatives.
|
||||
auto json_value = p.choice();
|
||||
if (types.has(common_chat_schema::TYPE_OBJECT)) {
|
||||
json_value |= p.json_object();
|
||||
}
|
||||
if (types.has(common_chat_schema::TYPE_ARRAY)) {
|
||||
json_value |= p.json_array();
|
||||
}
|
||||
if (types.has(common_chat_schema::TYPE_NUMBER) || types.has(common_chat_schema::TYPE_INTEGER)) {
|
||||
json_value |= p.json_number();
|
||||
}
|
||||
if (types.has(common_chat_schema::TYPE_BOOLEAN)) {
|
||||
json_value |= p.json_bool();
|
||||
}
|
||||
if (types.has(common_chat_schema::TYPE_NULL)) {
|
||||
json_value |= p.json_null();
|
||||
}
|
||||
arg_value = p.gbnf(p.atomic(p.tool_arg_json_value(json_value) + arg_close) | arg_string, "xml-arg-string");
|
||||
}
|
||||
|
||||
auto arg = p.space() + p.tool_arg(p.tool_arg_open(ARG_KEY + p.tool_arg_name(p.literal(param.name)) + ARG_KEY_END) <<
|
||||
arg_type + ARG_VAL + arg_value);
|
||||
(param.required ? required_args : optional_args).push_back(p.rule(rule_name, arg));
|
||||
});
|
||||
|
||||
auto args = p.permute("tool-" + std::to_string(tool_index) + "-args", required_args);
|
||||
if (!optional_args.empty()) {
|
||||
args = args + p.zero_or_more(p.choice(optional_args));
|
||||
}
|
||||
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), p.tool(
|
||||
p.tool_open(CALL_START + p.tool_name(p.literal(name)) + "\n") + p.tool_args(args) << p.tool_close(p.literal(CALL_END))));
|
||||
});
|
||||
}
|
||||
|
||||
auto required = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
auto calls = inputs.parallel_tool_calls ? tool_choice + p.zero_or_more(p.space() + tool_choice) : tool_choice;
|
||||
auto tool_calls = p.trigger_rule("tool-calls", p.repeat(SECTION_START << calls << SECTION_END, required ? 1 : 0, 1));
|
||||
|
||||
// Keep thinking inline when required calls bypass the content parser.
|
||||
if (required && !extract_reasoning) {
|
||||
reasoning = p.content(think_block(think_body));
|
||||
}
|
||||
|
||||
// A required call follows the reasoning directly, the models otherwise keep writing content
|
||||
auto content = required ? p.eps() : p.content(p.until(SECTION_START));
|
||||
|
||||
return generation_prompt + (reasoning << content << tool_calls);
|
||||
});
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = !(has_response_format || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED);
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
if (data.grammar_lazy) {
|
||||
data.grammar_triggers = {
|
||||
{ COMMON_GRAMMAR_TRIGGER_TYPE_WORD, SECTION_START },
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
return data;
|
||||
}
|
||||
@@ -48,7 +48,7 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
// Kimi K2 Thinking format:
|
||||
// - Reasoning: <think>{reasoning}</think>
|
||||
// - Content: text after reasoning
|
||||
@@ -79,7 +79,7 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
// The ID format is: functions.<name>:<index>
|
||||
// We need to match: functions.<name>:<digits>
|
||||
auto tool_choice = p.choice();
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const auto schema = common_chat_tool_parameters(function);
|
||||
@@ -89,11 +89,11 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
auto tool_id = p.tool_id(p.literal("functions.") + p.tool_name(p.literal(name)) + p.literal(":") + p.chars("[0-9]", 1, -1));
|
||||
auto tool_parser = p.tool(
|
||||
p.tool_open(tool_id + p.literal(ARGS_BEGIN)) +
|
||||
p.tool_args(p.schema(p.json(), "tool-" + name + "-schema", schema)) +
|
||||
p.tool_args(p.schema(p.json(), "tool-" + std::to_string(tool_index) + "-schema", schema)) +
|
||||
p.tool_close(p.optional((p.literal(CALL_END))))
|
||||
);
|
||||
|
||||
tool_choice |= p.rule("tool-" + name, tool_parser);
|
||||
tool_choice |= p.rule("tool-" + std::to_string(tool_index), tool_parser);
|
||||
});
|
||||
|
||||
// Tool calls section: <|tool_calls_section_begin|> tool_calls <|tool_calls_section_end|>
|
||||
@@ -111,12 +111,10 @@ common_chat_params common_chat_params_init_kimi_k2(const common_chat_template &
|
||||
return generation_prompt + reasoning + content_before_tools + tool_calls + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
@@ -66,7 +66,7 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
data.prompt += data.generation_prompt;
|
||||
}
|
||||
|
||||
auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
data.parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {
|
||||
auto end = p.end();
|
||||
|
||||
auto start = p.optional(p.literal(MSG_START));
|
||||
@@ -95,7 +95,7 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
}
|
||||
|
||||
auto tool_choices = p.choice();
|
||||
foreach_function(inputs.tools, [&](const json & tool) {
|
||||
foreach_function(inputs.tools, [&](size_t tool_index, const json & tool) {
|
||||
const auto & function = tool.at("function");
|
||||
std::string name = function.at("name");
|
||||
const json schema = common_chat_tool_parameters(function);
|
||||
@@ -106,6 +106,7 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
auto args = p.eps();
|
||||
if (schema.contains("properties") && !schema.at("properties").empty()) {
|
||||
auto arg_choices = p.choice();
|
||||
size_t param_index = 0;
|
||||
for (const auto & prop : schema.at("properties").items()) {
|
||||
const std::string & key = prop.key();
|
||||
|
||||
@@ -119,7 +120,7 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
p.tool_arg_value(p.until(ARG_END));
|
||||
|
||||
// skip the trailing type="..." attribute: anything up to <|sep|>
|
||||
arg_choices |= p.rule("kimi-k3-arg-" + name + "-" + key,
|
||||
arg_choices |= p.rule("kimi-k3-arg-" + std::to_string(tool_index) + "-" + std::to_string(param_index++),
|
||||
p.tool_arg(p.tool_arg_open(p.literal(ARG_START)) +
|
||||
p.tool_arg_name(p.literal(key)) + p.literal("\"") +
|
||||
p.until(SEP) + p.literal(SEP) + value +
|
||||
@@ -133,7 +134,7 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
p.until(SEP) + p.literal(SEP)) +
|
||||
p.tool_args(args) + p.tool_close(p.literal(CALL_END)));
|
||||
|
||||
tool_choices |= p.rule("kimi-k3-tool-" + name, call);
|
||||
tool_choices |= p.rule("kimi-k3-tool-" + std::to_string(tool_index), call);
|
||||
});
|
||||
|
||||
// all calls go inside one tools section, then the message is closed. the
|
||||
@@ -150,12 +151,10 @@ common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &
|
||||
return start + reasoning + response + tools + trailer + end;
|
||||
});
|
||||
|
||||
data.parser = parser.save();
|
||||
|
||||
if (include_grammar) {
|
||||
data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;
|
||||
data.grammar = build_grammar([&](const common_grammar_builder & builder) {
|
||||
parser.build_grammar(builder, data.grammar_lazy);
|
||||
data.parser.build_grammar(builder, data.grammar_lazy);
|
||||
});
|
||||
|
||||
data.grammar_triggers = {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user