mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-09-30 01:47:39 -05:00
* ci : update the oneAPI toolkit to 2026.1 oneDNN is removed from Intel Deep Learning Essentials in 2026.0, so staying on the deep-learning-essentials path would silently lose oneDNN support when the toolkit version is updated. Switch both the Ubuntu and Windows CI jobs to the new unified Intel oneAPI Toolkit installer, which still includes oneDNN (until 2027.0) and keeps the component IDs unchanged for the Windows install script. Measured with the same code (b10899) built with oneAPI 2026.1 vs the 2025.3-based release build on Arc B570: prompt processing 1331 vs 434 t/s (3.1x), token generation 50.1 vs 45.3-48.0 t/s. Assisted-by: GLM (z-ai/glm-5.3-flash) * docs : update the SYCL backend build requirements for oneAPI 2026.1 With the 2026.0 release the Base toolkit and the HPC toolkit are combined into the oneAPI Toolkit, and oneDNN is removed from the Deep Learning Essentials package. Update the install instructions, the verified release table and the news section accordingly. Assisted-by: GLM (z-ai/glm-5.3-flash) * ci : update the release workflow for oneAPI 2026.1 and Level Zero SDK 1.33.1 Align the release package build with the CI build update: - oneAPI toolkit 2025.3.3 -> 2026.1 (the unified oneAPI Toolkit) - Level Zero SDK 1.28.2 -> 1.33.1, and the Debian package names (level-zero/level-zero-devel -> libze1/libze-dev) - The Windows DLL copy list for the 2026.1 runtime: sycl9.dll and the .6/.3 MKL library versions Assisted-by: GLM (z-ai/glm-5.3-flash) * ci : remove the removed .spv fallback files from the Windows DLL copy list oneAPI 2026.1 no longer ships libsycl-fallback-bfloat16.spv and libsycl-native-bfloat16.spv (the OpenCL fallback mechanism changed), so the copy step failed with exit 1. Assisted-by: GLM (z-ai/glm-5.3-flash) * devops : update the oneAPI toolkit image in the Intel Dockerfile Assisted-by: GLM (z-ai/glm-5.3-flash) --------- Co-authored-by: Asahi-Prv <Asahi-Prv@users.noreply.github.com>
163 lines
6.0 KiB
Docker
163 lines
6.0 KiB
Docker
ARG ONEAPI_VERSION=2026.1.1-devel-ubuntu24.04
|
|
ARG BUILD_DATE=N/A
|
|
ARG APP_VERSION=N/A
|
|
ARG APP_REVISION=N/A
|
|
|
|
## Build Image
|
|
|
|
ARG NODE_VERSION=24
|
|
|
|
FROM docker.io/node:$NODE_VERSION AS web
|
|
|
|
ARG APP_VERSION
|
|
|
|
WORKDIR /app/tools/ui
|
|
|
|
COPY tools/ui/package.json tools/ui/package-lock.json ./
|
|
RUN npm ci
|
|
|
|
COPY tools/ui/ ./
|
|
RUN LLAMA_BUILD_NUMBER="$APP_VERSION" npm run build
|
|
|
|
FROM docker.io/intel/oneapi-toolkit:$ONEAPI_VERSION AS build
|
|
|
|
ARG GGML_SYCL_F16=ON
|
|
ARG LEVEL_ZERO_VERSION=1.28.2
|
|
ARG LEVEL_ZERO_UBUNTU_VERSION=u24.04
|
|
RUN apt-get update && \
|
|
apt-get install -y git libssl-dev wget ca-certificates && \
|
|
cd /tmp && \
|
|
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero.deb && \
|
|
wget -q "https://github.com/oneapi-src/level-zero/releases/download/v${LEVEL_ZERO_VERSION}/level-zero-devel_${LEVEL_ZERO_VERSION}%2B${LEVEL_ZERO_UBUNTU_VERSION}_amd64.deb" -O level-zero-devel.deb && \
|
|
apt-get -o Dpkg::Options::="--force-overwrite" install -y ./level-zero.deb ./level-zero-devel.deb && \
|
|
rm -f /tmp/level-zero.deb /tmp/level-zero-devel.deb
|
|
|
|
WORKDIR /app
|
|
|
|
COPY . .
|
|
|
|
COPY --from=web /app/tools/ui/dist tools/ui/dist
|
|
|
|
RUN if [ "${GGML_SYCL_F16}" = "ON" ]; then \
|
|
echo "GGML_SYCL_F16 is set" \
|
|
&& export OPT_SYCL_F16="-DGGML_SYCL_F16=ON" \
|
|
&& export SYCL_PROGRAM_COMPILE_OPTIONS="-cl-fp32-correctly-rounded-divide-sqrt"; \
|
|
fi && \
|
|
echo "Building with dynamic libs" && \
|
|
cmake -B build -DGGML_NATIVE=OFF -DGGML_SYCL=ON -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_BACKEND_DL=ON -DGGML_CPU_ALL_VARIANTS=ON -DLLAMA_BUILD_TESTS=OFF ${OPT_SYCL_F16} && \
|
|
cmake --build build --config Release -j$(nproc)
|
|
|
|
RUN mkdir -p /app/lib && \
|
|
find build -name "*.so*" -exec cp -P {} /app/lib \;
|
|
|
|
RUN mkdir -p /app/full \
|
|
&& cp build/bin/* /app/full \
|
|
&& cp *.py /app/full \
|
|
&& cp -r conversion /app/full \
|
|
&& cp -r gguf-py /app/full \
|
|
&& cp -r requirements /app/full \
|
|
&& cp requirements.txt /app/full \
|
|
&& cp .devops/tools.sh /app/full/tools.sh
|
|
|
|
FROM docker.io/intel/oneapi-toolkit:$ONEAPI_VERSION AS base
|
|
|
|
ARG BUILD_DATE=N/A
|
|
ARG APP_VERSION=N/A
|
|
ARG APP_REVISION=N/A
|
|
ARG IMAGE_URL=https://github.com/ggml-org/llama.cpp
|
|
ARG IMAGE_SOURCE=https://github.com/ggml-org/llama.cpp
|
|
LABEL org.opencontainers.image.created=$BUILD_DATE \
|
|
org.opencontainers.image.version=$APP_VERSION \
|
|
org.opencontainers.image.revision=$APP_REVISION \
|
|
org.opencontainers.image.title="llama.cpp" \
|
|
org.opencontainers.image.description="LLM inference in C/C++" \
|
|
org.opencontainers.image.url=$IMAGE_URL \
|
|
org.opencontainers.image.source=$IMAGE_SOURCE
|
|
|
|
#Following versions are for multiple GPUs, since 26.x has known issue:
|
|
# https://github.com/ggml-org/llama.cpp/issues/21747,
|
|
# https://github.com/intel/compute-runtime/issues/921.
|
|
#ARG IGC_VERSION=v2.20.5
|
|
#ARG IGC_VERSION_FULL=2_2.20.5+19972
|
|
#ARG COMPUTE_RUNTIME_VERSION=25.40.35563.10
|
|
#ARG COMPUTE_RUNTIME_VERSION_FULL=25.40.35563.10-0
|
|
#ARG IGDGMM_VERSION=22.8.2
|
|
|
|
|
|
ARG IGC_VERSION=v2.34.4
|
|
ARG IGC_VERSION_FULL=2_2.34.4+21428
|
|
ARG COMPUTE_RUNTIME_VERSION=26.18.38308.1
|
|
ARG COMPUTE_RUNTIME_VERSION_FULL=26.18.38308.1-0
|
|
ARG IGDGMM_VERSION=22.10.0
|
|
RUN mkdir /tmp/neo/ && cd /tmp/neo/ \
|
|
&& wget https://github.com/intel/intel-graphics-compiler/releases/download/$IGC_VERSION/intel-igc-core-${IGC_VERSION_FULL}_amd64.deb \
|
|
&& wget https://github.com/intel/intel-graphics-compiler/releases/download/$IGC_VERSION/intel-igc-opencl-${IGC_VERSION_FULL}_amd64.deb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-ocloc-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-ocloc_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-opencl-icd-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/intel-opencl-icd_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libigdgmm12_${IGDGMM_VERSION}_amd64.deb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libze-intel-gpu1-dbgsym_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.ddeb \
|
|
&& wget https://github.com/intel/compute-runtime/releases/download/$COMPUTE_RUNTIME_VERSION/libze-intel-gpu1_${COMPUTE_RUNTIME_VERSION_FULL}_amd64.deb \
|
|
&& dpkg --install *.deb
|
|
|
|
RUN apt-get update \
|
|
&& apt-get install -y libgomp1 curl ffmpeg \
|
|
&& apt autoremove -y \
|
|
&& apt clean -y \
|
|
&& rm -rf /tmp/* /var/tmp/* \
|
|
&& find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete \
|
|
&& find /var/cache -type f -delete
|
|
|
|
### Full
|
|
FROM base AS full
|
|
|
|
COPY --from=build /app/lib/ /app
|
|
COPY --from=build /app/full /app
|
|
|
|
WORKDIR /app
|
|
|
|
RUN apt-get update && \
|
|
apt-get install -y \
|
|
git \
|
|
python3 \
|
|
python3-pip \
|
|
python3-venv && \
|
|
python3 -m venv /opt/venv && \
|
|
. /opt/venv/bin/activate && \
|
|
pip install --upgrade pip setuptools wheel && \
|
|
pip install -r requirements.txt && \
|
|
apt autoremove -y && \
|
|
apt clean -y && \
|
|
rm -rf /tmp/* /var/tmp/* && \
|
|
find /var/cache/apt/archives /var/lib/apt/lists -not -name lock -type f -delete && \
|
|
find /var/cache -type f -delete
|
|
|
|
ENV PATH="/opt/venv/bin:$PATH"
|
|
|
|
ENTRYPOINT ["/app/tools.sh"]
|
|
|
|
### Light, CLI only
|
|
FROM base AS light
|
|
|
|
COPY --from=build /app/lib/ /app
|
|
COPY --from=build /app/full/llama /app/full/llama-cli /app/full/llama-completion /app
|
|
|
|
WORKDIR /app
|
|
|
|
ENTRYPOINT [ "/app/llama-cli" ]
|
|
|
|
### Server, Server only
|
|
FROM base AS server
|
|
|
|
ENV LLAMA_ARG_HOST=0.0.0.0
|
|
|
|
COPY --from=build /app/lib/ /app
|
|
COPY --from=build /app/full/llama /app/full/llama-server /app
|
|
|
|
WORKDIR /app
|
|
|
|
HEALTHCHECK CMD [ "curl", "-f", "http://localhost:8080/health" ]
|
|
|
|
ENTRYPOINT [ "/app/llama-server" ]
|