From 307869af285d7f6f689ba100b3515e2d1b3feb05 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=C3=81lvaro=20Justen?= Date: Mon, 21 Sep 2026 03:19:14 -0300 Subject: [PATCH] devops : reduce Vulkan Docker image to 39.4% of its original size (now 692 MB) (#4038) * devops : do not install recommended packages Also: - Replace apt-get with apt - Explicitly install ca-certificates * devops : remove development packages from runtime * devops : copy only CLI executables and shared libraries Won't copy compiled tests, source code and other build artifacts. * devops : remove bash as entrypoint With this change, arguments passed to `docker run` are forwarded directly to the selected command. Having `bash -c` in the entrypoint will cause additional arguments to be interpreted according to shell command-string semantics, which makes commands such as model downloaders and CLI tools behave unexpectedly unless `--entrypoint` is specified explicitly or the whole command is quoted. Removing it restores the conventional Docker behavior of: `docker run IMAGE COMMAND ARG...` passing `ARG` to `COMMAND` (not to `bash`). * docs : fix + enhance Docker examples --- .devops/main-vulkan.Dockerfile | 26 +++++++++++----- README.md | 57 ++++++++++++++++++++++++---------- 2 files changed, 59 insertions(+), 24 deletions(-) diff --git a/.devops/main-vulkan.Dockerfile b/.devops/main-vulkan.Dockerfile index 16ee19dc6..a4be06e30 100644 --- a/.devops/main-vulkan.Dockerfile +++ b/.devops/main-vulkan.Dockerfile @@ -1,20 +1,32 @@ FROM ubuntu:24.04 AS build WORKDIR /app -RUN apt-get update && \ - apt-get install -y build-essential wget cmake git libvulkan-dev spirv-headers glslc \ +RUN apt update && \ + apt install --no-install-recommends -y build-essential ca-certificates cmake git glslc libvulkan-dev spirv-headers wget \ && rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* COPY .. . RUN --mount=type=secret,id=HF_TOKEN,required=false,env=HF_TOKEN make base.en CMAKE_ARGS="-DGGML_VULKAN=1" +# Copy only CLI executables and shared libraries (not compiled tests, nor source code or other build artifacts) +RUN mkdir -p /runtime/usr/local/bin /runtime/usr/local/lib && \ + for path in build/bin/*; do \ + name="${path##*/}"; \ + case "$name" in \ + *.so|*.so.*) cp -a "$path" /runtime/usr/local/lib/ ;; \ + test-*|main|bench) ;; \ + *) cp -a "$path" /runtime/usr/local/bin/ ;; \ + esac; \ + done + FROM ubuntu:24.04 AS runtime WORKDIR /app -RUN apt-get update && \ - apt-get install -y curl ffmpeg libsdl2-dev wget cmake git libvulkan1 mesa-vulkan-drivers \ +RUN apt update && \ + apt install --no-install-recommends -y \ + ca-certificates curl ffmpeg libvulkan1 mesa-vulkan-drivers \ && rm -rf /var/lib/apt/lists/* /var/cache/apt/archives/* -COPY --from=build /app /app -ENV PATH=/app/build/bin:$PATH -ENTRYPOINT [ "bash", "-c" ] +COPY --from=build /runtime/ / +COPY --from=build /app/models/download-* /usr/local/bin/ +RUN ldconfig diff --git a/README.md b/README.md index a037e9545..e58c4ba10 100644 --- a/README.md +++ b/README.md @@ -586,31 +586,54 @@ We have multiple Docker images available for this project: ### Usage ```shell +# Use the main tag or: cublas, main-cuda, main-intel, main-musa, main-rocm, main-vulkan. +IMAGE="ghcr.io/ggml-org/whisper.cpp:main" +MODEL_PATH="/tmp/whisper.cpp-models" +AUDIO_PATH="/tmp/whisper.cpp-audio" +AUDIO_URL="https://github.com/ggml-org/whisper.cpp/raw/refs/heads/master/samples/jfk.wav" +mkdir -p "$MODEL_PATH" "$AUDIO_PATH" +wget -O "$AUDIO_PATH/jfk.wav" "$AUDIO_URL" + # download model and persist it in a local folder docker run -it --rm \ - -v path/to/models:/models \ - whisper.cpp:main "./models/download-ggml-model.sh base /models" + -v $MODEL_PATH:/models \ + $IMAGE \ + download-ggml-model.sh base /models # transcribe an audio file docker run -it --rm \ - -v path/to/models:/models \ - -v path/to/audios:/audios \ - whisper.cpp:main "whisper-cli -m /models/ggml-base.bin -f /audios/jfk.wav" - -# transcribe an audio file in samples folder -docker run -it --rm \ - -v path/to/models:/models \ - whisper.cpp:main "whisper-cli -m /models/ggml-base.bin -f ./samples/jfk.wav" + -v $MODEL_PATH:/models \ + -v $AUDIO_PATH:/audios \ + $IMAGE \ + whisper-cli -m /models/ggml-base.bin -f /audios/jfk.wav # run the web server -docker run -it --rm -p "8080:8080" \ - -v path/to/models:/models \ - whisper.cpp:main "whisper-server --host 127.0.0.1 -m /models/ggml-base.bin" - -# run the bench too on the small.en model using 4 threads docker run -it --rm \ - -v path/to/models:/models \ - whisper.cpp:main "whisper-bench -m /models/ggml-small.en.bin -t 4" + -p "8080:8080" \ + -v $MODEL_PATH:/models \ + $IMAGE \ + whisper-server --host 0.0.0.0 -m /models/ggml-base.bin +# then: +curl -v http://127.0.0.1:8080/inference \ + -F "file=@${AUDIO_PATH}/jfk.wav" \ + -F 'response_format=json' + +# download small.en and run the bench on it using 4 threads +docker run -it --rm \ + -v $MODEL_PATH:/models \ + $IMAGE \ + download-ggml-model.sh small.en /models +docker run -it --rm \ + -v $MODEL_PATH:/models \ + $IMAGE \ + whisper-bench -m /models/ggml-small.en.bin -t 4 + +# the methods above use the CPU - use your GPU by sharing the device, for example, for an AMD iGPU via Vulkan: +docker run --rm \ + --device /dev/dri \ + -v $MODEL_PATH:/models \ + "ghcr.io/ggml-org/whisper.cpp:main-vulkan" \ + whisper-bench -m /models/ggml-small.en.bin -t 4 ``` ## Installing with Conan