diff --git a/.agents/ci-caching.md b/.agents/ci-caching.md index 4b490575b..1aa9c54ac 100644 --- a/.agents/ci-caching.md +++ b/.agents/ci-caching.md @@ -125,7 +125,7 @@ The per-backend prefix match only sees files under a backend's own directory, so | `backend/backend.proto` | nothing if the edit is additive-only, otherwise everything (see below) | | `backend/Dockerfile.` | the Linux entries whose `dockerfile:` names it | | `backend/python/common/` | Python, Linux + Darwin | -| `scripts/build/package-gpu-libs.sh` | Python, Linux only | +| `scripts/build/package-gpu-libs.sh` | every Linux entry (Python, Go and C++ all run it) | | `scripts/build/-darwin.sh` | the Darwin entries that build target routes to | | `.github/workflows/backend_build[_darwin].yml` | everything on that OS | | anything else under `scripts/build/` (except `*_test.sh`) | everything — conservative default for unclassified packaging inputs | diff --git a/.docker/install-base-deps.sh b/.docker/install-base-deps.sh index 2b0e7e0c6..4331921ca 100755 --- a/.docker/install-base-deps.sh +++ b/.docker/install-base-deps.sh @@ -113,6 +113,54 @@ if [ "${BUILD_TYPE:-}" = "vulkan" ] && [ "${SKIP_DRIVERS:-false}" = "false" ]; t rm -rf /var/lib/apt/lists/* fi +# --- 2b. Intel graphics driver (BUILD_TYPE=sycl*) --- +# The Intel oneAPI base image brings the compilers and the oneAPI libraries, but +# not the driver that talks to the graphics card. The packaging step copies that +# driver into the backend, so that the backend works on a machine which has no +# Intel graphics packages of its own, for the same reason the Vulkan section +# above installs the Mesa drivers. Install it here so there is something to copy. +# +# Only the sycl builds are covered, because those are the ones whose packaging +# copies the driver. See package_intel_libs in scripts/build/package-gpu-libs.sh. +# +# The driver comes from Intel's own package repository, not from the Ubuntu +# archive. The archive has 23.43 from late 2023, which does not know any card +# released since, so a machine with a recent Intel GPU would end up carrying a +# driver that cannot drive it. Intel's repository has 25.18 for the same Ubuntu +# release. +# +# Anything that goes wrong here fails the build, on purpose. An unreachable +# repository is a passing problem that a retry fixes, whereas carrying a +# different driver than intended, or none, is a difference nobody would notice +# until a user reports an idle GPU. +if case "${BUILD_TYPE:-}" in sycl*) true;; *) false;; esac \ + && [ "${SKIP_DRIVERS:-false}" = "false" ]; then + # Ubuntu release name, which is what the repository is indexed by. + ubuntu_codename=$(. /etc/os-release && echo "${VERSION_CODENAME:-}") + if [ -z "$ubuntu_codename" ]; then + echo "ERROR: cannot tell which Ubuntu release this image is, so cannot pick the Intel driver repository" >&2 + exit 1 + fi + + # The key is armored text, which apt reads directly from a .asc file, so + # there is no need for gnupg here. "unified" is the component Intel ships + # its current driver in. + mkdir -p /usr/share/keyrings + curl -fsSL https://repositories.intel.com/gpu/intel-graphics.key \ + -o /usr/share/keyrings/intel-graphics.asc + echo "deb [arch=amd64 signed-by=/usr/share/keyrings/intel-graphics.asc] https://repositories.intel.com/gpu/ubuntu ${ubuntu_codename} unified" \ + > /etc/apt/sources.list.d/intel-graphics.list + apt-get update + # The first package holds the driver OpenCL talks to, the second the driver + # Level Zero talks to. Between them they pull in the compiler and the memory + # manager that both need. + apt-get install -y --no-install-recommends \ + intel-opencl-icd \ + libze-intel-gpu1 + apt-get clean + rm -rf /var/lib/apt/lists/* +fi + # --- 3. CUDA toolkit (BUILD_TYPE=cublas|l4t) --- if { [ "${BUILD_TYPE:-}" = "cublas" ] || [ "${BUILD_TYPE:-}" = "l4t" ]; } && [ "${SKIP_DRIVERS:-false}" = "false" ]; then apt-get update diff --git a/backend/cpp/bonsai/run.sh b/backend/cpp/bonsai/run.sh index 4c49f9e40..e59c5954e 100755 --- a/backend/cpp/bonsai/run.sh +++ b/backend/cpp/bonsai/run.sh @@ -40,6 +40,27 @@ else if [ -d "$CURDIR/lib/hipblaslt/library" ]; then export HIPBLASLT_TENSILE_LIBPATH="$CURDIR"/lib/hipblaslt/library fi + # Backends built for Intel GPUs carry a copy of the Intel graphics driver, + # and libze_loader is only there in those builds. Level Zero looks for a + # driver on its own, so point it at the copy that came with this backend: it + # was built against the same C library, while the machine's own driver may + # not have been, and loading that one can crash on start. + # + # Anything the user set is left alone, so a machine with a graphics card + # newer than the driver carried here can still be told to use its own. + # Nothing is said about OpenCL: no OpenCL driver is carried, so anything we + # set there would leave OpenCL worse off than the machine's own setup. + if [ -e "$CURDIR/lib/libze_loader.so.1" ]; then + if [ -e "$CURDIR/lib/libze_intel_gpu.so.1" ] && [ -z "${ZE_ENABLE_ALT_DRIVERS:-}" ]; then + export ZE_ENABLE_ALT_DRIVERS="$CURDIR"/lib/libze_intel_gpu.so.1 + fi + # Ask the driver how much graphics memory is free. Without this, the + # backend reads zero on an integrated graphics chip, because such a chip + # shares the system memory instead of having its own. + if [ -z "${ZES_ENABLE_SYSMAN:-}" ]; then + export ZES_ENABLE_SYSMAN=1 + fi + fi fi # If there is a lib/ld.so, use it diff --git a/backend/cpp/llama-cpp/run.sh b/backend/cpp/llama-cpp/run.sh index 1ccc1a37b..761098319 100755 --- a/backend/cpp/llama-cpp/run.sh +++ b/backend/cpp/llama-cpp/run.sh @@ -42,6 +42,27 @@ else if [ -d "$CURDIR/lib/hipblaslt/library" ]; then export HIPBLASLT_TENSILE_LIBPATH="$CURDIR"/lib/hipblaslt/library fi + # Backends built for Intel GPUs carry a copy of the Intel graphics driver, + # and libze_loader is only there in those builds. Level Zero looks for a + # driver on its own, so point it at the copy that came with this backend: it + # was built against the same C library, while the machine's own driver may + # not have been, and loading that one can crash on start. + # + # Anything the user set is left alone, so a machine with a graphics card + # newer than the driver carried here can still be told to use its own. + # Nothing is said about OpenCL: no OpenCL driver is carried, so anything we + # set there would leave OpenCL worse off than the machine's own setup. + if [ -e "$CURDIR/lib/libze_loader.so.1" ]; then + if [ -e "$CURDIR/lib/libze_intel_gpu.so.1" ] && [ -z "${ZE_ENABLE_ALT_DRIVERS:-}" ]; then + export ZE_ENABLE_ALT_DRIVERS="$CURDIR"/lib/libze_intel_gpu.so.1 + fi + # Ask the driver how much graphics memory is free. Without this, + # llama.cpp reads zero on an integrated graphics chip, because such a + # chip shares the system memory instead of having its own. + if [ -z "${ZES_ENABLE_SYSMAN:-}" ]; then + export ZES_ENABLE_SYSMAN=1 + fi + fi fi # If there is a lib/ld.so, use it diff --git a/backend/cpp/turboquant/run.sh b/backend/cpp/turboquant/run.sh index 84db6985a..869797faf 100755 --- a/backend/cpp/turboquant/run.sh +++ b/backend/cpp/turboquant/run.sh @@ -40,6 +40,27 @@ else if [ -d "$CURDIR/lib/hipblaslt/library" ]; then export HIPBLASLT_TENSILE_LIBPATH="$CURDIR"/lib/hipblaslt/library fi + # Backends built for Intel GPUs carry a copy of the Intel graphics driver, + # and libze_loader is only there in those builds. Level Zero looks for a + # driver on its own, so point it at the copy that came with this backend: it + # was built against the same C library, while the machine's own driver may + # not have been, and loading that one can crash on start. + # + # Anything the user set is left alone, so a machine with a graphics card + # newer than the driver carried here can still be told to use its own. + # Nothing is said about OpenCL: no OpenCL driver is carried, so anything we + # set there would leave OpenCL worse off than the machine's own setup. + if [ -e "$CURDIR/lib/libze_loader.so.1" ]; then + if [ -e "$CURDIR/lib/libze_intel_gpu.so.1" ] && [ -z "${ZE_ENABLE_ALT_DRIVERS:-}" ]; then + export ZE_ENABLE_ALT_DRIVERS="$CURDIR"/lib/libze_intel_gpu.so.1 + fi + # Ask the driver how much graphics memory is free. Without this, the + # backend reads zero on an integrated graphics chip, because such a chip + # shares the system memory instead of having its own. + if [ -z "${ZES_ENABLE_SYSMAN:-}" ]; then + export ZES_ENABLE_SYSMAN=1 + fi + fi fi # If there is a lib/ld.so, use it diff --git a/docs/content/features/GPU-acceleration.md b/docs/content/features/GPU-acceleration.md index 647aa84e1..94a97b8f5 100644 --- a/docs/content/features/GPU-acceleration.md +++ b/docs/content/features/GPU-acceleration.md @@ -329,7 +329,23 @@ This configuration has been tested on a 'custom' cluster managed by SUSE Rancher ### Requirements -If building from source, you need to install [Intel oneAPI Base Toolkit](https://software.intel.com/content/www/us/en/develop/tools/oneapi/base-toolkit/download.html) and have the Intel drivers available in the system. +You need a machine with an Intel GPU and a kernel that drives it, which every current Linux kernel does. You do not need to install any Intel graphics packages: the backends carry their own copy of the Intel graphics driver, so they work on a machine that has none installed, and on a machine whose own driver was built against a newer C library than the backend. + +If you build from source instead of using the images, you need the [Intel oneAPI Base Toolkit](https://software.intel.com/content/www/us/en/develop/tools/oneapi/base-toolkit/download.html). + +#### Using your own Intel driver instead + +The carried driver comes from Intel's own package repository, so it knows the cards released up to the point the image was built. If your GPU is newer than that, or you would rather use the driver your distribution ships, point the backend at it: + +```bash +docker run --rm -ti --device /dev/dri -p 8080:8080 \ + -e ZE_ENABLE_ALT_DRIVERS=/usr/lib/x86_64-linux-gnu/libze_intel_gpu.so.1 \ + -v $PWD/models:/models quay.io/go-skynet/local-ai:{{< version >}}-gpu-intel +``` + +Set the path to wherever your distribution keeps that file. Whatever you set is used as is, and the carried driver is left alone. + +The backends carry only the driver Level Zero uses, which is how llama.cpp reaches an Intel GPU. They do not carry an OpenCL driver, so OpenCL inside a container continues to use whatever the image itself provides. ### Container images @@ -355,6 +371,8 @@ docker run --rm -ti --device /dev/dri -p 8080:8080 -e DEBUG=true -e MODELS_PATH= Note also that sycl does have a known issue to hang with `mmap: true`. You have to disable it in the model configuration if explicitly enabled. +On an integrated Intel GPU, the amount of free graphics memory can only be read if the driver is asked to report it. The backends do that for you by setting `ZES_ENABLE_SYSMAN=1`. If you set that variable yourself, your value is kept, and setting it to `0` makes the backend read zero free memory, because an integrated GPU has no memory of its own and shares the system's. + ## Vulkan acceleration ### Requirements @@ -456,7 +474,7 @@ sycl-ls - **NVIDIA**: Ensure `nvidia-container-toolkit` is installed and the Docker runtime is configured. Test with `docker run --rm --gpus all nvidia/cuda:12.8.0-base-ubuntu24.04 nvidia-smi`. - **AMD**: Ensure `/dev/dri` and `/dev/kfd` are passed to the container and that `amdgpu-dkms` is installed on the host. -- **Intel**: Ensure `/dev/dri` is passed to the container and Intel GPU drivers are installed on the host. +- **Intel**: Ensure `/dev/dri` is passed to the container. No Intel graphics packages are needed on the host, since the backends bring their own driver. If the GPU is a recent model that the carried driver does not know, point the backend at the host's own driver as shown in [Intel acceleration](#intel-acceleration-sycl). ### Model loads on CPU instead of GPU diff --git a/scripts/build/backend-run-intel-env_test.sh b/scripts/build/backend-run-intel-env_test.sh new file mode 100755 index 000000000..5c78bec25 --- /dev/null +++ b/scripts/build/backend-run-intel-env_test.sh @@ -0,0 +1,139 @@ +#!/bin/bash +# Checks how the run.sh of each C++ backend sets up the Intel graphics driver. +# +# A backend built for Intel GPUs carries its own copy of the Intel graphics +# driver. run.sh has to tell Level Zero, which is how llama.cpp reaches the +# card, to use that copy. Three things must hold, and all three have broken in +# the past: +# +# 1. If the user already chose a driver, keep the user's choice. Otherwise a +# machine with a graphics card too new for the carried driver stops +# working, with no way to get back to the driver that did work. +# 2. Say nothing about OpenCL. No OpenCL driver is carried, so pointing +# OpenCL at the backend's own directory would leave it with no driver at +# all, where saying nothing leaves it the machine's own. +# 3. Ask the driver for the amount of free memory. Without this, llama.cpp +# reads zero free memory on an integrated graphics chip, because such a +# chip has no memory of its own and shares the system's. +# +# The test builds a fake backend directory for each run.sh, runs it, and reads +# back the variables it exported. +set -euo pipefail + +WORK=$(mktemp -d) +trap 'rm -rf "$WORK"' EXIT + +REPO_ROOT=$(dirname "$(dirname "$(dirname "$(realpath "$0")")")") + +RUN_SCRIPTS=( + "backend/cpp/llama-cpp/run.sh llama-cpp" + "backend/cpp/turboquant/run.sh turboquant" + "backend/cpp/bonsai/run.sh bonsai" +) + +failures=0 + +fail() { + echo "FAIL: $*" + failures=$((failures + 1)) +} + +# Builds a fake backend directory: the real run.sh, a stand-in for the backend +# program that prints the variables we care about, and whichever libraries the +# caller asked for. +# +# Usage: make_backend [library ...] +make_backend() { + local dir="$1" prefix="$2" + shift 2 + + mkdir -p "$dir/lib" + cp "$RUN_SH" "$dir/run.sh" + chmod +x "$dir/run.sh" + + local lib + for lib in "$@"; do + : > "$dir/lib/$lib" + done + + cat > "$dir/${prefix}-fallback" <<'PROGRAM' +#!/bin/bash +echo "level_zero_driver=${ZE_ENABLE_ALT_DRIVERS:-}" +echo "opencl_driver_list=${OCL_ICD_VENDORS:-}" +echo "report_free_memory=${ZES_ENABLE_SYSMAN:-}" +PROGRAM + chmod +x "$dir/${prefix}-fallback" +} + +# Runs a fake backend and prints the one variable asked for. +# Usage: read_variable +read_variable() { + local dir="$1" name="$2" + bash "$dir/run.sh" 2>/dev/null | sed -n "s/^${name}=//p" +} + +for entry in "${RUN_SCRIPTS[@]}"; do + read -r script prefix <<< "$entry" + RUN_SH="$REPO_ROOT/$script" + + if [ ! -f "$RUN_SH" ]; then + fail "$script does not exist" + continue + fi + + # An Intel build with its own graphics driver: point Level Zero and OpenCL + # at the bundled copies and ask for the free memory reading. + bundled="$WORK/$prefix-bundled" + make_backend "$bundled" "$prefix" \ + libze_loader.so.1 libze_intel_gpu.so.1 libigdrcl.so + mkdir -p "$bundled/etc/OpenCL/vendors" + echo "libigdrcl.so" > "$bundled/etc/OpenCL/vendors/intel.icd" + + got=$(read_variable "$bundled" level_zero_driver) + if [ "$got" != "$bundled/lib/libze_intel_gpu.so.1" ]; then + fail "$script: expected Level Zero to use the bundled driver, got '$got'" + fi + + # Even with an OpenCL driver and a driver list sitting in the backend, which + # is what an older packaging left behind, OpenCL must be left alone. + got=$(read_variable "$bundled" opencl_driver_list) + if [ -n "$got" ]; then + fail "$script: OpenCL was pointed at the backend's own directory ('$got')" + fi + + got=$(read_variable "$bundled" report_free_memory) + if [ "$got" != "1" ]; then + fail "$script: expected the free memory reading to be turned on, got '$got'" + fi + + # The user picked a driver already. Both choices must survive. + got=$(ZE_ENABLE_ALT_DRIVERS=/usr/lib/host-driver.so \ + read_variable "$bundled" level_zero_driver) + if [ "$got" != "/usr/lib/host-driver.so" ]; then + fail "$script: the user's Level Zero driver was overwritten with '$got'" + fi + + got=$(ZES_ENABLE_SYSMAN=0 read_variable "$bundled" report_free_memory) + if [ "$got" != "0" ]; then + fail "$script: the user's free memory setting was overwritten with '$got'" + fi + + # A build for some other kind of graphics card. None of the Intel + # variables belong here. + other="$WORK/$prefix-other" + make_backend "$other" "$prefix" libcublas.so.12 + + for name in level_zero_driver opencl_driver_list report_free_memory; do + got=$(read_variable "$other" "$name") + if [ -n "$got" ]; then + fail "$script: $name was set on a build with no Intel libraries ('$got')" + fi + done +done + +if [ "$failures" -gt 0 ]; then + echo "$failures check(s) failed" + exit 1 +fi + +echo "PASS: every run.sh sets up the Intel graphics driver correctly" diff --git a/scripts/build/package-gpu-libs-intel_test.sh b/scripts/build/package-gpu-libs-intel_test.sh new file mode 100755 index 000000000..e0af4289e --- /dev/null +++ b/scripts/build/package-gpu-libs-intel_test.sh @@ -0,0 +1,171 @@ +#!/bin/bash +# Checks what package_intel_libs puts in a backend built for Intel GPUs. +# +# The packager copies the libraries a backend needs next to the backend itself, +# so it can run on a machine that has none of them installed. Four things have +# to happen, and each one has been missing at some point: +# +# 1. Copy the libraries the backend program is linked against. Some of them +# are only reachable from the program, not from any other copied library, +# so looking at the copied libraries alone is not enough. +# 2. Copy the libraries that are opened by name while the program runs. Those +# are invisible to any tool that reads the list of libraries a file is +# linked against, so they have to be named one by one. +# 3. Copy the Intel graphics driver, which is also opened by name at run +# time. +# 4. Copy only the driver Level Zero talks to, and leave the OpenCL one out. +# llama.cpp reaches an Intel GPU through Level Zero; the OpenCL driver +# brings a second copy of the graphics compiler with it, which is about +# 139 MB for a path nothing here uses. +# +# The test builds a stand-in for an oneAPI installation, a stand-in for a +# driver installation and two fake backend programs, runs the real packager and +# checks the result. +set -euo pipefail + +CURDIR=$(dirname "$(realpath "$0")") +SCRIPT="$CURDIR/package-gpu-libs.sh" + +if ! command -v gcc >/dev/null 2>&1 || ! command -v ldd >/dev/null 2>&1; then + echo "SKIP: gcc/ldd not available" + exit 0 +fi + +WORK=$(mktemp -d) +trap 'rm -rf "$WORK"' EXIT + +# Stand-in for /opt/intel/oneapi/*/lib. +ONEAPI="$WORK/oneapi/lib" +mkdir -p "$ONEAPI" + +# Two libraries the backend programs are linked against, one each. Nothing else +# refers to them, so they can only be found by looking at the programs. +echo 'int first_fn(void){return 1;}' > "$WORK/first.c" +gcc -shared -fPIC -o "$ONEAPI/libfakeoneapifirst.so.2" "$WORK/first.c" +echo 'int second_fn(void){return 2;}' > "$WORK/second.c" +gcc -shared -fPIC -o "$ONEAPI/libfakeoneapisecond.so.2" "$WORK/second.c" + +# A library that is opened by name while the program runs. Nothing is linked +# against it, so only the list of names in the packager can find it. +echo 'int adapter_fn(void){return 3;}' > "$WORK/adapter.c" +gcc -shared -fPIC -o "$ONEAPI/libur_adapter_level_zero.so.0" "$WORK/adapter.c" + +# Two fake backend programs, in the directory the real packaging script uses: +# package/, one level above package/lib. One is named after llama.cpp, the +# other is not, because the same packager serves several backends. +PKG="$WORK/package" +TARGET="$PKG/lib" +mkdir -p "$TARGET" +echo 'int first_fn(void); int main(void){return first_fn();}' > "$WORK/main1.c" +gcc -o "$PKG/llama-cpp-grpc" "$WORK/main1.c" \ + -L"$ONEAPI" -l:libfakeoneapifirst.so.2 -Wl,-rpath,"$ONEAPI" +echo 'int second_fn(void); int main(void){return second_fn();}' > "$WORK/main2.c" +gcc -o "$PKG/bonsai-grpc" "$WORK/main2.c" \ + -L"$ONEAPI" -l:libfakeoneapisecond.so.2 -Wl,-rpath,"$ONEAPI" + +# The real directory also holds the script that starts the backend. Looking at a +# shell script for libraries has to be harmless. +printf '#!/bin/bash\necho started\n' > "$PKG/run.sh" +chmod +x "$PKG/run.sh" + +# Stand-in for the Intel graphics driver installation. These files are opened by +# name at run time rather than linked, so the packager has to name the ones it +# wants. +DRV="$WORK/driver" +mkdir -p "$DRV/intel-opencl" +echo 'int ze_drv(void){return 4;}' > "$WORK/zedrv.c" +gcc -shared -fPIC -o "$DRV/libze_intel_gpu.so.1" "$WORK/zedrv.c" +echo 'int cl_drv(void){return 5;}' > "$WORK/cldrv.c" +gcc -shared -fPIC -o "$DRV/intel-opencl/libigdrcl.so" "$WORK/cldrv.c" + +# The compiler front end the OpenCL driver needs, and the large library it is +# linked against. The link is what makes the big one arrive on its own if the +# front end is ever copied again, so the fake mirrors it. +echo 'int clang_fn(void){return 6;}' > "$WORK/clang.c" +gcc -shared -fPIC -o "$DRV/libopencl-clang.so.15" "$WORK/clang.c" +echo 'int clang_fn(void); int fcl_fn(void){return clang_fn();}' > "$WORK/fcl.c" +gcc -shared -fPIC -o "$DRV/libigdfcl.so.2" "$WORK/fcl.c" \ + -L"$DRV" -l:libopencl-clang.so.15 -Wl,-rpath,"$DRV" + +# Let the fake oneAPI libraries be found the way the real ones are on the build +# machine. +export LD_LIBRARY_PATH="$ONEAPI:${LD_LIBRARY_PATH:-}" + +# shellcheck source=/dev/null +source "$SCRIPT" "$TARGET" + +export BUILD_TYPE=sycl_f16 +export INTEL_ONEAPI_LIB_DIRS="$ONEAPI" +export INTEL_DRIVER_LIB_DIRS="$DRV $DRV/intel-opencl" +package_intel_libs + +fail=false + +for lib in libfakeoneapifirst.so.2 libfakeoneapisecond.so.2; do + if [ ! -e "$TARGET/$lib" ]; then + echo "FAIL: $lib is missing; the backend programs' own libraries were not copied" + fail=true + fi +done + +if [ ! -e "$TARGET/libur_adapter_level_zero.so.0" ]; then + echo "FAIL: the Level Zero adapter, which is opened by name, was not copied" + fail=true +fi + +if [ ! -e "$TARGET/libze_intel_gpu.so.1" ]; then + echo "FAIL: the Level Zero graphics driver was not copied" + fail=true +fi + +# The OpenCL driver and the compiler front end that hangs off it are left out, +# and so is the driver list that would name them. +for lib in libigdrcl.so libigdfcl.so.2 libopencl-clang.so.15; do + if [ -e "$TARGET/$lib" ]; then + echo "FAIL: $lib was copied, but nothing here uses the OpenCL path" + fail=true + fi +done +if [ -e "$TARGET/../etc/OpenCL" ]; then + echo "FAIL: an OpenCL driver list was created for a path nothing uses" + fail=true +fi + +# The Python backends for Intel GPUs, built as BUILD_TYPE=intel, start without +# run.sh and so never load a copied driver. Copying one for them would add +# several hundred megabytes that nothing reads. +PYTHON_STYLE="$WORK/python-backend/lib" +mkdir -p "$PYTHON_STYLE" +( + BUILD_TYPE=intel \ + INTEL_ONEAPI_LIB_DIRS="$ONEAPI" \ + INTEL_DRIVER_LIB_DIRS="$DRV $DRV/intel-opencl" \ + bash -c 'source "$0" "$1"; package_intel_libs' "$SCRIPT" "$PYTHON_STYLE" +) >/dev/null 2>&1 +if [ -e "$PYTHON_STYLE/libze_intel_gpu.so.1" ]; then + echo "FAIL: the graphics driver was copied into a backend that cannot load it" + fail=true +fi + +# A build for Intel GPUs that ends up with no driver still works, but only on a +# machine that has its own. That is easy to cause by accident and impossible to +# see afterwards, so the packager has to say so. +warning=$( + BUILD_TYPE=sycl_f16 \ + INTEL_ONEAPI_LIB_DIRS="$ONEAPI" \ + INTEL_DRIVER_LIB_DIRS="$WORK/empty" \ + bash -c 'source "$0" "$1"; package_intel_libs' \ + "$SCRIPT" "$WORK/nodriver/lib" 2>&1 >/dev/null || true +) +if ! grep -qi "no intel graphics driver" <<< "$warning"; then + echo "FAIL: no warning when the graphics driver could not be copied" + fail=true +fi + +if [ "$fail" = true ]; then + ls -la "$TARGET" || true + exit 1 +fi + +echo "PASS: the oneAPI libraries, the adapter and the Level Zero graphics driver were all handled" +exit 0 diff --git a/scripts/build/package-gpu-libs.sh b/scripts/build/package-gpu-libs.sh index 44d544f37..0a16bb584 100755 --- a/scripts/build/package-gpu-libs.sh +++ b/scripts/build/package-gpu-libs.sh @@ -675,21 +675,43 @@ package_rocm_libs() { package_intel_libs() { echo "Packaging Intel oneAPI/SYCL libraries for BUILD_TYPE=${BUILD_TYPE}..." - local intel_lib_paths=( - "/opt/intel/oneapi/compiler/latest/lib" - "/opt/intel/oneapi/mkl/latest/lib/intel64" - "/opt/intel/oneapi/tbb/latest/lib/intel64/gcc4.8" - ) + # Where to look for the oneAPI libraries. The default is the standard install + # layout. The list can be overridden with a space-separated one, which lets + # the tests run without a real oneAPI install, the same way ROCM_BASE_DIRS + # works. Both the current and the older math library layouts are listed, and + # the check below skips whichever of them is absent. + local intel_lib_paths + if [ -n "${INTEL_ONEAPI_LIB_DIRS:-}" ]; then + # shellcheck disable=SC2206 # intentional word-split of the override + intel_lib_paths=(${INTEL_ONEAPI_LIB_DIRS}) + else + intel_lib_paths=( + "/opt/intel/oneapi/compiler/latest/lib" + "/opt/intel/oneapi/mkl/latest/lib" + "/opt/intel/oneapi/mkl/latest/lib/intel64" + "/opt/intel/oneapi/dnnl/latest/lib" + "/opt/intel/oneapi/tbb/latest/lib/intel64/gcc4.8" + ) + fi - # Core Intel oneAPI runtime libraries + # The oneAPI libraries a backend needs at run time. The math library entries + # cover both of its number formats and both of its threading layers, because + # the llama.cpp build for Intel GPUs uses a different combination than the + # rest. The libur_adapter_* entries have to be named here even though nothing + # is linked against them: oneAPI opens them by name while the program runs, + # so the dependency scan later in this function cannot see them. local intel_libs=( "libsycl.so*" "libOpenCL.so*" "libmkl_core.so*" "libmkl_intel_lp64.so*" + "libmkl_intel_ilp64.so*" "libmkl_intel_thread.so*" + "libmkl_tbb_thread.so*" "libmkl_sequential.so*" "libmkl_sycl.so*" + "libmkl_sycl_blas.so*" + "libdnnl.so*" "libiomp5.so*" "libsvml.so*" "libirng.so*" @@ -697,6 +719,10 @@ package_intel_libs() { "libintlc.so*" "libtbb.so*" "libtbbmalloc.so*" + "libur_loader.so*" + "libur_adapter_level_zero.so*" + "libur_adapter_level_zero_v2.so*" + "libur_adapter_opencl.so*" "libpi_level_zero.so*" "libpi_opencl.so*" "libze_loader.so*" @@ -710,10 +736,92 @@ package_intel_libs() { fi done - # Pull in transitive deps the allowlist misses so the backend is - # self-contained (same class of failure as #10537). + # Copy the libraries the backend programs themselves are linked against. The + # list above is not enough on its own: the programs are linked directly + # against several oneAPI libraries that no copied library refers to, so + # without this step the backend only ran inside the build image, where oneAPI + # happens to be on the library path. + # + # The programs sit one level above the target directory, in package/, next to + # the run.sh that starts them. Every backend that builds for Intel GPUs is + # covered by looking at all of them, rather than at one set of names, because + # llama.cpp, turboquant and bonsai all come through here. + local pkg_dir="$TARGET_LIB_DIR/.." + local bin + for bin in "$pkg_dir"/*; do + if [ -f "$bin" ] && [ -x "$bin" ]; then + copy_elf_deps "$bin" + fi + done + + # Copy the Intel graphics driver itself, the way the Vulkan packaging copies + # the Mesa driver. Level Zero opens the driver by name while the program + # runs, so no dependency scan can find it and it has to be named here. + # + # This is what lets the backend run on a machine with no Intel graphics + # packages installed, and also on a machine whose own driver was built + # against a newer C library than the one this backend carries, where loading + # the host's driver crashes. Carrying the driver is safe across kernel + # versions because it reaches the graphics hardware through an interface the + # kernel keeps stable. The NVIDIA driver has no such interface, which is why + # that one is never copied. + # + # Only the builds that start through run.sh get a driver: run.sh is what + # tells Level Zero to use it. The Python backends built for Intel GPUs start + # differently and keep using the host's driver, so copying one for them would + # add several hundred megabytes that nothing would ever load. + # + # Only the Level Zero side is copied. llama.cpp reaches an Intel GPU through + # Level Zero, which hands the driver ready-compiled programs and so needs + # only the compiler's back end. The OpenCL driver can be handed source code + # instead, so it also needs the compiler's front end, and that pulls in a + # copy of clang: around 139 MB for a path nothing here takes. A user who + # wants OpenCL has their machine's own. + case "${BUILD_TYPE:-}" in + sycl*) + local intel_driver_lib_dirs + if [ -n "${INTEL_DRIVER_LIB_DIRS:-}" ]; then + # shellcheck disable=SC2206 # split the override into words on purpose + intel_driver_lib_dirs=(${INTEL_DRIVER_LIB_DIRS}) + else + intel_driver_lib_dirs=( + "/usr/lib/x86_64-linux-gnu" + "/usr/lib" + ) + fi + local driver_libs=( + "libze_intel_gpu.so*" # the driver Level Zero talks to + "libigc.so*" # turns compute programs into instructions for the card + "libigdgmm.so*" # manages graphics memory + ) + local drv_dir pat + for drv_dir in "${intel_driver_lib_dirs[@]}"; do + [ -d "$drv_dir" ] || continue + for pat in "${driver_libs[@]}"; do + copy_libs_glob "${drv_dir}/${pat}" + done + done + ;; + esac + + # Copy whatever the steps above still missed. Each library copied so far can + # need further libraries of its own, and a missing one stops the backend from + # starting at all (issue #10537). sweep_transitive_deps "$TARGET_LIB_DIR" + # Say so when a build meant for Intel GPUs ends up without a driver. It still + # works on a machine that has its own, so nothing fails here, and the only + # other way to notice is a user reporting that their GPU is not used. The + # usual cause is a build image that predates the driver being installed in + # .docker/install-base-deps.sh. + case "${BUILD_TYPE:-}" in + sycl*) + if [ ! -e "$TARGET_LIB_DIR/libze_intel_gpu.so.1" ]; then + echo "WARNING: no Intel graphics driver was found to copy. This backend will only use a GPU on a machine that has its own Intel driver installed." >&2 + fi + ;; + esac + echo "Intel oneAPI libraries packaged successfully" } diff --git a/scripts/lib/backend-filter.mjs b/scripts/lib/backend-filter.mjs index c5f7aab52..21c2800e4 100644 --- a/scripts/lib/backend-filter.mjs +++ b/scripts/lib/backend-filter.mjs @@ -389,10 +389,14 @@ export const SHARED_BUILD_INPUTS = [ darwin: always, }, { - // Stages the CUDA/ROCm runtime libraries into every Python image's lib/. - // COPY'd and run by Dockerfile.python only. This is the #10946 case. + // Decides which GPU libraries end up inside an image. Every Linux image + // runs it: Dockerfile.python calls it directly, and the Go and C++ backends + // call it from their own package.sh. Naming only the Python images here is + // how a packaging fix for the Intel llama.cpp backend could merge and reach + // no image, which is the #10946 case all over again. The Darwin builds have + // their own packaging scripts and never call this one. matches: file => file === "scripts/build/package-gpu-libs.sh", - linux: isLinuxPython, + linux: always, darwin: never, }, { diff --git a/scripts/lib/backend-filter_test.mjs b/scripts/lib/backend-filter_test.mjs index 7cb314324..f419afd3a 100644 --- a/scripts/lib/backend-filter_test.mjs +++ b/scripts/lib/backend-filter_test.mjs @@ -86,20 +86,25 @@ const run = (changedFiles, previousMatrix) => const names = entries => entries.map(e => e.backend).sort(); -test("a change to only package-gpu-libs.sh rebuilds every Python image", () => { - // The PR #10946 regression: this script is COPY'd and run by - // Dockerfile.python for every Python backend, but lives under scripts/, so - // the per-backend prefix match produced an empty matrix and the cuDNN - // packaging fix shipped to nothing. +test("a change to only package-gpu-libs.sh rebuilds every Linux image", () => { + // The PR #10946 regression: this script decides which GPU libraries end up + // inside an image, but lives under scripts/, so the per-backend prefix match + // produced an empty matrix and the packaging fix shipped to nothing. + // + // Every Linux image runs it, not only the Python ones: Dockerfile.python + // calls it directly, and the Go and C++ backends call it from their own + // package.sh (see backend/cpp/llama-cpp/package.sh and backend/go/*/ + // package.sh). Leaving those out is how a fix aimed at the Intel llama.cpp + // backend could merge and reach no image. const { filtered, filteredDarwin, changedBackends } = run([ "scripts/build/package-gpu-libs.sh", ]); - assert.notEqual(filtered.length, 0, "expected a non-empty Linux matrix"); - assert.deepEqual(names(filtered), ["diffusers", "vllm"]); + assert.equal(filtered.length, includes.length); assert.ok(changedBackends.has("vllm")); + assert.ok(changedBackends.has("llama-cpp")); - // Darwin Python builds never invoke it (see scripts/build/python-darwin.sh). + // The Darwin builds have their own packaging scripts and never call this one. assert.deepEqual(filteredDarwin, []); });