Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 11 additions & 0 deletions .docker/llama-cpp-build-target.sh
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,17 @@ set -euo pipefail
arch=${1:?target architecture is required}
build_type=${2-}

# SYCL compiles the whole tree with icpx -fsycl, and icpx never finishes
# ggml-cpu/arch/x86/repack.cpp at -march=sapphirerapids: the job sits on that one
# translation unit until GitHub kills it at 6h. gcc builds the same file in
# seconds, so only the SYCL images have to give up the CPU variant matrix.
case "$build_type" in
sycl*)
echo llama-cpp-fallback
exit 0
;;
esac

# GPU arm64 base images do not consistently provide the gcc-14 toolchain needed
# to compile ggml's armv9.2 CPU variants. Keep their portable fallback until the
# builder images can supply that compiler.
Expand Down
11 changes: 11 additions & 0 deletions .docker/turboquant-build-target.sh
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,17 @@ set -euo pipefail
arch=${1:?target architecture is required}
build_type=${2-}

# SYCL compiles the whole tree with icpx -fsycl, and icpx never finishes
# ggml-cpu/arch/x86/repack.cpp at -march=sapphirerapids: the job sits on that one
# translation unit until GitHub kills it at 6h. gcc builds the same file in
# seconds, so only the SYCL images have to give up the CPU variant matrix.
case "$build_type" in
sycl*)
echo turboquant-fallback
exit 0
;;
esac

# GPU arm64 base images do not consistently provide the gcc-14 toolchain needed
# to compile ggml's armv9.2 CPU variants. Keep their portable fallback until the
# builder images can supply that compiler.
Expand Down
5 changes: 3 additions & 2 deletions backend/cpp/llama-cpp/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -12,10 +12,11 @@ grep -e "flags" /proc/cpuinfo | head -1

BINARY=llama-cpp-fallback

# CPU images and x86 GPU images ship a single llama-cpp-cpu-all built with ggml
# CPU images and most x86 GPU images ship a single llama-cpp-cpu-all built with ggml
# CPU_ALL_VARIANTS: ggml's backend registry dlopens the best libggml-cpu-*.so for this
# host, so no shell-side AVX probing. GPU arm64 images still ship llama-cpp-fallback
# until their builder toolchains support ggml's complete arm variant matrix.
# until their builder toolchains support ggml's complete arm variant matrix, and so do
# the SYCL images, whose icpx compiler hangs on the sapphirerapids variant.
if [ -e "$CURDIR"/llama-cpp-cpu-all ]; then
BINARY=llama-cpp-cpu-all
fi
Expand Down
5 changes: 3 additions & 2 deletions backend/cpp/turboquant/run.sh
Original file line number Diff line number Diff line change
Expand Up @@ -12,11 +12,12 @@ grep -e "flags" /proc/cpuinfo | head -1

BINARY=turboquant-fallback

# CPU images and x86 GPU images ship a single turboquant-cpu-all built with ggml
# CPU images and most x86 GPU images ship a single turboquant-cpu-all built with ggml
# CPU_ALL_VARIANTS: ggml's
# backend registry dlopens the best libggml-cpu-*.so for this host, so no shell-side
# probing. GPU arm64 images still ship turboquant-fallback until their builder toolchains
# support ggml's complete arm variant matrix.
# support ggml's complete arm variant matrix, and so do the SYCL images, whose icpx
# compiler hangs on the sapphirerapids variant.
if [ -e "$CURDIR"/turboquant-cpu-all ]; then
BINARY=turboquant-cpu-all
fi
Expand Down
6 changes: 6 additions & 0 deletions scripts/build/llama-cpp-build-target_test.sh
Original file line number Diff line number Diff line change
Expand Up @@ -23,4 +23,10 @@ assert_target amd64 "" llama-cpp-cpu-all
assert_target arm64 cublas llama-cpp-fallback
assert_target arm64 "" llama-cpp-cpu-all

# SYCL builds the whole tree with icpx -fsycl, and icpx never finishes
# ggml-cpu/arch/x86/repack.cpp at -march=sapphirerapids: every sycl job sat on
# that one translation unit until GitHub killed it at its 6h limit.
assert_target amd64 sycl_f16 llama-cpp-fallback
assert_target amd64 sycl_f32 llama-cpp-fallback

echo "PASS: llama.cpp build target preserves CPU variants where supported"
6 changes: 6 additions & 0 deletions scripts/build/turboquant-build-target_test.sh
Original file line number Diff line number Diff line change
Expand Up @@ -23,4 +23,10 @@ assert_target amd64 "" turboquant-cpu-all
assert_target arm64 cublas turboquant-fallback
assert_target arm64 "" turboquant-cpu-all

# SYCL builds the whole tree with icpx -fsycl, and icpx never finishes
# ggml-cpu/arch/x86/repack.cpp at -march=sapphirerapids: every sycl job sat on
# that one translation unit until GitHub killed it at its 6h limit.
assert_target amd64 sycl_f16 turboquant-fallback
assert_target amd64 sycl_f32 turboquant-fallback

echo "PASS: turboquant build target preserves CPU variants where supported"
Loading