feat: add MTP (Multi-Token Prediction) support via new ROCm 7.2.3 and Vulkan RADV toolboxes

2026-05-14 20:09:03 +01:00
parent 7320eb3f00
commit 3e3f3674a8
4 changed files with 186 additions and 1 deletions
@@ -44,7 +44,7 @@ jobs:
        run: |
          IN='${{ github.event.inputs.backends }}'
          if [[ "$IN" == "all" || -z "$IN" ]]; then
-            JSON='["rocm-6.4.2","rocm-6.4.3","rocm-6.4.4","rocm-7.1.1","rocm-7.2","rocm-7.2.1","rocm-7.2.1-pr21344","rocm-7.2.2","rocm-7.2.3","rocm-7beta","rocm7-nightlies","vulkan-amdvlk","vulkan-radv"]'
+            JSON='["rocm-6.4.2","rocm-6.4.3","rocm-6.4.4","rocm-7.1.1","rocm-7.2","rocm-7.2.1","rocm-7.2.1-pr21344","rocm-7.2.2","rocm-7.2.3","rocm-7.2.3-mtp","rocm-7beta","rocm7-nightlies","vulkan-amdvlk","vulkan-radv","vulkan-radv-mtp"]'
          else
            IN_CLEAN=$(echo "$IN" | tr -d '[:space:]')
            JSON='["'${IN_CLEAN//,/\",\"}'"]'
@@ -11,6 +11,10 @@ TOOLBOXES["llama-rocm-6.4.4"]="docker.io/kyuz0/amd-strix-halo-toolboxes:rocm-6.4
 TOOLBOXES["llama-rocm-7.2.3"]="docker.io/kyuz0/amd-strix-halo-toolboxes:rocm-7.2.3 --device /dev/dri --device /dev/kfd --group-add video --group-add render --group-add sudo --security-opt seccomp=unconfined"
 TOOLBOXES["llama-rocm7-nightlies"]="docker.io/kyuz0/amd-strix-halo-toolboxes:rocm7-nightlies --device /dev/dri --device /dev/kfd --group-add video --group-add render --group-add sudo --security-opt seccomp=unconfined"
 # MTP (Multi-Token Prediction) — am17an/llama.cpp mtp-clean fork
 TOOLBOXES["llama-vulkan-radv-mtp"]="docker.io/kyuz0/amd-strix-halo-toolboxes:vulkan-radv-mtp --device /dev/dri --group-add video --security-opt seccomp=unconfined"
 TOOLBOXES["llama-rocm-7.2.3-mtp"]="docker.io/kyuz0/amd-strix-halo-toolboxes:rocm-7.2.3-mtp --device /dev/dri --device /dev/kfd --group-add video --group-add render --group-add sudo --security-opt seccomp=unconfined"
 function usage() {
  echo "Usage: $0 [all|toolbox-name1 toolbox-name2 ...]"
  echo "Available toolboxes:"
@@ -0,0 +1,113 @@
 # build stage
 FROM registry.fedoraproject.org/fedora:43 AS builder
 # rocm 7.2.3 repo
 RUN <<'EOF'
 tee /etc/yum.repos.d/rocm.repo <<REPO
 [ROCm-7.2.3]
 name=ROCm7.2.3
 baseurl=https://repo.radeon.com/rocm/rhel10/7.2.3/main
 enabled=1
 priority=50
 gpgcheck=1
 gpgkey=https://repo.radeon.com/rocm/rocm.gpg.key
 REPO
 EOF
 # deps
 RUN dnf -y --nodocs --setopt=install_weak_deps=False \
  --exclude='*sdk*' --exclude='*samples*' --exclude='*-doc*' --exclude='*-docs*' \
  install \
  make gcc cmake lld clang clang-devel compiler-rt libcurl-devel ninja-build \
  rocm-llvm rocm-device-libs hip-runtime-amd hip-devel \
  rocblas rocblas-devel hipblas hipblas-devel rocm-cmake libomp-devel libomp \
  rocminfo radeontop \
  git-core vim sudo rsync patch \
  && dnf clean all && rm -rf /var/cache/dnf/*
 # rocm env
 ENV ROCM_PATH=/opt/rocm \
  HIP_PATH=/opt/rocm \
  HIP_CLANG_PATH=/opt/rocm/llvm/bin \
  HIP_DEVICE_LIB_PATH=/opt/rocm/amdgcn/bitcode \
  PATH=/opt/rocm/bin:/opt/rocm/llvm/bin:$PATH
 # llama.cpp (am17an mtp-clean fork — Multi-Token Prediction)
 WORKDIR /opt/llama.cpp
 RUN git clone -b mtp-clean --single-branch https://github.com/am17an/llama.cpp.git .
 COPY llama-grammar.patch /tmp/llama-grammar.patch
 # build
 RUN git clean -xdf \
  && git submodule update --recursive \
  && patch -p1 < /tmp/llama-grammar.patch \
  && cmake -S . -B build \
  -DGGML_HIP=ON \
  -DCMAKE_HIP_FLAGS="--rocm-path=/opt/rocm -mllvm --amdgpu-unroll-threshold-local=600" \
  -DAMDGPU_TARGETS=gfx1151 \
  -DCMAKE_BUILD_TYPE=Release \
  -DGGML_RPC=ON \
  -DLLAMA_HIP_UMA=ON \
  -DGGML_CUDA_ENABLE_UNIFIED_MEMORY=ON \
  -DROCM_PATH=/opt/rocm \
  -DHIP_PATH=/opt/rocm \
  -DHIP_PLATFORM=amd \
  && cmake --build build --config Release -- -j$(nproc) \
  && cmake --install build --config Release
 # libs
 RUN find /opt/llama.cpp/build -type f -name 'lib*.so*' -exec cp {} /usr/lib64/ \; \
  && ldconfig
 # helper
 COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
 RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
 # runtime stage
 FROM registry.fedoraproject.org/fedora-minimal:43
 # rocm 7.2.3 repo
 RUN <<'EOF'
 tee /etc/yum.repos.d/rocm.repo <<REPO
 [ROCm-7.2.3]
 name=ROCm7.2.3
 baseurl=https://repo.radeon.com/rocm/rhel10/7.2.3/main
 enabled=1
 priority=50
 gpgcheck=1
 gpgkey=https://repo.radeon.com/rocm/rocm.gpg.key
 REPO
 EOF
 # runtime deps
 RUN microdnf -y --nodocs --setopt=install_weak_deps=0 \
  --exclude='*sdk*' --exclude='*samples*' --exclude='*-doc*' --exclude='*-docs*' \
  install \
  bash ca-certificates libatomic libstdc++ libgcc libgomp sudo \
  hip-runtime-amd rocblas hipblas \
  rocminfo radeontop procps-ng \
  && microdnf clean all && rm -rf /var/cache/dnf/*
 # copy
 COPY --from=builder /usr/local/ /usr/local/
 COPY --from=builder /opt/llama.cpp/build/bin/rpc-* /usr/local/bin/
 # ld
 RUN echo "/usr/local/lib"  > /etc/ld.so.conf.d/local.conf \
  && echo "/usr/local/lib64" >> /etc/ld.so.conf.d/local.conf \
  && ldconfig \
  && cp -n /usr/local/lib/libllama*.so* /usr/lib64/ 2>/dev/null || true \
  && ldconfig
 # helper
 COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
 RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
 # profile
 RUN printf '%s\n' \
  > /etc/profile.d/rocm.sh && chmod +x /etc/profile.d/rocm.sh \
  && echo 'source /etc/profile.d/rocm.sh' >> /etc/bashrc
 # shell
 CMD ["/bin/bash"]
@@ -0,0 +1,68 @@
 # build stage
 FROM registry.fedoraproject.org/fedora:43 AS builder
 # deps
 RUN dnf -y --nodocs --setopt=install_weak_deps=False install \
  git vim \
  make gcc cmake ninja-build lld clang clang-devel compiler-rt libcurl-devel \
  vulkan-loader-devel vulkaninfo mesa-vulkan-drivers \
  spirv-headers-devel radeontop glslc patch \
  && dnf clean all && rm -rf /var/cache/dnf/*
 # llama.cpp (am17an mtp-clean fork — Multi-Token Prediction)
 WORKDIR /opt/llama.cpp
 RUN git clone -b mtp-clean --single-branch https://github.com/am17an/llama.cpp.git .
 COPY llama-grammar.patch /tmp/llama-grammar.patch
 # build
 RUN git clean -xdf \
  && git submodule update --recursive \
  && patch -p1 < /tmp/llama-grammar.patch \
  && cmake -S . -B build -G Ninja \
  -DGGML_VULKAN=ON \
  -DCMAKE_BUILD_TYPE=Release \
  -DGGML_RPC=ON \
  -DCMAKE_INSTALL_PREFIX=/usr \
  -DLLAMA_BUILD_TESTS=OFF \
  -DLLAMA_BUILD_EXAMPLES=ON \
  -DLLAMA_BUILD_SERVER=ON \
  && cmake --build build --config Release \
  && cmake --install build --config Release
 # libs
 RUN find /opt/llama.cpp/build -type f -name 'lib*.so*' -exec cp {} /usr/lib64/ \; \
  && ldconfig
 # helper
 COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
 RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
 # runtime stage
 FROM registry.fedoraproject.org/fedora-minimal:43
 # runtime deps
 RUN microdnf -y --nodocs --setopt=install_weak_deps=0 install \
  bash ca-certificates libatomic libstdc++ libgcc \
  vulkan-loader vulkan-loader-devel vulkaninfo mesa-vulkan-drivers radeontop procps-ng \
  && microdnf clean all && rm -rf /var/cache/dnf/*
 # copy
 COPY --from=builder /usr/ /usr/
 COPY --from=builder /usr/local/ /usr/local/
 COPY --from=builder /opt/llama.cpp/build/bin/rpc-* /usr/local/bin/
 # ld
 RUN echo "/usr/local/lib"  > /etc/ld.so.conf.d/local.conf \
  && echo "/usr/local/lib64" >> /etc/ld.so.conf.d/local.conf \
  && ldconfig \
  && cp -n /usr/local/lib/libllama*.so* /usr/lib64/ 2>/dev/null || true \
  && ldconfig
 # helper
 COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
 RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
 # shell
 CMD ["/bin/bash"]