add rocm-7.2.1-pr21344 toolbox (gfx1151 MMQ/MMVQ tile + nwarp tuning)

Adds a new toolbox variant based on PR #21344 (pedapudi/llama.cpp@gfx1151-opt) which tunes MMQ tile sizes (x_max=48, y=64) and warp counts (nwarps=4) for RDNA3_5 gfx1151, yielding up to +100% prefill throughput at small batch sizes. Also adds BMI2/FMA/F16C CPU SIMD flags and GGML_CUDA_FA_ALL_QUANTS=ON to match the benchmark build used in the PR. Wire up CI (build matrix + prune), the refresh script, and run_benchmarks.sh so results land alongside rocm-7.2.1.
2026-04-15 09:23:58 +01:00
parent 14fae26ad0
commit 2c2c36d3da
6 changed files with 154 additions and 2 deletions
@@ -0,0 +1,124 @@
+# build stage
+# Based on Dockerfile.rocm-7.2.1, but clones pedapudi/llama.cpp@gfx1151-opt
+# (PR #21344: gfx1151 nwarps, tile sizing to curb VGPR pressure)
+FROM registry.fedoraproject.org/fedora:43 AS builder
+
+# rocm 7.2.1 repo
+RUN <<'EOF'
+tee /etc/yum.repos.d/rocm.repo <<REPO
+[ROCm-7.2.1]
+name=ROCm7.2.1
+baseurl=https://repo.radeon.com/rocm/rhel10/7.2.1/main
+enabled=1
+priority=50
+gpgcheck=1
+gpgkey=https://repo.radeon.com/rocm/rocm.gpg.key
+REPO
+EOF
+
+# deps
+RUN dnf -y --nodocs --setopt=install_weak_deps=False \
+  --exclude='*sdk*' --exclude='*samples*' --exclude='*-doc*' --exclude='*-docs*' \
+  install \
+  make gcc cmake lld clang clang-devel compiler-rt libcurl-devel ninja-build \
+  rocm-llvm rocm-device-libs hip-runtime-amd hip-devel \
+  rocblas rocblas-devel hipblas hipblas-devel rocm-cmake libomp-devel libomp \
+  rocminfo radeontop \
+  git-core vim sudo rsync patch \
+  && dnf clean all && rm -rf /var/cache/dnf/*
+
+# rocm env
+ENV ROCM_PATH=/opt/rocm \
+  HIP_PATH=/opt/rocm \
+  HIP_CLANG_PATH=/opt/rocm/llvm/bin \
+  HIP_DEVICE_LIB_PATH=/opt/rocm/amdgcn/bitcode \
+  PATH=/opt/rocm/bin:/opt/rocm/llvm/bin:$PATH
+
+# llama.cpp — PR #21344 fork (gfx1151 MMQ/MMVQ tile + nwarp tuning)
+WORKDIR /opt/llama.cpp
+ARG REPO=https://github.com/pedapudi/llama.cpp.git
+ARG BRANCH=gfx1151-opt
+RUN git clone -b ${BRANCH} --single-branch --recursive ${REPO} .
+
+COPY llama-grammar.patch /tmp/llama-grammar.patch
+
+# build
+RUN git clean -xdf \
+  && git submodule update --recursive \
+  && patch -p1 < /tmp/llama-grammar.patch \
+  && cmake -S . -B build \
+  -DGGML_HIP=ON \
+  -DCMAKE_HIP_COMPILER=${HIP_CLANG_PATH}/clang \
+  -DCMAKE_HIP_FLAGS="--rocm-path=/opt/rocm -mllvm --amdgpu-unroll-threshold-local=600" \
+  -DAMDGPU_TARGETS=gfx1151 \
+  -DCMAKE_BUILD_TYPE=Release \
+  -DGGML_RPC=ON \
+  -DLLAMA_HIP_UMA=ON \
+  -DGGML_CUDA_ENABLE_UNIFIED_MEMORY=ON \
+  -DGGML_BMI2=ON \
+  -DGGML_FMA=ON \
+  -DGGML_F16C=ON \
+  -DGGML_CUDA_FA_ALL_QUANTS=ON \
+  -DLLAMA_BUILD_TESTS=OFF \
+  -DLLAMA_BUILD_EXAMPLES=OFF \
+  -DROCM_PATH=/opt/rocm \
+  -DHIP_PATH=/opt/rocm \
+  -DHIP_PLATFORM=amd \
+  && cmake --build build --config Release -- -j$(nproc) \
+  && cmake --install build --config Release
+
+# libs
+RUN find /opt/llama.cpp/build -type f -name 'lib*.so*' -exec cp {} /usr/lib64/ \; \
+  && ldconfig
+
+# helper
+COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
+RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
+
+# runtime stage
+FROM registry.fedoraproject.org/fedora-minimal:43
+
+# rocm 7.2.1 repo
+RUN <<'EOF'
+tee /etc/yum.repos.d/rocm.repo <<REPO
+[ROCm-7.2.1]
+name=ROCm7.2.1
+baseurl=https://repo.radeon.com/rocm/rhel10/7.2.1/main
+enabled=1
+priority=50
+gpgcheck=1
+gpgkey=https://repo.radeon.com/rocm/rocm.gpg.key
+REPO
+EOF
+
+# runtime deps
+RUN microdnf -y --nodocs --setopt=install_weak_deps=0 \
+  --exclude='*sdk*' --exclude='*samples*' --exclude='*-doc*' --exclude='*-docs*' \
+  install \
+  bash ca-certificates libatomic libstdc++ libgcc libgomp sudo \
+  hip-runtime-amd rocblas hipblas \
+  rocminfo radeontop procps-ng \
+  && microdnf clean all && rm -rf /var/cache/dnf/*
+
+# copy
+COPY --from=builder /usr/local/ /usr/local/
+COPY --from=builder /opt/llama.cpp/build/bin/rpc-* /usr/local/bin/
+
+# ld
+RUN echo "/usr/local/lib"  > /etc/ld.so.conf.d/local.conf \
+  && echo "/usr/local/lib64" >> /etc/ld.so.conf.d/local.conf \
+  && ldconfig \
+  && cp -n /usr/local/lib/libllama*.so* /usr/lib64/ 2>/dev/null || true \
+  && ldconfig
+
+# helper
+COPY gguf-vram-estimator.py /usr/local/bin/gguf-vram-estimator.py
+RUN chmod +x /usr/local/bin/gguf-vram-estimator.py
+
+# profile
+RUN printf '%s\n' \
+  > /etc/profile.d/rocm.sh && chmod +x /etc/profile.d/rocm.sh \
+  && echo 'source /etc/profile.d/rocm.sh' >> /etc/bashrc
+
+# shell
+CMD ["/bin/bash"]